From 63d1ba4e74fe6a22d51bc39330fe187e3e7a6e11 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 27 Feb 2026 22:24:50 -0800 Subject: [PATCH 001/113] Fix bugs causing primal simplex to cycle. Enable primal simplex cleanup after dual simplex Fixed the following bugs that were causing primal simplex to cycle: 1) Swapped input/output arguments in b_solve() 2) Incorrectly setting variable status of leaving variable 3) Primal step length was not limited by bounds of entering variable. Also fixed a bug/typo where the basis was reorderd twice after factorization. Added code to switch to phase I if we loose primal feasibility, and switch back to phase II once feasibility is regained. Tested on NETLIB LPs. Only 2 LPs pilot87 and pilot_ja need primal simplex to remove perturbations at the end of the dual simplex solve. Tested on the 14 MIPLIB root relaxations that need primal simplex to remove perturbations at the end of the dual simplex solve. --- cpp/src/dual_simplex/phase2.cpp | 7 +- cpp/src/dual_simplex/primal.cpp | 311 +++++++++++++++++++++----------- cpp/src/dual_simplex/solve.cpp | 8 +- 3 files changed, 219 insertions(+), 107 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 2e3c1e05c5..c15f7f554c 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2367,6 +2367,9 @@ void prepare_optimality(i_t info, perturbation = 0.0; } else { settings.log.printf("Failed to remove perturbation of %.2e.\n", perturbation); + settings.log.printf("Unperturbed dual infeasibility: %.2e\n", dual_infeas); + settings.log.printf("Objective: %+.16e\n", sol.user_objective); + settings.log.printf("Num updates: %d\n", ft.num_updates()); } } } @@ -3737,10 +3740,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, 100.0 * dense_delta_z / (sparse_delta_z + dense_delta_z)); ft.print_stats(); } - if (settings.inside_mip == 1 && settings.concurrent_halt != nullptr) { - settings.log.debug("Setting concurrent halt in Dual Simplex Phase 2\n"); - *settings.concurrent_halt = 1; - } } return status; } diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 78c7107ca3..ec7a9bc25c 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -21,7 +21,6 @@ namespace { template void set_primal_variables_on_bounds(const lp_problem_t& lp, const simplex_solver_settings_t& settings, - const std::vector& z, std::vector& vstatus, std::vector& x) { @@ -158,7 +157,9 @@ i_t ratio_test(const lp_problem_t& lp, std::vector& x, std::vector& delta_x, f_t& step_length, - i_t& basic_leaving) + i_t& basic_leaving, + i_t entering_index, + i_t direction) { const i_t m = lp.num_rows; const i_t n = lp.num_cols; @@ -166,28 +167,51 @@ i_t ratio_test(const lp_problem_t& lp, i_t leaving_index = -1; f_t min_val = inf; constexpr f_t pivot_tol = 1e-8; + + // Entering variable can hit its opposite bound: limit step by that + if (direction > 0 && lp.upper[entering_index] < inf) { + const f_t limit = lp.upper[entering_index] - x[entering_index]; + if (limit >= 0 && limit < min_val) { + min_val = limit; + leaving_index = -1; // no basic leaves; will be handled by caller + basic_leaving = -1; + } + } else if (direction < 0 && lp.lower[entering_index] > -inf) { + const f_t limit = x[entering_index] - lp.lower[entering_index]; + if (limit >= 0 && limit < min_val) { + min_val = limit; + leaving_index = -1; + basic_leaving = -1; + } + } + for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; if (delta_x[j] == 0.0) { continue; } - if (lp.lower[j] > -inf && x[j] >= lp.lower[j] && delta_x[j] < -pivot_tol) { + if (lp.lower[j] > -inf && delta_x[j] < -pivot_tol) { // xj + step * delta_x[j] >= lp.lower[j] // step * delta_x[j] >= lp.lower[j] - x[j] // step <= (lp.lower[j] - x[j]) / delta_x[j], delta_x[j] < 0 const f_t neum = lp.lower[j] - x[j]; f_t ratio = neum / delta_x[j]; - if (ratio < min_val) { + // Already below lower and moving further (delta_x < 0): cap step at 0 + if (x[j] < lp.lower[j]) { ratio = 0; } + if (ratio >= 0 && ratio < min_val) { min_val = ratio; basic_leaving = k; leaving_index = j; } } - if (lp.upper[j] < inf && x[j] <= lp.upper[j] && delta_x[j] > pivot_tol) { + if (lp.upper[j] < inf && delta_x[j] > pivot_tol) { // xj + step * delta_x[j] <= lp.upper[j] // step * delta_x[j] <= lp.upper[j] - x[j] // step <= (lp.upper[j] - x[j]) / delta_x[j], delta_x[j] > 0 const f_t neum = lp.upper[j] - x[j]; f_t ratio = neum / delta_x[j]; - if (ratio < min_val) { + // If x[j] > upper (slightly infeasible), ratio < 0; skip so we don't use it. + // But if we're already above upper and would move further (delta_x > 0), cap step at 0. + if (x[j] > lp.upper[j]) { ratio = 0; } + if (ratio >= 0 && ratio < min_val) { min_val = ratio; basic_leaving = k; leaving_index = j; @@ -207,7 +231,7 @@ f_t primal_infeasibility(const lp_problem_t& lp, const i_t n = lp.num_cols; f_t primal_inf = 0; for (i_t j = 0; j < n; ++j) { - if (x[j] < lp.lower[j]) { + if (x[j] < lp.lower[j] - settings.primal_tol) { // x_j < l_j => -x_j > -l_j => -x_j + l_j > 0 const f_t infeas = -x[j] + lp.lower[j]; primal_inf += infeas; @@ -221,7 +245,7 @@ f_t primal_infeasibility(const lp_problem_t& lp, vstatus[j]); } } - if (x[j] > lp.upper[j]) { + if (x[j] > lp.upper[j] + settings.primal_tol) { // x_j > u_j => x_j - u_j > 0 const f_t infeas = x[j] - lp.upper[j]; primal_inf += infeas; @@ -239,12 +263,69 @@ f_t primal_infeasibility(const lp_problem_t& lp, return primal_inf; } +template +void compute_phase1_objective(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& x, + std::vector& objective) +{ + const i_t n = lp.num_cols; + for (i_t j = 0; j < n; ++j) { + if (x[j] < lp.lower[j] - settings.primal_tol) { + objective[j] = -1.0; + } else if (x[j] > lp.upper[j] + settings.primal_tol) { + objective[j] = 1.0; + } else { + objective[j] = 0.0; + } + } +} + +template +void compute_dual_variables(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& objective, + const std::vector& basic_list, + const std::vector& nonbasic_list, + basis_update_t& ft, + std::vector& c_basic, + std::vector& y, + std::vector& z) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + // Solve for y such that B'*y = c_B + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + c_basic[k] = objective[j]; + } + ft.b_transpose_solve(c_basic, y); + // zN = cN - N'*y + for (i_t k = 0; k < n - m; k++) { + const i_t j = nonbasic_list[k]; + // z_j <- c_j + z[j] = objective[j]; + + // z_j <- z_j - A(:, j)'*y + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + f_t dot = 0.0; + for (i_t p = col_start; p < col_end; ++p) { + dot += lp.A.x[p] * y[lp.A.i[p]]; + } + z[j] -= dot; + } + // zB = 0 + for (i_t k = 0; k < m; ++k) { + z[basic_list[k]] = 0.0; + } +} + } // namespace // Note this implementation of primal simplex is experimental // It is meant only to serve as a method to remove the perturbation to the objective // after dual simplex has found a primal feasible solution -// The implementation currently cycles. So is not enabled at this time. template primal_status_t primal_phase2(i_t phase, f_t start_time, @@ -308,6 +389,7 @@ primal_status_t primal_phase2(i_t phase, slacks_needed, work_estimate); if (rank == CONCURRENT_HALT_RETURN) { + settings.log.printf("Concurrent halt in primal phase2\n"); return primal_status_t::CONCURRENT_LIMIT; } else if (rank == TIME_LIMIT_RETURN) { return primal_status_t::TIME_LIMIT; @@ -352,46 +434,8 @@ primal_status_t primal_phase2(i_t phase, } } reorder_basic_list(q, basic_list); - reorder_basic_list(q, basic_list); basis_update_t ft(L, U, p); - std::vector c_basic(m); - for (i_t k = 0; k < m; ++k) { - const i_t j = basic_list[k]; - c_basic[k] = lp.objective[j]; - } - - // Solve B'*y = cB - ft.b_transpose_solve(c_basic, y); - settings.log.printf( - "|| y || %e || cB || %e\n", vector_norm_inf(y), vector_norm_inf(c_basic)); - - // zN = cN - N'*y - for (i_t k = 0; k < n - m; k++) { - const i_t j = nonbasic_list[k]; - // z_j <- c_j - z[j] = lp.objective[j]; - - // z_j <- z_j - A(:, j)'*y - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - f_t dot = 0.0; - for (i_t p = col_start; p < col_end; ++p) { - dot += lp.A.x[p] * y[lp.A.i[p]]; - } - z[j] -= dot; - } - // zB = 0 - for (i_t k = 0; k < m; ++k) { - z[basic_list[k]] = 0.0; - } - settings.log.printf("|| z || %e\n", vector_norm_inf(z)); - - set_primal_variables_on_bounds(lp, settings, z, vstatus, x); - - const f_t init_dual_inf = dual_infeasibility(lp, vstatus, z); - settings.log.printf("Initial dual infeasibility %e\n", init_dual_inf); - std::vector rhs = lp.rhs; // rhs = b - sum_{j : x_j = l_j} A(:, j) l(j) - sum_{j : x_j = u_j} A(:, j) * // u(j) @@ -412,6 +456,7 @@ primal_status_t primal_phase2(i_t phase, const i_t j = basic_list[k]; x[j] = xB[k]; } + set_primal_variables_on_bounds(lp, settings, vstatus, x); settings.log.printf("|| x || %e\n", vector_norm2(x)); std::vector residual = lp.rhs; @@ -421,6 +466,23 @@ primal_status_t primal_phase2(i_t phase, f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); settings.log.printf("Initial primal infeasibility %e\n", primal_inf); + std::vector objective = lp.objective; + const f_t primal_tol = settings.primal_tol; + if (primal_inf > primal_tol) { + // We are primal infeasible. Switch to phase 1 + compute_phase1_objective(lp, settings, x, objective); + phase = 1; + } else { + phase = 2; + } + + std::vector c_basic(m); + compute_dual_variables(lp, settings, objective, basic_list, nonbasic_list, ft, c_basic, y, z); + settings.log.printf("|| z || %e\n", vector_norm_inf(z)); + + const f_t init_dual_inf = dual_infeasibility(lp, vstatus, z); + settings.log.printf("Initial dual infeasibility %e\n", init_dual_inf); + const i_t iter_limit = iter + 1000; std::vector delta_y(m); std::vector delta_z(n); @@ -434,16 +496,34 @@ primal_status_t primal_phase2(i_t phase, i_t entering_index = phase2_pricing(lp, z, nonbasic_list, vstatus, direction, nonbasic_entering, dual_inf); if (entering_index == -1) { - f_t obj = compute_objective(lp, x); - f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); - settings.log.printf( - "Optimal solution found. Objective %e. Dual infeas %e. Primal " - "infeasibility %e. Iterations %d\n", - compute_user_objective(lp, obj), - dual_inf, - primal_inf, - iter); - return primal_status_t::OPTIMAL; + if (phase == 2) { + f_t obj = compute_objective(lp, x); + f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); + settings.log.printf( + "Optimal solution found. Objective %+.16e. Dual infeas %e. Primal " + "infeasibility %e. Iterations %d\n", + compute_user_objective(lp, obj), + dual_inf, + primal_inf, + iter); + return primal_status_t::OPTIMAL; + } else { + primal_inf = primal_infeasibility(lp, settings, vstatus, x); + + if (primal_inf > primal_tol) { + settings.log.printf("Primal infeasibility %e. No entering\n", primal_inf); + return primal_status_t::NUMERICAL; + } else { + // Restore the objective to the original objective + objective = lp.objective; + phase = 2; + settings.log.printf("Switching to phase 2\n"); + compute_dual_variables( + lp, settings, objective, basic_list, nonbasic_list, ft, c_basic, y, z); + iter++; + continue; + } + } } std::vector scaled_delta_xB(m); @@ -473,70 +553,97 @@ primal_status_t primal_phase2(i_t phase, i_t basic_leaving; f_t step_length; - i_t leaving_index = ratio_test(lp, vstatus, basic_list, x, delta_x, step_length, basic_leaving); - if (leaving_index == -1) { + i_t leaving_index = ratio_test( + lp, vstatus, basic_list, x, delta_x, step_length, basic_leaving, entering_index, direction); + if (leaving_index == -1 && step_length >= inf) { settings.log.printf("No leaving variable. Primal unbounded?\n"); return primal_status_t::PRIMAL_UNBOUNDED; } - assert(step_length >= 0.0); - // Update the primal variables + const bool basis_updated = (leaving_index != -1); for (i_t j = 0; j < n; ++j) { x[j] += step_length * delta_x[j]; } + if (basis_updated) { + assert(step_length >= 0.0); + basic_list[basic_leaving] = entering_index; + nonbasic_list[nonbasic_entering] = leaving_index; + vstatus[entering_index] = variable_status_t::BASIC; + if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { + vstatus[leaving_index] = variable_status_t::NONBASIC_FIXED; + } else if (delta_x[leaving_index] < 0) { + vstatus[leaving_index] = variable_status_t::NONBASIC_LOWER; + } else { + vstatus[leaving_index] = variable_status_t::NONBASIC_UPPER; + } - // Update the factorization - ft.update(utilde, basic_leaving); - - // Update the basis - basic_list[basic_leaving] = entering_index; - nonbasic_list[nonbasic_entering] = leaving_index; - vstatus[entering_index] = variable_status_t::BASIC; - if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { - vstatus[leaving_index] = variable_status_t::NONBASIC_FIXED; - } else if (direction == 1) { - vstatus[leaving_index] = variable_status_t::NONBASIC_LOWER; + bool should_refactor = ft.num_updates() > settings.refactor_frequency; + if (!should_refactor) { + i_t recommend_refactor = ft.update(utilde, basic_leaving); + should_refactor = recommend_refactor == 1; + } + if (should_refactor) { + i_t rank = factorize_basis(lp.A, + settings, + basic_list, + start_time, + L, + U, + p, + pinv, + q, + deficient, + slacks_needed, + work_estimate); + if (rank == CONCURRENT_HALT_RETURN) { return primal_status_t::CONCURRENT_LIMIT; } + if (rank == TIME_LIMIT_RETURN) { return primal_status_t::TIME_LIMIT; } + if (rank < 0) { + settings.log.printf("Failed to refactor basis. Iteration %d\n", iter); + return primal_status_t::NUMERICAL; + } + if (rank != m) { + settings.log.printf("Failed to refactor basis. rank %d m %d\n", rank, m); + return primal_status_t::NUMERICAL; + } + reorder_basic_list(q, basic_list); + ft.reset(L, U, p); + } } else { - vstatus[leaving_index] = variable_status_t::NONBASIC_UPPER; + if (direction > 0) { + vstatus[entering_index] = variable_status_t::NONBASIC_UPPER; + x[entering_index] = lp.upper[entering_index]; + } else { + vstatus[entering_index] = variable_status_t::NONBASIC_LOWER; + x[entering_index] = lp.lower[entering_index]; + } } - // Solve for y such that B'*y = c_B - for (i_t k = 0; k < m; ++k) { - const i_t j = basic_list[k]; - c_basic[k] = lp.objective[j]; - } - ft.b_transpose_solve(y, c_basic); - // zN = cN - N'*y - for (i_t k = 0; k < n - m; k++) { - const i_t j = nonbasic_list[k]; - // z_j <- c_j - z[j] = lp.objective[j]; - - // z_j <- z_j - A(:, j)'*y - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - f_t dot = 0.0; - for (i_t p = col_start; p < col_end; ++p) { - dot += lp.A.x[p] * y[lp.A.i[p]]; - } - z[j] -= dot; + // Check if we need to switch to phase 1 + const f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); + if (primal_inf > primal_tol) { + compute_phase1_objective(lp, settings, x, objective); + phase = 1; + } else if (phase == 1) { + objective = lp.objective; + phase = 2; } - // zB = 0 - for (i_t k = 0; k < m; ++k) { - z[basic_list[k]] = 0.0; + + if (basis_updated || primal_inf > primal_tol) { + compute_dual_variables(lp, settings, objective, basic_list, nonbasic_list, ft, c_basic, y, z); } - const f_t obj = compute_objective(lp, x); - const f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); - settings.log.printf("%3d %.10e %.2e %.2e %.2e %d %d\n", + const f_t obj = compute_objective(lp, x); + dual_inf = dual_infeasibility(lp, vstatus, z); + settings.log.printf("%3d %.10e %8.2e %8.2e %8.2e %8d %8d %d %.2f\n", iter, compute_user_objective(lp, obj), primal_inf, dual_inf, - step_length, + step_length == 0.0 ? 0.0 : step_length, entering_index, - leaving_index); - + leaving_index, + phase, + toc(start_time)); iter++; } diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index 697af9e869..c13e35c525 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -288,9 +288,15 @@ lp_status_t solve_linear_program_with_advanced_basis( edge_norms, work_unit_context); } - constexpr bool primal_cleanup = false; + constexpr bool primal_cleanup = true; if (status == dual_status_t::OPTIMAL && primal_cleanup) { + settings.log.printf("Running primal cleanup\n"); primal_phase2(2, start_time, lp, settings, vstatus, solution, iter); + // TODO: We need to update ft if the basis changed + } + if (settings.inside_mip && settings.concurrent_halt != nullptr) { + settings.log.printf("Setting concurrent halt to 1 inside_mip\n"); + *settings.concurrent_halt = 1; } if (status == dual_status_t::OPTIMAL) { std::vector unscaled_x(lp.num_cols); From 8b1e60bcf2d3522113ce7b09e088addf1f234922 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 21 Jul 2026 11:48:59 -0700 Subject: [PATCH 002/113] Primal simplex pivots on dual degenerate problems to reduce integer infeasibility --- cpp/src/branch_and_bound/branch_and_bound.cpp | 200 ++++++++++++++++++ cpp/src/branch_and_bound/branch_and_bound.hpp | 9 + cpp/src/dual_simplex/primal.cpp | 162 +++++++------- cpp/src/dual_simplex/primal.hpp | 13 ++ cpp/src/dual_simplex/solve.cpp | 2 +- 5 files changed, 311 insertions(+), 75 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index e4ce4dfd7b..0a3629e7f8 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -23,6 +23,7 @@ #include #include #include +#include #include #include #include @@ -88,6 +89,7 @@ i_t fractional_variables(const simplex_solver_settings_t& settings, { const i_t n = x.size(); assert(x.size() == var_types.size()); + fractional.clear(); for (i_t j = 0; j < n; ++j) { if (is_fractional(x[j], var_types[j], settings.integer_tol)) { fractional.push_back(j); } } @@ -763,6 +765,9 @@ void branch_and_bound_t::set_final_solution(mip_solution_t& exploration_stats_.lexical_reduction_fixings_applied.load(), exploration_stats_.lexical_reduction_pruned_nodes.load()); } + if (integer_pivots_.load() > 0) { + settings_.log.print_format("Number of integer pivots: {}\n", integer_pivots_.load()); + } if (gap <= settings_.absolute_mip_gap_tol || gap_rel <= settings_.relative_mip_gap_tol) { solver_status_ = mip_status_t::OPTIMAL; @@ -1545,6 +1550,20 @@ dual_status_t branch_and_bound_t::solve_node_lp( stats.total_lp_solve_time += toc(lp_start_time); stats.total_lp_iters += node_iter; + + if (lp_status == dual_status_t::OPTIMAL) { + std::vector fractional; + i_t num_fractional = + fractional_variables(settings_, worker->leaf_solution.x, var_types_, fractional); + pivot_out_integer_variables(worker->leaf_problem, + worker->basic_list, + worker->nonbasic_list, + worker->leaf_vstatus, + worker->leaf_solution, + worker->basis_factors, + num_fractional, + fractional); + } } } @@ -2390,6 +2409,19 @@ auto branch_and_bound_t::do_cut_pass( } root_objective_ = compute_objective(original_lp_, root_relax_soln_.x); + // Refresh fractional info after re-solving with cuts; the pre-cut count is stale. + num_fractional = + fractional_variables(settings_, root_relax_soln_.x, var_types_, fractional); + + pivot_out_integer_variables(original_lp_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional); + if (settings_.benchmark_info_ptr != nullptr) { settings_.benchmark_info_ptr->root_lp_with_cuts = compute_user_objective(original_lp_, root_objective_); @@ -2455,6 +2487,165 @@ auto branch_and_bound_t::do_cut_pass( return {cut_pass_action_t::CONTINUE, mip_status_t::UNSET}; } +template +void branch_and_bound_t::pivot_out_integer_variables( + const simplex::lp_problem_t& lp, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + i_t& num_fractional, + std::vector& fractional) +{ + std::vector zero_reduced_costs_vars; + std::vector zero_reduced_costs_vars_nonbasic_index; + for (i_t k = 0; k < static_cast(nonbasic_list.size()); k++) { + const i_t j = nonbasic_list[k]; + if (std::abs(solution.z[j]) <= 1e-10) { + zero_reduced_costs_vars.push_back(j); + zero_reduced_costs_vars_nonbasic_index.push_back(k); + } + } + if (zero_reduced_costs_vars.empty()) { return; } + + lp_solution_t soln_copy = solution; + std::vector basic_list_copy = basic_list; + std::vector nonbasic_list_copy = nonbasic_list; + std::vector vstatus_copy = vstatus; + simplex::basis_update_mpf_t basis_update_copy = basis_update; + + const i_t start_num_fractional = num_fractional; + + for (i_t k = 0; k < static_cast(zero_reduced_costs_vars.size()); k++) { + const i_t j = zero_reduced_costs_vars[k]; + if (var_types_[j] == variable_type_t::INTEGER) { continue; } + if (vstatus_copy[j] == variable_status_t::BASIC) { continue; } + + const i_t direction = + (vstatus_copy[j] == variable_status_t::NONBASIC_LOWER || + vstatus_copy[j] == variable_status_t::NONBASIC_FIXED) + ? 1 + : -1; + const i_t entering_index = j; + const i_t nonbasic_entering = zero_reduced_costs_vars_nonbasic_index[k]; + if (nonbasic_list_copy[nonbasic_entering] != j) { continue; } + + // Solve B * dxB = A(:, j) so utilde is valid for the MPF update. + // Apply direction when forming delta_x (same convention as primal_phase2). + sparse_vector_t rhs(lp.A, j); + sparse_vector_t delta_xB; + sparse_vector_t utilde_sparse; + basis_update_copy.b_solve(rhs, delta_xB, utilde_sparse); + + std::vector delta_xB_dense; + delta_xB.to_dense(delta_xB_dense); + std::vector delta_x(lp.num_cols, 0.0); + for (i_t h = 0; h < static_cast(basic_list_copy.size()); h++) { + delta_x[basic_list_copy[h]] = -direction * delta_xB_dense[h]; + } + delta_x[j] = direction; + + f_t step_length; + i_t basic_leaving; + const i_t leaving_index = simplex::primal_ratio_test(lp, + settings_, + vstatus_copy, + basic_list_copy, + soln_copy.x, + delta_x, + step_length, + basic_leaving, + entering_index, + direction); + bool binding_integer = + leaving_index != -1 && + is_fractional(soln_copy.x[leaving_index], var_types_[leaving_index], settings_.integer_tol); + if (!binding_integer) { continue; } + + std::vector test_x = soln_copy.x; + i_t integer_destroyed = 0; + for (i_t h = 0; h < lp.num_cols; ++h) { + test_x[h] += step_length * delta_x[h]; + if (var_types_[h] != variable_type_t::INTEGER) { continue; } + const bool was_fractional = + is_fractional(soln_copy.x[h], var_types_[h], settings_.integer_tol); + const bool now_fractional = + is_fractional(test_x[h], var_types_[h], settings_.integer_tol); + if (now_fractional && !was_fractional) { + integer_destroyed++; + } else if (!now_fractional && was_fractional) { + integer_destroyed--; + } + } + // Require a strict net decrease in fractional integers. + if (integer_destroyed >= 0) { continue; } + + soln_copy.x = test_x; + basic_list_copy[basic_leaving] = entering_index; + nonbasic_list_copy[nonbasic_entering] = leaving_index; + vstatus_copy[entering_index] = variable_status_t::BASIC; + if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { + vstatus_copy[leaving_index] = variable_status_t::NONBASIC_FIXED; + } else if (delta_x[leaving_index] < 0) { + vstatus_copy[leaving_index] = variable_status_t::NONBASIC_LOWER; + } else { + vstatus_copy[leaving_index] = variable_status_t::NONBASIC_UPPER; + } + + const i_t m = lp.num_rows; + sparse_vector_t es_sparse(m, 1); + es_sparse.i[0] = basic_leaving; + es_sparse.x[0] = 1.0; + sparse_vector_t UTsol_sparse(m, 1); + sparse_vector_t solution_sparse(m, 1); + basis_update_copy.b_transpose_solve(es_sparse, solution_sparse, UTsol_sparse); + const i_t recommend_refactor = + basis_update_copy.update(utilde_sparse, UTsol_sparse, basic_leaving); + if (recommend_refactor == 1) { + csc_matrix_t L(m, m, 1); + csc_matrix_t U(m, m, 1); + std::vector pinv(m); + std::vector p(m); + std::vector q(m); + std::vector deficient; + std::vector slacks_needed; + f_t factorize_work_estimate = 0.0; + const i_t rank = factorize_basis(lp.A, + settings_, + basic_list_copy, + exploration_stats_.start_time, + L, + U, + p, + pinv, + q, + deficient, + slacks_needed, + factorize_work_estimate); + if (rank == CONCURRENT_HALT_RETURN || rank == TIME_LIMIT_RETURN) { return; } + if (rank < 0 || rank != lp.num_rows) { return; } + simplex::reorder_basic_list(q, basic_list_copy); + basis_update_copy.reset(L, U, p); + } + } + + std::vector new_fractional; + const i_t num_new_fractional = + fractional_variables(settings_, soln_copy.x, var_types_, new_fractional); + if (num_new_fractional < start_num_fractional) { + i_t num_integer_increased = start_num_fractional - num_new_fractional; + integer_pivots_.fetch_add(num_integer_increased, std::memory_order_release); + num_fractional = num_new_fractional; + fractional = new_fractional; + basic_list = basic_list_copy; + nonbasic_list = nonbasic_list_copy; + vstatus = vstatus_copy; + basis_update = basis_update_copy; + solution = soln_copy; + } +} + template mip_status_t branch_and_bound_t::solve(mip_solution_t& solution) { @@ -2656,6 +2847,15 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut is_running_ = true; lower_bound_numerical_ = inf; + pivot_out_integer_variables(original_lp_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional); + if (num_fractional != 0 && settings_.max_cut_passes > 0) { print_table_header(); } cut_pool_t cut_pool(original_lp_.num_cols, settings_); diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 12c93fcd91..1fe2b2b897 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -309,6 +309,15 @@ class branch_and_bound_t { const std::vector& leaf_solution, i_t leaf_depth, search_strategy_t thread_type); + omp_atomic_t integer_pivots_{0}; + void pivot_out_integer_variables(const simplex::lp_problem_t& lp, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + simplex::lp_solution_t& soln, + simplex::basis_update_mpf_t& basis_update, + i_t& num_fractional, + std::vector& fractional); // Repairs low-quality solutions from the heuristics, if it is applicable. void repair_heuristic_solutions(); diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index ec7a9bc25c..aca2e785d2 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -150,78 +150,6 @@ i_t phase2_pricing(const lp_problem_t& lp, return entering_index; } -template -i_t ratio_test(const lp_problem_t& lp, - const std::vector& vstatus, - const std::vector& basic_list, - std::vector& x, - std::vector& delta_x, - f_t& step_length, - i_t& basic_leaving, - i_t entering_index, - i_t direction) -{ - const i_t m = lp.num_rows; - const i_t n = lp.num_cols; - basic_leaving = -1; - i_t leaving_index = -1; - f_t min_val = inf; - constexpr f_t pivot_tol = 1e-8; - - // Entering variable can hit its opposite bound: limit step by that - if (direction > 0 && lp.upper[entering_index] < inf) { - const f_t limit = lp.upper[entering_index] - x[entering_index]; - if (limit >= 0 && limit < min_val) { - min_val = limit; - leaving_index = -1; // no basic leaves; will be handled by caller - basic_leaving = -1; - } - } else if (direction < 0 && lp.lower[entering_index] > -inf) { - const f_t limit = x[entering_index] - lp.lower[entering_index]; - if (limit >= 0 && limit < min_val) { - min_val = limit; - leaving_index = -1; - basic_leaving = -1; - } - } - - for (i_t k = 0; k < m; ++k) { - const i_t j = basic_list[k]; - if (delta_x[j] == 0.0) { continue; } - if (lp.lower[j] > -inf && delta_x[j] < -pivot_tol) { - // xj + step * delta_x[j] >= lp.lower[j] - // step * delta_x[j] >= lp.lower[j] - x[j] - // step <= (lp.lower[j] - x[j]) / delta_x[j], delta_x[j] < 0 - const f_t neum = lp.lower[j] - x[j]; - f_t ratio = neum / delta_x[j]; - // Already below lower and moving further (delta_x < 0): cap step at 0 - if (x[j] < lp.lower[j]) { ratio = 0; } - if (ratio >= 0 && ratio < min_val) { - min_val = ratio; - basic_leaving = k; - leaving_index = j; - } - } - if (lp.upper[j] < inf && delta_x[j] > pivot_tol) { - // xj + step * delta_x[j] <= lp.upper[j] - // step * delta_x[j] <= lp.upper[j] - x[j] - // step <= (lp.upper[j] - x[j]) / delta_x[j], delta_x[j] > 0 - const f_t neum = lp.upper[j] - x[j]; - f_t ratio = neum / delta_x[j]; - // If x[j] > upper (slightly infeasible), ratio < 0; skip so we don't use it. - // But if we're already above upper and would move further (delta_x > 0), cap step at 0. - if (x[j] > lp.upper[j]) { ratio = 0; } - if (ratio >= 0 && ratio < min_val) { - min_val = ratio; - basic_leaving = k; - leaving_index = j; - } - } - } - step_length = min_val; - return leaving_index; -} - template f_t primal_infeasibility(const lp_problem_t& lp, const simplex_solver_settings_t& settings, @@ -323,6 +251,80 @@ void compute_dual_variables(const lp_problem_t& lp, } // namespace + +template +i_t primal_ratio_test(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& vstatus, + const std::vector& basic_list, + std::vector& x, + std::vector& delta_x, + f_t& step_length, + i_t& basic_leaving, + i_t entering_index, + i_t direction) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + basic_leaving = -1; + i_t leaving_index = -1; + f_t min_val = inf; + constexpr f_t pivot_tol = 1e-8; + + // Entering variable can hit its opposite bound: limit step by that + if (direction > 0 && lp.upper[entering_index] < inf) { + const f_t limit = lp.upper[entering_index] - x[entering_index]; + if (limit >= 0 && limit < min_val) { + min_val = limit; + leaving_index = -1; // no basic leaves; will be handled by caller + basic_leaving = -1; + } + } else if (direction < 0 && lp.lower[entering_index] > -inf) { + const f_t limit = x[entering_index] - lp.lower[entering_index]; + if (limit >= 0 && limit < min_val) { + min_val = limit; + leaving_index = -1; + basic_leaving = -1; + } + } + + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + if (delta_x[j] == 0.0) { continue; } + if (lp.lower[j] > -inf && delta_x[j] < -pivot_tol) { + // xj + step * delta_x[j] >= lp.lower[j] + // step * delta_x[j] >= lp.lower[j] - x[j] + // step <= (lp.lower[j] - x[j]) / delta_x[j], delta_x[j] < 0 + const f_t neum = lp.lower[j] - x[j]; + f_t ratio = neum / delta_x[j]; + // Already below lower and moving further (delta_x < 0): cap step at 0 + if (x[j] < lp.lower[j]) { ratio = 0; } + if (ratio >= 0 && ratio < min_val) { + min_val = ratio; + basic_leaving = k; + leaving_index = j; + } + } + if (lp.upper[j] < inf && delta_x[j] > pivot_tol) { + // xj + step * delta_x[j] <= lp.upper[j] + // step * delta_x[j] <= lp.upper[j] - x[j] + // step <= (lp.upper[j] - x[j]) / delta_x[j], delta_x[j] > 0 + const f_t neum = lp.upper[j] - x[j]; + f_t ratio = neum / delta_x[j]; + // If x[j] > upper (slightly infeasible), ratio < 0; skip so we don't use it. + // But if we're already above upper and would move further (delta_x > 0), cap step at 0. + if (x[j] > lp.upper[j]) { ratio = 0; } + if (ratio >= 0 && ratio < min_val) { + min_val = ratio; + basic_leaving = k; + leaving_index = j; + } + } + } + step_length = min_val; + return leaving_index; +} + // Note this implementation of primal simplex is experimental // It is meant only to serve as a method to remove the perturbation to the objective // after dual simplex has found a primal feasible solution @@ -553,8 +555,8 @@ primal_status_t primal_phase2(i_t phase, i_t basic_leaving; f_t step_length; - i_t leaving_index = ratio_test( - lp, vstatus, basic_list, x, delta_x, step_length, basic_leaving, entering_index, direction); + i_t leaving_index = primal_ratio_test( + lp, settings, vstatus, basic_list, x, delta_x, step_length, basic_leaving, entering_index, direction); if (leaving_index == -1 && step_length >= inf) { settings.log.printf("No leaving variable. Primal unbounded?\n"); return primal_status_t::PRIMAL_UNBOUNDED; @@ -654,6 +656,18 @@ primal_status_t primal_phase2(i_t phase, #ifdef DUAL_SIMPLEX_INSTANTIATE_DOUBLE +template +int primal_ratio_test(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& vstatus, + const std::vector& basic_list, + std::vector& x, + std::vector& delta_x, + double& step_length, + int& basic_leaving, + int entering_index, + int direction); + template primal_status_t primal_phase2( int phase, double start_time, diff --git a/cpp/src/dual_simplex/primal.hpp b/cpp/src/dual_simplex/primal.hpp index 930958a802..34ffbd8ba5 100644 --- a/cpp/src/dual_simplex/primal.hpp +++ b/cpp/src/dual_simplex/primal.hpp @@ -27,6 +27,19 @@ enum class primal_status_t { CONCURRENT_LIMIT = 6 }; + +template +i_t primal_ratio_test(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& vstatus, + const std::vector& basic_list, + std::vector& x, + std::vector& delta_x, + f_t& step_length, + i_t& basic_leaving, + i_t entering_index, + i_t direction); + template primal_status_t primal_phase2(i_t phase, f_t start_time, diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index c13e35c525..da6834f60f 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -288,7 +288,7 @@ lp_status_t solve_linear_program_with_advanced_basis( edge_norms, work_unit_context); } - constexpr bool primal_cleanup = true; + constexpr bool primal_cleanup = false; if (status == dual_status_t::OPTIMAL && primal_cleanup) { settings.log.printf("Running primal cleanup\n"); primal_phase2(2, start_time, lp, settings, vstatus, solution, iter); From 17b2725eb5c43f285737c9810c1603da92b8ed52 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 21 Jul 2026 15:26:47 -0700 Subject: [PATCH 003/113] First stab at using the feasibility pump on a reduced problem on the optimal face --- cpp/src/branch_and_bound/branch_and_bound.cpp | 233 +++++++++++++++++- cpp/src/branch_and_bound/branch_and_bound.hpp | 16 ++ 2 files changed, 240 insertions(+), 9 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 0a3629e7f8..064367b60e 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -28,6 +28,7 @@ #include #include #include +#include #include #include @@ -2487,28 +2488,232 @@ auto branch_and_bound_t::do_cut_pass( return {cut_pass_action_t::CONTINUE, mip_status_t::UNSET}; } + template -void branch_and_bound_t::pivot_out_integer_variables( - const simplex::lp_problem_t& lp, +bool branch_and_bound_t::check_for_dual_degeneracy( + const simplex::lp_solution_t& solution, + const std::vector& nonbasic_list, + std::vector& zero_reduced_costs_vars, + std::vector& zero_reduced_costs_vars_nonbasic_index) +{ + const i_t num_nonbasics = nonbasic_list.size(); + for (i_t k = 0; k < num_nonbasics; k++) { + const i_t j = nonbasic_list[k]; + if (std::abs(solution.z[j]) <= 1e-10) { + zero_reduced_costs_vars.push_back(j); + zero_reduced_costs_vars_nonbasic_index.push_back(k); + } + } + return !zero_reduced_costs_vars.empty(); +} + +template +void branch_and_bound_t::dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, std::vector& basic_list, std::vector& nonbasic_list, std::vector& vstatus, - simplex::lp_solution_t& solution, + simplex::lp_solution_t& soln, simplex::basis_update_mpf_t& basis_update, i_t& num_fractional, std::vector& fractional) { std::vector zero_reduced_costs_vars; std::vector zero_reduced_costs_vars_nonbasic_index; - for (i_t k = 0; k < static_cast(nonbasic_list.size()); k++) { - const i_t j = nonbasic_list[k]; - if (std::abs(solution.z[j]) <= 1e-10) { - zero_reduced_costs_vars.push_back(j); - zero_reduced_costs_vars_nonbasic_index.push_back(k); + bool dual_degenerate = check_for_dual_degeneracy(soln, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); + if (!dual_degenerate) { return; } + + // Construct a new LP problem + // minimize p^T x + // subject to B x_B + N_z x_z = b - N x_N + // l_B <= x_B <= u_B + // l_z <= x_z <= u_z + // + // where B is the basic matrix, N is the nonbasic matrix, b is the right-hand side, + + const i_t m = lp.num_rows; + const i_t n = lp.num_rows + zero_reduced_costs_vars.size(); + + i_t nnz = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + nnz += lp.A.col_start[j + 1] - lp.A.col_start[j]; + } + } + simplex::lp_problem_t lp_reduced(lp.handle_ptr, m, n, nnz); + csc_matrix_t& A_reduced = lp_reduced.A; + i_t nz = 0; + i_t reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + A_reduced.col_start[reduced_col] = nz; + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + const f_t value = lp.A.x[p]; + A_reduced.i[nz] = i; + A_reduced.x[nz] = value; + nz++; + } + lp_reduced.lower[reduced_col] = lp.lower[j]; + lp_reduced.upper[reduced_col] = lp.upper[j]; + reduced_col++; + } + } + A_reduced.col_start[reduced_col] = nz; + + std::vector b_reduced = lp.rhs; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + // PASS + } else { + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + const f_t value = lp.A.x[p]; + b_reduced[i] -= value * soln.x[j]; + } + } + } + lp_reduced.rhs = b_reduced; + lp_reduced.obj_scale = 1.0; + + + + settings_.log.printf("Constructed dual degenerate feasibility pump LP with %d rows and %d columns\n", m, n); + + std::vector reduced_basic_list(m); + std::vector reduced_nonbasic_list(zero_reduced_costs_vars.size()); + std::vector reduced_vstatus(n); + i_t num_basic = 0; + i_t num_nonbasic = 0; + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC){ + reduced_basic_list[num_basic++] = reduced_col; + reduced_vstatus[reduced_col++] = variable_status_t::BASIC; + } else if (std::abs(soln.z[j]) <= 1e-10) { + reduced_nonbasic_list[num_nonbasic++] = reduced_col; + reduced_vstatus[reduced_col++] = vstatus[j]; } } - if (zero_reduced_costs_vars.empty()) { return; } + simplex::lp_solution_t reduced_solution(m, n); + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + reduced_solution.x[reduced_col++] = soln.x[j]; + } + } + + std::vector reduced_edge_norms(n); + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + reduced_edge_norms[reduced_col++] = edge_norms_[j]; + } + } + + simplex::basis_update_mpf_t reduced_basis_update = basis_update; + i_t iter = 0; + + i_t max_pump_iter = 100; + simplex::random_t rng(settings_.random_seed); + + i_t best_num_fractional = num_fractional; + bool stalled = false; + for (i_t pump_iter = 0; pump_iter < max_pump_iter; pump_iter++) { + + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + lp_reduced.objective[reduced_col] = 0; + if (var_types_[j] == variable_type_t::INTEGER) { + if (is_fractional(reduced_solution.x[reduced_col], var_types_[j], settings_.integer_tol)) { + // Default to the exact nearest-integer rounding. Only perturb the + // rounding direction when the previous pass made no progress (a + // zero-pivot solve), to break out of the stall. + const f_t random_value = stalled ? 0.25 * (2.0 * rng.random() - 1.0) : 0.0; // [-0.25, 0.25] + if (reduced_solution.x[reduced_col] + random_value < std::floor(reduced_solution.x[reduced_col]) + 0.5) { + lp_reduced.objective[reduced_col] = 1; + } else { + lp_reduced.objective[reduced_col] = -1; + } + } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_LOWER) { + lp_reduced.objective[reduced_col] = 1; + } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_UPPER) { + lp_reduced.objective[reduced_col] = -1; + } + } + reduced_col++; + } + } + + + bool recompute_basis = false; + const i_t iter_before = iter; + simplex::primal_status_t lp_status = simplex::primal_phase2(2, + exploration_stats_.start_time, + lp_reduced, + settings_, + reduced_vstatus, + reduced_solution, + iter); + // Detect a stall: the solve made no pivots, so the incumbent vertex was + // already optimal for this objective and x did not move. Perturb next pass. + stalled = (iter == iter_before); + + if (lp_status == simplex::primal_status_t::OPTIMAL) { + std::vector adjusted_solution(lp.num_cols, 0.0); + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + adjusted_solution[j] = reduced_solution.x[reduced_col++]; + } else { + adjusted_solution[j] = soln.x[j]; + } + } + + // Verify the solution is primal feasible + std::vector residual = lp.rhs; + matrix_vector_multiply(lp.A, 1.0, adjusted_solution, -1.0, residual); + + settings_.log.printf("Reduced LP residual|| A*x - b ||_inf = %.16e\n", vector_norm_inf(residual)); + + std::vector tmp_fractional; + i_t num_fractional_reduced = fractional_variables(settings_, adjusted_solution, var_types_, tmp_fractional); + settings_.log.printf("Reduced LP fractional variables %d/%d\n", num_fractional_reduced, num_fractional); + // Also treat a pass that fails to improve the best as a stall, so we perturb + // the next pass even when the solve pivoted (moved) without reducing the count. + stalled = stalled || (num_fractional_reduced >= best_num_fractional); + if (num_fractional_reduced < best_num_fractional) { + best_num_fractional = num_fractional_reduced; + } + } + +} + settings_.log.printf("Best number of fractional variables %d/%d\n", best_num_fractional, num_fractional); + +} + +template +void branch_and_bound_t::pivot_out_integer_variables( + const simplex::lp_problem_t& lp, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + i_t& num_fractional, + std::vector& fractional) +{ + + std::vector zero_reduced_costs_vars; + std::vector zero_reduced_costs_vars_nonbasic_index; + bool dual_degenerate = check_for_dual_degeneracy(solution, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); + if (!dual_degenerate) { return; } + lp_solution_t soln_copy = solution; std::vector basic_list_copy = basic_list; std::vector nonbasic_list_copy = nonbasic_list; @@ -2636,6 +2841,7 @@ void branch_and_bound_t::pivot_out_integer_variables( if (num_new_fractional < start_num_fractional) { i_t num_integer_increased = start_num_fractional - num_new_fractional; integer_pivots_.fetch_add(num_integer_increased, std::memory_order_release); + settings_.log.printf("Pivoted out %d integer variables: %d -> %d\n", num_integer_increased, start_num_fractional, num_new_fractional); num_fractional = num_new_fractional; fractional = new_fractional; basic_list = basic_list_copy; @@ -2856,6 +3062,15 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut num_fractional, fractional); + dual_degenerate_feasibility_pump(original_lp_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional); + if (num_fractional != 0 && settings_.max_cut_passes > 0) { print_table_header(); } cut_pool_t cut_pool(original_lp_.num_cols, settings_); diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 1fe2b2b897..9e3a23b440 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -309,7 +309,14 @@ class branch_and_bound_t { const std::vector& leaf_solution, i_t leaf_depth, search_strategy_t thread_type); + + omp_atomic_t integer_pivots_{0}; + bool check_for_dual_degeneracy(const simplex::lp_solution_t& solution, + const std::vector& nonbasic_list, + std::vector& zero_reduced_costs_vars, + std::vector& zero_reduced_costs_vars_nonbasic_index); + void pivot_out_integer_variables(const simplex::lp_problem_t& lp, std::vector& basic_list, std::vector& nonbasic_list, @@ -319,6 +326,15 @@ class branch_and_bound_t { i_t& num_fractional, std::vector& fractional); + void dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + simplex::lp_solution_t& soln, + simplex::basis_update_mpf_t& basis_update, + i_t& num_fractional, + std::vector& fractional); + // Repairs low-quality solutions from the heuristics, if it is applicable. void repair_heuristic_solutions(); From ff6f460ec5e05ea94f0e697115073d45a1c7d3fd Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 24 Jul 2026 06:08:43 -0700 Subject: [PATCH 004/113] Add check for fast pivot using slacks --- cpp/src/branch_and_bound/branch_and_bound.cpp | 101 +++++++++++++++++- 1 file changed, 100 insertions(+), 1 deletion(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 064367b60e..be3af49b99 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -2722,7 +2722,106 @@ void branch_and_bound_t::pivot_out_integer_variables( const i_t start_num_fractional = num_fractional; - for (i_t k = 0; k < static_cast(zero_reduced_costs_vars.size()); k++) { + const i_t num_zero_reduced_costs_vars = zero_reduced_costs_vars.size(); + + std::vector row_to_slack(lp.num_rows, -1); + for (i_t j : new_slacks_) { + if (lp.lower[j] != 0 || lp.upper[j] != inf) { continue; } + const i_t p = lp.A.col_start[j]; + row_to_slack[lp.A.i[p]] = j; + } + + std::vector fast_candidates; + std::vector fast_rows; + for (i_t j : fractional) { + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + const i_t num_rows = col_end - col_start; + i_t num_basic_slacks = 0; + i_t num_nonbasic_slacks_with_reduced_cost_zero = 0; + i_t nonbasic_slack = -1; + i_t slack_row = -1; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + const i_t slack = row_to_slack[i]; + if (slack >= 0) { + if (vstatus_copy[slack] == variable_status_t::BASIC) { + num_basic_slacks++; + } else if (std::abs(solution.z[slack]) <= 1e-10) { + num_nonbasic_slacks_with_reduced_cost_zero++; + nonbasic_slack = slack; + slack_row = i; + } + } + } + if (num_basic_slacks == num_rows - 1 && num_nonbasic_slacks_with_reduced_cost_zero == 1) { + fast_candidates.push_back(j); + fast_rows.push_back(slack_row); + } + } + + if (fast_candidates.size() > 0) { + settings_.log.printf("Found %ld fast candidates for pivot out integer variables\n", fast_candidates.size()); + } + + const i_t num_candidates = fast_candidates.size(); + for (i_t k = 0; k < num_candidates; k++) { + const i_t j = fast_candidates[k]; + const i_t row = fast_rows[k]; + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + const i_t num_rows = col_end - col_start; + f_t a_ij = 0.0; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + if (i == row) { + a_ij = lp.A.x[p]; + break; + } + } + if (a_ij == 0.0) { continue; } + + f_t bound = a_ij > 0 ? lp.lower[j] : lp.upper[j]; + if (std::abs(bound) == inf) { continue; } + + sparse_vector_t delta_x; + delta_x.n = lp.num_cols; + delta_x.i.reserve(num_rows + 1); + delta_x.x.reserve(num_rows + 1); + const f_t delta_xj = bound - solution.x[j]; + delta_x.i.push_back(j); + delta_x.x.push_back(delta_xj); + for (i_t p = col_start; p < col_end; p++) { + const i_t r = lp.A.i[p]; + const f_t a_rj = lp.A.x[p]; + const f_t delta_slack_r = -delta_xj * a_rj; + delta_x.i.push_back(row_to_slack[r]); + delta_x.x.push_back(delta_slack_r); + } + + bool ok = true; + const i_t ndx = delta_x.i.size(); + for (i_t h = 0; h < ndx; h++) { + const i_t jj = delta_x.i[h]; + if (jj == j) continue; + const f_t val = delta_x.x[h]; + const f_t slack_value = solution.x[jj]; + if (val < -slack_value) { + ok = false; + break; + } + } + + if (ok) { + std::vector delta_x_dense(lp.num_cols, 0.0); + delta_x.to_dense(delta_x_dense); + std::vector residual(lp.num_rows); + matrix_vector_multiply(lp.A, 1.0, delta_x_dense, 0.0, residual); + settings_.log.printf("Fast candidate ok || A*delta_x ||_inf = %e\n", vector_norm_inf(residual)); + } + } + + for (i_t k = 0; k < num_zero_reduced_costs_vars; k++) { const i_t j = zero_reduced_costs_vars[k]; if (var_types_[j] == variable_type_t::INTEGER) { continue; } if (vstatus_copy[j] == variable_status_t::BASIC) { continue; } From 900805a940d4f7166538c9fa6c037b186547e6ff Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Mon, 27 Jul 2026 14:06:19 -0700 Subject: [PATCH 005/113] Enable primal simplex. Solves 85/93 NETLIB LPs in under 1 minute --- .../mathematical_optimization/constants.h | 3 +- .../pdlp/solver_settings.hpp | 3 + cpp/src/branch_and_bound/branch_and_bound.cpp | 215 ++++-- cpp/src/dual_simplex/primal.cpp | 689 ++++++++++++++---- cpp/src/dual_simplex/primal.hpp | 16 +- cpp/src/dual_simplex/solve.cpp | 157 +++- cpp/src/dual_simplex/solve.hpp | 6 + cpp/src/math_optimization/solver_settings.cu | 2 +- cpp/src/pdlp/solve.cu | 67 +- 9 files changed, 955 insertions(+), 203 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/constants.h b/cpp/include/cuopt/mathematical_optimization/constants.h index f6be07aaa9..4ed3723aa2 100644 --- a/cpp/include/cuopt/mathematical_optimization/constants.h +++ b/cpp/include/cuopt/mathematical_optimization/constants.h @@ -192,7 +192,8 @@ #define CUOPT_METHOD_PDLP 1 #define CUOPT_METHOD_DUAL_SIMPLEX 2 #define CUOPT_METHOD_BARRIER 3 -#define CUOPT_METHOD_UNSET 4 +#define CUOPT_METHOD_PRIMAL 4 +#define CUOPT_METHOD_UNSET 5 /* @brief PDLP precision mode constants */ #define CUOPT_PDLP_DEFAULT_PRECISION -1 diff --git a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp index 96f548ec32..3bf3b6ab01 100644 --- a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp @@ -57,6 +57,7 @@ enum pdlp_solver_mode_t : int { * PDLP: Use the PDLP method. * DualSimplex: Use the dual simplex method. * Barrier: Use the barrier method + * Primal: Use the (experimental) primal simplex method. * Unset: The value was not set. * * @note Default method is Concurrent. @@ -66,6 +67,7 @@ enum method_t : int { PDLP = CUOPT_METHOD_PDLP, DualSimplex = CUOPT_METHOD_DUAL_SIMPLEX, Barrier = CUOPT_METHOD_BARRIER, + Primal = CUOPT_METHOD_PRIMAL, Unset = CUOPT_METHOD_UNSET }; @@ -77,6 +79,7 @@ inline std::string method_to_string(method_t method) case method_t::PDLP: return "PDLP"; case method_t::Barrier: return "Barrier"; case method_t::Concurrent: return "Concurrent"; + case method_t::Primal: return "Primal Simplex"; default: return "Unset"; } } diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 6a47deb079..1da2f0f249 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3171,6 +3171,15 @@ auto branch_and_bound_t::do_cut_pass( basis_update, num_fractional, fractional); + + dual_degenerate_feasibility_pump(original_lp_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional); if (settings_.benchmark_info_ptr != nullptr) { settings_.benchmark_info_ptr->root_lp_with_cuts = @@ -3290,10 +3299,12 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple } simplex::lp_problem_t lp_reduced(lp.handle_ptr, m, n, nnz); csc_matrix_t& A_reduced = lp_reduced.A; + std::vector original_col_to_reduced_col(lp.num_cols, -1); i_t nz = 0; i_t reduced_col = 0; for (i_t j = 0; j < lp.num_cols; j++) { if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + original_col_to_reduced_col[j] = reduced_col; A_reduced.col_start[reduced_col] = nz; const i_t col_start = lp.A.col_start[j]; const i_t col_end = lp.A.col_start[j + 1]; @@ -3340,11 +3351,10 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple reduced_col = 0; for (i_t j = 0; j < lp.num_cols; j++) { if (vstatus[j] == variable_status_t::BASIC){ - reduced_basic_list[num_basic++] = reduced_col; reduced_vstatus[reduced_col++] = variable_status_t::BASIC; } else if (std::abs(soln.z[j]) <= 1e-10) { - reduced_nonbasic_list[num_nonbasic++] = reduced_col; - reduced_vstatus[reduced_col++] = vstatus[j]; + reduced_nonbasic_list[num_nonbasic++] = reduced_col; // Does ordering of nonbasic variables matter? + reduced_vstatus[reduced_col++] = vstatus[j]; } } @@ -3365,85 +3375,170 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple } simplex::basis_update_mpf_t reduced_basis_update = basis_update; - i_t iter = 0; + for (i_t k = 0; k < m; k++) { + reduced_basic_list[k] = original_col_to_reduced_col[basic_list[k]]; + } - i_t max_pump_iter = 100; + i_t iter = 0; + i_t max_pump_iter = 10; simplex::random_t rng(settings_.random_seed); - i_t best_num_fractional = num_fractional; - bool stalled = false; + std::vector best_reduced_vstatus(n); + bool stalled = false; for (i_t pump_iter = 0; pump_iter < max_pump_iter; pump_iter++) { - reduced_col = 0; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { - lp_reduced.objective[reduced_col] = 0; - if (var_types_[j] == variable_type_t::INTEGER) { - if (is_fractional(reduced_solution.x[reduced_col], var_types_[j], settings_.integer_tol)) { - // Default to the exact nearest-integer rounding. Only perturb the - // rounding direction when the previous pass made no progress (a - // zero-pivot solve), to break out of the stall. - const f_t random_value = stalled ? 0.25 * (2.0 * rng.random() - 1.0) : 0.0; // [-0.25, 0.25] - if (reduced_solution.x[reduced_col] + random_value < std::floor(reduced_solution.x[reduced_col]) + 0.5) { + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + lp_reduced.objective[reduced_col] = 0; + if (var_types_[j] == variable_type_t::INTEGER) { + if (is_fractional( + reduced_solution.x[reduced_col], var_types_[j], settings_.integer_tol)) { + // Default to the exact nearest-integer rounding. Only perturb the + // rounding direction when the previous pass made no progress (a + // zero-pivot solve), to break out of the stall. + const f_t random_value = + stalled ? 0.25 * (2.0 * rng.random() - 1.0) : 0.0; // [-0.25, 0.25] + if (reduced_solution.x[reduced_col] + random_value < + std::floor(reduced_solution.x[reduced_col]) + 0.5) { + lp_reduced.objective[reduced_col] = 1; + } else { + lp_reduced.objective[reduced_col] = -1; + } + } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_LOWER) { lp_reduced.objective[reduced_col] = 1; - } else { + } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_UPPER) { lp_reduced.objective[reduced_col] = -1; } - } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_LOWER) { - lp_reduced.objective[reduced_col] = 1; - } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_UPPER) { - lp_reduced.objective[reduced_col] = -1; } + reduced_col++; } - reduced_col++; - } + } + + bool recompute_basis = false; + const i_t iter_before = iter; + f_t primal_work_estimate = 0; + simplex::primal_status_t lp_status = simplex::primal_phase2_with_advanced_basis(2, + exploration_stats_.start_time, + lp_reduced, + settings_, + reduced_vstatus, + reduced_basis_update, + reduced_basic_list, + reduced_nonbasic_list, + reduced_solution, + iter, + primal_work_estimate); + // Detect a stall: the solve made no pivots, so the incumbent vertex was + // already optimal for this objective and x did not move. Perturb next pass. + stalled = (iter == iter_before); + + if (lp_status == simplex::primal_status_t::OPTIMAL) { + std::vector adjusted_solution(lp.num_cols, 0.0); + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + adjusted_solution[j] = reduced_solution.x[reduced_col++]; + } else { + adjusted_solution[j] = soln.x[j]; + } + } + + // Verify the solution is primal feasible + std::vector residual = lp.rhs; + matrix_vector_multiply(lp.A, 1.0, adjusted_solution, -1.0, residual); + + settings_.log.printf("Reduced LP residual|| A*x - b ||_inf = %.16e\n", + vector_norm_inf(residual)); + + std::vector tmp_fractional; + i_t num_fractional_reduced = + fractional_variables(settings_, adjusted_solution, var_types_, tmp_fractional); + settings_.log.printf( + "Reduced LP fractional variables %d/%d\n", num_fractional_reduced, num_fractional); + // Also treat a pass that fails to improve the best as a stall, so we perturb + // the next pass even when the solve pivoted (moved) without reducing the count. + stalled = stalled || (num_fractional_reduced >= best_num_fractional); + if (num_fractional_reduced < best_num_fractional) { + best_num_fractional = num_fractional_reduced; + best_reduced_vstatus = reduced_vstatus; + } + } } - - bool recompute_basis = false; - const i_t iter_before = iter; - simplex::primal_status_t lp_status = simplex::primal_phase2(2, - exploration_stats_.start_time, - lp_reduced, - settings_, - reduced_vstatus, - reduced_solution, - iter); - // Detect a stall: the solve made no pivots, so the incumbent vertex was - // already optimal for this objective and x did not move. Perturb next pass. - stalled = (iter == iter_before); - - if (lp_status == simplex::primal_status_t::OPTIMAL) { - std::vector adjusted_solution(lp.num_cols, 0.0); - reduced_col = 0; + settings_.log.printf("Best number of fractional variables %d/%d\n", best_num_fractional, num_fractional); + if (best_num_fractional < num_fractional) { + // Translate the vstatus from the reduced problem to the vstatus for the original problem + i_t reduced_cols = 0; for (i_t j = 0; j < lp.num_cols; j++) { if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { - adjusted_solution[j] = reduced_solution.x[reduced_col++]; + vstatus[j] = best_reduced_vstatus[reduced_cols++]; + } + } + + std::vector superbasic_list; + nonbasic_list.clear(); + simplex::get_basis_from_vstatus(m, vstatus, basic_list, nonbasic_list, superbasic_list); + assert(superbasic_list.empty()); + const i_t refactor_status = basis_update.refactor_basis(lp.A, + settings_, + lp.lower, + lp.upper, + exploration_stats_.start_time, + basic_list, + nonbasic_list, + vstatus); + if (refactor_status == CONCURRENT_HALT_RETURN || refactor_status == TIME_LIMIT_RETURN) { + // TODO: On failure vstatus, basic_list, and nonbasic_list are in a bad state. + // We should save copies before the failure and restore them after the failure. + return; + } + if (refactor_status != 0) { + settings_.log.printf("Failed to refactor basis after dual degenerate feasibility pump. " + "%d deficient columns.\n", + refactor_status); + return; + } + + // Update the solution + // First set the nonbasic variables on their bounds + for (i_t k = 0; k < lp.num_cols - lp.num_rows; k++) { + const i_t j = nonbasic_list[k]; + if (vstatus[j] == variable_status_t::NONBASIC_LOWER || vstatus[j] == variable_status_t::NONBASIC_FIXED) { + soln.x[j] = lp.lower[j]; + } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER) { + soln.x[j] = lp.upper[j]; } else { - adjusted_solution[j] = soln.x[j]; + soln.x[j] = 0; } } + // Then compute the effective rhs + std::vector rhs = lp.rhs; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC) { continue; } + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; - // Verify the solution is primal feasible - std::vector residual = lp.rhs; - matrix_vector_multiply(lp.A, 1.0, adjusted_solution, -1.0, residual); + const f_t x_j = soln.x[j]; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + const f_t aij = lp.A.x[p]; + rhs[i] -= aij * x_j; + } + } - settings_.log.printf("Reduced LP residual|| A*x - b ||_inf = %.16e\n", vector_norm_inf(residual)); + // Then solve B xB = rhs + std::vector xB(lp.num_rows); + basis_update.b_solve(rhs, xB); - std::vector tmp_fractional; - i_t num_fractional_reduced = fractional_variables(settings_, adjusted_solution, var_types_, tmp_fractional); - settings_.log.printf("Reduced LP fractional variables %d/%d\n", num_fractional_reduced, num_fractional); - // Also treat a pass that fails to improve the best as a stall, so we perturb - // the next pass even when the solve pivoted (moved) without reducing the count. - stalled = stalled || (num_fractional_reduced >= best_num_fractional); - if (num_fractional_reduced < best_num_fractional) { - best_num_fractional = num_fractional_reduced; + // Then update the basic variables + for (i_t k = 0; k < lp.num_rows; k++) { + soln.x[basic_list[k]] = xB[k]; } - } + + fractional.clear(); + num_fractional = fractional_variables(settings_, soln.x, var_types_, fractional); -} - settings_.log.printf("Best number of fractional variables %d/%d\n", best_num_fractional, num_fractional); - + } } template diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index aca2e785d2..d12f24e98f 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -14,6 +14,8 @@ #include #include +#include + namespace cuopt::mathematical_optimization::simplex { namespace { @@ -57,13 +59,13 @@ void set_primal_variables_on_bounds(const lp_problem_t& lp, template f_t dual_infeasibility(const lp_problem_t& lp, const std::vector& vstatus, - const std::vector& z) + const std::vector& z, + f_t tight_tol, + i_t& num_infeasible) { const i_t n = lp.num_cols; - const i_t m = lp.num_rows; - i_t num_infeasible = 0; + num_infeasible = 0; f_t sum_infeasible = 0.0; - constexpr f_t tight_tol = 0; i_t lower_bound_inf = 0; i_t upper_bound_inf = 0; i_t free_inf = 0; @@ -110,6 +112,7 @@ i_t phase2_pricing(const lp_problem_t& lp, const std::vector& z, const std::vector& nonbasic_list, const std::vector& vstatus, + f_t dual_tol, i_t& direction, i_t& basic_entering, f_t& dual_inf) @@ -120,8 +123,7 @@ i_t phase2_pricing(const lp_problem_t& lp, f_t max_infeas = 0.0; dual_inf = 0.0; for (i_t k = 0; k < n - m; ++k) { - const i_t j = nonbasic_list[k]; - constexpr f_t dual_tol = 1e-6; + const i_t j = nonbasic_list[k]; if (vstatus[j] == variable_status_t::NONBASIC_FIXED) { continue; } if ((vstatus[j] == variable_status_t::NONBASIC_LOWER || vstatus[j] == variable_status_t::NONBASIC_FREE) && @@ -154,15 +156,20 @@ template f_t primal_infeasibility(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& vstatus, - const std::vector& x) + const std::vector& x, + i_t& num_infeasible) { const i_t n = lp.num_cols; f_t primal_inf = 0; + num_infeasible = 0; for (i_t j = 0; j < n; ++j) { + // Nonbasics are pinned to a bound; only basics can be (legitimately) infeasible. + if (vstatus[j] != variable_status_t::BASIC) { continue; } if (x[j] < lp.lower[j] - settings.primal_tol) { // x_j < l_j => -x_j > -l_j => -x_j + l_j > 0 const f_t infeas = -x[j] + lp.lower[j]; primal_inf += infeas; + num_infeasible++; if (infeas > 1e-6) { settings.log.debug("x %d infeas %e lo %e val %e up %e vstatus %hhd\n", j, @@ -177,6 +184,7 @@ f_t primal_infeasibility(const lp_problem_t& lp, // x_j > u_j => x_j - u_j > 0 const f_t infeas = x[j] - lp.upper[j]; primal_inf += infeas; + num_infeasible++; if (infeas > 1e-6) { settings.log.debug("x %d infeas %e lo %e val %e up %e vstatus %hhd\n", j, @@ -191,15 +199,28 @@ f_t primal_infeasibility(const lp_problem_t& lp, return primal_inf; } +template +f_t primal_infeasibility(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& vstatus, + const std::vector& x) +{ + i_t num_infeasible = 0; + return primal_infeasibility(lp, settings, vstatus, x, num_infeasible); +} + template void compute_phase1_objective(const lp_problem_t& lp, const simplex_solver_settings_t& settings, + const std::vector& vstatus, const std::vector& x, std::vector& objective) { const i_t n = lp.num_cols; for (i_t j = 0; j < n; ++j) { - if (x[j] < lp.lower[j] - settings.primal_tol) { + if (vstatus[j] != variable_status_t::BASIC) { + objective[j] = 0.0; + } else if (x[j] < lp.lower[j] - settings.primal_tol) { objective[j] = -1.0; } else if (x[j] > lp.upper[j] + settings.primal_tol) { objective[j] = 1.0; @@ -209,13 +230,79 @@ void compute_phase1_objective(const lp_problem_t& lp, } } +template +void compute_delta_y(const basis_update_mpf_t& basis_update, + i_t basic_leaving, + sparse_vector_t& delta_y, + sparse_vector_t& etilde) +{ + const i_t m = delta_y.n; + sparse_vector_t ei(m, 1); + ei.i[0] = basic_leaving; + ei.x[0] = 1.0; + delta_y.clear(); + etilde.clear(); + basis_update.b_transpose_solve(ei, delta_y, etilde); +} + +template +void compute_delta_z(const csr_matrix_t& Arow, + const std::vector& vstatus, + const sparse_vector_t& delta_y, + std::vector& delta_z) +{ + // A^T delta_y + delta_z = 0 + // delta_z = -A^T delta_y = - sum_i A(i, :) * delta_y_i + std::fill(delta_z.begin(), delta_z.end(), 0.0); + for (i_t k = 0; k < static_cast(delta_y.i.size()); ++k) { + const i_t i = delta_y.i[k]; + const f_t delta_y_i = delta_y.x[k]; + const i_t row_start = Arow.row_start[i]; + const i_t row_end = Arow.row_start[i + 1]; + for (i_t p = row_start; p < row_end; ++p) { + const i_t j = Arow.j[p]; + if (vstatus[j] != variable_status_t::BASIC) { delta_z[j] -= Arow.x[p] * delta_y_i; } + } + } +} + +template +f_t compute_dual_step_length(f_t entering_reduced_cost, f_t pivot) +{ + assert(pivot != 0.0); + return entering_reduced_cost / pivot; +} + +template +void update_y(f_t dual_step_length, const sparse_vector_t& delta_y, std::vector& y) +{ + for (i_t k = 0; k < static_cast(delta_y.i.size()); ++k) { + const i_t i = delta_y.i[k]; + y[i] += dual_step_length * delta_y.x[k]; + } +} + +template +void update_z(f_t dual_step_length, + const std::vector& nonbasic_list, + i_t entering_index, + const std::vector& delta_z, + std::vector& z) +{ + for (i_t k = 0; k < static_cast(nonbasic_list.size()); ++k) { + const i_t j = nonbasic_list[k]; + z[j] += dual_step_length * delta_z[j]; + } + z[entering_index] = 0.0; +} + template void compute_dual_variables(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& objective, const std::vector& basic_list, const std::vector& nonbasic_list, - basis_update_t& ft, + basis_update_mpf_t& ft, std::vector& c_basic, std::vector& y, std::vector& z) @@ -249,6 +336,40 @@ void compute_dual_variables(const lp_problem_t& lp, } } +template +void compute_basic_primal_variables(const lp_problem_t& lp, + const basis_update_mpf_t& basis_update, + const std::vector& basic_list, + const std::vector& nonbasic_list, + std::vector& x) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + std::vector rhs = lp.rhs; + for (i_t k = 0; k < n - m; ++k) { + const i_t j = nonbasic_list[k]; + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + const f_t xj = x[j]; + for (i_t p = col_start; p < col_end; ++p) { + rhs[lp.A.i[p]] -= xj * lp.A.x[p]; + } + } + std::vector xB(m); + basis_update.b_solve(rhs, xB); + for (i_t k = 0; k < m; ++k) { + x[basic_list[k]] = xB[k]; + } +} + +template +f_t primal_constraint_residual(const lp_problem_t& lp, const std::vector& x) +{ + std::vector residual = lp.rhs; + matrix_vector_multiply(lp.A, 1.0, x, -1.0, residual); + return vector_norm_inf(residual); +} + } // namespace @@ -269,6 +390,7 @@ i_t primal_ratio_test(const lp_problem_t& lp, basic_leaving = -1; i_t leaving_index = -1; f_t min_val = inf; + f_t current_dx = 0.0; constexpr f_t pivot_tol = 1e-8; // Entering variable can hit its opposite bound: limit step by that @@ -291,33 +413,72 @@ i_t primal_ratio_test(const lp_problem_t& lp, for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; if (delta_x[j] == 0.0) { continue; } + + // Already below lower and moving back up: stop when we reach the lower bound. + // Without this, phase I can take an unbounded step (false unbounded) or skip the + // breakpoint of the piecewise phase-I objective and stall still infeasible. + if (x[j] < lp.lower[j] && delta_x[j] > pivot_tol && lp.lower[j] > -inf) { + const f_t ratio = (lp.lower[j] - x[j]) / delta_x[j]; + if (ratio >= 0 && ratio < min_val) { + min_val = ratio; + basic_leaving = k; + leaving_index = j; + current_dx = delta_x[j]; + } + } + // Already above upper and moving back down: stop when we reach the upper bound. + if (x[j] > lp.upper[j] && delta_x[j] < -pivot_tol && lp.upper[j] < inf) { + const f_t ratio = (lp.upper[j] - x[j]) / delta_x[j]; + if (ratio >= 0 && ratio < min_val) { + min_val = ratio; + basic_leaving = k; + leaving_index = j; + current_dx = -delta_x[j]; + } + } + if (lp.lower[j] > -inf && delta_x[j] < -pivot_tol) { // xj + step * delta_x[j] >= lp.lower[j] // step * delta_x[j] >= lp.lower[j] - x[j] // step <= (lp.lower[j] - x[j]) / delta_x[j], delta_x[j] < 0 - const f_t neum = lp.lower[j] - x[j]; - f_t ratio = neum / delta_x[j]; - // Already below lower and moving further (delta_x < 0): cap step at 0 - if (x[j] < lp.lower[j]) { ratio = 0; } + f_t neum = lp.lower[j] - x[j]; + // A basic sitting a hair below its bound (within the primal tolerance) is on + // the bound numerically, but gives a tiny negative ratio. Dropping it lets + // the step run straight through the bound, so treat it as a zero-length + // block. A genuine violation is left to the branches above, which stop at + // the bound when the variable moves back toward it. + if (neum > 0 && neum <= settings.primal_tol) { neum = 0.0; } + f_t ratio = neum / delta_x[j]; if (ratio >= 0 && ratio < min_val) { min_val = ratio; basic_leaving = k; leaving_index = j; + current_dx = -delta_x[j]; + } else if (ratio >= 0 && ratio < min_val + 1e-9 && -delta_x[j] > current_dx) { + min_val = ratio; + basic_leaving = k; + leaving_index = j; + current_dx = -delta_x[j]; } } if (lp.upper[j] < inf && delta_x[j] > pivot_tol) { // xj + step * delta_x[j] <= lp.upper[j] // step * delta_x[j] <= lp.upper[j] - x[j] // step <= (lp.upper[j] - x[j]) / delta_x[j], delta_x[j] > 0 - const f_t neum = lp.upper[j] - x[j]; - f_t ratio = neum / delta_x[j]; - // If x[j] > upper (slightly infeasible), ratio < 0; skip so we don't use it. - // But if we're already above upper and would move further (delta_x > 0), cap step at 0. - if (x[j] > lp.upper[j]) { ratio = 0; } + f_t neum = lp.upper[j] - x[j]; + // Mirror of the lower bound case: a hair above the bound is on the bound. + if (neum < 0 && -neum <= settings.primal_tol) { neum = 0.0; } + f_t ratio = neum / delta_x[j]; if (ratio >= 0 && ratio < min_val) { min_val = ratio; basic_leaving = k; leaving_index = j; + current_dx = delta_x[j]; + } else if (ratio >= 0 && ratio < min_val + 1e-9 && delta_x[j] > current_dx) { + min_val = ratio; + basic_leaving = k; + leaving_index = j; + current_dx = delta_x[j]; } } } @@ -325,9 +486,7 @@ i_t primal_ratio_test(const lp_problem_t& lp, return leaving_index; } -// Note this implementation of primal simplex is experimental -// It is meant only to serve as a method to remove the perturbation to the objective -// after dual simplex has found a primal feasible solution + template primal_status_t primal_phase2(i_t phase, f_t start_time, @@ -339,32 +498,11 @@ primal_status_t primal_phase2(i_t phase, { const i_t m = lp.num_rows; const i_t n = lp.num_cols; - assert(m <= n); - assert(vstatus.size() == n); - assert(lp.A.m == m); - assert(lp.A.n == n); - assert(lp.objective.size() == n); - assert(lp.lower.size() == n); - assert(lp.upper.size() == n); - assert(lp.rhs.size() == m); + f_t work_estimate = 0; std::vector basic_list(m); std::vector nonbasic_list; std::vector superbasic_list; - std::vector bound_info(n - m); - - std::vector& x = sol.x; - std::vector& y = sol.y; - std::vector& z = sol.z; - - std::vector incoming_x = x; - std::vector incoming_vstatus = vstatus; - - settings.log.printf("Primal Simplex Phase %d\n", phase); - settings.log.printf("Solving a problem with %d constraints %d variables %d nonzeros\n", - lp.num_rows, - lp.num_cols, - lp.A.col_start[lp.num_cols]); get_basis_from_vstatus(m, vstatus, basic_list, nonbasic_list, superbasic_list); assert(superbasic_list.size() == 0); @@ -436,7 +574,58 @@ primal_status_t primal_phase2(i_t phase, } } reorder_basic_list(q, basic_list); - basis_update_t ft(L, U, p); + basis_update_mpf_t ft(L, U, p, settings.refactor_frequency); + + return primal_phase2_with_advanced_basis(phase, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + work_estimate); +} +// Note this implementation of primal simplex is experimental +// It is meant only to serve as a method to remove the perturbation to the objective +// after dual simplex has found a primal feasible solution +template +primal_status_t primal_phase2_with_advanced_basis( + i_t phase, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& basis_update, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + i_t& iter, + f_t& work_estimate) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + assert(m <= n); + assert(vstatus.size() == n); + assert(lp.A.m == m); + assert(lp.A.n == n); + assert(lp.objective.size() == n); + assert(lp.lower.size() == n); + assert(lp.upper.size() == n); + assert(lp.rhs.size() == m); + + std::vector& x = sol.x; + std::vector& y = sol.y; + std::vector& z = sol.z; + + std::vector incoming_x = x; + std::vector incoming_vstatus = vstatus; + settings.log.printf("Primal Simplex\n"); + // Nonbasics must be on their bounds before forming B x_B = b - A_N x_N. + // Setting them after the solve leaves ||A*x - b|| large whenever x_N != 0. + set_primal_variables_on_bounds(lp, settings, vstatus, x); std::vector rhs = lp.rhs; // rhs = b - sum_{j : x_j = l_j} A(:, j) l(j) - sum_{j : x_j = u_j} A(:, j) * @@ -452,91 +641,237 @@ primal_status_t primal_phase2(i_t phase, } std::vector xB(m); - ft.b_solve(rhs, xB); + basis_update.b_solve(rhs, xB); for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; x[j] = xB[k]; } - set_primal_variables_on_bounds(lp, settings, vstatus, x); - settings.log.printf("|| x || %e\n", vector_norm2(x)); + constexpr bool print_norms = false; + if constexpr (print_norms) { + settings.log.printf("|| x || %e\n", vector_norm2(x)); + } std::vector residual = lp.rhs; matrix_vector_multiply(lp.A, 1.0, x, -1.0, residual); f_t primal_residual = vector_norm_inf(residual); - if (primal_residual > 1e-6) { settings.log.printf("|| A*x - b || %e\n", primal_residual); } - f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); - settings.log.printf("Initial primal infeasibility %e\n", primal_inf); - + if (primal_residual > settings.primal_tol) { + settings.log.printf("|| A*x - b || %e\n", primal_residual); + } + + std::vector objective = lp.objective; const f_t primal_tol = settings.primal_tol; + f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); if (primal_inf > primal_tol) { // We are primal infeasible. Switch to phase 1 - compute_phase1_objective(lp, settings, x, objective); + compute_phase1_objective(lp, settings, vstatus, x, objective); + settings.log.printf("Phase 1\n"); + settings.log.printf("Initial primal infeasibility %e\n", primal_inf); phase = 1; } else { + settings.log.printf("Phase 2\n"); phase = 2; } std::vector c_basic(m); - compute_dual_variables(lp, settings, objective, basic_list, nonbasic_list, ft, c_basic, y, z); - settings.log.printf("|| z || %e\n", vector_norm_inf(z)); + compute_dual_variables( + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); + if constexpr (print_norms) { + settings.log.printf("|| z || %e\n", vector_norm_inf(z)); + } + + i_t num_dual_inf = 0; + i_t num_primal_inf = 0; + const f_t init_dual_inf = + dual_infeasibility(lp, vstatus, z, settings.dual_tol, num_dual_inf); + if (num_dual_inf > 0) { + settings.log.printf("Initial dual infeasibility %e\n", init_dual_inf); + } - const f_t init_dual_inf = dual_infeasibility(lp, vstatus, z); - settings.log.printf("Initial dual infeasibility %e\n", init_dual_inf); + csr_matrix_t Arow(m, n, lp.A.nnz()); + lp.A.to_compressed_row(Arow); - const i_t iter_limit = iter + 1000; - std::vector delta_y(m); + const i_t iter_limit = settings.iteration_limit; + const i_t start_iter = iter; + sparse_vector_t delta_y(m, 0); + sparse_vector_t etilde(m, 0); std::vector delta_z(n); std::vector delta_x(n); - settings.log.printf("Iter Objective Primal inf Dual Inf. Step Entering Leaving\n"); + f_t dual_inf = init_dual_inf; + f_t obj = compute_objective(lp, x); + f_t pricing_dual_tol = settings.dual_tol; + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf); + settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); + settings.log.printf("%5d %+.16e %7d %.8e %.2f\n", + iter, + compute_user_objective(lp, obj), + phase == 1 ? num_primal_inf : num_dual_inf, + phase == 1 ? primal_inf : dual_inf, + toc(start_time)); + bool switched_phase = false; while (iter < iter_limit) { i_t nonbasic_entering = -1; - f_t dual_inf; i_t direction; - i_t entering_index = - phase2_pricing(lp, z, nonbasic_list, vstatus, direction, nonbasic_entering, dual_inf); + i_t entering_index = phase2_pricing( + lp, z, nonbasic_list, vstatus, pricing_dual_tol, direction, nonbasic_entering, dual_inf); if (entering_index == -1) { if (phase == 2) { - f_t obj = compute_objective(lp, x); - f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); - settings.log.printf( - "Optimal solution found. Objective %+.16e. Dual infeas %e. Primal " - "infeasibility %e. Iterations %d\n", - compute_user_objective(lp, obj), - dual_inf, - primal_inf, - iter); + // Verify optimality with a consistent basic solution: refactor, put + // nonbasics exactly on their status bounds, rebuild x_B so Ax = b, and + // refresh duals. If that point is not primal/dual feasible, continue. + if (basis_update.num_updates() > 0) { + i_t rank = basis_update.refactor_basis( + lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); + if (rank == CONCURRENT_HALT_RETURN) { return primal_status_t::CONCURRENT_LIMIT; } + if (rank == TIME_LIMIT_RETURN) { return primal_status_t::TIME_LIMIT; } + if (rank != 0) { + settings.log.printf("Failed to refactor basis at optimality check. Iteration %d\n", + iter); + return primal_status_t::NUMERICAL; + } + work_estimate = basis_update.work_estimate(); + } + set_primal_variables_on_bounds(lp, settings, vstatus, x); + compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x); + compute_dual_variables( + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf); + dual_inf = + dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf); + if (primal_inf > primal_tol) { + compute_phase1_objective(lp, settings, vstatus, x, objective); + phase = 1; + pricing_dual_tol = settings.dual_tol; + compute_dual_variables( + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); + settings.log.printf( + "Switching to Primal Simplex Phase 1 after optimality refresh. " + "Primal infeasibility %e\n", + primal_inf); + settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); + switched_phase = true; + continue; + } + if (num_dual_inf > 0) { + // The refreshed reduced costs contain a candidate visible at the active + // pricing tolerance. + continue; + } + + i_t num_tight_dual_inf = 0; + const f_t tight_dual_inf = + dual_infeasibility(lp, vstatus, z, f_t(0.0), num_tight_dual_inf); + if (tight_dual_inf > settings.dual_tol) { + // No candidate is visible at the active pricing tolerance, but the + // zero-tolerance residual is still material. Try tighter pricing before + // accepting optimality. This is needed for problems such as cycle, + // where many small reduced-cost violations lead to improving pivots. + f_t retry_dual_tol = pricing_dual_tol; + f_t retry_dual_inf = 0.0; + i_t retry_entering = -1; + while (retry_entering == -1 && retry_dual_tol > f_t(1e-10)) { + retry_dual_tol *= f_t(0.1); + retry_entering = phase2_pricing(lp, + z, + nonbasic_list, + vstatus, + retry_dual_tol, + direction, + nonbasic_entering, + retry_dual_inf); + } + if (retry_entering != -1) { + pricing_dual_tol = retry_dual_tol; + continue; + } + } + // Report the unfiltered residual at the accepted solution. + dual_inf = tight_dual_inf; + num_dual_inf = num_tight_dual_inf; + obj = compute_objective(lp, x); + sol.objective = obj; + sol.user_objective = compute_user_objective(lp, obj); + if (!settings.inside_mip) { + settings.log.printf("\n"); + settings.log.printf( + "Optimal solution found in %d iterations and %.2fs\n", iter, toc(start_time)); + settings.log.printf("Objective %+.8e\n", sol.user_objective); + settings.log.printf("\n"); + settings.log.printf("Primal infeasibility (abs): %.2e\n", primal_inf); + settings.log.printf("Dual infeasibility (abs): %.2e\n", dual_inf); + settings.log.printf("Primal residual ||Ax-b||: %.2e\n", + primal_constraint_residual(lp, x)); + } return primal_status_t::OPTIMAL; } else { - primal_inf = primal_infeasibility(lp, settings, vstatus, x); + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf); if (primal_inf > primal_tol) { - settings.log.printf("Primal infeasibility %e. No entering\n", primal_inf); - return primal_status_t::NUMERICAL; + // Incremental duals may be stale relative to the current phase-I + // objective. Refresh objective and duals, then retry pricing with + // successively tighter dual tolerances. + settings.log.printf("Refreshing phase-I objective and duals. Num updates %d. Iter %d\n", + basis_update.num_updates(), iter); + compute_phase1_objective(lp, settings, vstatus, x, objective); + compute_dual_variables( + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); + f_t retry_dual_tol = pricing_dual_tol; + while (entering_index == -1 && retry_dual_tol > f_t(1e-10)) { + retry_dual_tol *= f_t(0.1); + settings.log.printf("Retrying phase-I pricing with dual_tol %e\n", retry_dual_tol); + entering_index = phase2_pricing(lp, + z, + nonbasic_list, + vstatus, + retry_dual_tol, + direction, + nonbasic_entering, + dual_inf); + } + if (entering_index == -1) { + settings.log.printf( + "Numerical issues encountered. No entering variable found with large " + "infeasibility %e (%d).\n", + primal_inf, + num_primal_inf); + return primal_status_t::NUMERICAL; + } + pricing_dual_tol = retry_dual_tol; } else { // Restore the objective to the original objective - objective = lp.objective; - phase = 2; - settings.log.printf("Switching to phase 2\n"); + objective = lp.objective; + phase = 2; + pricing_dual_tol = settings.dual_tol; + settings.log.printf( + "Primal phase I complete. Iterations %d. Time %.2f\n", iter, toc(start_time)); + settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, ft, c_basic, y, z); + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); + obj = compute_objective(lp, x); + dual_inf = + dual_infeasibility(lp, vstatus, z, settings.dual_tol, num_dual_inf); iter++; + // Print here: continue may hit dual-optimal Phase 2 and return before + // the end-of-loop log checks switched_phase. + settings.log.printf("%5d %+.16e %7d %.8e %.2f\n", + iter, + compute_user_objective(lp, obj), + num_dual_inf, + dual_inf, + toc(start_time)); continue; } } } + sparse_vector_t rhs_sparse(lp.A, entering_index); + sparse_vector_t scaled_delta_xB_sparse(m, 0); + sparse_vector_t utilde_sparse(m, 0); + basis_update.b_solve(rhs_sparse, scaled_delta_xB_sparse, utilde_sparse); std::vector scaled_delta_xB(m); - std::vector rhs(m); - const i_t col_start = lp.A.col_start[entering_index]; - const i_t col_end = lp.A.col_start[entering_index + 1]; - for (i_t p = col_start; p < col_end; ++p) { - rhs[lp.A.i[p]] = lp.A.x[p]; - } - std::vector utilde(m); - ft.b_solve(rhs, scaled_delta_xB, utilde); + scaled_delta_xB_sparse.to_dense(scaled_delta_xB); for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; @@ -548,67 +883,116 @@ primal_status_t primal_phase2(i_t phase, } delta_x[entering_index] = direction; - std::vector residual(m); - matrix_vector_multiply(lp.A, 1.0, delta_x, 1.0, residual); +#ifdef CHECK_NULLSPACE + std::vector residual(m, 0.0); + matrix_vector_multiply(lp.A, 1.0, delta_x, 0.0, residual); f_t primal_step_err = vector_norm_inf(residual); - if (primal_step_err > 1e-3) { printf("|| A * dx || %e\n", primal_step_err); } + if (primal_step_err > 1e-3) { + settings.log.printf("|| A * dx || %e at iter %d (updates %d)\n", + primal_step_err, + iter, + basis_update.num_updates()); + } +#endif i_t basic_leaving; f_t step_length; - i_t leaving_index = primal_ratio_test( - lp, settings, vstatus, basic_list, x, delta_x, step_length, basic_leaving, entering_index, direction); + i_t leaving_index = primal_ratio_test(lp, + settings, + vstatus, + basic_list, + x, + delta_x, + step_length, + basic_leaving, + entering_index, + direction); if (leaving_index == -1 && step_length >= inf) { settings.log.printf("No leaving variable. Primal unbounded?\n"); return primal_status_t::PRIMAL_UNBOUNDED; } const bool basis_updated = (leaving_index != -1); + bool recompute_duals = false; for (i_t j = 0; j < n; ++j) { x[j] += step_length * delta_x[j]; } + +#ifdef COMPUTE_RESIDUAL + f_t debug_primal_residual = primal_constraint_residual(lp, x); + if (debug_primal_residual > 1e-6) { + settings.log.printf("|| A * x - b || %e at iteration %d (updates %d)\n", debug_primal_residual, iter, basis_update.num_updates()); + } +#endif + + if (basis_updated) { assert(step_length >= 0.0); + + bool should_refactor = basis_update.num_updates() > settings.refactor_frequency; + f_t dual_step_length = 0.0; + if (!should_refactor) { + compute_delta_y(basis_update, basic_leaving, delta_y, etilde); + const f_t pivot = scaled_delta_xB[basic_leaving]; + dual_step_length = compute_dual_step_length(z[entering_index], pivot); + } + basic_list[basic_leaving] = entering_index; nonbasic_list[nonbasic_entering] = leaving_index; vstatus[entering_index] = variable_status_t::BASIC; + // Place the leaver on its leaving bound. If that bound is far from the + // current value (typical after a zero-step leave of an already-infeasible + // basic), rebuild x_B after the factor matches the new basis so Ax = b; + // phase handling below may then (re)enter Phase I if basics are infeasible. + bool rebuild_x_after_bound_snap = false; + f_t leave_bound = 0.0; if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { vstatus[leaving_index] = variable_status_t::NONBASIC_FIXED; - } else if (delta_x[leaving_index] < 0) { - vstatus[leaving_index] = variable_status_t::NONBASIC_LOWER; + leave_bound = lp.lower[leaving_index]; } else { - vstatus[leaving_index] = variable_status_t::NONBASIC_UPPER; + // Classify by which bound was hit. Using sign(delta_x) is wrong when the + // variable approached the bound from the infeasible side (phase I). + const f_t x_leave = x[leaving_index]; + const f_t dist_to_lower = std::abs(x_leave - lp.lower[leaving_index]); + const f_t dist_to_upper = std::abs(x_leave - lp.upper[leaving_index]); + if (lp.lower[leaving_index] > -inf && + (lp.upper[leaving_index] >= inf || dist_to_lower <= dist_to_upper)) { + vstatus[leaving_index] = variable_status_t::NONBASIC_LOWER; + leave_bound = lp.lower[leaving_index]; + } else { + vstatus[leaving_index] = variable_status_t::NONBASIC_UPPER; + leave_bound = lp.upper[leaving_index]; + } + } + if (std::abs(x[leaving_index] - leave_bound) > settings.primal_tol) { + rebuild_x_after_bound_snap = true; } + x[leaving_index] = leave_bound; - bool should_refactor = ft.num_updates() > settings.refactor_frequency; if (!should_refactor) { - i_t recommend_refactor = ft.update(utilde, basic_leaving); - should_refactor = recommend_refactor == 1; + compute_delta_z(Arow, vstatus, delta_y, delta_z); + update_y(dual_step_length, delta_y, y); + update_z(dual_step_length, nonbasic_list, entering_index, delta_z, z); + should_refactor = basis_update.update(utilde_sparse, etilde, basic_leaving) == 1; } if (should_refactor) { - i_t rank = factorize_basis(lp.A, - settings, - basic_list, - start_time, - L, - U, - p, - pinv, - q, - deficient, - slacks_needed, - work_estimate); + i_t rank = basis_update.refactor_basis( + lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); if (rank == CONCURRENT_HALT_RETURN) { return primal_status_t::CONCURRENT_LIMIT; } if (rank == TIME_LIMIT_RETURN) { return primal_status_t::TIME_LIMIT; } - if (rank < 0) { + if (rank != 0) { settings.log.printf("Failed to refactor basis. Iteration %d\n", iter); return primal_status_t::NUMERICAL; } - if (rank != m) { - settings.log.printf("Failed to refactor basis. rank %d m %d\n", rank, m); - return primal_status_t::NUMERICAL; - } - reorder_basic_list(q, basic_list); - ft.reset(L, U, p); + work_estimate = basis_update.work_estimate(); + recompute_duals = true; + // Factor matches basic_list: rebuild x_B so Ax = b exactly. + set_primal_variables_on_bounds(lp, settings, vstatus, x); + compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x); + } else if (rebuild_x_after_bound_snap) { + // FT update already matches the new basis; recompute x_B with the leaver + // snapped onto its bound. + compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x); } } else { if (direction > 0) { @@ -620,33 +1004,53 @@ primal_status_t primal_phase2(i_t phase, } } - // Check if we need to switch to phase 1 - const f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf); if (primal_inf > primal_tol) { - compute_phase1_objective(lp, settings, x, objective); - phase = 1; + if (phase != 1) { + settings.log.printf( + "Switching to Primal Simplex Phase 1. Iteration %d. Primal infeasibility %e\n", + iter, + primal_inf); + settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); + switched_phase = true; + } + compute_phase1_objective(lp, settings, vstatus, x, objective); + phase = 1; + recompute_duals = true; } else if (phase == 1) { - objective = lp.objective; - phase = 2; + objective = lp.objective; + phase = 2; + pricing_dual_tol = settings.dual_tol; + recompute_duals = true; + settings.log.printf( + "Primal phase I complete. Iterations %d. Time %.2f\n", iter, toc(start_time)); + settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); + switched_phase = true; } - if (basis_updated || primal_inf > primal_tol) { - compute_dual_variables(lp, settings, objective, basic_list, nonbasic_list, ft, c_basic, y, z); + if (recompute_duals) { + compute_dual_variables( + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); } - const f_t obj = compute_objective(lp, x); - dual_inf = dual_infeasibility(lp, vstatus, z); - settings.log.printf("%3d %.10e %8.2e %8.2e %8.2e %8d %8d %d %.2f\n", - iter, - compute_user_objective(lp, obj), - primal_inf, - dual_inf, - step_length == 0.0 ? 0.0 : step_length, - entering_index, - leaving_index, - phase, - toc(start_time)); + obj = compute_objective(lp, x); + dual_inf = + dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf); + iter++; + + f_t now = toc(start_time); + if (0|| (iter - start_iter) < settings.first_iteration_log || + (iter % settings.iteration_log_frequency) == 0 || switched_phase) { + const f_t user_obj = compute_user_objective(lp, obj); + settings.log.printf("%5d %+.16e %7d %.8e %.2f\n", + iter, + user_obj, + phase == 1 ? num_primal_inf : num_dual_inf, + phase == 1 ? primal_inf : dual_inf, + now); + switched_phase = false; + } } if (iter == iter_limit) { return primal_status_t::ITERATION_LIMIT; } @@ -677,6 +1081,19 @@ template primal_status_t primal_phase2( lp_solution_t& sol, int& iter); +template primal_status_t primal_phase2_with_advanced_basis( + int phase, + double start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& basis_update, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + int& iter, + double& work_estimate); + #endif } // namespace cuopt::mathematical_optimization::simplex diff --git a/cpp/src/dual_simplex/primal.hpp b/cpp/src/dual_simplex/primal.hpp index 34ffbd8ba5..79008829d5 100644 --- a/cpp/src/dual_simplex/primal.hpp +++ b/cpp/src/dual_simplex/primal.hpp @@ -7,6 +7,7 @@ #pragma once +#include #include #include #include @@ -27,7 +28,6 @@ enum class primal_status_t { CONCURRENT_LIMIT = 6 }; - template i_t primal_ratio_test(const lp_problem_t& lp, const simplex_solver_settings_t& settings, @@ -40,6 +40,20 @@ i_t primal_ratio_test(const lp_problem_t& lp, i_t entering_index, i_t direction); +template +primal_status_t primal_phase2_with_advanced_basis( + i_t phase, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& basis_update, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + i_t& iter, + f_t& work_estimate); + template primal_status_t primal_phase2(i_t phase, f_t start_time, diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index da6834f60f..adad745109 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -61,6 +61,52 @@ void write_matlab(const std::string& filename, const simplex::lp_problem_t +void initialize_slack_basis_vstatus(const lp_problem_t& lp, + std::vector& vstatus) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + vstatus.resize(n); + for (i_t j = 0; j < n; ++j) { + if (lp.lower[j] == -inf && lp.upper[j] == inf) { + vstatus[j] = variable_status_t::NONBASIC_FREE; + } else if (std::abs(lp.upper[j] - lp.lower[j]) < 1e-12) { + vstatus[j] = variable_status_t::NONBASIC_FIXED; + } else if (lp.lower[j] > -inf) { + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } else { + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } + } + i_t num_basic = 0; + for (i_t j = n - 1; j >= 0; --j) { + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + const i_t nz = col_end - col_start; + if (nz == 1 && std::abs(lp.A.x[col_start]) == 1.0) { + vstatus[j] = variable_status_t::BASIC; + num_basic++; + } + if (num_basic == m) { break; } + } + assert(num_basic == m); +} + } // namespace template @@ -288,7 +334,7 @@ lp_status_t solve_linear_program_with_advanced_basis( edge_norms, work_unit_context); } - constexpr bool primal_cleanup = false; + constexpr bool primal_cleanup = true; if (status == dual_status_t::OPTIMAL && primal_cleanup) { settings.log.printf("Running primal cleanup\n"); primal_phase2(2, start_time, lp, settings, vstatus, solution, iter); @@ -684,6 +730,109 @@ lp_status_t solve_linear_program_with_barrier(const user_problem_t& us return solve_linear_program_with_barrier(user_problem, settings, start_time, solution); } +template +lp_status_t solve_linear_program_with_primal(const user_problem_t& user_problem, + const simplex_solver_settings_t& settings, + f_t start_time, + lp_solution_t& solution) +{ + raft::common::nvtx::range scope("PrimalSimplex::solve_lp"); + lp_problem_t original_lp(user_problem.handle_ptr, 1, 1, 1); + std::vector new_slacks; + dualize_info_t dualize_info; + convert_user_problem(user_problem, settings, original_lp, new_slacks, dualize_info); + + solution.resize(user_problem.num_rows, user_problem.num_cols); + lp_solution_t original_solution(original_lp.num_rows, original_lp.num_cols); + + // Presolve adds/retains artificial variables so a full slack basis exists. + lp_problem_t presolved_lp(original_lp.handle_ptr, 1, 1, 1); + presolve_info_t presolve_info; + const i_t ok = presolve(original_lp, settings, presolved_lp, presolve_info); + if (ok == CONCURRENT_HALT_RETURN) { return lp_status_t::CONCURRENT_LIMIT; } + if (ok == TIME_LIMIT_RETURN) { return lp_status_t::TIME_LIMIT; } + if (ok == -1) { return lp_status_t::INFEASIBLE; } + + lp_problem_t lp(original_lp.handle_ptr, + presolved_lp.num_rows, + presolved_lp.num_cols, + presolved_lp.A.col_start[presolved_lp.num_cols]); + std::vector column_scales; + std::vector row_scales; + scaling(presolved_lp, settings, lp, column_scales, row_scales); + + std::vector vstatus; + initialize_slack_basis_vstatus(lp, vstatus); + + lp_solution_t lp_solution(lp.num_rows, lp.num_cols); + i_t iter = 0; + const primal_status_t primal_status = + primal_phase2(2, start_time, lp, settings, vstatus, lp_solution, iter); + lp_solution.iterations = iter; + original_solution.iterations = iter; + + if (primal_status == primal_status_t::CONCURRENT_LIMIT) { + solution.iterations = iter; + return lp_status_t::CONCURRENT_LIMIT; + } + + if (primal_status == primal_status_t::OPTIMAL) { + lp_solution.objective = compute_objective(lp, lp_solution.x); + lp_solution.user_objective = compute_user_objective(lp, lp_solution.objective); + + std::vector residual = lp.rhs; + matrix_vector_multiply(lp.A, 1.0, lp_solution.x, -1.0, residual); + lp_solution.l2_primal_residual = vector_norm2(residual); + + std::vector dual_residual = lp_solution.z; + for (i_t j = 0; j < lp.num_cols; ++j) { + dual_residual[j] -= lp.objective[j]; + } + matrix_transpose_vector_multiply(lp.A, 1.0, lp_solution.y, 1.0, dual_residual); + lp_solution.l2_dual_residual = vector_norm2(dual_residual); + + std::vector unscaled_x(lp.num_cols); + std::vector unscaled_y(lp.num_rows); + std::vector unscaled_z(lp.num_cols); + unscale_solution(column_scales, + row_scales, + lp_solution.x, + lp_solution.y, + lp_solution.z, + unscaled_x, + unscaled_y, + unscaled_z); + uncrush_solution(presolve_info, + settings, + original_lp, + unscaled_x, + unscaled_y, + unscaled_z, + original_solution.x, + original_solution.y, + original_solution.z); + original_solution.objective = lp_solution.objective; + original_solution.user_objective = lp_solution.user_objective; + original_solution.l2_primal_residual = lp_solution.l2_primal_residual; + original_solution.l2_dual_residual = lp_solution.l2_dual_residual; + } + + uncrush_primal_solution(user_problem, original_lp, original_solution.x, solution.x); + uncrush_dual_solution(user_problem, + original_lp, + original_solution.y, + original_solution.z, + solution.y, + solution.z); + solution.objective = original_solution.objective; + solution.user_objective = original_solution.user_objective; + solution.iterations = original_solution.iterations; + solution.l2_primal_residual = original_solution.l2_primal_residual; + solution.l2_dual_residual = original_solution.l2_dual_residual; + return map_primal_status_to_lp_status(primal_status); +} + + template lp_status_t solve_linear_program(const user_problem_t& user_problem, const simplex_solver_settings_t& settings, @@ -831,6 +980,12 @@ template lp_status_t solve_linear_program_with_barrier( double start_time, lp_solution_t& solution); +template lp_status_t solve_linear_program_with_primal( + const user_problem_t& user_problem, + const simplex_solver_settings_t& settings, + double start_time, + lp_solution_t& solution); + template lp_status_t solve_linear_program(const user_problem_t& user_problem, const simplex_solver_settings_t& settings, lp_solution_t& solution); diff --git a/cpp/src/dual_simplex/solve.hpp b/cpp/src/dual_simplex/solve.hpp index 7cc9a9f5cf..f4807306e0 100644 --- a/cpp/src/dual_simplex/solve.hpp +++ b/cpp/src/dual_simplex/solve.hpp @@ -98,6 +98,12 @@ lp_status_t solve_linear_program_with_barrier(const user_problem_t& us f_t start_time, lp_solution_t& solution); +template +lp_status_t solve_linear_program_with_primal(const user_problem_t& user_problem, + const simplex_solver_settings_t& settings, + f_t start_time, + lp_solution_t& solution); + template lp_status_t solve_linear_program(const user_problem_t& user_problem, const simplex_solver_settings_t& settings, diff --git a/cpp/src/math_optimization/solver_settings.cu b/cpp/src/math_optimization/solver_settings.cu index 9193112d71..ab129b9c89 100644 --- a/cpp/src/math_optimization/solver_settings.cu +++ b/cpp/src/math_optimization/solver_settings.cu @@ -131,7 +131,7 @@ solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings {CUOPT_ITERATION_LIMIT, &pdlp_settings.iteration_limit, 0, std::numeric_limits::max(), std::numeric_limits::max()}, {CUOPT_NODE_LIMIT, &mip_settings.node_limit, 0, std::numeric_limits::max(), std::numeric_limits::max()}, {CUOPT_PDLP_SOLVER_MODE, reinterpret_cast(&pdlp_settings.pdlp_solver_mode), CUOPT_PDLP_SOLVER_MODE_STABLE1, CUOPT_PDLP_SOLVER_MODE_STABLE3, CUOPT_PDLP_SOLVER_MODE_STABLE3}, - {CUOPT_METHOD, reinterpret_cast(&pdlp_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_BARRIER, CUOPT_METHOD_CONCURRENT}, + {CUOPT_METHOD, reinterpret_cast(&pdlp_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_PRIMAL, CUOPT_METHOD_CONCURRENT}, {CUOPT_NUM_CPU_THREADS, &mip_settings.num_cpu_threads, -1, std::numeric_limits::max(), -1}, {CUOPT_AUGMENTED, &pdlp_settings.augmented, -1, 1, -1}, {CUOPT_FOLDING, &pdlp_settings.folding, -1, 1, -1}, diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index 25b427fa9a..173619c5d1 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -431,7 +431,7 @@ optimization_problem_solution_t convert_dual_simplex_sol( termination_status != pdlp_termination_status_t::TimeLimit && termination_status != pdlp_termination_status_t::ConcurrentLimit) { CUOPT_LOG_INFO("%s Solve status %s", - method == method_t::DualSimplex ? "Dual Simplex" : "Barrier", + method_to_string(method).c_str(), sol.get_termination_status_string().c_str()); } @@ -630,6 +630,59 @@ optimization_problem_solution_t run_dual_simplex( method_t::DualSimplex); } +template +std::tuple, simplex::lp_status_t, f_t, f_t, f_t> run_primal( + simplex::user_problem_t& user_problem, + pdlp_solver_settings_t const& settings, + const timer_t& timer) +{ + f_t norm_user_objective = vector_norm2(user_problem.objective); + f_t norm_rhs = vector_norm2(user_problem.rhs); + + simplex::simplex_solver_settings_t primal_settings; + primal_settings.time_limit = settings.time_limit; + primal_settings.iteration_limit = settings.iteration_limit; + primal_settings.concurrent_halt = settings.concurrent_halt; + if (primal_settings.concurrent_halt != nullptr) { + // Don't show the primal simplex log in concurrent mode. Show the PDLP log instead + primal_settings.log.log = false; + } + + simplex::lp_solution_t solution(user_problem.num_rows, user_problem.num_cols); + auto status = simplex::solve_linear_program_with_primal( + user_problem, primal_settings, timer.get_tic_start(), solution); + + CUOPT_LOG_CONDITIONAL_INFO( + !settings.inside_mip, "Primal simplex finished in %.2f seconds", timer.elapsed_time()); + + if (settings.concurrent_halt != nullptr && + (status == simplex::lp_status_t::OPTIMAL || status == simplex::lp_status_t::UNBOUNDED || + status == simplex::lp_status_t::INFEASIBLE || + status == simplex::lp_status_t::UNBOUNDED_OR_INFEASIBLE)) { + // We finished. Tell PDLP to stop if it is still running. + *settings.concurrent_halt = 1; + } + + return {std::move(solution), status, timer.elapsed_time(), norm_user_objective, norm_rhs}; +} + +template +optimization_problem_solution_t run_primal(mip::problem_t& problem, + pdlp_solver_settings_t const& settings, + const timer_t& timer) +{ + simplex::user_problem_t primal_problem = + cuopt_problem_to_user_problem(problem.handle_ptr, problem); + auto sol_primal = run_primal(primal_problem, settings, timer); + return convert_dual_simplex_sol(problem, + std::get<0>(sol_primal), + std::get<1>(sol_primal), + std::get<2>(sol_primal), + std::get<3>(sol_primal), + std::get<4>(sol_primal), + method_t::Primal); +} + #if PDLP_INSTANTIATE_FLOAT || CUOPT_INSTANTIATE_FLOAT template @@ -1754,19 +1807,27 @@ optimization_problem_solution_t solve_lp_with_method( if constexpr (std::is_same_v) { if (settings.method == method_t::DualSimplex) { return run_dual_simplex(problem, settings, timer); + } else if (settings.method == method_t::Primal) { + return run_primal(problem, settings, timer); } else if (settings.method == method_t::Barrier) { return run_barrier(problem, settings, timer); } else if (settings.method == method_t::Concurrent) { return run_concurrent(problem, settings, timer, is_batch_mode); + } else if (settings.method == method_t::PDLP) { + return run_pdlp(problem, settings, timer, is_batch_mode); } else { + cuopt_expects(false, + error_type_t::ValidationError, + "Invalid LP method. Valid values: Concurrent(0), PDLP(1), DualSimplex(2), " + "Barrier(3), Primal(4)."); return run_pdlp(problem, settings, timer, is_batch_mode); } } else { // Float precision only supports PDLP without presolve/crossover cuopt_expects(settings.method == method_t::PDLP, error_type_t::ValidationError, - "Float precision only supports PDLP method. DualSimplex, Barrier, and Concurrent " - "require double precision."); + "Float precision only supports PDLP method. DualSimplex, Primal, Barrier, and " + "Concurrent require double precision."); return run_pdlp(problem, settings, timer, is_batch_mode); } } From efc9fb021fba49bbef3506f4c217b44bcb6f17dd Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 28 Jul 2026 10:47:40 -0700 Subject: [PATCH 006/113] Use primal simplex to remove a perturbation from dual simplex --- cpp/src/dual_simplex/phase2.cpp | 59 ++++++++++++++++++++++++++++++--- cpp/src/dual_simplex/primal.cpp | 14 ++++---- cpp/src/dual_simplex/primal.hpp | 5 ++- cpp/src/dual_simplex/solve.cpp | 2 +- 4 files changed, 68 insertions(+), 12 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index c15f7f554c..d3867086f3 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -11,6 +11,7 @@ #include #include #include +#include #include #include #include @@ -2331,13 +2332,15 @@ void prepare_optimality(i_t info, const simplex_solver_settings_t& settings, basis_update_mpf_t& ft, const std::vector& objective, - const std::vector& basic_list, - const std::vector& nonbasic_list, - const std::vector& vstatus, + // Primal cleanup below pivots, so the basis, the statuses + // and the iteration count are updated in place. + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, int phase, f_t start_time, f_t max_val, - i_t iter, + i_t& iter, const std::vector& x, std::vector& y, std::vector& z, @@ -2370,6 +2373,54 @@ void prepare_optimality(i_t info, settings.log.printf("Unperturbed dual infeasibility: %.2e\n", dual_infeas); settings.log.printf("Objective: %+.16e\n", sol.user_objective); settings.log.printf("Num updates: %d\n", ft.num_updates()); + + // Primal pivots in place, so keep the perturbed solution to fall back on. + // The factor is snapshot rather than refactorized on failure: the copy is + // exact, keeps ft consistent with the restored basis, and cannot itself + // fail the way a refactorization can. + const basis_update_mpf_t saved_ft = ft; + const std::vector saved_x = sol.x; + const std::vector saved_y = sol.y; + const std::vector saved_z = sol.z; + const std::vector saved_vstatus = vstatus; + const std::vector saved_basic_list = basic_list; + const std::vector saved_nonbasic_list = nonbasic_list; + + // Reoptimize the unperturbed objective from this basis. The point is + // primal feasible, so primal simplex stays in phase 2 and pivots only to + // restore dual feasibility. It writes through sol, so x, y and z here see + // the cleaned up solution. It prints no summary; the one below reports the + // final result. + primal_status_t primal_status = primal_phase2_with_advanced_basis(2, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + work_estimate, + false); + if (primal_status == primal_status_t::OPTIMAL) { + // z now prices the original objective, so no perturbation remains. + settings.log.printf("Primal cleanup successful.\n"); + perturbation = 0.0; + sol.objective = compute_objective(lp, sol.x); + sol.user_objective = compute_user_objective(lp, sol.objective); + } else { + // Restore the perturbed optimum; a partially pivoted basis is worse than + // the dual feasible point we started from. + settings.log.printf("Primal cleanup failed. Reporting the perturbed solution.\n"); + ft = saved_ft; + sol.x = saved_x; + sol.y = saved_y; + sol.z = saved_z; + vstatus = saved_vstatus; + basic_list = saved_basic_list; + nonbasic_list = saved_nonbasic_list; + } } } } diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index d12f24e98f..7633f9a1b7 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -442,7 +442,7 @@ i_t primal_ratio_test(const lp_problem_t& lp, // step * delta_x[j] >= lp.lower[j] - x[j] // step <= (lp.lower[j] - x[j]) / delta_x[j], delta_x[j] < 0 f_t neum = lp.lower[j] - x[j]; - // A basic sitting a hair below its bound (within the primal tolerance) is on + // A basic sitting below its bound (within the primal tolerance) is on // the bound numerically, but gives a tiny negative ratio. Dropping it lets // the step run straight through the bound, so treat it as a zero-length // block. A genuine violation is left to the branches above, which stop at @@ -466,7 +466,7 @@ i_t primal_ratio_test(const lp_problem_t& lp, // step * delta_x[j] <= lp.upper[j] - x[j] // step <= (lp.upper[j] - x[j]) / delta_x[j], delta_x[j] > 0 f_t neum = lp.upper[j] - x[j]; - // Mirror of the lower bound case: a hair above the bound is on the bound. + // Mirror of the lower bound case: slightly above the bound is considered on the bound. if (neum < 0 && -neum <= settings.primal_tol) { neum = 0.0; } f_t ratio = neum / delta_x[j]; if (ratio >= 0 && ratio < min_val) { @@ -603,7 +603,8 @@ primal_status_t primal_phase2_with_advanced_basis( std::vector& nonbasic_list, lp_solution_t& sol, i_t& iter, - f_t& work_estimate) + f_t& work_estimate, + bool print_summary) { const i_t m = lp.num_rows; const i_t n = lp.num_cols; @@ -793,7 +794,7 @@ primal_status_t primal_phase2_with_advanced_basis( obj = compute_objective(lp, x); sol.objective = obj; sol.user_objective = compute_user_objective(lp, obj); - if (!settings.inside_mip) { + if (!settings.inside_mip && print_summary) { settings.log.printf("\n"); settings.log.printf( "Optimal solution found in %d iterations and %.2fs\n", iter, toc(start_time)); @@ -990,7 +991,7 @@ primal_status_t primal_phase2_with_advanced_basis( set_primal_variables_on_bounds(lp, settings, vstatus, x); compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x); } else if (rebuild_x_after_bound_snap) { - // FT update already matches the new basis; recompute x_B with the leaver + // FT update already matches the new basis; recompute x_B with the leaving variable // snapped onto its bound. compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x); } @@ -1092,7 +1093,8 @@ template primal_status_t primal_phase2_with_advanced_basis( std::vector& nonbasic_list, lp_solution_t& sol, int& iter, - double& work_estimate); + double& work_estimate, + bool print_summary); #endif diff --git a/cpp/src/dual_simplex/primal.hpp b/cpp/src/dual_simplex/primal.hpp index 79008829d5..63f4761c0e 100644 --- a/cpp/src/dual_simplex/primal.hpp +++ b/cpp/src/dual_simplex/primal.hpp @@ -52,7 +52,10 @@ primal_status_t primal_phase2_with_advanced_basis( std::vector& nonbasic_list, lp_solution_t& sol, i_t& iter, - f_t& work_estimate); + f_t& work_estimate, + // Callers that print their own summary (dual simplex perturbation cleanup) + // suppress this one, so optimality is not reported twice. + bool print_summary = true); template primal_status_t primal_phase2(i_t phase, diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index adad745109..a24d12fc55 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -334,7 +334,7 @@ lp_status_t solve_linear_program_with_advanced_basis( edge_norms, work_unit_context); } - constexpr bool primal_cleanup = true; + constexpr bool primal_cleanup = false; if (status == dual_status_t::OPTIMAL && primal_cleanup) { settings.log.printf("Running primal cleanup\n"); primal_phase2(2, start_time, lp, settings, vstatus, solution, iter); From cf3d4d4821f52292653a9d2a81720696a7672d13 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 29 Jul 2026 14:04:05 -0700 Subject: [PATCH 007/113] Clean up logging of degenerate feasibility pump --- cpp/src/branch_and_bound/branch_and_bound.cpp | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 1da2f0f249..5e5e8540df 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3417,10 +3417,12 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple bool recompute_basis = false; const i_t iter_before = iter; f_t primal_work_estimate = 0; + simplex_solver_settings_t primal_settings = settings_; + primal_settings.log.log = false; simplex::primal_status_t lp_status = simplex::primal_phase2_with_advanced_basis(2, exploration_stats_.start_time, lp_reduced, - settings_, + primal_settings, reduced_vstatus, reduced_basis_update, reduced_basic_list, @@ -3446,15 +3448,17 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple // Verify the solution is primal feasible std::vector residual = lp.rhs; matrix_vector_multiply(lp.A, 1.0, adjusted_solution, -1.0, residual); + const f_t primal_residual = vector_norm_inf(residual); - settings_.log.printf("Reduced LP residual|| A*x - b ||_inf = %.16e\n", - vector_norm_inf(residual)); + if (primal_residual > 1e-6) { + settings_.log.printf("Reduced LP residual|| A*x - b ||_inf = %.4e\n", primal_residual); + } std::vector tmp_fractional; i_t num_fractional_reduced = fractional_variables(settings_, adjusted_solution, var_types_, tmp_fractional); settings_.log.printf( - "Reduced LP fractional variables %d/%d\n", num_fractional_reduced, num_fractional); + "Degenerate feasibility pump (%d/%d): fractional variables %d/%d\n", pump_iter, max_pump_iter, num_fractional_reduced, num_fractional); // Also treat a pass that fails to improve the best as a stall, so we perturb // the next pass even when the solve pivoted (moved) without reducing the count. stalled = stalled || (num_fractional_reduced >= best_num_fractional); @@ -3465,7 +3469,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple } } - settings_.log.printf("Best number of fractional variables %d/%d\n", best_num_fractional, num_fractional); + settings_.log.printf("Degenerate feasibility pump: Simplex iterations %d, Best number of fractional variables %d/%d\n", iter, best_num_fractional, num_fractional); if (best_num_fractional < num_fractional) { // Translate the vstatus from the reduced problem to the vstatus for the original problem i_t reduced_cols = 0; From a7cfd191962139fae9f34148543a97ea03a5627b Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 29 Jul 2026 15:03:22 -0700 Subject: [PATCH 008/113] Add work estimates to primal simplex --- cpp/src/branch_and_bound/branch_and_bound.cpp | 4 +- cpp/src/dual_simplex/primal.cpp | 184 +++++++++++++----- cpp/src/dual_simplex/primal.hpp | 3 +- 3 files changed, 136 insertions(+), 55 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 5e5e8540df..57a93eaf47 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3700,6 +3700,7 @@ void branch_and_bound_t::pivot_out_integer_variables( f_t step_length; i_t basic_leaving; + f_t work_estimate = 0.0; const i_t leaving_index = simplex::primal_ratio_test(lp, settings_, vstatus_copy, @@ -3709,7 +3710,8 @@ void branch_and_bound_t::pivot_out_integer_variables( step_length, basic_leaving, entering_index, - direction); + direction, + work_estimate); bool binding_integer = leaving_index != -1 && is_fractional(soln_copy.x[leaving_index], var_types_[leaving_index], settings_.integer_tol); diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 7633f9a1b7..f5a756a78a 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -24,9 +24,12 @@ template void set_primal_variables_on_bounds(const lp_problem_t& lp, const simplex_solver_settings_t& settings, std::vector& vstatus, - std::vector& x) + std::vector& x, + f_t& work_estimate) { - const i_t n = lp.num_cols; + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + constexpr f_t diff_tol = 1e-6; for (i_t j = 0; j < n; ++j) { if (vstatus[j] == variable_status_t::BASIC) { continue; } @@ -54,6 +57,7 @@ void set_primal_variables_on_bounds(const lp_problem_t& lp, assert(1 == 0); } } + work_estimate += n + 3.0*(n - m); } template @@ -61,7 +65,8 @@ f_t dual_infeasibility(const lp_problem_t& lp, const std::vector& vstatus, const std::vector& z, f_t tight_tol, - i_t& num_infeasible) + i_t& num_infeasible, + f_t& work_estimate) { const i_t n = lp.num_cols; num_infeasible = 0; @@ -103,6 +108,7 @@ f_t dual_infeasibility(const lp_problem_t& lp, non_basic_upper_inf++; } } + work_estimate += 8 * n; return sum_infeasible; } @@ -115,7 +121,8 @@ i_t phase2_pricing(const lp_problem_t& lp, f_t dual_tol, i_t& direction, i_t& basic_entering, - f_t& dual_inf) + f_t& dual_inf, + f_t& work_estimate) { const i_t m = lp.num_rows; const i_t n = lp.num_cols; @@ -149,6 +156,7 @@ i_t phase2_pricing(const lp_problem_t& lp, } } } + work_estimate += 4 * (n - m); return entering_index; } @@ -157,8 +165,10 @@ f_t primal_infeasibility(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& vstatus, const std::vector& x, - i_t& num_infeasible) + i_t& num_infeasible, + f_t& work_estimate) { + const i_t m = lp.num_rows; const i_t n = lp.num_cols; f_t primal_inf = 0; num_infeasible = 0; @@ -196,6 +206,7 @@ f_t primal_infeasibility(const lp_problem_t& lp, } } } + work_estimate += n + 4*m; return primal_inf; } @@ -203,19 +214,23 @@ template f_t primal_infeasibility(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& vstatus, - const std::vector& x) + const std::vector& x, + f_t& work_estimate) { i_t num_infeasible = 0; - return primal_infeasibility(lp, settings, vstatus, x, num_infeasible); + return primal_infeasibility(lp, settings, vstatus, x, num_infeasible, work_estimate); } +// work estimate: n-m + 4 * m template void compute_phase1_objective(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& vstatus, const std::vector& x, - std::vector& objective) + std::vector& objective, + f_t& work_estimate) { + const i_t m = lp.num_rows; const i_t n = lp.num_cols; for (i_t j = 0; j < n; ++j) { if (vstatus[j] != variable_status_t::BASIC) { @@ -228,6 +243,7 @@ void compute_phase1_objective(const lp_problem_t& lp, objective[j] = 0.0; } } + work_estimate += n-m + 4 * m; } template @@ -249,11 +265,13 @@ template void compute_delta_z(const csr_matrix_t& Arow, const std::vector& vstatus, const sparse_vector_t& delta_y, - std::vector& delta_z) + std::vector& delta_z, + f_t& work_estimate) { // A^T delta_y + delta_z = 0 // delta_z = -A^T delta_y = - sum_i A(i, :) * delta_y_i std::fill(delta_z.begin(), delta_z.end(), 0.0); + work_estimate += delta_z.size(); for (i_t k = 0; k < static_cast(delta_y.i.size()); ++k) { const i_t i = delta_y.i[k]; const f_t delta_y_i = delta_y.x[k]; @@ -263,7 +281,9 @@ void compute_delta_z(const csr_matrix_t& Arow, const i_t j = Arow.j[p]; if (vstatus[j] != variable_status_t::BASIC) { delta_z[j] -= Arow.x[p] * delta_y_i; } } + work_estimate += 4*(row_end - row_start); } + work_estimate += 4 * delta_y.i.size(); } template @@ -274,12 +294,16 @@ f_t compute_dual_step_length(f_t entering_reduced_cost, f_t pivot) } template -void update_y(f_t dual_step_length, const sparse_vector_t& delta_y, std::vector& y) +void update_y(f_t dual_step_length, + const sparse_vector_t& delta_y, + std::vector& y, + f_t& work_estimate) { for (i_t k = 0; k < static_cast(delta_y.i.size()); ++k) { const i_t i = delta_y.i[k]; y[i] += dual_step_length * delta_y.x[k]; } + work_estimate += 3 * delta_y.i.size(); } template @@ -287,12 +311,14 @@ void update_z(f_t dual_step_length, const std::vector& nonbasic_list, i_t entering_index, const std::vector& delta_z, - std::vector& z) + std::vector& z, + f_t& work_estimate) { for (i_t k = 0; k < static_cast(nonbasic_list.size()); ++k) { const i_t j = nonbasic_list[k]; z[j] += dual_step_length * delta_z[j]; } + work_estimate += 3 * nonbasic_list.size(); z[entering_index] = 0.0; } @@ -305,7 +331,8 @@ void compute_dual_variables(const lp_problem_t& lp, basis_update_mpf_t& ft, std::vector& c_basic, std::vector& y, - std::vector& z) + std::vector& z, + f_t& work_estimate) { const i_t m = lp.num_rows; const i_t n = lp.num_cols; @@ -314,6 +341,7 @@ void compute_dual_variables(const lp_problem_t& lp, const i_t j = basic_list[k]; c_basic[k] = objective[j]; } + work_estimate += 3 * m; ft.b_transpose_solve(c_basic, y); // zN = cN - N'*y for (i_t k = 0; k < n - m; k++) { @@ -328,12 +356,15 @@ void compute_dual_variables(const lp_problem_t& lp, for (i_t p = col_start; p < col_end; ++p) { dot += lp.A.x[p] * y[lp.A.i[p]]; } + work_estimate += 3.0*(col_end - col_start); z[j] -= dot; } + work_estimate += 6 * (n - m); // zB = 0 for (i_t k = 0; k < m; ++k) { z[basic_list[k]] = 0.0; } + work_estimate += 2*m; } template @@ -341,7 +372,8 @@ void compute_basic_primal_variables(const lp_problem_t& lp, const basis_update_mpf_t& basis_update, const std::vector& basic_list, const std::vector& nonbasic_list, - std::vector& x) + std::vector& x, + f_t& work_estimate) { const i_t m = lp.num_rows; const i_t n = lp.num_cols; @@ -354,12 +386,16 @@ void compute_basic_primal_variables(const lp_problem_t& lp, for (i_t p = col_start; p < col_end; ++p) { rhs[lp.A.i[p]] -= xj * lp.A.x[p]; } + work_estimate += 3.0*(col_end - col_start); } + work_estimate += 4 * (n - m); std::vector xB(m); + work_estimate += m; basis_update.b_solve(rhs, xB); for (i_t k = 0; k < m; ++k) { x[basic_list[k]] = xB[k]; } + work_estimate += 3 * m; } template @@ -383,7 +419,8 @@ i_t primal_ratio_test(const lp_problem_t& lp, f_t& step_length, i_t& basic_leaving, i_t entering_index, - i_t direction) + i_t direction, + f_t& work_estimate) { const i_t m = lp.num_rows; const i_t n = lp.num_cols; @@ -482,6 +519,7 @@ i_t primal_ratio_test(const lp_problem_t& lp, } } } + work_estimate += 10*m; step_length = min_val; return leaving_index; } @@ -505,6 +543,7 @@ primal_status_t primal_phase2(i_t phase, std::vector superbasic_list; get_basis_from_vstatus(m, vstatus, basic_list, nonbasic_list, superbasic_list); + work_estimate += 2*n; assert(superbasic_list.size() == 0); assert(nonbasic_list.size() == n - m); @@ -623,12 +662,14 @@ primal_status_t primal_phase2_with_advanced_basis( std::vector incoming_x = x; std::vector incoming_vstatus = vstatus; + work_estimate += 2.0 * n; settings.log.printf("Primal Simplex\n"); // Nonbasics must be on their bounds before forming B x_B = b - A_N x_N. // Setting them after the solve leaves ||A*x - b|| large whenever x_N != 0. - set_primal_variables_on_bounds(lp, settings, vstatus, x); + set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); std::vector rhs = lp.rhs; + work_estimate += m; // rhs = b - sum_{j : x_j = l_j} A(:, j) l(j) - sum_{j : x_j = u_j} A(:, j) * // u(j) for (i_t k = 0; k < n - m; ++k) { @@ -639,34 +680,45 @@ primal_status_t primal_phase2_with_advanced_basis( for (i_t p = col_start; p < col_end; ++p) { rhs[lp.A.i[p]] -= xj * lp.A.x[p]; } + work_estimate += 3.0*(col_end - col_start); } + work_estimate += 4 * (n - m); + std::vector xB(m); + work_estimate += m; + basis_update.b_solve(rhs, xB); for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; x[j] = xB[k]; } + work_estimate += 3 * m; + constexpr bool print_norms = false; if constexpr (print_norms) { settings.log.printf("|| x || %e\n", vector_norm2(x)); } std::vector residual = lp.rhs; + work_estimate += m; matrix_vector_multiply(lp.A, 1.0, x, -1.0, residual); + work_estimate += m + 2*n + 4.0*lp.A.col_start[lp.A.n]; f_t primal_residual = vector_norm_inf(residual); + work_estimate += m; if (primal_residual > settings.primal_tol) { settings.log.printf("|| A*x - b || %e\n", primal_residual); } std::vector objective = lp.objective; + work_estimate += 2*n; const f_t primal_tol = settings.primal_tol; - f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); + f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x, work_estimate); if (primal_inf > primal_tol) { // We are primal infeasible. Switch to phase 1 - compute_phase1_objective(lp, settings, vstatus, x, objective); + compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); settings.log.printf("Phase 1\n"); settings.log.printf("Initial primal infeasibility %e\n", primal_inf); phase = 1; @@ -676,8 +728,9 @@ primal_status_t primal_phase2_with_advanced_basis( } std::vector c_basic(m); + work_estimate += m; compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); if constexpr (print_norms) { settings.log.printf("|| z || %e\n", vector_norm_inf(z)); } @@ -685,13 +738,15 @@ primal_status_t primal_phase2_with_advanced_basis( i_t num_dual_inf = 0; i_t num_primal_inf = 0; const f_t init_dual_inf = - dual_infeasibility(lp, vstatus, z, settings.dual_tol, num_dual_inf); + dual_infeasibility(lp, vstatus, z, settings.dual_tol, num_dual_inf, work_estimate); if (num_dual_inf > 0) { settings.log.printf("Initial dual infeasibility %e\n", init_dual_inf); } csr_matrix_t Arow(m, n, lp.A.nnz()); + work_estimate += n + 2*lp.A.nnz(); lp.A.to_compressed_row(Arow); + work_estimate += m + 6*lp.A.nnz(); const i_t iter_limit = settings.iteration_limit; const i_t start_iter = iter; @@ -699,11 +754,13 @@ primal_status_t primal_phase2_with_advanced_basis( sparse_vector_t etilde(m, 0); std::vector delta_z(n); std::vector delta_x(n); + work_estimate += 2*m + 2*n; f_t dual_inf = init_dual_inf; f_t obj = compute_objective(lp, x); + work_estimate += 2*n; f_t pricing_dual_tol = settings.dual_tol; - primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf); + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); settings.log.printf("%5d %+.16e %7d %.8e %.2f\n", iter, @@ -712,11 +769,15 @@ primal_status_t primal_phase2_with_advanced_basis( phase == 1 ? primal_inf : dual_inf, toc(start_time)); bool switched_phase = false; + + work_estimate += basis_update.work_estimate(); + basis_update.clear_work_estimate(); + while (iter < iter_limit) { i_t nonbasic_entering = -1; i_t direction; i_t entering_index = phase2_pricing( - lp, z, nonbasic_list, vstatus, pricing_dual_tol, direction, nonbasic_entering, dual_inf); + lp, z, nonbasic_list, vstatus, pricing_dual_tol, direction, nonbasic_entering, dual_inf, work_estimate); if (entering_index == -1) { if (phase == 2) { // Verify optimality with a consistent basic solution: refactor, put @@ -732,23 +793,24 @@ primal_status_t primal_phase2_with_advanced_basis( iter); return primal_status_t::NUMERICAL; } - work_estimate = basis_update.work_estimate(); + work_estimate += basis_update.work_estimate(); + basis_update.clear_work_estimate(); } - set_primal_variables_on_bounds(lp, settings, vstatus, x); - compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x); + set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); + compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x, work_estimate); compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); - primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf); + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); dual_inf = - dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf); + dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf, work_estimate); if (primal_inf > primal_tol) { - compute_phase1_objective(lp, settings, vstatus, x, objective); + compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); phase = 1; pricing_dual_tol = settings.dual_tol; compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); settings.log.printf( - "Switching to Primal Simplex Phase 1 after optimality refresh. " + "Switching to Primal Simplex Phase 1 after near optimality. " "Primal infeasibility %e\n", primal_inf); settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); @@ -763,7 +825,7 @@ primal_status_t primal_phase2_with_advanced_basis( i_t num_tight_dual_inf = 0; const f_t tight_dual_inf = - dual_infeasibility(lp, vstatus, z, f_t(0.0), num_tight_dual_inf); + dual_infeasibility(lp, vstatus, z, f_t(0.0), num_tight_dual_inf, work_estimate); if (tight_dual_inf > settings.dual_tol) { // No candidate is visible at the active pricing tolerance, but the // zero-tolerance residual is still material. Try tighter pricing before @@ -781,7 +843,8 @@ primal_status_t primal_phase2_with_advanced_basis( retry_dual_tol, direction, nonbasic_entering, - retry_dual_inf); + retry_dual_inf, + work_estimate); } if (retry_entering != -1) { pricing_dual_tol = retry_dual_tol; @@ -792,6 +855,7 @@ primal_status_t primal_phase2_with_advanced_basis( dual_inf = tight_dual_inf; num_dual_inf = num_tight_dual_inf; obj = compute_objective(lp, x); + work_estimate += 2*n; sol.objective = obj; sol.user_objective = compute_user_objective(lp, obj); if (!settings.inside_mip && print_summary) { @@ -807,7 +871,7 @@ primal_status_t primal_phase2_with_advanced_basis( } return primal_status_t::OPTIMAL; } else { - primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf); + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); if (primal_inf > primal_tol) { // Incremental duals may be stale relative to the current phase-I @@ -815,9 +879,9 @@ primal_status_t primal_phase2_with_advanced_basis( // successively tighter dual tolerances. settings.log.printf("Refreshing phase-I objective and duals. Num updates %d. Iter %d\n", basis_update.num_updates(), iter); - compute_phase1_objective(lp, settings, vstatus, x, objective); + compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); f_t retry_dual_tol = pricing_dual_tol; while (entering_index == -1 && retry_dual_tol > f_t(1e-10)) { retry_dual_tol *= f_t(0.1); @@ -829,7 +893,8 @@ primal_status_t primal_phase2_with_advanced_basis( retry_dual_tol, direction, nonbasic_entering, - dual_inf); + dual_inf, + work_estimate); } if (entering_index == -1) { settings.log.printf( @@ -849,10 +914,11 @@ primal_status_t primal_phase2_with_advanced_basis( "Primal phase I complete. Iterations %d. Time %.2f\n", iter, toc(start_time)); settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); obj = compute_objective(lp, x); + work_estimate += 2*n; dual_inf = - dual_infeasibility(lp, vstatus, z, settings.dual_tol, num_dual_inf); + dual_infeasibility(lp, vstatus, z, settings.dual_tol, num_dual_inf, work_estimate); iter++; // Print here: continue may hit dual-optimal Phase 2 and return before // the end-of-loop log checks switched_phase. @@ -868,20 +934,24 @@ primal_status_t primal_phase2_with_advanced_basis( } sparse_vector_t rhs_sparse(lp.A, entering_index); + work_estimate += 3*rhs_sparse.i.size(); sparse_vector_t scaled_delta_xB_sparse(m, 0); sparse_vector_t utilde_sparse(m, 0); basis_update.b_solve(rhs_sparse, scaled_delta_xB_sparse, utilde_sparse); std::vector scaled_delta_xB(m); scaled_delta_xB_sparse.to_dense(scaled_delta_xB); - + work_estimate += m + scaled_delta_xB_sparse.i.size(); + for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; delta_x[j] = -direction * scaled_delta_xB[k]; } + work_estimate += 3*m; for (i_t k = 0; k < n - m; ++k) { const i_t j = nonbasic_list[k]; delta_x[j] = 0.0; } + work_estimate += 2*(n - m); delta_x[entering_index] = direction; #ifdef CHECK_NULLSPACE @@ -907,7 +977,8 @@ primal_status_t primal_phase2_with_advanced_basis( step_length, basic_leaving, entering_index, - direction); + direction, + work_estimate); if (leaving_index == -1 && step_length >= inf) { settings.log.printf("No leaving variable. Primal unbounded?\n"); return primal_status_t::PRIMAL_UNBOUNDED; @@ -918,6 +989,7 @@ primal_status_t primal_phase2_with_advanced_basis( for (i_t j = 0; j < n; ++j) { x[j] += step_length * delta_x[j]; } + work_estimate += 2*n; #ifdef COMPUTE_RESIDUAL f_t debug_primal_residual = primal_constraint_residual(lp, x); @@ -971,9 +1043,9 @@ primal_status_t primal_phase2_with_advanced_basis( x[leaving_index] = leave_bound; if (!should_refactor) { - compute_delta_z(Arow, vstatus, delta_y, delta_z); - update_y(dual_step_length, delta_y, y); - update_z(dual_step_length, nonbasic_list, entering_index, delta_z, z); + compute_delta_z(Arow, vstatus, delta_y, delta_z, work_estimate); + update_y(dual_step_length, delta_y, y, work_estimate); + update_z(dual_step_length, nonbasic_list, entering_index, delta_z, z, work_estimate); should_refactor = basis_update.update(utilde_sparse, etilde, basic_leaving) == 1; } if (should_refactor) { @@ -985,15 +1057,16 @@ primal_status_t primal_phase2_with_advanced_basis( settings.log.printf("Failed to refactor basis. Iteration %d\n", iter); return primal_status_t::NUMERICAL; } - work_estimate = basis_update.work_estimate(); + work_estimate += basis_update.work_estimate(); + basis_update.clear_work_estimate(); recompute_duals = true; // Factor matches basic_list: rebuild x_B so Ax = b exactly. - set_primal_variables_on_bounds(lp, settings, vstatus, x); - compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x); + set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); + compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x, work_estimate); } else if (rebuild_x_after_bound_snap) { // FT update already matches the new basis; recompute x_B with the leaving variable // snapped onto its bound. - compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x); + compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x, work_estimate); } } else { if (direction > 0) { @@ -1005,7 +1078,7 @@ primal_status_t primal_phase2_with_advanced_basis( } } - primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf); + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); if (primal_inf > primal_tol) { if (phase != 1) { settings.log.printf( @@ -1015,7 +1088,7 @@ primal_status_t primal_phase2_with_advanced_basis( settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); switched_phase = true; } - compute_phase1_objective(lp, settings, vstatus, x, objective); + compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); phase = 1; recompute_duals = true; } else if (phase == 1) { @@ -1031,17 +1104,18 @@ primal_status_t primal_phase2_with_advanced_basis( if (recompute_duals) { compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z); + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); } obj = compute_objective(lp, x); + work_estimate += 2*n; dual_inf = - dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf); + dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf, work_estimate); iter++; f_t now = toc(start_time); - if (0|| (iter - start_iter) < settings.first_iteration_log || + if ((iter - start_iter) < settings.first_iteration_log || (iter % settings.iteration_log_frequency) == 0 || switched_phase) { const f_t user_obj = compute_user_objective(lp, obj); settings.log.printf("%5d %+.16e %7d %.8e %.2f\n", @@ -1052,6 +1126,9 @@ primal_status_t primal_phase2_with_advanced_basis( now); switched_phase = false; } + + work_estimate += basis_update.work_estimate(); + basis_update.clear_work_estimate(); } if (iter == iter_limit) { return primal_status_t::ITERATION_LIMIT; } @@ -1071,7 +1148,8 @@ int primal_ratio_test(const lp_problem_t& lp, double& step_length, int& basic_leaving, int entering_index, - int direction); + int direction, + double& work_estimate); template primal_status_t primal_phase2( int phase, diff --git a/cpp/src/dual_simplex/primal.hpp b/cpp/src/dual_simplex/primal.hpp index 63f4761c0e..df2c998db5 100644 --- a/cpp/src/dual_simplex/primal.hpp +++ b/cpp/src/dual_simplex/primal.hpp @@ -38,7 +38,8 @@ i_t primal_ratio_test(const lp_problem_t& lp, f_t& step_length, i_t& basic_leaving, i_t entering_index, - i_t direction); + i_t direction, + f_t& work_estimate); template primal_status_t primal_phase2_with_advanced_basis( From 6dfebbfcbf77e0a92ed784724df356bae0b424be Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 29 Jul 2026 15:10:31 -0700 Subject: [PATCH 009/113] Display work estimate and simplex iterations --- cpp/src/branch_and_bound/branch_and_bound.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 57a93eaf47..c7338f74aa 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3379,6 +3379,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple reduced_basic_list[k] = original_col_to_reduced_col[basic_list[k]]; } + f_t primal_work_estimate = 0.0; i_t iter = 0; i_t max_pump_iter = 10; simplex::random_t rng(settings_.random_seed); @@ -3416,7 +3417,6 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple bool recompute_basis = false; const i_t iter_before = iter; - f_t primal_work_estimate = 0; simplex_solver_settings_t primal_settings = settings_; primal_settings.log.log = false; simplex::primal_status_t lp_status = simplex::primal_phase2_with_advanced_basis(2, @@ -3458,7 +3458,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple i_t num_fractional_reduced = fractional_variables(settings_, adjusted_solution, var_types_, tmp_fractional); settings_.log.printf( - "Degenerate feasibility pump (%d/%d): fractional variables %d/%d\n", pump_iter, max_pump_iter, num_fractional_reduced, num_fractional); + "Degenerate feasibility pump (%d/%d): primal work estimate %.2e, iter %d, fractional variables %d/%d\n", pump_iter, max_pump_iter, primal_work_estimate, iter, num_fractional_reduced, num_fractional); // Also treat a pass that fails to improve the best as a stall, so we perturb // the next pass even when the solve pivoted (moved) without reducing the count. stalled = stalled || (num_fractional_reduced >= best_num_fractional); From 99d7eade28398413fbe80b92a514706f1495466a Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Mon, 3 Aug 2026 17:15:38 -0700 Subject: [PATCH 010/113] Primal in crossover. Crossover tolerance mismatch fix. Pipe work estimates for root relaxation. Add initial perturbation parameter --- .../mathematical_optimization/constants.h | 1 + .../pdlp/solver_settings.hpp | 1 + cpp/src/branch_and_bound/branch_and_bound.cpp | 424 ++++++++++++------ cpp/src/branch_and_bound/branch_and_bound.hpp | 27 +- cpp/src/branch_and_bound/pseudo_costs.cpp | 6 +- cpp/src/dual_simplex/basis_updates.cpp | 29 ++ cpp/src/dual_simplex/basis_updates.hpp | 8 + cpp/src/dual_simplex/crossover.cpp | 65 ++- cpp/src/dual_simplex/phase2.cpp | 25 +- cpp/src/dual_simplex/phase2.hpp | 2 + .../dual_simplex/simplex_solver_settings.hpp | 1 + cpp/src/dual_simplex/solve.cpp | 17 +- cpp/src/dual_simplex/solve.hpp | 2 + cpp/src/math_optimization/solver_settings.cu | 1 + cpp/src/pdlp/solve.cu | 1 + 15 files changed, 448 insertions(+), 162 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/constants.h b/cpp/include/cuopt/mathematical_optimization/constants.h index 4ed3723aa2..9752b41937 100644 --- a/cpp/include/cuopt/mathematical_optimization/constants.h +++ b/cpp/include/cuopt/mathematical_optimization/constants.h @@ -52,6 +52,7 @@ #define CUOPT_ELIMINATE_DENSE_COLUMNS "eliminate_dense_columns" #define CUOPT_CUDSS_DETERMINISTIC "cudss_deterministic" #define CUOPT_PRESOLVE "presolve" +#define CUOPT_INITIAL_PERTURBATION "initial_perturbation" #define CUOPT_MIP_PROBING "mip_probing" #define CUOPT_DUAL_POSTSOLVE "dual_postsolve" #define CUOPT_MIP_DETERMINISM_MODE "mip_determinism_mode" diff --git a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp index 3bf3b6ab01..521b234b52 100644 --- a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp @@ -282,6 +282,7 @@ class pdlp_solver_settings_t { i_t augmented{-1}; i_t dualize{-1}; i_t ordering{-1}; + i_t initial_perturbation{-1}; i_t barrier_dual_initial_point{-1}; bool eliminate_dense_columns{true}; pdlp_precision_t pdlp_precision{pdlp_precision_t::DefaultPrecision}; diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index c7338f74aa..9303159465 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -708,9 +708,18 @@ bool branch_and_bound_t::repair_solution(const std::vector& edge_ lp_settings.set_log(false); lp_settings.inside_mip = 2; std::vector leaf_edge_norms = edge_norms; + f_t repair_work_estimate = 0.0; // should probably set the cut off here lp_settings.cut_off - dual_status_t lp_status = simplex::dual_phase2( - 2, 0, lp_start_time, repair_lp, lp_settings, vstatus, lp_solution, iter, leaf_edge_norms); + dual_status_t lp_status = simplex::dual_phase2(2, + 0, + lp_start_time, + repair_lp, + lp_settings, + vstatus, + lp_solution, + iter, + repair_work_estimate, + leaf_edge_norms); repaired_solution = lp_solution.x; if (lp_status == dual_status_t::OPTIMAL) { @@ -1608,8 +1617,9 @@ dual_status_t branch_and_bound_t::solve_node_lp( feasible = apply_symmetry_reductions(node_ptr, worker, stats); if (feasible) { - i_t node_iter = 0; - f_t lp_start_time = tic(); + i_t node_iter = 0; + f_t lp_start_time = tic(); + f_t node_work_estimate = 0.0; lp_status = dual_phase2_with_advanced_basis(2, 0, @@ -1623,6 +1633,7 @@ dual_status_t branch_and_bound_t::solve_node_lp( worker->nonbasic_list, worker->leaf_solution, node_iter, + node_work_estimate, worker->leaf_edge_norms); if (lp_status == dual_status_t::NUMERICAL) { @@ -1636,7 +1647,8 @@ dual_status_t branch_and_bound_t::solve_node_lp( worker->basic_list, worker->nonbasic_list, worker->leaf_vstatus, - worker->leaf_edge_norms); + worker->leaf_edge_norms, + node_work_estimate); lp_status = convert_lp_status_to_dual_status(second_status); } @@ -2796,7 +2808,8 @@ lp_status_t branch_and_bound_t::solve_root_relaxation( basis_update_mpf_t& basis_update, std::vector& basic_list, std::vector& nonbasic_list, - std::vector& edge_norms) + std::vector& edge_norms, + f_t& work_estimate) { lp_status_t root_status; @@ -2812,6 +2825,7 @@ lp_status_t branch_and_bound_t::solve_root_relaxation( nonbasic_list, root_vstatus_, edge_norms_, + work_estimate, nullptr); } @@ -3108,6 +3122,7 @@ auto branch_and_bound_t::do_cut_pass( bool initialize_basis = false; lp_settings.concurrent_halt = NULL; f_t dual_phase2_start_time = tic(); + f_t cut_work_estimate = 0.0; dual_status_t cut_status = dual_phase2_with_advanced_basis(2, 0, initialize_basis, @@ -3120,6 +3135,7 @@ auto branch_and_bound_t::do_cut_pass( nonbasic_list, root_relax_soln_, iter, + cut_work_estimate, edge_norms_); exploration_stats_.total_simplex_iters += iter; f_t dual_phase2_time = toc(dual_phase2_start_time); @@ -3143,7 +3159,8 @@ auto branch_and_bound_t::do_cut_pass( basic_list, nonbasic_list, root_vstatus_, - edge_norms_); + edge_norms_, + cut_work_estimate); if (scratch_status == lp_status_t::OPTIMAL) { // We recovered cut_status = convert_lp_status_to_dual_status(scratch_status); @@ -3171,7 +3188,7 @@ auto branch_and_bound_t::do_cut_pass( basis_update, num_fractional, fractional); - + dual_degenerate_feasibility_pump(original_lp_, basic_list, nonbasic_list, @@ -3275,13 +3292,14 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple i_t& num_fractional, std::vector& fractional) { + f_t dual_degenerate_feasibility_pump_start_time = tic(); std::vector zero_reduced_costs_vars; std::vector zero_reduced_costs_vars_nonbasic_index; bool dual_degenerate = check_for_dual_degeneracy(soln, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); if (!dual_degenerate) { return; } - // Construct a new LP problem - // minimize p^T x + // Construct a new LP problem + // minimize p^T x // subject to B x_B + N_z x_z = b - N x_N // l_B <= x_B <= u_B // l_z <= x_z <= u_z @@ -3339,7 +3357,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple lp_reduced.rhs = b_reduced; lp_reduced.obj_scale = 1.0; - + settings_.log.printf("Constructed dual degenerate feasibility pump LP with %d rows and %d columns\n", m, n); @@ -3354,7 +3372,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple reduced_vstatus[reduced_col++] = variable_status_t::BASIC; } else if (std::abs(soln.z[j]) <= 1e-10) { reduced_nonbasic_list[num_nonbasic++] = reduced_col; // Does ordering of nonbasic variables matter? - reduced_vstatus[reduced_col++] = vstatus[j]; + reduced_vstatus[reduced_col++] = vstatus[j]; } } @@ -3458,7 +3476,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple i_t num_fractional_reduced = fractional_variables(settings_, adjusted_solution, var_types_, tmp_fractional); settings_.log.printf( - "Degenerate feasibility pump (%d/%d): primal work estimate %.2e, iter %d, fractional variables %d/%d\n", pump_iter, max_pump_iter, primal_work_estimate, iter, num_fractional_reduced, num_fractional); + "Degenerate feasibility pump (%d/%d): primal work estimate %.2e, iter %d, fractional variables %d/%d. Time %.2f\n", pump_iter, max_pump_iter, primal_work_estimate, iter, num_fractional_reduced, num_fractional, toc(dual_degenerate_feasibility_pump_start_time)); // Also treat a pass that fails to improve the best as a stall, so we perturb // the next pass even when the solve pivoted (moved) without reducing the count. stalled = stalled || (num_fractional_reduced >= best_num_fractional); @@ -3469,7 +3487,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple } } - settings_.log.printf("Degenerate feasibility pump: Simplex iterations %d, Best number of fractional variables %d/%d\n", iter, best_num_fractional, num_fractional); + settings_.log.printf("Degenerate feasibility pump: Simplex iterations %d, Best number of fractional variables %d/%d. Time %.2f\n", iter, best_num_fractional, num_fractional, toc(dual_degenerate_feasibility_pump_start_time)); if (best_num_fractional < num_fractional) { // Translate the vstatus from the reduced problem to the vstatus for the original problem i_t reduced_cols = 0; @@ -3538,13 +3556,120 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple for (i_t k = 0; k < lp.num_rows; k++) { soln.x[basic_list[k]] = xB[k]; } - + fractional.clear(); num_fractional = fractional_variables(settings_, soln.x, var_types_, fractional); } } + +template +void branch_and_bound_t::apply_delta_x_for_integer_pivot( + const simplex::lp_problem_t& lp, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& nonbasic_index, + std::vector& vstatus, + i_t entering_index, + i_t nonbasic_entering, + i_t direction, + std::vector& delta_x, + const sparse_vector_t& utilde_sparse, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + f_t& work_estimate) +{ + f_t step_length; + i_t basic_leaving; + const i_t leaving_index = simplex::primal_ratio_test(lp, + settings_, + vstatus, + basic_list, + solution.x, + delta_x, + step_length, + basic_leaving, + entering_index, + direction, + work_estimate); + bool binding_integer = + leaving_index != -1 && + is_fractional(solution.x[leaving_index], var_types_[leaving_index], settings_.integer_tol); + if (!binding_integer) { return; } + + std::vector test_x = solution.x; + i_t integer_destroyed = 0; + for (i_t h = 0; h < lp.num_cols; ++h) { + test_x[h] += step_length * delta_x[h]; + if (var_types_[h] != variable_type_t::INTEGER) { continue; } + const bool was_fractional = + is_fractional(solution.x[h], var_types_[h], settings_.integer_tol); + const bool now_fractional = + is_fractional(test_x[h], var_types_[h], settings_.integer_tol); + if (now_fractional && !was_fractional) { + integer_destroyed++; + } else if (!now_fractional && was_fractional) { + integer_destroyed--; + } + } + // Require a strict net decrease in fractional integers. + if (integer_destroyed >= 0) { return; } + + solution.x = test_x; + basic_list[basic_leaving] = entering_index; + nonbasic_list[nonbasic_entering] = leaving_index; + vstatus[entering_index] = variable_status_t::BASIC; + if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { + vstatus[leaving_index] = variable_status_t::NONBASIC_FIXED; + } else if (delta_x[leaving_index] < 0) { + vstatus[leaving_index] = variable_status_t::NONBASIC_LOWER; + } else { + vstatus[leaving_index] = variable_status_t::NONBASIC_UPPER; + } + + // Keep nonbasic_index consistent with nonbasic_list: entering_index is now basic, + // and leaving_index has taken its slot in nonbasic_list. + nonbasic_index[entering_index] = -1; + nonbasic_index[leaving_index] = nonbasic_entering; + + const i_t m = lp.num_rows; + sparse_vector_t es_sparse(m, 1); + es_sparse.i[0] = basic_leaving; + es_sparse.x[0] = 1.0; + sparse_vector_t UTsol_sparse(m, 1); + sparse_vector_t solution_sparse(m, 1); + basis_update.b_transpose_solve(es_sparse, solution_sparse, UTsol_sparse); + const i_t recommend_refactor = + basis_update.update(utilde_sparse, UTsol_sparse, basic_leaving); + if (recommend_refactor == 1) { + csc_matrix_t L(m, m, 1); + csc_matrix_t U(m, m, 1); + std::vector pinv(m); + std::vector p(m); + std::vector q(m); + std::vector deficient; + std::vector slacks_needed; + f_t factorize_work_estimate = 0.0; + const i_t rank = factorize_basis(lp.A, + settings_, + basic_list, + exploration_stats_.start_time, + L, + U, + p, + pinv, + q, + deficient, + slacks_needed, + factorize_work_estimate); + if (rank == CONCURRENT_HALT_RETURN || rank == TIME_LIMIT_RETURN) { return; } + if (rank < 0 || rank != lp.num_rows) { return; } + simplex::reorder_basic_list(q, basic_list); + basis_update.reset(L, U, p); + } +} + template void branch_and_bound_t::pivot_out_integer_variables( const simplex::lp_problem_t& lp, @@ -3556,12 +3681,12 @@ void branch_and_bound_t::pivot_out_integer_variables( i_t& num_fractional, std::vector& fractional) { - + f_t pivot_out_integer_variables_start_time = tic(); std::vector zero_reduced_costs_vars; std::vector zero_reduced_costs_vars_nonbasic_index; bool dual_degenerate = check_for_dual_degeneracy(solution, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); if (!dual_degenerate) { return; } - + lp_solution_t soln_copy = solution; std::vector basic_list_copy = basic_list; std::vector nonbasic_list_copy = nonbasic_list; @@ -3571,7 +3696,7 @@ void branch_and_bound_t::pivot_out_integer_variables( const i_t start_num_fractional = num_fractional; const i_t num_zero_reduced_costs_vars = zero_reduced_costs_vars.size(); - + std::vector row_to_slack(lp.num_rows, -1); for (i_t j : new_slacks_) { if (lp.lower[j] != 0 || lp.upper[j] != inf) { continue; } @@ -3579,18 +3704,21 @@ void branch_and_bound_t::pivot_out_integer_variables( row_to_slack[lp.A.i[p]] = j; } + f_t work_estimate = 0.0; + std::vector fast_candidates; std::vector fast_rows; + std::vector fast_nonbasic_slacks; for (i_t j : fractional) { - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - const i_t num_rows = col_end - col_start; - i_t num_basic_slacks = 0; + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + const i_t num_rows = col_end - col_start; + i_t num_basic_slacks = 0; i_t num_nonbasic_slacks_with_reduced_cost_zero = 0; - i_t nonbasic_slack = -1; - i_t slack_row = -1; + i_t nonbasic_slack = -1; + i_t slack_row = -1; for (i_t p = col_start; p < col_end; p++) { - const i_t i = lp.A.i[p]; + const i_t i = lp.A.i[p]; const i_t slack = row_to_slack[i]; if (slack >= 0) { if (vstatus_copy[slack] == variable_status_t::BASIC) { @@ -3598,28 +3726,43 @@ void branch_and_bound_t::pivot_out_integer_variables( } else if (std::abs(solution.z[slack]) <= 1e-10) { num_nonbasic_slacks_with_reduced_cost_zero++; nonbasic_slack = slack; - slack_row = i; + slack_row = i; } } } if (num_basic_slacks == num_rows - 1 && num_nonbasic_slacks_with_reduced_cost_zero == 1) { fast_candidates.push_back(j); fast_rows.push_back(slack_row); + fast_nonbasic_slacks.push_back(nonbasic_slack); } } if (fast_candidates.size() > 0) { - settings_.log.printf("Found %ld fast candidates for pivot out integer variables\n", fast_candidates.size()); + settings_.log.printf("Found %ld fast candidates for pivot out integer variables\n", + fast_candidates.size()); + } + + // Build a reverse index nonbasic_index[v] = position of v in nonbasic_list_copy, or -1 if not + // present. Used to locate the entering variable's slot in the fast-candidate path. + // apply_delta_x_for_integer_pivot keeps this index consistent by applying an O(1) fix-up + // on each successful pivot; the two variables whose (non)basic status changes are the only + // entries that need to be updated. + std::vector nonbasic_index(lp.num_cols, -1); + for (i_t p = 0; p < static_cast(nonbasic_list_copy.size()); ++p) { + nonbasic_index[nonbasic_list_copy[p]] = p; } const i_t num_candidates = fast_candidates.size(); for (i_t k = 0; k < num_candidates; k++) { - const i_t j = fast_candidates[k]; - const i_t row = fast_rows[k]; + const i_t j = fast_candidates[k]; + const i_t row = fast_rows[k]; + const i_t nonbasic_slack = fast_nonbasic_slacks[k]; + // Skip if state changed by a prior successful pivot. + if (vstatus_copy[j] != variable_status_t::BASIC) { continue; } + if (vstatus_copy[nonbasic_slack] == variable_status_t::BASIC) { continue; } const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - const i_t num_rows = col_end - col_start; - f_t a_ij = 0.0; + const i_t col_end = lp.A.col_start[j + 1]; + f_t a_ij = 0.0; for (i_t p = col_start; p < col_end; p++) { const i_t i = lp.A.i[p]; if (i == row) { @@ -3632,40 +3775,90 @@ void branch_and_bound_t::pivot_out_integer_variables( f_t bound = a_ij > 0 ? lp.lower[j] : lp.upper[j]; if (std::abs(bound) == inf) { continue; } - sparse_vector_t delta_x; - delta_x.n = lp.num_cols; - delta_x.i.reserve(num_rows + 1); - delta_x.x.reserve(num_rows + 1); - const f_t delta_xj = bound - solution.x[j]; - delta_x.i.push_back(j); - delta_x.x.push_back(delta_xj); + // Build delta_x describing "move x[j] to its bound, let the basic slacks compensate to + // keep A*x = b". This is a feasible direction (A*delta_x = 0). The nonzero pattern lives + // on the entries of column A(:, j) plus j itself. In the nonbasic_slack slot, + // delta_x[nonbasic_slack] = -delta_xj * a_ij > 0, i.e. the entering slack moves up from + // its lower bound 0. We build the sparse version to feed the feasibility scan, then + // scatter into a dense vector and normalize so that delta_x[nonbasic_slack] == 1 (the + // convention primal_ratio_test expects for entering variables). + sparse_vector_t delta_x_sparse; + delta_x_sparse.n = lp.num_cols; + delta_x_sparse.i.reserve(col_end - col_start + 1); + delta_x_sparse.x.reserve(col_end - col_start + 1); + const f_t delta_xj = bound - soln_copy.x[j]; + delta_x_sparse.i.push_back(j); + delta_x_sparse.x.push_back(delta_xj); for (i_t p = col_start; p < col_end; p++) { - const i_t r = lp.A.i[p]; - const f_t a_rj = lp.A.x[p]; + const i_t r = lp.A.i[p]; + const f_t a_rj = lp.A.x[p]; const f_t delta_slack_r = -delta_xj * a_rj; - delta_x.i.push_back(row_to_slack[r]); - delta_x.x.push_back(delta_slack_r); + delta_x_sparse.i.push_back(row_to_slack[r]); + delta_x_sparse.x.push_back(delta_slack_r); } - bool ok = true; - const i_t ndx = delta_x.i.size(); + // Reject if the full unit step would drive any basic slack below zero. + bool ok = true; + const i_t ndx = delta_x_sparse.i.size(); for (i_t h = 0; h < ndx; h++) { - const i_t jj = delta_x.i[h]; + const i_t jj = delta_x_sparse.i[h]; if (jj == j) continue; - const f_t val = delta_x.x[h]; - const f_t slack_value = solution.x[jj]; + const f_t val = delta_x_sparse.x[h]; + const f_t slack_value = soln_copy.x[jj]; if (val < -slack_value) { ok = false; break; } } + if (!ok) { continue; } - if (ok) { - std::vector delta_x_dense(lp.num_cols, 0.0); - delta_x.to_dense(delta_x_dense); - std::vector residual(lp.num_rows); - matrix_vector_multiply(lp.A, 1.0, delta_x_dense, 0.0, residual); - settings_.log.printf("Fast candidate ok || A*delta_x ||_inf = %e\n", vector_norm_inf(residual)); + std::vector delta_x(lp.num_cols, 0.0); + delta_x_sparse.to_dense(delta_x); + + // Normalize so that delta_x[nonbasic_slack] == 1 (the standard entering-direction + // convention). Also confirms A*delta_x = 0 at debug log time. + const f_t scale = delta_x[nonbasic_slack]; + if (!(std::abs(scale) > 1e-12)) { continue; } + for (i_t h = 0; h < lp.num_cols; ++h) { delta_x[h] /= scale; } + + // Entering variable is the nonbasic slack, moving up from its lower bound 0. + const i_t entering_index = nonbasic_slack; + const i_t nonbasic_entering = nonbasic_index[nonbasic_slack]; + if (nonbasic_entering < 0) { continue; } + const i_t direction = 1; + + // Recover B^{-1} * abar from the full-vector delta_x. In our sign convention, + // delta_x[basic_list[h]] = -direction * (B^{-1} abar)[h], so + // (B^{-1} abar)[h] = -direction * delta_x[basic_list[h]]. + // Then utilde = L^{-1} P abar = U * (B^{-1} abar). In MPF, U == U0 (rank-1 updates all + // live in L), so u_multiply is a single sparse matvec against U0. + std::vector b_inv_abar(lp.num_rows); + for (i_t h = 0; h < lp.num_rows; ++h) { + b_inv_abar[h] = -direction * delta_x[basic_list_copy[h]]; + } + std::vector utilde_dense; + basis_update_copy.u_multiply(b_inv_abar, utilde_dense); + sparse_vector_t utilde_sparse; + utilde_sparse.from_dense(utilde_dense); + + apply_delta_x_for_integer_pivot(lp, + basic_list_copy, + nonbasic_list_copy, + nonbasic_index, + vstatus_copy, + entering_index, + nonbasic_entering, + direction, + delta_x, + utilde_sparse, + soln_copy, + basis_update_copy, + work_estimate); + // apply_delta_x_for_integer_pivot only mutates vstatus when the pivot actually fires, + // so entering_index transitioning to BASIC is a reliable success signal. + if (vstatus_copy[entering_index] == variable_status_t::BASIC) { + settings_.log.printf( + "Fast candidate pivot succeeded: j=%d entering slack=%d row=%d\n", j, entering_index, row); } } @@ -3681,7 +3874,11 @@ void branch_and_bound_t::pivot_out_integer_variables( : -1; const i_t entering_index = j; const i_t nonbasic_entering = zero_reduced_costs_vars_nonbasic_index[k]; - if (nonbasic_list_copy[nonbasic_entering] != j) { continue; } + if (nonbasic_entering < 0 || + nonbasic_entering >= static_cast(nonbasic_list_copy.size()) || + nonbasic_list_copy[nonbasic_entering] != j) { + continue; + } // Solve B * dxB = A(:, j) so utilde is valid for the MPF update. // Apply direction when forming delta_x (same convention as primal_phase2). @@ -3698,90 +3895,19 @@ void branch_and_bound_t::pivot_out_integer_variables( } delta_x[j] = direction; - f_t step_length; - i_t basic_leaving; - f_t work_estimate = 0.0; - const i_t leaving_index = simplex::primal_ratio_test(lp, - settings_, - vstatus_copy, - basic_list_copy, - soln_copy.x, - delta_x, - step_length, - basic_leaving, - entering_index, - direction, - work_estimate); - bool binding_integer = - leaving_index != -1 && - is_fractional(soln_copy.x[leaving_index], var_types_[leaving_index], settings_.integer_tol); - if (!binding_integer) { continue; } - - std::vector test_x = soln_copy.x; - i_t integer_destroyed = 0; - for (i_t h = 0; h < lp.num_cols; ++h) { - test_x[h] += step_length * delta_x[h]; - if (var_types_[h] != variable_type_t::INTEGER) { continue; } - const bool was_fractional = - is_fractional(soln_copy.x[h], var_types_[h], settings_.integer_tol); - const bool now_fractional = - is_fractional(test_x[h], var_types_[h], settings_.integer_tol); - if (now_fractional && !was_fractional) { - integer_destroyed++; - } else if (!now_fractional && was_fractional) { - integer_destroyed--; - } - } - // Require a strict net decrease in fractional integers. - if (integer_destroyed >= 0) { continue; } - - soln_copy.x = test_x; - basic_list_copy[basic_leaving] = entering_index; - nonbasic_list_copy[nonbasic_entering] = leaving_index; - vstatus_copy[entering_index] = variable_status_t::BASIC; - if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { - vstatus_copy[leaving_index] = variable_status_t::NONBASIC_FIXED; - } else if (delta_x[leaving_index] < 0) { - vstatus_copy[leaving_index] = variable_status_t::NONBASIC_LOWER; - } else { - vstatus_copy[leaving_index] = variable_status_t::NONBASIC_UPPER; - } - - const i_t m = lp.num_rows; - sparse_vector_t es_sparse(m, 1); - es_sparse.i[0] = basic_leaving; - es_sparse.x[0] = 1.0; - sparse_vector_t UTsol_sparse(m, 1); - sparse_vector_t solution_sparse(m, 1); - basis_update_copy.b_transpose_solve(es_sparse, solution_sparse, UTsol_sparse); - const i_t recommend_refactor = - basis_update_copy.update(utilde_sparse, UTsol_sparse, basic_leaving); - if (recommend_refactor == 1) { - csc_matrix_t L(m, m, 1); - csc_matrix_t U(m, m, 1); - std::vector pinv(m); - std::vector p(m); - std::vector q(m); - std::vector deficient; - std::vector slacks_needed; - f_t factorize_work_estimate = 0.0; - const i_t rank = factorize_basis(lp.A, - settings_, - basic_list_copy, - exploration_stats_.start_time, - L, - U, - p, - pinv, - q, - deficient, - slacks_needed, - factorize_work_estimate); - if (rank == CONCURRENT_HALT_RETURN || rank == TIME_LIMIT_RETURN) { return; } - if (rank < 0 || rank != lp.num_rows) { return; } - simplex::reorder_basic_list(q, basic_list_copy); - basis_update_copy.reset(L, U, p); - } + apply_delta_x_for_integer_pivot(lp, + basic_list_copy, + nonbasic_list_copy, + nonbasic_index, + vstatus_copy, + entering_index, + nonbasic_entering, + direction, + delta_x, + utilde_sparse, + soln_copy, + basis_update_copy, + work_estimate); } std::vector new_fractional; @@ -3790,7 +3916,7 @@ void branch_and_bound_t::pivot_out_integer_variables( if (num_new_fractional < start_num_fractional) { i_t num_integer_increased = start_num_fractional - num_new_fractional; integer_pivots_.fetch_add(num_integer_increased, std::memory_order_release); - settings_.log.printf("Pivoted out %d integer variables: %d -> %d\n", num_integer_increased, start_num_fractional, num_new_fractional); + settings_.log.printf("Pivoted out %d integer variables: %d -> %d in %.2f\n", num_integer_increased, start_num_fractional, num_new_fractional, toc(pivot_out_integer_variables_start_time)); num_fractional = num_new_fractional; fractional = new_fractional; basic_list = basic_list_copy; @@ -3883,7 +4009,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut solving_root_relaxation_ = true; f_t root_relax_start_time = tic(); - + f_t root_relax_work_estimate = 0.0; if (!enable_concurrent_lp_root_solve()) { // RINS/SUBMIP path settings_.log.printf("\n"); @@ -3896,7 +4022,8 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut basic_list, nonbasic_list, root_vstatus_, - edge_norms_); + edge_norms_, + root_relax_work_estimate); root_relax_solved_by = DualSimplex; exploration_stats_.total_simplex_iters = root_relax_soln_.iterations; @@ -3909,12 +4036,15 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut basis_update, basic_list, nonbasic_list, - edge_norms_); + edge_norms_, + root_relax_work_estimate); } solving_root_relaxation_ = false; f_t root_relax_elapsed_time = toc(root_relax_start_time); exploration_stats_.total_lp_solve_time = root_relax_elapsed_time; + i_t root_iterations = exploration_stats_.total_simplex_iters; + if (root_status == lp_status_t::INFEASIBLE) { settings_.log.printf("\nThe root LP relaxation is infeasible\n", @@ -3968,6 +4098,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut root_relax_soln_.iterations, root_relax_elapsed_time, method_to_string(root_relax_solved_by)); + settings_.log.printf("Dual simplex iteration %d work estimate %.2e work per second %.2e\n", root_iterations, root_relax_work_estimate, root_relax_work_estimate / root_relax_elapsed_time); settings_.log.printf("Root relaxation objective %+.8e\n\n", root_relax_soln_.user_objective); assert(root_vstatus_.size() == original_lp_.num_cols); @@ -4917,7 +5048,7 @@ node_status_t branch_and_bound_t::solve_node_deterministic( i_t node_iter = 0; f_t lp_start_time = tic(); std::vector leaf_edge_norms = edge_norms_; - + f_t dual_work_estimate = 0.0; dual_status_t lp_status = dual_phase2_with_advanced_basis(2, 0, worker.recompute_bounds_and_basis, @@ -4930,6 +5061,7 @@ node_status_t branch_and_bound_t::solve_node_deterministic( worker.nonbasic_list, worker.leaf_solution, node_iter, + dual_work_estimate, leaf_edge_norms, &worker.work_context); @@ -4945,6 +5077,7 @@ node_status_t branch_and_bound_t::solve_node_deterministic( worker.nonbasic_list, worker.leaf_vstatus, leaf_edge_norms, + dual_work_estimate, &worker.work_context); lp_status = convert_lp_status_to_dual_status(second_status); } @@ -5530,6 +5663,7 @@ void branch_and_bound_t::deterministic_dive( worker.leaf_solution.resize(worker.leaf_problem.num_rows, worker.leaf_problem.num_cols); i_t node_iter = 0; f_t lp_start_time = tic(); + f_t dual_work_estimate = 0.0; std::vector leaf_edge_norms = edge_norms_; decompress_vstatus(node_ptr->packed_vstatus, worker.leaf_problem.num_cols, worker.leaf_vstatus); @@ -5545,6 +5679,7 @@ void branch_and_bound_t::deterministic_dive( worker.nonbasic_list, worker.leaf_solution, node_iter, + dual_work_estimate, leaf_edge_norms, &worker.work_context); @@ -5558,6 +5693,7 @@ void branch_and_bound_t::deterministic_dive( worker.nonbasic_list, worker.leaf_vstatus, leaf_edge_norms, + dual_work_estimate, &worker.work_context); lp_status = convert_lp_status_to_dual_status(second_status); } diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index ed4af6d6bc..c0ff1761a2 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -173,7 +173,8 @@ class branch_and_bound_t { simplex::basis_update_mpf_t& basis_update, std::vector& basic_list, std::vector& nonbasic_list, - std::vector& edge_norms); + std::vector& edge_norms, + f_t& work_estimate); i_t find_reduced_cost_fixings(f_t upper_bound, std::vector& lower_bounds, @@ -347,6 +348,7 @@ class branch_and_bound_t { std::vector& zero_reduced_costs_vars, std::vector& zero_reduced_costs_vars_nonbasic_index); + void pivot_out_integer_variables(const simplex::lp_problem_t& lp, std::vector& basic_list, std::vector& nonbasic_list, @@ -356,6 +358,29 @@ class branch_and_bound_t { i_t& num_fractional, std::vector& fractional); + // Try to pivot the nonbasic variable `entering_index` (currently at position + // `nonbasic_entering` in `nonbasic_list`) into the basis along the direction + // `delta_x`, with `utilde` = U * B^{-1} A(:, entering_index) satisfying + // L * utilde = P * A(:, entering_index). Applies the pivot only if it yields a + // strict net decrease in the number of fractional integer variables. On success, + // mutates basic_list/nonbasic_list/vstatus/solution/basis_update in place and applies + // an O(1) fix-up to `nonbasic_index` for the two variables whose (non)basic status + // changed. On skip, leaves all outputs untouched. + void apply_delta_x_for_integer_pivot( + const simplex::lp_problem_t& lp, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& nonbasic_index, + std::vector& vstatus, + i_t entering_index, + i_t nonbasic_entering, + i_t direction, + std::vector& delta_x, + const sparse_vector_t& utilde_sparse, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + f_t& work_estimate); + void dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, std::vector& basic_list, std::vector& nonbasic_list, diff --git a/cpp/src/branch_and_bound/pseudo_costs.cpp b/cpp/src/branch_and_bound/pseudo_costs.cpp index cdba90f219..eaa60cf475 100644 --- a/cpp/src/branch_and_bound/pseudo_costs.cpp +++ b/cpp/src/branch_and_bound/pseudo_costs.cpp @@ -370,6 +370,7 @@ void strong_branch_helper(i_t start, i_t iter = 0; std::vector vstatus = root_vstatus; std::vector child_edge_norms = edge_norms; + f_t child_work_estimate = 0.0; dual_status_t status = simplex::dual_phase2(2, 0, lp_start_time, @@ -378,6 +379,7 @@ void strong_branch_helper(i_t start, vstatus, solution, iter, + child_work_estimate, child_edge_norms); f_t obj = std::numeric_limits::quiet_NaN(); @@ -506,7 +508,8 @@ std::pair trial_branching(const lp_problem_t& orig // Only refactor the basis if we encounter numerical issues. child_basis_factors.set_refactor_frequency(iter_limit); - dual_status_t status = simplex::dual_phase2_with_advanced_basis(2, + f_t child_work_estimate = 0.0; + dual_status_t status = simplex::dual_phase2_with_advanced_basis(2, 0, initialize_basis, start_time, @@ -518,6 +521,7 @@ std::pair trial_branching(const lp_problem_t& orig child_nonbasic_list, solution, iter, + child_work_estimate, child_edge_norms); settings.log.debug("Trial branching on variable %d. Lo: %e Up: %e. Iter %d. Status %s. Obj %e\n", diff --git a/cpp/src/dual_simplex/basis_updates.cpp b/cpp/src/dual_simplex/basis_updates.cpp index f81962d054..1081cc4773 100644 --- a/cpp/src/dual_simplex/basis_updates.cpp +++ b/cpp/src/dual_simplex/basis_updates.cpp @@ -2009,6 +2009,35 @@ i_t basis_update_mpf_t::u_solve(sparse_vector_t& rhs) const return 0; } + +// Compute y = U*x. In the MPF factorization, the rank-1 update factors are absorbed into L, so +// U == U0 and U*x reduces to a sparse matvec against U0. +template +void basis_update_mpf_t::u_multiply(const std::vector& x, + std::vector& y) const +{ + const i_t m = L0_.m; + y.assign(m, 0.0); + matrix_vector_multiply(U0_, f_t(1.0), x, f_t(0.0), y); + work_estimate_ += 2 * U0_.col_start[U0_.n]; +} + +// Sparse-in/sparse-out overload of u_multiply. Same semantics as the dense version. +template +void basis_update_mpf_t::u_multiply(const sparse_vector_t& x, + sparse_vector_t& y) const +{ + const i_t m = L0_.m; + // Scatter x into a dense workspace, compute U0 * x, gather back to sparse. + std::vector x_dense; + x.to_dense(x_dense); + std::vector y_dense(m, 0.0); + matrix_vector_multiply(U0_, f_t(1.0), x_dense, f_t(0.0), y_dense); + work_estimate_ += 2 * U0_.col_start[U0_.n]; + y.from_dense(y_dense); + work_estimate_ += m; +} + // Solve for x such that L*x = y template i_t basis_update_mpf_t::l_solve(std::vector& rhs) const diff --git a/cpp/src/dual_simplex/basis_updates.hpp b/cpp/src/dual_simplex/basis_updates.hpp index d1c623db55..bdedcc4a18 100644 --- a/cpp/src/dual_simplex/basis_updates.hpp +++ b/cpp/src/dual_simplex/basis_updates.hpp @@ -353,6 +353,14 @@ class basis_update_mpf_t { // Solve for x such that U'*x = y i_t u_transpose_solve(sparse_vector_t& rhs) const; + // Compute y = U*x. In the MPF factorization the rank-1 update factors are absorbed into L, so + // U is unchanged from the initial factorization (U == U0), and U*x is just a sparse matvec + // against U0. + void u_multiply(const std::vector& x, std::vector& y) const; + + // Sparse-in/sparse-out overload of u_multiply. + void u_multiply(const sparse_vector_t& x, sparse_vector_t& y) const; + // Replace the column B(:, leaving_index) with the vector abar. Pass in utilde such that L*utilde // = abar i_t update(const std::vector& utilde, const std::vector& etilde, i_t leaving_index); diff --git a/cpp/src/dual_simplex/crossover.cpp b/cpp/src/dual_simplex/crossover.cpp index e1ba272adf..5b0dd451e3 100644 --- a/cpp/src/dual_simplex/crossover.cpp +++ b/cpp/src/dual_simplex/crossover.cpp @@ -168,9 +168,10 @@ f_t primal_infeasibility(const lp_problem_t& lp, f_t primal_inf = 0; constexpr bool verbose = false; constexpr f_t infeas_tol = 1e-3; + const f_t primal_tol = settings.primal_tol; for (i_t j = 0; j < n; ++j) { - if (x[j] < lp.lower[j]) { - // x_j < l_j => -x_j > -l_j => -x_j + l_j > 0 + if (x[j] < lp.lower[j] - primal_tol) { + // x_j < l_j - tol => violation exceeds per-variable threshold const f_t infeas = -x[j] + lp.lower[j]; primal_inf += infeas; if (verbose && infeas > infeas_tol) { @@ -183,8 +184,8 @@ f_t primal_infeasibility(const lp_problem_t& lp, vstatus[j]); } } - if (x[j] > lp.upper[j]) { - // x_j > u_j => x_j - u_j > 0 + if (x[j] > lp.upper[j] + primal_tol) { + // x_j > u_j + tol => violation exceeds per-variable threshold const f_t infeas = x[j] - lp.upper[j]; primal_inf += infeas; if (verbose && infeas > infeas_tol) { @@ -1423,8 +1424,11 @@ crossover_status_t crossover(const lp_problem_t& lp, } else if (dual_feasible && !primal_feasible) { i_t dual_iter = 0; std::vector edge_norms; + f_t work_estimate = 0.0; + simplex_solver_settings_t dual_settings = settings; + dual_settings.iteration_limit = std::numeric_limits::max(); dual_status_t status = - dual_phase2(2, 0, start_time, lp, settings, vstatus, solution, dual_iter, edge_norms); + dual_phase2(2, 0, start_time, lp, dual_settings, vstatus, solution, dual_iter, work_estimate, edge_norms); if (toc(start_time) > settings.time_limit) { settings.log.printf("Time limit exceeded\n"); return crossover_status_t::TIME_LIMIT; @@ -1443,7 +1447,32 @@ crossover_status_t crossover(const lp_problem_t& lp, solution.iterations += dual_iter; primal_feasible = primal_infeas <= primal_tol && primal_res <= primal_tol; dual_feasible = dual_infeas <= dual_tol && dual_res <= dual_tol; + } else if (primal_feasible && !dual_feasible) { + i_t primal_iter = 0; + simplex_solver_settings_t primal_settings = settings; + primal_settings.iteration_limit = std::numeric_limits::max(); + primal_status_t primal_status = primal_phase2(2, start_time, lp, primal_settings, vstatus, solution, primal_iter); + if (toc(start_time) > settings.time_limit) { + settings.log.printf("Time limit exceeded\n"); + return crossover_status_t::TIME_LIMIT; + } + if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { + if (!settings.inside_mip) { settings.log.printf("Concurrent halt\n"); } + return crossover_status_t::CONCURRENT_LIMIT; + } + primal_infeas = primal_infeasibility(lp, settings, vstatus, solution.x); + dual_infeas = dual_infeasibility(lp, settings, vstatus, solution.z); + primal_res = primal_residual(lp, solution); + dual_res = dual_residual(lp, solution); + if (primal_status != primal_status_t::OPTIMAL) { + print_crossover_info(lp, settings, vstatus, solution, "Primal phase 2 complete"); + } + solution.iterations += primal_iter; + primal_feasible = primal_infeas <= primal_tol && primal_res <= primal_tol; + dual_feasible = dual_infeas <= dual_tol && dual_res <= dual_tol; } else { + simplex_solver_settings_t dual_settings = settings; + dual_settings.iteration_limit = std::numeric_limits::max(); lp_problem_t phase1_problem(lp.handle_ptr, 1, 1, 1); create_phase1_problem(lp, phase1_problem); std::vector phase1_vstatus(n); @@ -1469,8 +1498,17 @@ crossover_status_t crossover(const lp_problem_t& lp, i_t iter = 0; lp_solution_t phase1_solution(phase1_problem.num_rows, phase1_problem.num_cols); std::vector junk; - dual_status_t phase1_status = dual_phase2( - 1, 1, start_time, phase1_problem, settings, phase1_vstatus, phase1_solution, iter, junk); + f_t phase1_work_estimate = 0.0; + dual_status_t phase1_status = dual_phase2(1, + 1, + start_time, + phase1_problem, + dual_settings, + phase1_vstatus, + phase1_solution, + iter, + phase1_work_estimate, + junk); if (phase1_status == dual_status_t::NUMERICAL || phase1_status == dual_status_t::DUAL_UNBOUNDED) { settings.log.printf("Failed in Phase 1\n"); @@ -1585,8 +1623,17 @@ crossover_status_t crossover(const lp_problem_t& lp, dual_status_t status = dual_status_t::NUMERICAL; if (dual_infeas <= settings.dual_tol) { std::vector edge_norms; - status = dual_phase2( - 2, iter == 0 ? 1 : 0, start_time, lp, settings, vstatus, solution, iter, edge_norms); + f_t phase2_work_estimate = 0.0; + status = dual_phase2(2, + iter == 0 ? 1 : 0, + start_time, + lp, + dual_settings, + vstatus, + solution, + iter, + phase2_work_estimate, + edge_norms); if (toc(start_time) > settings.time_limit) { settings.log.printf("Time limit exceeded\n"); return crossover_status_t::TIME_LIMIT; diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index d3867086f3..d803dab930 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -470,7 +470,7 @@ void initial_perturbation(const lp_problem_t& lp, f_t sum_perturb = 0.0; i_t num_perturb = 0; - random_t random(settings.seed); + random_t random(settings.random_seed); for (i_t j = 0; j < n; ++j) { f_t obj = objective[j] = lp.objective[j]; @@ -2340,6 +2340,7 @@ void prepare_optimality(i_t info, int phase, f_t start_time, f_t max_val, + f_t& work_estimate, i_t& iter, const std::vector& x, std::vector& y, @@ -2348,7 +2349,6 @@ void prepare_optimality(i_t info, { const i_t m = lp.num_rows; const i_t n = lp.num_cols; - f_t work_estimate = 0; // Work in this function is not captured sol.objective = compute_objective(lp, sol.x); sol.user_objective = compute_user_objective(lp, sol.objective); @@ -2373,6 +2373,9 @@ void prepare_optimality(i_t info, settings.log.printf("Unperturbed dual infeasibility: %.2e\n", dual_infeas); settings.log.printf("Objective: %+.16e\n", sol.user_objective); settings.log.printf("Num updates: %d\n", ft.num_updates()); + settings.log.printf("Iterations: %d\n", iter); + + i_t dual_iter = iter; // Primal pivots in place, so keep the perturbed solution to fall back on. // The factor is snapshot rather than refactorized on failure: the copy is @@ -2405,7 +2408,7 @@ void prepare_optimality(i_t info, false); if (primal_status == primal_status_t::OPTIMAL) { // z now prices the original objective, so no perturbation remains. - settings.log.printf("Primal cleanup successful.\n"); + settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); perturbation = 0.0; sol.objective = compute_objective(lp, sol.x); sol.user_objective = compute_user_objective(lp, sol.objective); @@ -2433,6 +2436,9 @@ void prepare_optimality(i_t info, settings.log.printf("Dual phase I complete. Iterations %d. Time %.2f\n", iter, toc(start_time)); } if (phase == 2) { + if (settings.inside_mip == 0 || settings.inside_mip == 1) { + settings.log.printf("Work estimate: %.2e\n", work_estimate); + } if (!settings.inside_mip) { settings.log.printf("\n"); settings.log.printf( @@ -2557,6 +2563,7 @@ dual_status_t dual_phase2(i_t phase, std::vector& vstatus, lp_solution_t& sol, i_t& iter, + f_t& work_estimate, std::vector& delta_y_steepest_edge, work_limit_context_t* work_unit_context) { @@ -2579,6 +2586,7 @@ dual_status_t dual_phase2(i_t phase, nonbasic_list, sol, iter, + work_estimate, delta_y_steepest_edge, work_unit_context); } @@ -2596,6 +2604,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, std::vector& nonbasic_list, lp_solution_t& sol, i_t& iter, + f_t& phase2_work_estimate, std::vector& delta_y_steepest_edge, work_limit_context_t* work_unit_context) { @@ -2610,7 +2619,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, assert(lp.lower.size() == n); assert(lp.upper.size() == n); assert(lp.rhs.size() == m); - f_t phase2_work_estimate = 0.0; ft.clear_work_estimate(); std::vector& x = sol.x; @@ -2663,6 +2671,10 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (toc(start_time) > settings.time_limit) { return dual_status_t::TIME_LIMIT; } } + if (settings.initial_perturbation == 1 && phase == 2) { + phase2::initial_perturbation(lp, settings, vstatus, objective); + } + // Populate c_basic after basis is initialized for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; @@ -3035,6 +3047,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase, start_time, max_val, + phase2_work_estimate, iter, x, y, @@ -3255,6 +3268,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase, start_time, max_val, + phase2_work_estimate, iter, x, y, @@ -3311,6 +3325,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase, start_time, max_val, + phase2_work_estimate, iter, x, y, @@ -3806,6 +3821,7 @@ template dual_status_t dual_phase2( std::vector& vstatus, lp_solution_t& sol, int& iter, + double& work_estimate, std::vector& steepest_edge_norms, work_limit_context_t* work_unit_context); @@ -3822,6 +3838,7 @@ template dual_status_t dual_phase2_with_advanced_basis( std::vector& nonbasic_list, lp_solution_t& sol, int& iter, + double& work_estimate, std::vector& steepest_edge_norms, work_limit_context_t* work_unit_context); diff --git a/cpp/src/dual_simplex/phase2.hpp b/cpp/src/dual_simplex/phase2.hpp index daa946e019..e5a4bacf62 100644 --- a/cpp/src/dual_simplex/phase2.hpp +++ b/cpp/src/dual_simplex/phase2.hpp @@ -60,6 +60,7 @@ dual_status_t dual_phase2(i_t phase, std::vector& vstatus, lp_solution_t& sol, i_t& iter, + f_t& work_estimate, std::vector& steepest_edge_norms, work_limit_context_t* work_unit_context = nullptr); @@ -76,6 +77,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, std::vector& nonbasic_list, lp_solution_t& sol, i_t& iter, + f_t& work_estimate, std::vector& delta_y_steepest_edge, work_limit_context_t* work_unit_context = nullptr); diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 6a69cdfcd2..a5f137e2e5 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -167,6 +167,7 @@ struct simplex_solver_settings_t { i_t augmented; // -1 automatic, 0 to solve with ADAT, 1 to solve with augmented system i_t dualize; // -1 automatic, 0 to not dualize, 1 to dualize i_t ordering; // -1 automatic, 0 to use nested dissection, 1 to use AMD + i_t initial_perturbation; // -1 automatic, 0 to not perturb, 1 to perturb i_t barrier_dual_initial_point; // -1 automatic, 0 to use Lustig, Marsten, and Shanno initial // point, 1 to use initial point form dual least squares problem bool check_Q; // true to check if Q is positive semidefinite diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index a24d12fc55..db064dabf8 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -157,6 +157,7 @@ lp_status_t solve_linear_program_advanced(const lp_problem_t& original lp_solution_t& original_solution, std::vector& vstatus, std::vector& edge_norms, + f_t& work_estimate, work_limit_context_t* work_unit_context) { raft::common::nvtx::range scope("DualSimplex::solve_lp"); @@ -175,6 +176,7 @@ lp_status_t solve_linear_program_advanced(const lp_problem_t& original nonbasic_list, vstatus, edge_norms, + work_estimate, work_unit_context); return result; } @@ -190,6 +192,7 @@ lp_status_t solve_linear_program_with_advanced_basis( std::vector& nonbasic_list, std::vector& vstatus, std::vector& edge_norms, + f_t& work_estimate, work_limit_context_t* work_unit_context) { lp_status_t lp_status = lp_status_t::UNSET; @@ -257,6 +260,7 @@ lp_status_t solve_linear_program_with_advanced_basis( phase1_vstatus, phase1_solution, iter, + work_estimate, edge_norms, work_unit_context); } @@ -295,6 +299,7 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, solution, iter, + work_estimate, edge_norms, work_unit_context); if (status == dual_status_t::NUMERICAL) { @@ -315,6 +320,7 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, phase1_solution, iter, + work_estimate, edge_norms, work_unit_context); vstatus = phase1_vstatus; @@ -331,6 +337,7 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, solution, iter, + work_estimate, edge_norms, work_unit_context); } @@ -341,7 +348,7 @@ lp_status_t solve_linear_program_with_advanced_basis( // TODO: We need to update ft if the basis changed } if (settings.inside_mip && settings.concurrent_halt != nullptr) { - settings.log.printf("Setting concurrent halt to 1 inside_mip\n"); + settings.log.debug("Setting concurrent halt to 1 inside_mip\n"); *settings.concurrent_halt = 1; } if (status == dual_status_t::OPTIMAL) { @@ -847,8 +854,9 @@ lp_status_t solve_linear_program(const user_problem_t& user_problem, lp_solution_t lp_solution(original_lp.num_rows, original_lp.num_cols); std::vector vstatus; std::vector edge_norms; + f_t work_estimate = 0.0; lp_status_t status = solve_linear_program_advanced( - original_lp, start_time, settings, lp_solution, vstatus, edge_norms); + original_lp, start_time, settings, lp_solution, vstatus, edge_norms, work_estimate); if (status == lp_status_t::CONCURRENT_LIMIT) { solution.iterations = lp_solution.iterations; return lp_status_t::CONCURRENT_LIMIT; @@ -900,8 +908,9 @@ i_t solve(const user_problem_t& problem, lp_solution_t solution(original_lp.num_rows, original_lp.num_cols); std::vector vstatus; std::vector edge_norms; + f_t work_estimate = 0.0; lp_status_t lp_status = solve_linear_program_advanced( - original_lp, start_time, settings, solution, vstatus, edge_norms); + original_lp, start_time, settings, solution, vstatus, edge_norms, work_estimate); primal_solution = solution.x; if (lp_status == lp_status_t::OPTIMAL) { status = 0; @@ -955,6 +964,7 @@ template lp_status_t solve_linear_program_advanced( lp_solution_t& original_solution, std::vector& vstatus, std::vector& edge_norms, + double& work_estimate, work_limit_context_t* work_unit_context); template lp_status_t solve_linear_program_with_advanced_basis( @@ -967,6 +977,7 @@ template lp_status_t solve_linear_program_with_advanced_basis( std::vector& nonbasic_list, std::vector& vstatus, std::vector& edge_norms, + double& work_estimate, work_limit_context_t* work_unit_context); template lp_status_t solve_linear_program_with_barrier( diff --git a/cpp/src/dual_simplex/solve.hpp b/cpp/src/dual_simplex/solve.hpp index f4807306e0..f295792369 100644 --- a/cpp/src/dual_simplex/solve.hpp +++ b/cpp/src/dual_simplex/solve.hpp @@ -70,6 +70,7 @@ lp_status_t solve_linear_program_advanced(const lp_problem_t& original lp_solution_t& original_solution, std::vector& vstatus, std::vector& edge_norms, + f_t& work_estimate, work_limit_context_t* work_unit_context = nullptr); // Solve the LP using dual simplex and keep the `basis_update_mpf_t` @@ -85,6 +86,7 @@ lp_status_t solve_linear_program_with_advanced_basis( std::vector& nonbasic_list, std::vector& vstatus, std::vector& edge_norms, + f_t& work_estimate, work_limit_context_t* work_unit_context = nullptr); template diff --git a/cpp/src/math_optimization/solver_settings.cu b/cpp/src/math_optimization/solver_settings.cu index ab129b9c89..a4b3d550a7 100644 --- a/cpp/src/math_optimization/solver_settings.cu +++ b/cpp/src/math_optimization/solver_settings.cu @@ -137,6 +137,7 @@ solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings {CUOPT_FOLDING, &pdlp_settings.folding, -1, 1, -1}, {CUOPT_DUALIZE, &pdlp_settings.dualize, -1, 1, -1}, {CUOPT_ORDERING, &pdlp_settings.ordering, -1, 1, -1}, + {CUOPT_INITIAL_PERTURBATION, &pdlp_settings.initial_perturbation, -1, 1, -1}, {CUOPT_BARRIER_DUAL_INITIAL_POINT, &pdlp_settings.barrier_dual_initial_point, -1, 1, -1}, {CUOPT_MIP_CUT_PASSES, &mip_settings.max_cut_passes, -1, std::numeric_limits::max(), 10}, {CUOPT_MIP_MIXED_INTEGER_ROUNDING_CUTS, &mip_settings.mir_cuts, -1, 1, -1}, diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index 173619c5d1..d09ea65052 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -588,6 +588,7 @@ std::tuple, simplex::lp_status_t, f_t, f_t, f_t dual_simplex_settings.time_limit = settings.time_limit; dual_simplex_settings.iteration_limit = settings.iteration_limit; dual_simplex_settings.concurrent_halt = settings.concurrent_halt; + dual_simplex_settings.initial_perturbation = settings.initial_perturbation; if (dual_simplex_settings.concurrent_halt != nullptr) { // Don't show the dual simplex log in concurrent mode. Show the PDLP log instead dual_simplex_settings.log.log = false; From 0cfc746f9593db4d0d8de5e50a53f893bcf50b2c Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 6 Aug 2026 14:27:32 -0700 Subject: [PATCH 011/113] Address coderabbit review comments --- cpp/src/branch_and_bound/branch_and_bound.cpp | 4 ++++ cpp/src/dual_simplex/primal.cpp | 24 ++++++++++--------- cpp/src/dual_simplex/primal.hpp | 15 ++++++------ cpp/src/dual_simplex/solve.cpp | 1 + cpp/src/dual_simplex/solve.hpp | 2 +- .../solver_settings/solver_settings.pyx | 1 + 6 files changed, 28 insertions(+), 19 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 706ce51c07..980f9bffb0 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3439,6 +3439,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple const i_t iter_before = iter; simplex_solver_settings_t primal_settings = settings_; primal_settings.log.log = false; + primal_settings.time_limit = settings_.time_limit - toc(exploration_stats_.start_time); simplex::primal_status_t lp_status = simplex::primal_phase2_with_advanced_basis(2, exploration_stats_.start_time, lp_reduced, @@ -3486,6 +3487,8 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple best_num_fractional = num_fractional_reduced; best_reduced_vstatus = reduced_vstatus; } + } else { + break; } } @@ -3683,6 +3686,7 @@ void branch_and_bound_t::pivot_out_integer_variables( i_t& num_fractional, std::vector& fractional) { + if (num_fractional == 0) { return; } f_t pivot_out_integer_variables_start_time = tic(); std::vector zero_reduced_costs_vars; std::vector zero_reduced_costs_vars_nonbasic_index; diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index f5a756a78a..1778299c79 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -29,7 +29,7 @@ void set_primal_variables_on_bounds(const lp_problem_t& lp, { const i_t m = lp.num_rows; const i_t n = lp.num_cols; - + constexpr f_t diff_tol = 1e-6; for (i_t j = 0; j < n; ++j) { if (vstatus[j] == variable_status_t::BASIC) { continue; } @@ -221,7 +221,7 @@ f_t primal_infeasibility(const lp_problem_t& lp, return primal_infeasibility(lp, settings, vstatus, x, num_infeasible, work_estimate); } -// work estimate: n-m + 4 * m +// work estimate: n-m + 4 * m template void compute_phase1_objective(const lp_problem_t& lp, const simplex_solver_settings_t& settings, @@ -294,7 +294,7 @@ f_t compute_dual_step_length(f_t entering_reduced_cost, f_t pivot) } template -void update_y(f_t dual_step_length, +void update_y(f_t dual_step_length, const sparse_vector_t& delta_y, std::vector& y, f_t& work_estimate) @@ -710,8 +710,8 @@ primal_status_t primal_phase2_with_advanced_basis( if (primal_residual > settings.primal_tol) { settings.log.printf("|| A*x - b || %e\n", primal_residual); } - - + + std::vector objective = lp.objective; work_estimate += 2*n; const f_t primal_tol = settings.primal_tol; @@ -877,7 +877,7 @@ primal_status_t primal_phase2_with_advanced_basis( // Incremental duals may be stale relative to the current phase-I // objective. Refresh objective and duals, then retry pricing with // successively tighter dual tolerances. - settings.log.printf("Refreshing phase-I objective and duals. Num updates %d. Iter %d\n", + settings.log.printf("Refreshing phase-I objective and duals. Num updates %d. Iter %d\n", basis_update.num_updates(), iter); compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); compute_dual_variables( @@ -898,11 +898,11 @@ primal_status_t primal_phase2_with_advanced_basis( } if (entering_index == -1) { settings.log.printf( - "Numerical issues encountered. No entering variable found with large " + "No entering variable found with large " "infeasibility %e (%d).\n", primal_inf, num_primal_inf); - return primal_status_t::NUMERICAL; + return primal_status_t::PRIMAL_INFEASIBLE; } pricing_dual_tol = retry_dual_tol; } else { @@ -941,7 +941,7 @@ primal_status_t primal_phase2_with_advanced_basis( std::vector scaled_delta_xB(m); scaled_delta_xB_sparse.to_dense(scaled_delta_xB); work_estimate += m + scaled_delta_xB_sparse.i.size(); - + for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; delta_x[j] = -direction * scaled_delta_xB[k]; @@ -996,7 +996,7 @@ primal_status_t primal_phase2_with_advanced_basis( if (debug_primal_residual > 1e-6) { settings.log.printf("|| A * x - b || %e at iteration %d (updates %d)\n", debug_primal_residual, iter, basis_update.num_updates()); } -#endif +#endif if (basis_updated) { @@ -1129,9 +1129,11 @@ primal_status_t primal_phase2_with_advanced_basis( work_estimate += basis_update.work_estimate(); basis_update.clear_work_estimate(); + + if (now > settings.time_limit) { return primal_status_t::TIME_LIMIT; } } - if (iter == iter_limit) { return primal_status_t::ITERATION_LIMIT; } + if (iter >= iter_limit) { return primal_status_t::ITERATION_LIMIT; } return primal_status_t::NUMERICAL; } diff --git a/cpp/src/dual_simplex/primal.hpp b/cpp/src/dual_simplex/primal.hpp index df2c998db5..7e4d280655 100644 --- a/cpp/src/dual_simplex/primal.hpp +++ b/cpp/src/dual_simplex/primal.hpp @@ -19,13 +19,14 @@ namespace cuopt::mathematical_optimization::simplex { enum class primal_status_t { - OPTIMAL = 0, - PRIMAL_UNBOUNDED = 1, - NUMERICAL = 2, - NOT_LOADED = 3, - TIME_LIMIT = 4, - ITERATION_LIMIT = 5, - CONCURRENT_LIMIT = 6 + OPTIMAL = 0, + PRIMAL_UNBOUNDED = 1, + PRIMAL_INFEASIBLE = 2, + NUMERICAL = 3, + TIME_LIMIT = 5, + ITERATION_LIMIT = 6, + CONCURRENT_LIMIT = 7, + NOT_LOADED = 8 }; template diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index 8266b545df..dce731d4d1 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -66,6 +66,7 @@ lp_status_t map_primal_status_to_lp_status(primal_status_t status) switch (status) { case primal_status_t::OPTIMAL: return lp_status_t::OPTIMAL; case primal_status_t::PRIMAL_UNBOUNDED: return lp_status_t::UNBOUNDED; + case primal_status_t::PRIMAL_INFEASIBLE: return lp_status_t::INFEASIBLE; case primal_status_t::TIME_LIMIT: return lp_status_t::TIME_LIMIT; case primal_status_t::ITERATION_LIMIT: return lp_status_t::ITERATION_LIMIT; case primal_status_t::CONCURRENT_LIMIT: return lp_status_t::CONCURRENT_LIMIT; diff --git a/cpp/src/dual_simplex/solve.hpp b/cpp/src/dual_simplex/solve.hpp index 8ac0c0194d..291675a67b 100644 --- a/cpp/src/dual_simplex/solve.hpp +++ b/cpp/src/dual_simplex/solve.hpp @@ -105,7 +105,7 @@ lp_status_t solve_linear_program_with_primal(const user_problem_t& use const simplex_solver_settings_t& settings, f_t start_time, lp_solution_t& solution); - +template lp_status_t solve_linear_program_with_barrier(const user_problem_t& user_problem, const simplex_solver_settings_t& settings, f_t start_time, diff --git a/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx b/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx index a5dcc78d18..ce3ef6fef3 100644 --- a/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx +++ b/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx @@ -62,6 +62,7 @@ class SolverMethod(IntEnum): PDLP = auto() DualSimplex = auto() Barrier = auto() + Primal = auto() Unset = auto() def __str__(self): From 1c2c01a46036b71136d22d28a9bb58aca15e82e5 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 6 Aug 2026 14:36:18 -0700 Subject: [PATCH 012/113] Address coderabbit review comments --- cpp/src/dual_simplex/simplex_solver_settings.hpp | 1 + 1 file changed, 1 insertion(+) diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 120a4bcb99..c4338810bc 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -77,6 +77,7 @@ struct simplex_solver_settings_t { augmented(0), dualize(-1), ordering(-1), + initial_perturbation(-1), barrier_dual_initial_point(-1), postsolve_info(-1), qcqp_ruiz_equilibration(-1), From b2de00d19f684a6ff4a233b69a9359249feedaba Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 6 Aug 2026 14:48:23 -0700 Subject: [PATCH 013/113] Use tight tol for reduced costs zero check --- cpp/src/branch_and_bound/branch_and_bound.cpp | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 980f9bffb0..e346f33c5e 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3276,7 +3276,7 @@ bool branch_and_bound_t::check_for_dual_degeneracy( const i_t num_nonbasics = nonbasic_list.size(); for (i_t k = 0; k < num_nonbasics; k++) { const i_t j = nonbasic_list[k]; - if (std::abs(solution.z[j]) <= 1e-10) { + if (std::abs(solution.z[j]) <= settings_.tight_tol) { zero_reduced_costs_vars.push_back(j); zero_reduced_costs_vars_nonbasic_index.push_back(k); } @@ -3313,7 +3313,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple i_t nnz = 0; for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { nnz += lp.A.col_start[j + 1] - lp.A.col_start[j]; } } @@ -3323,7 +3323,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple i_t nz = 0; i_t reduced_col = 0; for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { original_col_to_reduced_col[j] = reduced_col; A_reduced.col_start[reduced_col] = nz; const i_t col_start = lp.A.col_start[j]; @@ -3344,7 +3344,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple std::vector b_reduced = lp.rhs; for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { // PASS } else { const i_t col_start = lp.A.col_start[j]; @@ -3381,7 +3381,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple simplex::lp_solution_t reduced_solution(m, n); reduced_col = 0; for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { reduced_solution.x[reduced_col++] = soln.x[j]; } } @@ -3389,7 +3389,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple std::vector reduced_edge_norms(n); reduced_col = 0; for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { reduced_edge_norms[reduced_col++] = edge_norms_[j]; } } @@ -3409,7 +3409,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple for (i_t pump_iter = 0; pump_iter < max_pump_iter; pump_iter++) { reduced_col = 0; for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { lp_reduced.objective[reduced_col] = 0; if (var_types_[j] == variable_type_t::INTEGER) { if (is_fractional( @@ -3459,7 +3459,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple std::vector adjusted_solution(lp.num_cols, 0.0); reduced_col = 0; for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { adjusted_solution[j] = reduced_solution.x[reduced_col++]; } else { adjusted_solution[j] = soln.x[j]; @@ -3497,7 +3497,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple // Translate the vstatus from the reduced problem to the vstatus for the original problem i_t reduced_cols = 0; for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= 1e-10) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { vstatus[j] = best_reduced_vstatus[reduced_cols++]; } } From a92825c96d341f90b17e67dd2b883197bdccd7e3 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 6 Aug 2026 15:04:17 -0700 Subject: [PATCH 014/113] Try to clean up normalization --- cpp/src/branch_and_bound/branch_and_bound.cpp | 22 +++++++++---------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index e346f33c5e..dadf7c2e72 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3776,23 +3776,24 @@ void branch_and_bound_t::pivot_out_integer_variables( break; } } - if (a_ij == 0.0) { continue; } - f_t bound = a_ij > 0 ? lp.lower[j] : lp.upper[j]; if (std::abs(bound) == inf) { continue; } + const f_t delta_xj = bound - soln_copy.x[j]; + const f_t scale = -delta_xj * a_ij; + if (std::abs(scale) <= 1e-12) { continue; } + // Build delta_x describing "move x[j] to its bound, let the basic slacks compensate to // keep A*x = b". This is a feasible direction (A*delta_x = 0). The nonzero pattern lives // on the entries of column A(:, j) plus j itself. In the nonbasic_slack slot, // delta_x[nonbasic_slack] = -delta_xj * a_ij > 0, i.e. the entering slack moves up from // its lower bound 0. We build the sparse version to feed the feasibility scan, then - // scatter into a dense vector and normalize so that delta_x[nonbasic_slack] == 1 (the - // convention primal_ratio_test expects for entering variables). + // normalize so that delta_x[nonbasic_slack] == 1 (the convention primal_ratio_test expects + // for entering variables) and scatter into a dense vector. sparse_vector_t delta_x_sparse; delta_x_sparse.n = lp.num_cols; delta_x_sparse.i.reserve(col_end - col_start + 1); delta_x_sparse.x.reserve(col_end - col_start + 1); - const f_t delta_xj = bound - soln_copy.x[j]; delta_x_sparse.i.push_back(j); delta_x_sparse.x.push_back(delta_xj); for (i_t p = col_start; p < col_end; p++) { @@ -3818,15 +3819,14 @@ void branch_and_bound_t::pivot_out_integer_variables( } if (!ok) { continue; } + // Normalize so that delta_x[nonbasic_slack] == 1 (the standard entering-direction + // convention). Done on the sparse vector, after the feasibility scan above, which reads + // the unnormalized values. + for (f_t& val : delta_x_sparse.x) { val /= scale; } + std::vector delta_x(lp.num_cols, 0.0); delta_x_sparse.to_dense(delta_x); - // Normalize so that delta_x[nonbasic_slack] == 1 (the standard entering-direction - // convention). Also confirms A*delta_x = 0 at debug log time. - const f_t scale = delta_x[nonbasic_slack]; - if (!(std::abs(scale) > 1e-12)) { continue; } - for (i_t h = 0; h < lp.num_cols; ++h) { delta_x[h] /= scale; } - // Entering variable is the nonbasic slack, moving up from its lower bound 0. const i_t entering_index = nonbasic_slack; const i_t nonbasic_entering = nonbasic_index[nonbasic_slack]; From 31436ba4f93b32226811b41dc34aa76ab97333c4 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 6 Aug 2026 15:06:21 -0700 Subject: [PATCH 015/113] Remove AI slop --- cpp/src/branch_and_bound/branch_and_bound.hpp | 8 -------- 1 file changed, 8 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index c0ff1761a2..7bc5d905fc 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -358,14 +358,6 @@ class branch_and_bound_t { i_t& num_fractional, std::vector& fractional); - // Try to pivot the nonbasic variable `entering_index` (currently at position - // `nonbasic_entering` in `nonbasic_list`) into the basis along the direction - // `delta_x`, with `utilde` = U * B^{-1} A(:, entering_index) satisfying - // L * utilde = P * A(:, entering_index). Applies the pivot only if it yields a - // strict net decrease in the number of fractional integer variables. On success, - // mutates basic_list/nonbasic_list/vstatus/solution/basis_update in place and applies - // an O(1) fix-up to `nonbasic_index` for the two variables whose (non)basic status - // changed. On skip, leaves all outputs untouched. void apply_delta_x_for_integer_pivot( const simplex::lp_problem_t& lp, std::vector& basic_list, From ef752075a332ea08f1ae54d496231924dc8e9e44 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 6 Aug 2026 15:07:41 -0700 Subject: [PATCH 016/113] Style fixes --- cpp/src/branch_and_bound/branch_and_bound.cpp | 193 +++++++++-------- cpp/src/branch_and_bound/branch_and_bound.hpp | 29 ++- cpp/src/dual_simplex/basis_updates.cpp | 3 +- cpp/src/dual_simplex/crossover.cpp | 17 +- cpp/src/dual_simplex/phase2.cpp | 4 +- cpp/src/dual_simplex/primal.cpp | 199 +++++++++++------- .../dual_simplex/simplex_solver_settings.hpp | 2 +- cpp/src/dual_simplex/solve.cpp | 11 +- cpp/src/pdlp/solve.cu | 13 +- .../solver_settings/solver_settings.pyx | 2 +- 10 files changed, 263 insertions(+), 210 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index dadf7c2e72..99e4c2416d 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -24,9 +24,9 @@ #include #include #include -#include #include #include +#include #include #include #include @@ -710,7 +710,7 @@ bool branch_and_bound_t::repair_solution(const std::vector& edge_ std::vector leaf_edge_norms = edge_norms; f_t repair_work_estimate = 0.0; // should probably set the cut off here lp_settings.cut_off - dual_status_t lp_status = simplex::dual_phase2(2, + dual_status_t lp_status = simplex::dual_phase2(2, 0, lp_start_time, repair_lp, @@ -720,7 +720,7 @@ bool branch_and_bound_t::repair_solution(const std::vector& edge_ iter, repair_work_estimate, leaf_edge_norms); - repaired_solution = lp_solution.x; + repaired_solution = lp_solution.x; if (lp_status == dual_status_t::OPTIMAL) { f_t primal_error; @@ -3179,8 +3179,7 @@ auto branch_and_bound_t::do_cut_pass( root_objective_ = compute_objective(original_lp_, root_relax_soln_.x); // Refresh fractional info after re-solving with cuts; the pre-cut count is stale. - num_fractional = - fractional_variables(settings_, root_relax_soln_.x, var_types_, fractional); + num_fractional = fractional_variables(settings_, root_relax_soln_.x, var_types_, fractional); pivot_out_integer_variables(original_lp_, basic_list, @@ -3265,7 +3264,6 @@ auto branch_and_bound_t::do_cut_pass( return {cut_pass_action_t::CONTINUE, mip_status_t::UNSET}; } - template bool branch_and_bound_t::check_for_dual_degeneracy( const simplex::lp_solution_t& solution, @@ -3285,7 +3283,8 @@ bool branch_and_bound_t::check_for_dual_degeneracy( } template -void branch_and_bound_t::dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, +void branch_and_bound_t::dual_degenerate_feasibility_pump( + const simplex::lp_problem_t& lp, std::vector& basic_list, std::vector& nonbasic_list, std::vector& vstatus, @@ -3297,7 +3296,8 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple f_t dual_degenerate_feasibility_pump_start_time = tic(); std::vector zero_reduced_costs_vars; std::vector zero_reduced_costs_vars_nonbasic_index; - bool dual_degenerate = check_for_dual_degeneracy(soln, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); + bool dual_degenerate = check_for_dual_degeneracy( + soln, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); if (!dual_degenerate) { return; } // Construct a new LP problem @@ -3320,16 +3320,16 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple simplex::lp_problem_t lp_reduced(lp.handle_ptr, m, n, nnz); csc_matrix_t& A_reduced = lp_reduced.A; std::vector original_col_to_reduced_col(lp.num_cols, -1); - i_t nz = 0; + i_t nz = 0; i_t reduced_col = 0; for (i_t j = 0; j < lp.num_cols; j++) { if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { - original_col_to_reduced_col[j] = reduced_col; + original_col_to_reduced_col[j] = reduced_col; A_reduced.col_start[reduced_col] = nz; - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; for (i_t p = col_start; p < col_end; p++) { - const i_t i = lp.A.i[p]; + const i_t i = lp.A.i[p]; const f_t value = lp.A.x[p]; A_reduced.i[nz] = i; A_reduced.x[nz] = value; @@ -3348,32 +3348,32 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple // PASS } else { const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; + const i_t col_end = lp.A.col_start[j + 1]; for (i_t p = col_start; p < col_end; p++) { - const i_t i = lp.A.i[p]; + const i_t i = lp.A.i[p]; const f_t value = lp.A.x[p]; b_reduced[i] -= value * soln.x[j]; } } } - lp_reduced.rhs = b_reduced; + lp_reduced.rhs = b_reduced; lp_reduced.obj_scale = 1.0; - - - settings_.log.printf("Constructed dual degenerate feasibility pump LP with %d rows and %d columns\n", m, n); + settings_.log.printf( + "Constructed dual degenerate feasibility pump LP with %d rows and %d columns\n", m, n); std::vector reduced_basic_list(m); std::vector reduced_nonbasic_list(zero_reduced_costs_vars.size()); std::vector reduced_vstatus(n); - i_t num_basic = 0; + i_t num_basic = 0; i_t num_nonbasic = 0; - reduced_col = 0; + reduced_col = 0; for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC){ + if (vstatus[j] == variable_status_t::BASIC) { reduced_vstatus[reduced_col++] = variable_status_t::BASIC; } else if (std::abs(soln.z[j]) <= 1e-10) { - reduced_nonbasic_list[num_nonbasic++] = reduced_col; // Does ordering of nonbasic variables matter? + reduced_nonbasic_list[num_nonbasic++] = + reduced_col; // Does ordering of nonbasic variables matter? reduced_vstatus[reduced_col++] = vstatus[j]; } } @@ -3400,8 +3400,8 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple } f_t primal_work_estimate = 0.0; - i_t iter = 0; - i_t max_pump_iter = 10; + i_t iter = 0; + i_t max_pump_iter = 10; simplex::random_t rng(settings_.random_seed); i_t best_num_fractional = num_fractional; std::vector best_reduced_vstatus(n); @@ -3435,22 +3435,23 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple } } - bool recompute_basis = false; - const i_t iter_before = iter; + bool recompute_basis = false; + const i_t iter_before = iter; simplex_solver_settings_t primal_settings = settings_; - primal_settings.log.log = false; + primal_settings.log.log = false; primal_settings.time_limit = settings_.time_limit - toc(exploration_stats_.start_time); - simplex::primal_status_t lp_status = simplex::primal_phase2_with_advanced_basis(2, - exploration_stats_.start_time, - lp_reduced, - primal_settings, - reduced_vstatus, - reduced_basis_update, - reduced_basic_list, - reduced_nonbasic_list, - reduced_solution, - iter, - primal_work_estimate); + simplex::primal_status_t lp_status = + simplex::primal_phase2_with_advanced_basis(2, + exploration_stats_.start_time, + lp_reduced, + primal_settings, + reduced_vstatus, + reduced_basis_update, + reduced_basic_list, + reduced_nonbasic_list, + reduced_solution, + iter, + primal_work_estimate); // Detect a stall: the solve made no pivots, so the incumbent vertex was // already optimal for this objective and x did not move. Perturb next pass. stalled = (iter == iter_before); @@ -3479,7 +3480,15 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple i_t num_fractional_reduced = fractional_variables(settings_, adjusted_solution, var_types_, tmp_fractional); settings_.log.printf( - "Degenerate feasibility pump (%d/%d): primal work estimate %.2e, iter %d, fractional variables %d/%d. Time %.2f\n", pump_iter, max_pump_iter, primal_work_estimate, iter, num_fractional_reduced, num_fractional, toc(dual_degenerate_feasibility_pump_start_time)); + "Degenerate feasibility pump (%d/%d): primal work estimate %.2e, iter %d, fractional " + "variables %d/%d. Time %.2f\n", + pump_iter, + max_pump_iter, + primal_work_estimate, + iter, + num_fractional_reduced, + num_fractional, + toc(dual_degenerate_feasibility_pump_start_time)); // Also treat a pass that fails to improve the best as a stall, so we perturb // the next pass even when the solve pivoted (moved) without reducing the count. stalled = stalled || (num_fractional_reduced >= best_num_fractional); @@ -3492,7 +3501,13 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple } } - settings_.log.printf("Degenerate feasibility pump: Simplex iterations %d, Best number of fractional variables %d/%d. Time %.2f\n", iter, best_num_fractional, num_fractional, toc(dual_degenerate_feasibility_pump_start_time)); + settings_.log.printf( + "Degenerate feasibility pump: Simplex iterations %d, Best number of fractional variables " + "%d/%d. Time %.2f\n", + iter, + best_num_fractional, + num_fractional, + toc(dual_degenerate_feasibility_pump_start_time)); if (best_num_fractional < num_fractional) { // Translate the vstatus from the reduced problem to the vstatus for the original problem i_t reduced_cols = 0; @@ -3520,9 +3535,10 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple return; } if (refactor_status != 0) { - settings_.log.printf("Failed to refactor basis after dual degenerate feasibility pump. " - "%d deficient columns.\n", - refactor_status); + settings_.log.printf( + "Failed to refactor basis after dual degenerate feasibility pump. " + "%d deficient columns.\n", + refactor_status); return; } @@ -3530,7 +3546,8 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple // First set the nonbasic variables on their bounds for (i_t k = 0; k < lp.num_cols - lp.num_rows; k++) { const i_t j = nonbasic_list[k]; - if (vstatus[j] == variable_status_t::NONBASIC_LOWER || vstatus[j] == variable_status_t::NONBASIC_FIXED) { + if (vstatus[j] == variable_status_t::NONBASIC_LOWER || + vstatus[j] == variable_status_t::NONBASIC_FIXED) { soln.x[j] = lp.lower[j]; } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER) { soln.x[j] = lp.upper[j]; @@ -3543,11 +3560,11 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple for (i_t j = 0; j < lp.num_cols; j++) { if (vstatus[j] == variable_status_t::BASIC) { continue; } const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; + const i_t col_end = lp.A.col_start[j + 1]; const f_t x_j = soln.x[j]; for (i_t p = col_start; p < col_end; p++) { - const i_t i = lp.A.i[p]; + const i_t i = lp.A.i[p]; const f_t aij = lp.A.x[p]; rhs[i] -= aij * x_j; } @@ -3564,11 +3581,9 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump(const simple fractional.clear(); num_fractional = fractional_variables(settings_, soln.x, var_types_, fractional); - } } - template void branch_and_bound_t::apply_delta_x_for_integer_pivot( const simplex::lp_problem_t& lp, @@ -3608,10 +3623,8 @@ void branch_and_bound_t::apply_delta_x_for_integer_pivot( for (i_t h = 0; h < lp.num_cols; ++h) { test_x[h] += step_length * delta_x[h]; if (var_types_[h] != variable_type_t::INTEGER) { continue; } - const bool was_fractional = - is_fractional(solution.x[h], var_types_[h], settings_.integer_tol); - const bool now_fractional = - is_fractional(test_x[h], var_types_[h], settings_.integer_tol); + const bool was_fractional = is_fractional(solution.x[h], var_types_[h], settings_.integer_tol); + const bool now_fractional = is_fractional(test_x[h], var_types_[h], settings_.integer_tol); if (now_fractional && !was_fractional) { integer_destroyed++; } else if (!now_fractional && was_fractional) { @@ -3645,8 +3658,7 @@ void branch_and_bound_t::apply_delta_x_for_integer_pivot( sparse_vector_t UTsol_sparse(m, 1); sparse_vector_t solution_sparse(m, 1); basis_update.b_transpose_solve(es_sparse, solution_sparse, UTsol_sparse); - const i_t recommend_refactor = - basis_update.update(utilde_sparse, UTsol_sparse, basic_leaving); + const i_t recommend_refactor = basis_update.update(utilde_sparse, UTsol_sparse, basic_leaving); if (recommend_refactor == 1) { csc_matrix_t L(m, m, 1); csc_matrix_t U(m, m, 1); @@ -3657,17 +3669,17 @@ void branch_and_bound_t::apply_delta_x_for_integer_pivot( std::vector slacks_needed; f_t factorize_work_estimate = 0.0; const i_t rank = factorize_basis(lp.A, - settings_, - basic_list, - exploration_stats_.start_time, - L, - U, - p, - pinv, - q, - deficient, - slacks_needed, - factorize_work_estimate); + settings_, + basic_list, + exploration_stats_.start_time, + L, + U, + p, + pinv, + q, + deficient, + slacks_needed, + factorize_work_estimate); if (rank == CONCURRENT_HALT_RETURN || rank == TIME_LIMIT_RETURN) { return; } if (rank < 0 || rank != lp.num_rows) { return; } simplex::reorder_basic_list(q, basic_list); @@ -3690,7 +3702,8 @@ void branch_and_bound_t::pivot_out_integer_variables( f_t pivot_out_integer_variables_start_time = tic(); std::vector zero_reduced_costs_vars; std::vector zero_reduced_costs_vars_nonbasic_index; - bool dual_degenerate = check_for_dual_degeneracy(solution, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); + bool dual_degenerate = check_for_dual_degeneracy( + solution, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); if (!dual_degenerate) { return; } lp_solution_t soln_copy = solution; @@ -3706,7 +3719,7 @@ void branch_and_bound_t::pivot_out_integer_variables( std::vector row_to_slack(lp.num_rows, -1); for (i_t j : new_slacks_) { if (lp.lower[j] != 0 || lp.upper[j] != inf) { continue; } - const i_t p = lp.A.col_start[j]; + const i_t p = lp.A.col_start[j]; row_to_slack[lp.A.i[p]] = j; } @@ -3745,7 +3758,7 @@ void branch_and_bound_t::pivot_out_integer_variables( if (fast_candidates.size() > 0) { settings_.log.printf("Found %ld fast candidates for pivot out integer variables\n", - fast_candidates.size()); + fast_candidates.size()); } // Build a reverse index nonbasic_index[v] = position of v in nonbasic_list_copy, or -1 if not @@ -3780,7 +3793,7 @@ void branch_and_bound_t::pivot_out_integer_variables( if (std::abs(bound) == inf) { continue; } const f_t delta_xj = bound - soln_copy.x[j]; - const f_t scale = -delta_xj * a_ij; + const f_t scale = -delta_xj * a_ij; if (std::abs(scale) <= 1e-12) { continue; } // Build delta_x describing "move x[j] to its bound, let the basic slacks compensate to @@ -3797,8 +3810,8 @@ void branch_and_bound_t::pivot_out_integer_variables( delta_x_sparse.i.push_back(j); delta_x_sparse.x.push_back(delta_xj); for (i_t p = col_start; p < col_end; p++) { - const i_t r = lp.A.i[p]; - const f_t a_rj = lp.A.x[p]; + const i_t r = lp.A.i[p]; + const f_t a_rj = lp.A.x[p]; const f_t delta_slack_r = -delta_xj * a_rj; delta_x_sparse.i.push_back(row_to_slack[r]); delta_x_sparse.x.push_back(delta_slack_r); @@ -3822,7 +3835,9 @@ void branch_and_bound_t::pivot_out_integer_variables( // Normalize so that delta_x[nonbasic_slack] == 1 (the standard entering-direction // convention). Done on the sparse vector, after the feasibility scan above, which reads // the unnormalized values. - for (f_t& val : delta_x_sparse.x) { val /= scale; } + for (f_t& val : delta_x_sparse.x) { + val /= scale; + } std::vector delta_x(lp.num_cols, 0.0); delta_x_sparse.to_dense(delta_x); @@ -3873,15 +3888,13 @@ void branch_and_bound_t::pivot_out_integer_variables( if (var_types_[j] == variable_type_t::INTEGER) { continue; } if (vstatus_copy[j] == variable_status_t::BASIC) { continue; } - const i_t direction = - (vstatus_copy[j] == variable_status_t::NONBASIC_LOWER || - vstatus_copy[j] == variable_status_t::NONBASIC_FIXED) - ? 1 - : -1; + const i_t direction = (vstatus_copy[j] == variable_status_t::NONBASIC_LOWER || + vstatus_copy[j] == variable_status_t::NONBASIC_FIXED) + ? 1 + : -1; const i_t entering_index = j; const i_t nonbasic_entering = zero_reduced_costs_vars_nonbasic_index[k]; - if (nonbasic_entering < 0 || - nonbasic_entering >= static_cast(nonbasic_list_copy.size()) || + if (nonbasic_entering < 0 || nonbasic_entering >= static_cast(nonbasic_list_copy.size()) || nonbasic_list_copy[nonbasic_entering] != j) { continue; } @@ -3922,7 +3935,11 @@ void branch_and_bound_t::pivot_out_integer_variables( if (num_new_fractional < start_num_fractional) { i_t num_integer_increased = start_num_fractional - num_new_fractional; integer_pivots_.fetch_add(num_integer_increased, std::memory_order_release); - settings_.log.printf("Pivoted out %d integer variables: %d -> %d in %.2f\n", num_integer_increased, start_num_fractional, num_new_fractional, toc(pivot_out_integer_variables_start_time)); + settings_.log.printf("Pivoted out %d integer variables: %d -> %d in %.2f\n", + num_integer_increased, + start_num_fractional, + num_new_fractional, + toc(pivot_out_integer_variables_start_time)); num_fractional = num_new_fractional; fractional = new_fractional; basic_list = basic_list_copy; @@ -4014,7 +4031,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut lp_status_t root_status = lp_status_t::UNSET; solving_root_relaxation_ = true; - f_t root_relax_start_time = tic(); + f_t root_relax_start_time = tic(); f_t root_relax_work_estimate = 0.0; if (!enable_concurrent_lp_root_solve()) { // RINS/SUBMIP path @@ -4049,8 +4066,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut solving_root_relaxation_ = false; f_t root_relax_elapsed_time = toc(root_relax_start_time); exploration_stats_.total_lp_solve_time = root_relax_elapsed_time; - i_t root_iterations = exploration_stats_.total_simplex_iters; - + i_t root_iterations = exploration_stats_.total_simplex_iters; if (root_status == lp_status_t::INFEASIBLE) { settings_.log.printf("\nThe root LP relaxation is infeasible\n", @@ -4104,7 +4120,10 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut root_relax_soln_.iterations, root_relax_elapsed_time, method_to_string(root_relax_solved_by)); - settings_.log.printf("Dual simplex iteration %d work estimate %.2e work per second %.2e\n", root_iterations, root_relax_work_estimate, root_relax_work_estimate / root_relax_elapsed_time); + settings_.log.printf("Dual simplex iteration %d work estimate %.2e work per second %.2e\n", + root_iterations, + root_relax_work_estimate, + root_relax_work_estimate / root_relax_elapsed_time); settings_.log.printf("Root relaxation objective %+.8e\n\n", root_relax_soln_.user_objective); assert(root_vstatus_.size() == original_lp_.num_cols); @@ -5054,8 +5073,8 @@ node_status_t branch_and_bound_t::solve_node_deterministic( i_t node_iter = 0; f_t lp_start_time = tic(); std::vector leaf_edge_norms = edge_norms_; - f_t dual_work_estimate = 0.0; - dual_status_t lp_status = dual_phase2_with_advanced_basis(2, + f_t dual_work_estimate = 0.0; + dual_status_t lp_status = dual_phase2_with_advanced_basis(2, 0, worker.recompute_bounds_and_basis, lp_start_time, diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 7bc5d905fc..17f6f7a3a3 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -341,14 +341,12 @@ class branch_and_bound_t { i_t leaf_depth, search_strategy_t thread_type); - omp_atomic_t integer_pivots_{0}; bool check_for_dual_degeneracy(const simplex::lp_solution_t& solution, const std::vector& nonbasic_list, std::vector& zero_reduced_costs_vars, std::vector& zero_reduced_costs_vars_nonbasic_index); - void pivot_out_integer_variables(const simplex::lp_problem_t& lp, std::vector& basic_list, std::vector& nonbasic_list, @@ -358,20 +356,19 @@ class branch_and_bound_t { i_t& num_fractional, std::vector& fractional); - void apply_delta_x_for_integer_pivot( - const simplex::lp_problem_t& lp, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& nonbasic_index, - std::vector& vstatus, - i_t entering_index, - i_t nonbasic_entering, - i_t direction, - std::vector& delta_x, - const sparse_vector_t& utilde_sparse, - simplex::lp_solution_t& solution, - simplex::basis_update_mpf_t& basis_update, - f_t& work_estimate); + void apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& nonbasic_index, + std::vector& vstatus, + i_t entering_index, + i_t nonbasic_entering, + i_t direction, + std::vector& delta_x, + const sparse_vector_t& utilde_sparse, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + f_t& work_estimate); void dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, std::vector& basic_list, diff --git a/cpp/src/dual_simplex/basis_updates.cpp b/cpp/src/dual_simplex/basis_updates.cpp index 1081cc4773..a3d3787183 100644 --- a/cpp/src/dual_simplex/basis_updates.cpp +++ b/cpp/src/dual_simplex/basis_updates.cpp @@ -2013,8 +2013,7 @@ i_t basis_update_mpf_t::u_solve(sparse_vector_t& rhs) const // Compute y = U*x. In the MPF factorization, the rank-1 update factors are absorbed into L, so // U == U0 and U*x reduces to a sparse matvec against U0. template -void basis_update_mpf_t::u_multiply(const std::vector& x, - std::vector& y) const +void basis_update_mpf_t::u_multiply(const std::vector& x, std::vector& y) const { const i_t m = L0_.m; y.assign(m, 0.0); diff --git a/cpp/src/dual_simplex/crossover.cpp b/cpp/src/dual_simplex/crossover.cpp index 5b0dd451e3..977f5e5511 100644 --- a/cpp/src/dual_simplex/crossover.cpp +++ b/cpp/src/dual_simplex/crossover.cpp @@ -1424,11 +1424,11 @@ crossover_status_t crossover(const lp_problem_t& lp, } else if (dual_feasible && !primal_feasible) { i_t dual_iter = 0; std::vector edge_norms; - f_t work_estimate = 0.0; + f_t work_estimate = 0.0; simplex_solver_settings_t dual_settings = settings; - dual_settings.iteration_limit = std::numeric_limits::max(); - dual_status_t status = - dual_phase2(2, 0, start_time, lp, dual_settings, vstatus, solution, dual_iter, work_estimate, edge_norms); + dual_settings.iteration_limit = std::numeric_limits::max(); + dual_status_t status = dual_phase2( + 2, 0, start_time, lp, dual_settings, vstatus, solution, dual_iter, work_estimate, edge_norms); if (toc(start_time) > settings.time_limit) { settings.log.printf("Time limit exceeded\n"); return crossover_status_t::TIME_LIMIT; @@ -1448,10 +1448,11 @@ crossover_status_t crossover(const lp_problem_t& lp, primal_feasible = primal_infeas <= primal_tol && primal_res <= primal_tol; dual_feasible = dual_infeas <= dual_tol && dual_res <= dual_tol; } else if (primal_feasible && !dual_feasible) { - i_t primal_iter = 0; + i_t primal_iter = 0; simplex_solver_settings_t primal_settings = settings; - primal_settings.iteration_limit = std::numeric_limits::max(); - primal_status_t primal_status = primal_phase2(2, start_time, lp, primal_settings, vstatus, solution, primal_iter); + primal_settings.iteration_limit = std::numeric_limits::max(); + primal_status_t primal_status = + primal_phase2(2, start_time, lp, primal_settings, vstatus, solution, primal_iter); if (toc(start_time) > settings.time_limit) { settings.log.printf("Time limit exceeded\n"); return crossover_status_t::TIME_LIMIT; @@ -1472,7 +1473,7 @@ crossover_status_t crossover(const lp_problem_t& lp, dual_feasible = dual_infeas <= dual_tol && dual_res <= dual_tol; } else { simplex_solver_settings_t dual_settings = settings; - dual_settings.iteration_limit = std::numeric_limits::max(); + dual_settings.iteration_limit = std::numeric_limits::max(); lp_problem_t phase1_problem(lp.handle_ptr, 1, 1, 1); create_phase1_problem(lp, phase1_problem); std::vector phase1_vstatus(n); diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 233bc1ea6a..f86aeb0333 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2347,8 +2347,8 @@ void prepare_optimality(i_t info, std::vector& z, lp_solution_t& sol) { - const i_t m = lp.num_rows; - const i_t n = lp.num_cols; + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; sol.objective = compute_objective(lp, sol.x); sol.user_objective = compute_user_objective(lp, sol.objective); diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 1778299c79..27351a3685 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -57,7 +57,7 @@ void set_primal_variables_on_bounds(const lp_problem_t& lp, assert(1 == 0); } } - work_estimate += n + 3.0*(n - m); + work_estimate += n + 3.0 * (n - m); } template @@ -168,7 +168,7 @@ f_t primal_infeasibility(const lp_problem_t& lp, i_t& num_infeasible, f_t& work_estimate) { - const i_t m = lp.num_rows; + const i_t m = lp.num_rows; const i_t n = lp.num_cols; f_t primal_inf = 0; num_infeasible = 0; @@ -206,7 +206,7 @@ f_t primal_infeasibility(const lp_problem_t& lp, } } } - work_estimate += n + 4*m; + work_estimate += n + 4 * m; return primal_inf; } @@ -243,7 +243,7 @@ void compute_phase1_objective(const lp_problem_t& lp, objective[j] = 0.0; } } - work_estimate += n-m + 4 * m; + work_estimate += n - m + 4 * m; } template @@ -281,7 +281,7 @@ void compute_delta_z(const csr_matrix_t& Arow, const i_t j = Arow.j[p]; if (vstatus[j] != variable_status_t::BASIC) { delta_z[j] -= Arow.x[p] * delta_y_i; } } - work_estimate += 4*(row_end - row_start); + work_estimate += 4 * (row_end - row_start); } work_estimate += 4 * delta_y.i.size(); } @@ -356,7 +356,7 @@ void compute_dual_variables(const lp_problem_t& lp, for (i_t p = col_start; p < col_end; ++p) { dot += lp.A.x[p] * y[lp.A.i[p]]; } - work_estimate += 3.0*(col_end - col_start); + work_estimate += 3.0 * (col_end - col_start); z[j] -= dot; } work_estimate += 6 * (n - m); @@ -364,7 +364,7 @@ void compute_dual_variables(const lp_problem_t& lp, for (i_t k = 0; k < m; ++k) { z[basic_list[k]] = 0.0; } - work_estimate += 2*m; + work_estimate += 2 * m; } template @@ -386,7 +386,7 @@ void compute_basic_primal_variables(const lp_problem_t& lp, for (i_t p = col_start; p < col_end; ++p) { rhs[lp.A.i[p]] -= xj * lp.A.x[p]; } - work_estimate += 3.0*(col_end - col_start); + work_estimate += 3.0 * (col_end - col_start); } work_estimate += 4 * (n - m); std::vector xB(m); @@ -408,7 +408,6 @@ f_t primal_constraint_residual(const lp_problem_t& lp, const std::vect } // namespace - template i_t primal_ratio_test(const lp_problem_t& lp, const simplex_solver_settings_t& settings, @@ -519,12 +518,11 @@ i_t primal_ratio_test(const lp_problem_t& lp, } } } - work_estimate += 10*m; + work_estimate += 10 * m; step_length = min_val; return leaving_index; } - template primal_status_t primal_phase2(i_t phase, f_t start_time, @@ -543,7 +541,7 @@ primal_status_t primal_phase2(i_t phase, std::vector superbasic_list; get_basis_from_vstatus(m, vstatus, basic_list, nonbasic_list, superbasic_list); - work_estimate += 2*n; + work_estimate += 2 * n; assert(superbasic_list.size() == 0); assert(nonbasic_list.size() == n - m); @@ -680,11 +678,10 @@ primal_status_t primal_phase2_with_advanced_basis( for (i_t p = col_start; p < col_end; ++p) { rhs[lp.A.i[p]] -= xj * lp.A.x[p]; } - work_estimate += 3.0*(col_end - col_start); + work_estimate += 3.0 * (col_end - col_start); } work_estimate += 4 * (n - m); - std::vector xB(m); work_estimate += m; @@ -697,25 +694,22 @@ primal_status_t primal_phase2_with_advanced_basis( work_estimate += 3 * m; constexpr bool print_norms = false; - if constexpr (print_norms) { - settings.log.printf("|| x || %e\n", vector_norm2(x)); - } + if constexpr (print_norms) { settings.log.printf("|| x || %e\n", vector_norm2(x)); } std::vector residual = lp.rhs; work_estimate += m; matrix_vector_multiply(lp.A, 1.0, x, -1.0, residual); - work_estimate += m + 2*n + 4.0*lp.A.col_start[lp.A.n]; + work_estimate += m + 2 * n + 4.0 * lp.A.col_start[lp.A.n]; f_t primal_residual = vector_norm_inf(residual); work_estimate += m; if (primal_residual > settings.primal_tol) { settings.log.printf("|| A*x - b || %e\n", primal_residual); } - std::vector objective = lp.objective; - work_estimate += 2*n; - const f_t primal_tol = settings.primal_tol; - f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x, work_estimate); + work_estimate += 2 * n; + const f_t primal_tol = settings.primal_tol; + f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x, work_estimate); if (primal_inf > primal_tol) { // We are primal infeasible. Switch to phase 1 compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); @@ -731,22 +725,18 @@ primal_status_t primal_phase2_with_advanced_basis( work_estimate += m; compute_dual_variables( lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); - if constexpr (print_norms) { - settings.log.printf("|| z || %e\n", vector_norm_inf(z)); - } + if constexpr (print_norms) { settings.log.printf("|| z || %e\n", vector_norm_inf(z)); } - i_t num_dual_inf = 0; - i_t num_primal_inf = 0; + i_t num_dual_inf = 0; + i_t num_primal_inf = 0; const f_t init_dual_inf = dual_infeasibility(lp, vstatus, z, settings.dual_tol, num_dual_inf, work_estimate); - if (num_dual_inf > 0) { - settings.log.printf("Initial dual infeasibility %e\n", init_dual_inf); - } + if (num_dual_inf > 0) { settings.log.printf("Initial dual infeasibility %e\n", init_dual_inf); } csr_matrix_t Arow(m, n, lp.A.nnz()); - work_estimate += n + 2*lp.A.nnz(); + work_estimate += n + 2 * lp.A.nnz(); lp.A.to_compressed_row(Arow); - work_estimate += m + 6*lp.A.nnz(); + work_estimate += m + 6 * lp.A.nnz(); const i_t iter_limit = settings.iteration_limit; const i_t start_iter = iter; @@ -754,13 +744,13 @@ primal_status_t primal_phase2_with_advanced_basis( sparse_vector_t etilde(m, 0); std::vector delta_z(n); std::vector delta_x(n); - work_estimate += 2*m + 2*n; + work_estimate += 2 * m + 2 * n; f_t dual_inf = init_dual_inf; f_t obj = compute_objective(lp, x); - work_estimate += 2*n; + work_estimate += 2 * n; f_t pricing_dual_tol = settings.dual_tol; - primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); settings.log.printf("%5d %+.16e %7d %.8e %.2f\n", iter, @@ -776,8 +766,15 @@ primal_status_t primal_phase2_with_advanced_basis( while (iter < iter_limit) { i_t nonbasic_entering = -1; i_t direction; - i_t entering_index = phase2_pricing( - lp, z, nonbasic_list, vstatus, pricing_dual_tol, direction, nonbasic_entering, dual_inf, work_estimate); + i_t entering_index = phase2_pricing(lp, + z, + nonbasic_list, + vstatus, + pricing_dual_tol, + direction, + nonbasic_entering, + dual_inf, + work_estimate); if (entering_index == -1) { if (phase == 2) { // Verify optimality with a consistent basic solution: refactor, put @@ -797,9 +794,18 @@ primal_status_t primal_phase2_with_advanced_basis( basis_update.clear_work_estimate(); } set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); - compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x, work_estimate); - compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); + compute_basic_primal_variables( + lp, basis_update, basic_list, nonbasic_list, x, work_estimate); + compute_dual_variables(lp, + settings, + objective, + basic_list, + nonbasic_list, + basis_update, + c_basic, + y, + z, + work_estimate); primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); dual_inf = dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf, work_estimate); @@ -807,8 +813,16 @@ primal_status_t primal_phase2_with_advanced_basis( compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); phase = 1; pricing_dual_tol = settings.dual_tol; - compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); + compute_dual_variables(lp, + settings, + objective, + basic_list, + nonbasic_list, + basis_update, + c_basic, + y, + z, + work_estimate); settings.log.printf( "Switching to Primal Simplex Phase 1 after near optimality. " "Primal infeasibility %e\n", @@ -852,10 +866,10 @@ primal_status_t primal_phase2_with_advanced_basis( } } // Report the unfiltered residual at the accepted solution. - dual_inf = tight_dual_inf; - num_dual_inf = num_tight_dual_inf; - obj = compute_objective(lp, x); - work_estimate += 2*n; + dual_inf = tight_dual_inf; + num_dual_inf = num_tight_dual_inf; + obj = compute_objective(lp, x); + work_estimate += 2 * n; sol.objective = obj; sol.user_objective = compute_user_objective(lp, obj); if (!settings.inside_mip && print_summary) { @@ -878,10 +892,19 @@ primal_status_t primal_phase2_with_advanced_basis( // objective. Refresh objective and duals, then retry pricing with // successively tighter dual tolerances. settings.log.printf("Refreshing phase-I objective and duals. Num updates %d. Iter %d\n", - basis_update.num_updates(), iter); + basis_update.num_updates(), + iter); compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); - compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); + compute_dual_variables(lp, + settings, + objective, + basic_list, + nonbasic_list, + basis_update, + c_basic, + y, + z, + work_estimate); f_t retry_dual_tol = pricing_dual_tol; while (entering_index == -1 && retry_dual_tol > f_t(1e-10)) { retry_dual_tol *= f_t(0.1); @@ -913,10 +936,18 @@ primal_status_t primal_phase2_with_advanced_basis( settings.log.printf( "Primal phase I complete. Iterations %d. Time %.2f\n", iter, toc(start_time)); settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); - compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); - obj = compute_objective(lp, x); - work_estimate += 2*n; + compute_dual_variables(lp, + settings, + objective, + basic_list, + nonbasic_list, + basis_update, + c_basic, + y, + z, + work_estimate); + obj = compute_objective(lp, x); + work_estimate += 2 * n; dual_inf = dual_infeasibility(lp, vstatus, z, settings.dual_tol, num_dual_inf, work_estimate); iter++; @@ -934,7 +965,7 @@ primal_status_t primal_phase2_with_advanced_basis( } sparse_vector_t rhs_sparse(lp.A, entering_index); - work_estimate += 3*rhs_sparse.i.size(); + work_estimate += 3 * rhs_sparse.i.size(); sparse_vector_t scaled_delta_xB_sparse(m, 0); sparse_vector_t utilde_sparse(m, 0); basis_update.b_solve(rhs_sparse, scaled_delta_xB_sparse, utilde_sparse); @@ -946,12 +977,12 @@ primal_status_t primal_phase2_with_advanced_basis( const i_t j = basic_list[k]; delta_x[j] = -direction * scaled_delta_xB[k]; } - work_estimate += 3*m; + work_estimate += 3 * m; for (i_t k = 0; k < n - m; ++k) { const i_t j = nonbasic_list[k]; delta_x[j] = 0.0; } - work_estimate += 2*(n - m); + work_estimate += 2 * (n - m); delta_x[entering_index] = direction; #ifdef CHECK_NULLSPACE @@ -989,16 +1020,18 @@ primal_status_t primal_phase2_with_advanced_basis( for (i_t j = 0; j < n; ++j) { x[j] += step_length * delta_x[j]; } - work_estimate += 2*n; + work_estimate += 2 * n; #ifdef COMPUTE_RESIDUAL f_t debug_primal_residual = primal_constraint_residual(lp, x); if (debug_primal_residual > 1e-6) { - settings.log.printf("|| A * x - b || %e at iteration %d (updates %d)\n", debug_primal_residual, iter, basis_update.num_updates()); + settings.log.printf("|| A * x - b || %e at iteration %d (updates %d)\n", + debug_primal_residual, + iter, + basis_update.num_updates()); } #endif - if (basis_updated) { assert(step_length >= 0.0); @@ -1062,11 +1095,13 @@ primal_status_t primal_phase2_with_advanced_basis( recompute_duals = true; // Factor matches basic_list: rebuild x_B so Ax = b exactly. set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); - compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x, work_estimate); + compute_basic_primal_variables( + lp, basis_update, basic_list, nonbasic_list, x, work_estimate); } else if (rebuild_x_after_bound_snap) { // FT update already matches the new basis; recompute x_B with the leaving variable // snapped onto its bound. - compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x, work_estimate); + compute_basic_primal_variables( + lp, basis_update, basic_list, nonbasic_list, x, work_estimate); } } else { if (direction > 0) { @@ -1103,14 +1138,21 @@ primal_status_t primal_phase2_with_advanced_basis( } if (recompute_duals) { - compute_dual_variables( - lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); + compute_dual_variables(lp, + settings, + objective, + basic_list, + nonbasic_list, + basis_update, + c_basic, + y, + z, + work_estimate); } - obj = compute_objective(lp, x); - work_estimate += 2*n; - dual_inf = - dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf, work_estimate); + obj = compute_objective(lp, x); + work_estimate += 2 * n; + dual_inf = dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf, work_estimate); iter++; @@ -1140,18 +1182,17 @@ primal_status_t primal_phase2_with_advanced_basis( #ifdef DUAL_SIMPLEX_INSTANTIATE_DOUBLE -template -int primal_ratio_test(const lp_problem_t& lp, - const simplex_solver_settings_t& settings, - const std::vector& vstatus, - const std::vector& basic_list, - std::vector& x, - std::vector& delta_x, - double& step_length, - int& basic_leaving, - int entering_index, - int direction, - double& work_estimate); +template int primal_ratio_test(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& vstatus, + const std::vector& basic_list, + std::vector& x, + std::vector& delta_x, + double& step_length, + int& basic_leaving, + int entering_index, + int direction, + double& work_estimate); template primal_status_t primal_phase2( int phase, diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index c4338810bc..50e8ca15a7 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -172,7 +172,7 @@ struct simplex_solver_settings_t { i_t augmented; // -1 automatic, 0 to solve with ADAT, 1 to solve with augmented system i_t dualize; // -1 automatic, 0 to not dualize, 1 to dualize i_t ordering; // -1 automatic, 0 to use nested dissection, 1 to use AMD - i_t initial_perturbation; // -1 automatic, 0 to not perturb, 1 to perturb + i_t initial_perturbation; // -1 automatic, 0 to not perturb, 1 to perturb i_t barrier_dual_initial_point; // -1 automatic, 0 to use Lustig, Marsten, and Shanno initial // point, 1 to use initial point form dual least squares problem i_t postsolve_info; // -1 automatic (disabled), 0 disabled, 1 enabled diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index dce731d4d1..fedb9de356 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -796,7 +796,7 @@ lp_status_t solve_linear_program_with_primal(const user_problem_t& use i_t iter = 0; const primal_status_t primal_status = primal_phase2(2, start_time, lp, settings, vstatus, lp_solution, iter); - lp_solution.iterations = iter; + lp_solution.iterations = iter; original_solution.iterations = iter; if (primal_status == primal_status_t::CONCURRENT_LIMIT) { @@ -846,12 +846,8 @@ lp_status_t solve_linear_program_with_primal(const user_problem_t& use } uncrush_primal_solution(user_problem, original_lp, original_solution.x, solution.x); - uncrush_dual_solution(user_problem, - original_lp, - original_solution.y, - original_solution.z, - solution.y, - solution.z); + uncrush_dual_solution( + user_problem, original_lp, original_solution.y, original_solution.z, solution.y, solution.z); solution.objective = original_solution.objective; solution.user_objective = original_solution.user_objective; solution.iterations = original_solution.iterations; @@ -860,7 +856,6 @@ lp_status_t solve_linear_program_with_primal(const user_problem_t& use return map_primal_status_to_lp_status(primal_status); } - template lp_status_t solve_linear_program(const user_problem_t& user_problem, const simplex_solver_settings_t& settings, diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index 4a7a6bec31..beb53e8a17 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -594,9 +594,9 @@ std::tuple, simplex::lp_status_t, f_t, f_t, f_t f_t norm_rhs = vector_norm2(user_problem.rhs); simplex::simplex_solver_settings_t dual_simplex_settings; - dual_simplex_settings.time_limit = settings.time_limit; - dual_simplex_settings.iteration_limit = settings.iteration_limit; - dual_simplex_settings.concurrent_halt = settings.concurrent_halt; + dual_simplex_settings.time_limit = settings.time_limit; + dual_simplex_settings.iteration_limit = settings.iteration_limit; + dual_simplex_settings.concurrent_halt = settings.concurrent_halt; dual_simplex_settings.initial_perturbation = settings.initial_perturbation; if (dual_simplex_settings.concurrent_halt != nullptr) { // Don't show the dual simplex log in concurrent mode. Show the PDLP log instead @@ -677,9 +677,10 @@ std::tuple, simplex::lp_status_t, f_t, f_t, f_t } template -optimization_problem_solution_t run_primal(mip::problem_t& problem, - pdlp_solver_settings_t const& settings, - const timer_t& timer) +optimization_problem_solution_t run_primal( + mip::problem_t& problem, + pdlp_solver_settings_t const& settings, + const timer_t& timer) { simplex::user_problem_t primal_problem = cuopt_problem_to_user_problem(problem.handle_ptr, problem); diff --git a/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx b/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx index ce3ef6fef3..73a2ccccf9 100644 --- a/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx +++ b/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx @@ -1,4 +1,4 @@ -# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # noqa +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 # cython: profile=False From 117b2b4348bc75a620d36120773fe05492360581 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 6 Aug 2026 16:49:40 -0700 Subject: [PATCH 017/113] Fix work estimate bug in BFRT. And improve work estimates --- cpp/src/dual_simplex/basis_updates.cpp | 2 +- .../bound_flipping_ratio_test.cpp | 5 +- .../bound_flipping_ratio_test.hpp | 2 +- cpp/src/dual_simplex/phase2.cpp | 183 +++++++++++------- 4 files changed, 119 insertions(+), 73 deletions(-) diff --git a/cpp/src/dual_simplex/basis_updates.cpp b/cpp/src/dual_simplex/basis_updates.cpp index a3d3787183..84468ba097 100644 --- a/cpp/src/dual_simplex/basis_updates.cpp +++ b/cpp/src/dual_simplex/basis_updates.cpp @@ -2230,7 +2230,7 @@ i_t basis_update_mpf_t::update(const sparse_vector_t& utilde // Ensure the workspace is sorted. Otherwise, the sparse dot will be incorrect. std::sort(xi_workspace_.begin() + m, xi_workspace_.begin() + m + nz, std::less()); - work_estimate_ += (m + nz) * std::log2(m + nz); + work_estimate_ += nz > 1 ? nz * std::log2(nz) : 0; // Gather the workspace into a column of S i_t S_start; diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index cb0964dc05..3fbfbd1f82 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -229,14 +229,14 @@ void bound_flipping_ratio_test_t::heap_passes(const std::vector& }; std::make_heap(bare_idx.begin(), bare_idx.end(), compare); - work_estimate_ += 3 * bare_idx.size(); + work_estimate_ += 10 * bare_idx.size(); while (bare_idx.size() > 0 && slope > 0) { // Remove minimum ratio from the heap and rebalance i_t heap_index = bare_idx.front(); std::pop_heap(bare_idx.begin(), bare_idx.end(), compare); - work_estimate_ += 2 * std::log2(bare_idx.size()); bare_idx.pop_back(); + work_estimate_ += 7 * std::log2(bare_idx.size() + 1); nonbasic_entering = current_indicies[heap_index]; const i_t j = entering_index = nonbasic_list_[nonbasic_entering]; @@ -264,6 +264,7 @@ void bound_flipping_ratio_test_t::heap_passes(const std::vector& // The variable is not bounded. Stop the search. break; } + work_estimate_ += 10; if (toc(start_time_) > settings_.time_limit) { entering_index = RATIO_TEST_TIME_LIMIT; diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp index 2e73d05eff..2f73069451 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp @@ -100,7 +100,7 @@ class bound_flipping_ratio_test_t { i_t n_; i_t m_; - f_t work_estimate_; + f_t work_estimate_{0.0}; }; } // namespace cuopt::mathematical_optimization::simplex diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index f86aeb0333..6e8ef4bbdd 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -161,7 +161,7 @@ void compute_delta_z(const csr_matrix_t& Arow, } } work_estimate += 4 * nz_delta_y; - work_estimate += 4 * nnz_processed; + work_estimate += 5 * nnz_processed; work_estimate += 2 * delta_z_indices.size(); // delta_zB = sigma*ei @@ -905,7 +905,7 @@ bool update_primal_infeasibilities(const lp_problem_t& lp, primal_inf); if (old_val != 0.0 && squared_infeasibilities[j] == 0.0) { became_feasible = true; } } - work_estimate += 8 * nz; + work_estimate += 9 * nz; return became_feasible; } @@ -1257,6 +1257,7 @@ i_t flip_bounds(const lp_problem_t& lp, num_flipped++; } } + work_estimate += 4 * delta_z_indices.size(); return num_flipped; } @@ -2473,6 +2474,21 @@ void prepare_optimality(i_t info, #endif } +template +struct work_timer_t { + work_timer_t(f_t t) : time(t) {} + f_t time{0.0}; + f_t work{0.0}; +}; + +template +work_timer_t& operator+=(work_timer_t& lhs, const work_timer_t& rhs) +{ + lhs.time += rhs.time; + lhs.work += rhs.work; + return lhs; +} + template class phase2_timers_t { public: @@ -2495,60 +2511,89 @@ class phase2_timers_t { { } - void start_timer() + void start_timer(f_t work) { if (!record_time) { return; } start_time = tic(); + start_work = work; + } + + work_timer_t stop_timer(f_t stop_work) + { + if (!record_time) { return work_timer_t(0.0); } + work_timer_t result(toc(start_time)); + result.work = stop_work - start_work; + return result; } - f_t stop_timer() + void print_one(const simplex_solver_settings_t& settings, + const char* name, + const work_timer_t& t, + f_t total_time, + f_t total_work) const { - if (!record_time) { return 0.0; } - return toc(start_time); + const f_t work_per_sec = t.time > 0.0 ? t.work / t.time : f_t(0); + settings.log.printf("%-15s %.2fs %4.1f%% (%.2e work %4.1f%% %.2e/s)\n", + name, + t.time, + total_time > 0.0 ? 100.0 * t.time / total_time : 0.0, + t.work, + total_work > 0.0 ? 100.0 * t.work / total_work : 0.0, + work_per_sec); } void print_timers(const simplex_solver_settings_t& settings) const { if (!record_time) { return; } - const f_t total_time = bfrt_time + pricing_time + btran_time + ftran_time + flip_time + - delta_z_time + lu_update_time + lu_factorization_time + se_norms_time + - se_entering_time + perturb_time + vector_time + objective_time + - update_infeasibility_time; + const f_t total_time = bfrt_time.time + pricing_time.time + btran_time.time + ftran_time.time + + flip_time.time + delta_z_time.time + lu_update_time.time + + lu_factorization_time.time + se_norms_time.time + se_entering_time.time + + perturb_time.time + vector_time.time + objective_time.time + + update_infeasibility_time.time; + const f_t total_work = bfrt_time.work + pricing_time.work + btran_time.work + ftran_time.work + + flip_time.work + delta_z_time.work + lu_update_time.work + + lu_factorization_time.work + se_norms_time.work + se_entering_time.work + + perturb_time.work + vector_time.work + objective_time.work + + update_infeasibility_time.work; // clang-format off - settings.log.printf("BFRT time %.2fs %4.1f%\n", bfrt_time, 100.0 * bfrt_time / total_time); - settings.log.printf("Pricing time %.2fs %4.1f%\n", pricing_time, 100.0 * pricing_time / total_time); - settings.log.printf("BTran time %.2fs %4.1f%\n", btran_time, 100.0 * btran_time / total_time); - settings.log.printf("FTran time %.2fs %4.1f%\n", ftran_time, 100.0 * ftran_time / total_time); - settings.log.printf("Flip time %.2fs %4.1f%\n", flip_time, 100.0 * flip_time / total_time); - settings.log.printf("Delta_z time %.2fs %4.1f%\n", delta_z_time, 100.0 * delta_z_time / total_time); - settings.log.printf("LU update time %.2fs %4.1f%\n", lu_update_time, 100.0 * lu_update_time / total_time); - settings.log.printf("LU factor time %.2fs %4.1f%\n", lu_factorization_time, 100.0 * lu_factorization_time / total_time); - settings.log.printf("SE norms time %.2fs %4.1f%\n", se_norms_time, 100.0 * se_norms_time / total_time); - settings.log.printf("SE enter time %.2fs %4.1f%\n", se_entering_time, 100.0 * se_entering_time / total_time); - settings.log.printf("Perturb time %.2fs %4.1f%\n", perturb_time, 100.0 * perturb_time / total_time); - settings.log.printf("Vector time %.2fs %4.1f%\n", vector_time, 100.0 * vector_time / total_time); - settings.log.printf("Objective time %.2fs %4.1f%\n", objective_time, 100.0 * objective_time / total_time); - settings.log.printf("Inf update time %.2fs %4.1f%\n", update_infeasibility_time, 100.0 * update_infeasibility_time / total_time); - settings.log.printf("Sum %.2fs\n", total_time); + print_one(settings, "BFRT time", bfrt_time, total_time, total_work); + print_one(settings, "Pricing time", pricing_time, total_time, total_work); + print_one(settings, "BTran time", btran_time, total_time, total_work); + print_one(settings, "FTran time", ftran_time, total_time, total_work); + print_one(settings, "Flip time", flip_time, total_time, total_work); + print_one(settings, "Delta_z time", delta_z_time, total_time, total_work); + print_one(settings, "LU update time", lu_update_time, total_time, total_work); + print_one(settings, "LU factor time", lu_factorization_time, total_time, total_work); + print_one(settings, "SE norms time", se_norms_time, total_time, total_work); + print_one(settings, "SE enter time", se_entering_time, total_time, total_work); + print_one(settings, "Perturb time", perturb_time, total_time, total_work); + print_one(settings, "Vector time", vector_time, total_time, total_work); + print_one(settings, "Objective time", objective_time, total_time, total_work); + print_one(settings, "Inf update time", update_infeasibility_time, total_time, total_work); + settings.log.printf("Sum %.2fs (%.2e work %.2e/s)\n", + total_time, + total_work, + total_time > 0.0 ? total_work / total_time : f_t(0)); // clang-format on } - f_t bfrt_time; - f_t pricing_time; - f_t btran_time; - f_t ftran_time; - f_t flip_time; - f_t delta_z_time; - f_t se_norms_time; - f_t se_entering_time; - f_t lu_update_time; - f_t lu_factorization_time; - f_t perturb_time; - f_t vector_time; - f_t objective_time; - f_t update_infeasibility_time; + work_timer_t bfrt_time; + work_timer_t pricing_time; + work_timer_t btran_time; + work_timer_t ftran_time; + work_timer_t flip_time; + work_timer_t delta_z_time; + work_timer_t se_norms_time; + work_timer_t se_entering_time; + work_timer_t lu_update_time; + work_timer_t lu_factorization_time; + work_timer_t perturb_time; + work_timer_t vector_time; + work_timer_t objective_time; + work_timer_t update_infeasibility_time; private: f_t start_time; + f_t start_work; bool record_time; }; @@ -2901,7 +2946,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t basic_leaving_index = -1; i_t leaving_index = -1; f_t max_val; - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); { PHASE2_NVTX_RANGE("DualSimplex::pricing"); if (settings.use_steepest_edge_pricing) { @@ -2922,7 +2967,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, lp, settings, x, basic_list, direction, basic_leaving_index, primal_infeasibility); } } - timers.pricing_time += timers.stop_timer(); + timers.pricing_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); if (leaving_index == -1) { #ifdef CHECK_BASIS_UPDATE for (i_t k = 0; k < basic_list.size(); k++) { @@ -3065,7 +3110,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, // BTran // BT*delta_y = -delta_zB = -sigma*ei - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); delta_y_sparse.clear(); UTsol_sparse.clear(); f_t btran_start_work = ft.work_estimate(); @@ -3073,7 +3118,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, PHASE2_NVTX_RANGE("DualSimplex::btran"); phase2::compute_delta_y(ft, basic_leaving_index, direction, delta_y_sparse, UTsol_sparse); } - timers.btran_time += timers.stop_timer(); + timers.btran_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); solve_work += (ft.work_estimate() - btran_start_work); if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { @@ -3097,7 +3142,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, continue; } - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); i_t delta_y_nz0 = 0; const i_t nz_delta_y = delta_y_sparse.i.size(); for (i_t k = 0; k < nz_delta_y; k++) { @@ -3136,7 +3181,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate); } } - timers.delta_z_time += timers.stop_timer(); + timers.delta_z_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { return dual_status_t::CONCURRENT_LIMIT; } @@ -3172,7 +3217,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, step_length, nonbasic_entering_index); } else if (bound_flip_ratio) { - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); f_t slope = direction == 1 ? (lp.lower[leaving_index] - x[leaving_index]) : (x[leaving_index] - lp.upper[leaving_index]); bound_flipping_ratio_test_t bfrt(settings, @@ -3195,7 +3240,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, settings.log.printf("Numerical issues encountered in ratio test.\n"); return dual_status_t::NUMERICAL; } - timers.bfrt_time += timers.stop_timer(); + timers.bfrt_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); } else { entering_index = phase2::phase2_ratio_test( lp, settings, vstatus, nonbasic_list, z, delta_z, step_length, nonbasic_entering_index); @@ -3386,7 +3431,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, return dual_status_t::DUAL_UNBOUNDED; } - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // Update dual variables // y <- y + steplength * delta_y // z <- z + steplength * delta_z @@ -3402,7 +3447,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, settings.log.printf("Numerical issues encountered in update_dual_variables.\n"); return dual_status_t::NUMERICAL; } - timers.vector_time += timers.stop_timer(); + timers.vector_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); #ifdef COMPUTE_DUAL_RESIDUAL std::vector dual_res1; @@ -3413,7 +3458,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } #endif - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // Update primal variable const i_t num_flipped = phase2::flip_bounds(lp, settings, @@ -3430,12 +3475,12 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, atilde_index, phase2_work_estimate); - timers.flip_time += timers.stop_timer(); + timers.flip_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); total_bound_flips += num_flipped; delta_xB_0_sparse.clear(); if (num_flipped > 0) { - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); phase2::adjust_for_flips(ft, basic_list, delta_z_indices, @@ -3447,10 +3492,10 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, delta_x_flip, x, phase2_work_estimate); - timers.ftran_time += timers.stop_timer(); + timers.ftran_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); } - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); utilde_sparse.clear(); scaled_delta_xB_sparse.clear(); rhs_sparse.from_csc_column(lp.A, entering_index); @@ -3477,7 +3522,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } } solve_work += (ft.work_estimate() - ftran_start_work); - timers.ftran_time += timers.stop_timer(); + timers.ftran_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { return dual_status_t::CONCURRENT_LIMIT; } @@ -3489,7 +3534,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (primal_step_err > 1e-4) { settings.log.printf("|| A * dx || %e\n", primal_step_err); } #endif - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); f_t se_norms_start_work = ft.work_estimate(); const i_t steepest_edge_status = phase2::update_steepest_edge_norms(settings, basic_list, @@ -3511,18 +3556,18 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } #endif assert(steepest_edge_status == 0); - timers.se_norms_time += timers.stop_timer(); + timers.se_norms_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); solve_work += (ft.work_estimate() - se_norms_start_work); if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { return dual_status_t::CONCURRENT_LIMIT; } - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // x <- x + delta_x phase2::update_primal_variables( scaled_delta_xB_sparse, basic_list, delta_x, entering_index, x, phase2_work_estimate); - timers.vector_time += timers.stop_timer(); + timers.vector_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); #ifdef COMPUTE_PRIMAL_RESIDUAL residual = lp.rhs; @@ -3533,7 +3578,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } #endif - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // TODO(CMM): Do I also need to update the objective due to the bound flips? // TODO(CMM): I'm using the unperturbed objective here, should this be the perturbed objective? phase2::update_objective(basic_list, @@ -3543,9 +3588,9 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, entering_index, obj, phase2_work_estimate); - timers.objective_time += timers.stop_timer(); + timers.objective_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // Update primal infeasibilities due to changes in basic variables // from flipping bounds #ifdef CHECK_BASIC_INFEASIBILITIES @@ -3598,17 +3643,17 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::check_primal_infeasibilities( lp, settings, basic_list, x, squared_infeasibilities, infeasibility_indices); #endif - timers.update_infeasibility_time += timers.stop_timer(); + timers.update_infeasibility_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); // Clear delta_x phase2::clear_delta_x( basic_list, entering_index, scaled_delta_xB_sparse, delta_x, phase2_work_estimate); - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); f_t sum_perturb = 0.0; phase2::compute_perturbation( lp, settings, delta_z_indices, z, objective, sum_perturb, phase2_work_estimate); - timers.perturb_time += timers.stop_timer(); + timers.perturb_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); // Update basis information vstatus[entering_index] = variable_status_t::BASIC; @@ -3631,7 +3676,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::check_basic_infeasibilities(basic_list, basic_mark, infeasibility_indices, 5); #endif - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // Refactor or update the basis factorization { PHASE2_NVTX_RANGE("DualSimplex::basis_update"); @@ -3647,8 +3692,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::check_update(lp, settings, ft, basic_list, basic_leaving_index); #endif should_refactor = recommend_refactor == 1; - timers.lu_update_time += timers.stop_timer(); - timers.start_timer(); + timers.lu_update_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); } #ifdef CHECK_BASIC_INFEASIBILITIES @@ -3726,7 +3771,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::check_basic_infeasibilities(basic_list, basic_mark, infeasibility_indices, 7); #endif } - timers.lu_factorization_time += timers.stop_timer(); + timers.lu_factorization_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); #ifdef STEEPEST_EDGE_DEBUG if (iter < 100 || iter % 100 == 0)) From 99207b22c0da28146cab92ac09d97e4960a5842b Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 6 Aug 2026 19:16:22 -0700 Subject: [PATCH 018/113] Harris ratio test; timers in primal; limit feasibility pump to do less work than root relaxation --- cpp/src/branch_and_bound/branch_and_bound.cpp | 14 +- cpp/src/branch_and_bound/branch_and_bound.hpp | 1 + cpp/src/dual_simplex/primal.cpp | 269 +++++++++++++++--- cpp/src/dual_simplex/primal.hpp | 3 +- 4 files changed, 237 insertions(+), 50 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 99e4c2416d..6471071fbd 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3440,6 +3440,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( simplex_solver_settings_t primal_settings = settings_; primal_settings.log.log = false; primal_settings.time_limit = settings_.time_limit - toc(exploration_stats_.start_time); + primal_settings.work_limit = root_relax_work_estimate_; simplex::primal_status_t lp_status = simplex::primal_phase2_with_advanced_basis(2, exploration_stats_.start_time, @@ -3503,10 +3504,11 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( settings_.log.printf( "Degenerate feasibility pump: Simplex iterations %d, Best number of fractional variables " - "%d/%d. Time %.2f\n", + "%d/%d. Work estimate %.2e, Time %.2f\n", iter, best_num_fractional, num_fractional, + primal_work_estimate, toc(dual_degenerate_feasibility_pump_start_time)); if (best_num_fractional < num_fractional) { // Translate the vstatus from the reduced problem to the vstatus for the original problem @@ -4032,7 +4034,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut solving_root_relaxation_ = true; f_t root_relax_start_time = tic(); - f_t root_relax_work_estimate = 0.0; + root_relax_work_estimate_ = 0.0; if (!enable_concurrent_lp_root_solve()) { // RINS/SUBMIP path settings_.log.printf("\n"); @@ -4046,7 +4048,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut nonbasic_list, root_vstatus_, edge_norms_, - root_relax_work_estimate); + root_relax_work_estimate_); root_relax_solved_by = DualSimplex; exploration_stats_.total_simplex_iters = root_relax_soln_.iterations; @@ -4060,7 +4062,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut basic_list, nonbasic_list, edge_norms_, - root_relax_work_estimate); + root_relax_work_estimate_); } solving_root_relaxation_ = false; @@ -4122,8 +4124,8 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut method_to_string(root_relax_solved_by)); settings_.log.printf("Dual simplex iteration %d work estimate %.2e work per second %.2e\n", root_iterations, - root_relax_work_estimate, - root_relax_work_estimate / root_relax_elapsed_time); + root_relax_work_estimate_, + root_relax_work_estimate_ / root_relax_elapsed_time); settings_.log.printf("Root relaxation objective %+.8e\n\n", root_relax_soln_.user_objective); assert(root_vstatus_.size() == original_lp_.num_cols); diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 17f6f7a3a3..eaf622b1e3 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -251,6 +251,7 @@ class branch_and_bound_t { simplex::lp_solution_t root_relax_soln_; simplex::lp_solution_t root_crossover_soln_; method_t root_relax_solved_by{Unset}; + f_t root_relax_work_estimate_; std::vector edge_norms_; std::atomic root_crossover_solution_set_{false}; omp_atomic_t root_lp_current_lower_bound_; diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 27351a3685..8867c8c9b4 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -18,6 +18,112 @@ namespace cuopt::mathematical_optimization::simplex { +template +struct primal_work_timer_t { + primal_work_timer_t(f_t t) : time(t) {} + f_t time{0.0}; + f_t work{0.0}; +}; + +template +primal_work_timer_t& operator+=(primal_work_timer_t& lhs, + const primal_work_timer_t& rhs) +{ + lhs.time += rhs.time; + lhs.work += rhs.work; + return lhs; +} + +template +class primal_timers_t { + public: + primal_timers_t(bool should_time) + : record_time(should_time), + pricing_time(0), + ftran_time(0), + ratio_test_time(0), + btran_time(0), + delta_z_time(0), + update_duals_time(0), + lu_update_time(0), + lu_factorization_time(0), + update_x_time(0) + { + } + + void start_timer(f_t work) + { + if (!record_time) { return; } + start_time_ = tic(); + start_work_ = work; + } + + primal_work_timer_t stop_timer(f_t stop_work) + { + if (!record_time) { return primal_work_timer_t(0.0); } + primal_work_timer_t result(toc(start_time_)); + result.work = stop_work - start_work_; + return result; + } + + void print_one(const simplex_solver_settings_t& settings, + const char* name, + const primal_work_timer_t& t, + f_t total_time, + f_t total_work) const + { + const f_t work_per_sec = t.time > 0.0 ? t.work / t.time : f_t(0); + settings.log.printf("%-15s %.2fs %4.1f%% (%.2e work %4.1f%% %.2e/s)\n", + name, + t.time, + total_time > 0.0 ? 100.0 * t.time / total_time : 0.0, + t.work, + total_work > 0.0 ? 100.0 * t.work / total_work : 0.0, + work_per_sec); + } + + void print_timers(const simplex_solver_settings_t& settings) const + { + if (!record_time) { return; } + const f_t total_time = pricing_time.time + ftran_time.time + ratio_test_time.time + + btran_time.time + delta_z_time.time + update_duals_time.time + + lu_update_time.time + lu_factorization_time.time + update_x_time.time; + const f_t total_work = pricing_time.work + ftran_time.work + ratio_test_time.work + + btran_time.work + delta_z_time.work + update_duals_time.work + + lu_update_time.work + lu_factorization_time.work + update_x_time.work; + // clang-format off + print_one(settings, "Pricing time", pricing_time, total_time, total_work); + print_one(settings, "FTran time", ftran_time, total_time, total_work); + print_one(settings, "Ratio test", ratio_test_time, total_time, total_work); + print_one(settings, "BTran time", btran_time, total_time, total_work); + print_one(settings, "Delta_z time", delta_z_time, total_time, total_work); + print_one(settings, "Update duals", update_duals_time, total_time, total_work); + print_one(settings, "LU update time", lu_update_time, total_time, total_work); + print_one(settings, "LU factor time", lu_factorization_time, total_time, total_work); + print_one(settings, "Update x time", update_x_time, total_time, total_work); + settings.log.printf("Sum %.2fs (%.2e work %.2e/s)\n", + total_time, + total_work, + total_time > 0.0 ? total_work / total_time : f_t(0)); + // clang-format on + } + + primal_work_timer_t pricing_time; + primal_work_timer_t ftran_time; + primal_work_timer_t ratio_test_time; + primal_work_timer_t btran_time; + primal_work_timer_t delta_z_time; + primal_work_timer_t update_duals_time; + primal_work_timer_t lu_update_time; + primal_work_timer_t lu_factorization_time; + primal_work_timer_t update_x_time; + + private: + f_t start_time_; + f_t start_work_; + bool record_time; +}; + namespace { template @@ -156,7 +262,7 @@ i_t phase2_pricing(const lp_problem_t& lp, } } } - work_estimate += 4 * (n - m); + work_estimate += 5 * (n - m); return entering_index; } @@ -281,7 +387,7 @@ void compute_delta_z(const csr_matrix_t& Arow, const i_t j = Arow.j[p]; if (vstatus[j] != variable_status_t::BASIC) { delta_z[j] -= Arow.x[p] * delta_y_i; } } - work_estimate += 4 * (row_end - row_start); + work_estimate += 5 * (row_end - row_start); } work_estimate += 4 * delta_y.i.size(); } @@ -422,104 +528,147 @@ i_t primal_ratio_test(const lp_problem_t& lp, f_t& work_estimate) { const i_t m = lp.num_rows; - const i_t n = lp.num_cols; basic_leaving = -1; i_t leaving_index = -1; - f_t min_val = inf; - f_t current_dx = 0.0; constexpr f_t pivot_tol = 1e-8; + constexpr f_t harris_tol = 1e-8; + + // Harris ratio test: two passes. + // Pass 1: find the maximum step length alpha_1 such that no variable + // moves more than harris_tol past its bound. + // Pass 2: among all candidates with ratio <= alpha_1, pick the one + // with the largest pivot (|delta_x[j]|). + + f_t alpha_1 = inf; // Entering variable can hit its opposite bound: limit step by that if (direction > 0 && lp.upper[entering_index] < inf) { const f_t limit = lp.upper[entering_index] - x[entering_index]; - if (limit >= 0 && limit < min_val) { - min_val = limit; - leaving_index = -1; // no basic leaves; will be handled by caller + if (limit >= 0 && limit < alpha_1) { alpha_1 = limit; } + } else if (direction < 0 && lp.lower[entering_index] > -inf) { + const f_t limit = x[entering_index] - lp.lower[entering_index]; + if (limit >= 0 && limit < alpha_1) { alpha_1 = limit; } + } + + // Pass 1: compute alpha_1 (Harris step) + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + if (std::abs(delta_x[j]) <= pivot_tol) { continue; } + + // Already below lower and moving back up: stop exactly at the bound. + // No harris tolerance here — these variables are already infeasible + // and must not overshoot their bound (needed for Phase I correctness). + if (x[j] < lp.lower[j] && delta_x[j] > pivot_tol && lp.lower[j] > -inf) { + const f_t ratio = (lp.lower[j] - x[j]) / delta_x[j]; + if (ratio >= 0 && ratio < alpha_1) { alpha_1 = ratio; } + } + // Already above upper and moving back down: stop exactly at the bound. + if (x[j] > lp.upper[j] && delta_x[j] < -pivot_tol && lp.upper[j] < inf) { + const f_t ratio = (lp.upper[j] - x[j]) / delta_x[j]; + if (ratio >= 0 && ratio < alpha_1) { alpha_1 = ratio; } + } + + if (lp.lower[j] > -inf && delta_x[j] < -pivot_tol) { + // xj + step * delta_x[j] >= lp.lower[j] - harris_tol + f_t neum = lp.lower[j] - x[j] - harris_tol; + if (neum > 0 && neum <= settings.primal_tol) { neum = 0.0; } + f_t ratio = neum / delta_x[j]; + if (ratio >= 0 && ratio < alpha_1) { alpha_1 = ratio; } + } + if (lp.upper[j] < inf && delta_x[j] > pivot_tol) { + // xj + step * delta_x[j] <= lp.upper[j] + harris_tol + f_t neum = lp.upper[j] - x[j] + harris_tol; + if (neum < 0 && -neum <= settings.primal_tol) { neum = 0.0; } + f_t ratio = neum / delta_x[j]; + if (ratio >= 0 && ratio < alpha_1) { alpha_1 = ratio; } + } + } + + // Pass 2: among candidates with exact ratio <= alpha_1, pick largest pivot + f_t best_pivot = 0.0; + step_length = alpha_1; + + // Check entering variable bound (no pivot selection needed — it's fixed at direction) + if (direction > 0 && lp.upper[entering_index] < inf) { + const f_t limit = lp.upper[entering_index] - x[entering_index]; + if (limit >= 0 && limit <= alpha_1) { + // Entering hits its own bound — this is always pivot = 1.0 effectively + step_length = limit; + leaving_index = -1; basic_leaving = -1; + best_pivot = inf; // Always prefer this if it's within alpha_1 } } else if (direction < 0 && lp.lower[entering_index] > -inf) { const f_t limit = x[entering_index] - lp.lower[entering_index]; - if (limit >= 0 && limit < min_val) { - min_val = limit; + if (limit >= 0 && limit <= alpha_1) { + step_length = limit; leaving_index = -1; basic_leaving = -1; + best_pivot = inf; } } for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; - if (delta_x[j] == 0.0) { continue; } + if (std::abs(delta_x[j]) <= pivot_tol) { continue; } + + const f_t abs_dx = std::abs(delta_x[j]); // Already below lower and moving back up: stop when we reach the lower bound. // Without this, phase I can take an unbounded step (false unbounded) or skip the // breakpoint of the piecewise phase-I objective and stall still infeasible. if (x[j] < lp.lower[j] && delta_x[j] > pivot_tol && lp.lower[j] > -inf) { const f_t ratio = (lp.lower[j] - x[j]) / delta_x[j]; - if (ratio >= 0 && ratio < min_val) { - min_val = ratio; + if (ratio >= 0 && ratio <= alpha_1 && abs_dx > best_pivot) { + best_pivot = abs_dx; + step_length = ratio; basic_leaving = k; leaving_index = j; - current_dx = delta_x[j]; } } - // Already above upper and moving back down: stop when we reach the upper bound. + // Already above upper and moving back down if (x[j] > lp.upper[j] && delta_x[j] < -pivot_tol && lp.upper[j] < inf) { const f_t ratio = (lp.upper[j] - x[j]) / delta_x[j]; - if (ratio >= 0 && ratio < min_val) { - min_val = ratio; + if (ratio >= 0 && ratio <= alpha_1 && abs_dx > best_pivot) { + best_pivot = abs_dx; + step_length = ratio; basic_leaving = k; leaving_index = j; - current_dx = -delta_x[j]; } } if (lp.lower[j] > -inf && delta_x[j] < -pivot_tol) { // xj + step * delta_x[j] >= lp.lower[j] - // step * delta_x[j] >= lp.lower[j] - x[j] // step <= (lp.lower[j] - x[j]) / delta_x[j], delta_x[j] < 0 f_t neum = lp.lower[j] - x[j]; // A basic sitting below its bound (within the primal tolerance) is on - // the bound numerically, but gives a tiny negative ratio. Dropping it lets - // the step run straight through the bound, so treat it as a zero-length - // block. A genuine violation is left to the branches above, which stop at - // the bound when the variable moves back toward it. + // the bound numerically. Treat it as a zero-length block. if (neum > 0 && neum <= settings.primal_tol) { neum = 0.0; } f_t ratio = neum / delta_x[j]; - if (ratio >= 0 && ratio < min_val) { - min_val = ratio; + if (ratio >= 0 && ratio <= alpha_1 && abs_dx > best_pivot) { + best_pivot = abs_dx; + step_length = ratio; basic_leaving = k; leaving_index = j; - current_dx = -delta_x[j]; - } else if (ratio >= 0 && ratio < min_val + 1e-9 && -delta_x[j] > current_dx) { - min_val = ratio; - basic_leaving = k; - leaving_index = j; - current_dx = -delta_x[j]; } } if (lp.upper[j] < inf && delta_x[j] > pivot_tol) { // xj + step * delta_x[j] <= lp.upper[j] - // step * delta_x[j] <= lp.upper[j] - x[j] // step <= (lp.upper[j] - x[j]) / delta_x[j], delta_x[j] > 0 f_t neum = lp.upper[j] - x[j]; - // Mirror of the lower bound case: slightly above the bound is considered on the bound. + // Mirror of the lower bound case: slightly above the bound is on the bound. if (neum < 0 && -neum <= settings.primal_tol) { neum = 0.0; } f_t ratio = neum / delta_x[j]; - if (ratio >= 0 && ratio < min_val) { - min_val = ratio; - basic_leaving = k; - leaving_index = j; - current_dx = delta_x[j]; - } else if (ratio >= 0 && ratio < min_val + 1e-9 && delta_x[j] > current_dx) { - min_val = ratio; + if (ratio >= 0 && ratio <= alpha_1 && abs_dx > best_pivot) { + best_pivot = abs_dx; + step_length = ratio; basic_leaving = k; leaving_index = j; - current_dx = delta_x[j]; } } } + work_estimate += 10 * m; - step_length = min_val; return leaving_index; } @@ -763,7 +912,14 @@ primal_status_t primal_phase2_with_advanced_basis( work_estimate += basis_update.work_estimate(); basis_update.clear_work_estimate(); + if (work_estimate > settings.work_limit) { + return primal_status_t::WORK_LIMIT; + } + + primal_timers_t timers(false); + while (iter < iter_limit) { + timers.start_timer(work_estimate + basis_update.work_estimate()); i_t nonbasic_entering = -1; i_t direction; i_t entering_index = phase2_pricing(lp, @@ -775,6 +931,7 @@ primal_status_t primal_phase2_with_advanced_basis( nonbasic_entering, dual_inf, work_estimate); + timers.pricing_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); if (entering_index == -1) { if (phase == 2) { // Verify optimality with a consistent basic solution: refactor, put @@ -883,6 +1040,7 @@ primal_status_t primal_phase2_with_advanced_basis( settings.log.printf("Primal residual ||Ax-b||: %.2e\n", primal_constraint_residual(lp, x)); } + timers.print_timers(settings); return primal_status_t::OPTIMAL; } else { primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); @@ -968,6 +1126,7 @@ primal_status_t primal_phase2_with_advanced_basis( work_estimate += 3 * rhs_sparse.i.size(); sparse_vector_t scaled_delta_xB_sparse(m, 0); sparse_vector_t utilde_sparse(m, 0); + timers.start_timer(work_estimate + basis_update.work_estimate()); basis_update.b_solve(rhs_sparse, scaled_delta_xB_sparse, utilde_sparse); std::vector scaled_delta_xB(m); scaled_delta_xB_sparse.to_dense(scaled_delta_xB); @@ -984,6 +1143,7 @@ primal_status_t primal_phase2_with_advanced_basis( } work_estimate += 2 * (n - m); delta_x[entering_index] = direction; + timers.ftran_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); #ifdef CHECK_NULLSPACE std::vector residual(m, 0.0); @@ -997,6 +1157,7 @@ primal_status_t primal_phase2_with_advanced_basis( } #endif + timers.start_timer(work_estimate + basis_update.work_estimate()); i_t basic_leaving; f_t step_length; i_t leaving_index = primal_ratio_test(lp, @@ -1010,6 +1171,7 @@ primal_status_t primal_phase2_with_advanced_basis( entering_index, direction, work_estimate); + timers.ratio_test_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); if (leaving_index == -1 && step_length >= inf) { settings.log.printf("No leaving variable. Primal unbounded?\n"); return primal_status_t::PRIMAL_UNBOUNDED; @@ -1017,10 +1179,12 @@ primal_status_t primal_phase2_with_advanced_basis( const bool basis_updated = (leaving_index != -1); bool recompute_duals = false; + timers.start_timer(work_estimate + basis_update.work_estimate()); for (i_t j = 0; j < n; ++j) { x[j] += step_length * delta_x[j]; } work_estimate += 2 * n; + timers.update_x_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); #ifdef COMPUTE_RESIDUAL f_t debug_primal_residual = primal_constraint_residual(lp, x); @@ -1038,7 +1202,9 @@ primal_status_t primal_phase2_with_advanced_basis( bool should_refactor = basis_update.num_updates() > settings.refactor_frequency; f_t dual_step_length = 0.0; if (!should_refactor) { + timers.start_timer(work_estimate + basis_update.work_estimate()); compute_delta_y(basis_update, basic_leaving, delta_y, etilde); + timers.btran_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); const f_t pivot = scaled_delta_xB[basic_leaving]; dual_step_length = compute_dual_step_length(z[entering_index], pivot); } @@ -1076,12 +1242,19 @@ primal_status_t primal_phase2_with_advanced_basis( x[leaving_index] = leave_bound; if (!should_refactor) { + timers.start_timer(work_estimate + basis_update.work_estimate()); compute_delta_z(Arow, vstatus, delta_y, delta_z, work_estimate); + timers.delta_z_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); + timers.start_timer(work_estimate + basis_update.work_estimate()); update_y(dual_step_length, delta_y, y, work_estimate); update_z(dual_step_length, nonbasic_list, entering_index, delta_z, z, work_estimate); + timers.update_duals_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); + timers.start_timer(work_estimate + basis_update.work_estimate()); should_refactor = basis_update.update(utilde_sparse, etilde, basic_leaving) == 1; + timers.lu_update_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); } if (should_refactor) { + timers.start_timer(work_estimate + basis_update.work_estimate()); i_t rank = basis_update.refactor_basis( lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); if (rank == CONCURRENT_HALT_RETURN) { return primal_status_t::CONCURRENT_LIMIT; } @@ -1097,6 +1270,8 @@ primal_status_t primal_phase2_with_advanced_basis( set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); compute_basic_primal_variables( lp, basis_update, basic_list, nonbasic_list, x, work_estimate); + timers.lu_factorization_time += + timers.stop_timer(work_estimate + basis_update.work_estimate()); } else if (rebuild_x_after_bound_snap) { // FT update already matches the new basis; recompute x_B with the leaving variable // snapped onto its bound. @@ -1172,9 +1347,17 @@ primal_status_t primal_phase2_with_advanced_basis( work_estimate += basis_update.work_estimate(); basis_update.clear_work_estimate(); - if (now > settings.time_limit) { return primal_status_t::TIME_LIMIT; } + if (now > settings.time_limit) { + timers.print_timers(settings); + return primal_status_t::TIME_LIMIT; + } + if (work_estimate > settings.work_limit) { + timers.print_timers(settings); + return primal_status_t::WORK_LIMIT; + } } + timers.print_timers(settings); if (iter >= iter_limit) { return primal_status_t::ITERATION_LIMIT; } return primal_status_t::NUMERICAL; diff --git a/cpp/src/dual_simplex/primal.hpp b/cpp/src/dual_simplex/primal.hpp index 7e4d280655..fc47d90368 100644 --- a/cpp/src/dual_simplex/primal.hpp +++ b/cpp/src/dual_simplex/primal.hpp @@ -26,7 +26,8 @@ enum class primal_status_t { TIME_LIMIT = 5, ITERATION_LIMIT = 6, CONCURRENT_LIMIT = 7, - NOT_LOADED = 8 + WORK_LIMIT = 8, + NOT_LOADED = 9 }; template From eb47c20cf8c6ddf69bfca7056869e2e17e90669d Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 7 Aug 2026 19:53:26 -0700 Subject: [PATCH 019/113] Add reduced cost bounds table. Change objective in feasibility pump. Fix work limit in feasibility pump. Check for reduced cost violation before calling primal simplex. Check if we reduced number of integer infeasibilites when we hit a work limit --- cpp/src/branch_and_bound/branch_and_bound.cpp | 268 +++++++++++++++--- cpp/src/branch_and_bound/branch_and_bound.hpp | 154 +++++++++- 2 files changed, 386 insertions(+), 36 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 6471071fbd..a41bc1bd95 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -445,6 +445,69 @@ void branch_and_bound_t::report( settings_.log.printf("%s\n", log_line.c_str()); } + +template +void branch_and_bound_t::update_reduced_cost_bounds( + f_t relaxation_objective, + const std::vector& reduced_costs, + const std::vector& var_status, + reduced_cost_bounds_t& reduced_cost_bounds) +{ + const i_t n = reduced_cost_bounds.num_cols(); + const f_t threshold = 100.0 * settings_.integer_tol; + const f_t tol = 1e-2; + for (i_t j = 0; j < n; ++j) { + if (std::isfinite(reduced_costs[j]) && std::abs(reduced_costs[j]) > threshold && + var_status[j] != variable_status_t::BASIC) { + const f_t lower_j = original_lp_.lower[j]; + const f_t upper_j = original_lp_.upper[j]; + + // x_j <= l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] + // Let u_tilde_j = u_j - epsilon, so that floor(u_tilde_j) = u_j - 1 + // We want to solve for want the incumbent objective needs to be to make + // x_j <= u_tilde_j + // This means l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] <= u_tilde_j + // Or equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * (u_tilde_j - l_j) when reduced_costs[j] > 0 + if (lower_j > -inf && reduced_costs[j] > 0) { + const f_t u_tilde_j = var_types_[j] == variable_type_t::INTEGER + ? upper_j - tol + : std::max(upper_j - 1.0, lower_j); + const f_t bound_j = + var_types_[j] == variable_type_t::INTEGER ? std::floor(u_tilde_j) : u_tilde_j; + const f_t diff = u_tilde_j - lower_j; + const f_t objective_j = relaxation_objective + diff * reduced_costs[j]; + if (((var_types_[j] == variable_type_t::INTEGER && bound_j == upper_j - 1.0) || + var_types_[j] != variable_type_t::INTEGER) && std::isfinite(objective_j) && std::isfinite(bound_j)) { + i_t info = reduced_cost_bounds.add_upper_bound(j, objective_j, bound_j); + //settings_.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. Info %d\n", objective_j, bound_j, j, info); + } + } + + // x_j >= u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] when reduced_costs[j] < 0 + // Let l_tilde_j = l_j + epsilon, so that ceil(l_tilde_j) = l_j + 1 + // We want to solve for want the incumbent objective needs to be to make + // x_j >= l_tilde_j + // This means u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j + // Or equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * (l_tilde_j - u_j) when reduced_costs[j] < 0 + if (upper_j < inf && reduced_costs[j] < 0) { + const f_t l_tilde_j = var_types_[j] == variable_type_t::INTEGER + ? lower_j + tol + : std::min(lower_j + 1.0, upper_j); + const f_t bound_j = + var_types_[j] == variable_type_t::INTEGER ? std::ceil(l_tilde_j) : l_tilde_j; + const f_t diff = l_tilde_j - upper_j; + const f_t objective_j = relaxation_objective + diff * reduced_costs[j]; + if (((var_types_[j] == variable_type_t::INTEGER && bound_j == lower_j + 1.0) || + var_types_[j] != variable_type_t::INTEGER) && std::isfinite(objective_j) && std::isfinite(bound_j)) { + i_t info = reduced_cost_bounds.add_lower_bound(j, objective_j, bound_j); + //settings_.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. Info %d\n", objective_j, bound_j, j, info); + } + } + } + } +} + + template i_t branch_and_bound_t::find_reduced_cost_fixings(f_t upper_bound, std::vector& lower_bounds, @@ -2946,7 +3009,7 @@ lp_status_t branch_and_bound_t::solve_root_relaxation( } template -auto branch_and_bound_t::do_cut_pass( +typename branch_and_bound_t::cut_pass_result_t branch_and_bound_t::do_cut_pass( [[maybe_unused]] i_t cut_pass, mip_solution_t& solution, i_t& num_fractional, @@ -2963,8 +3026,9 @@ auto branch_and_bound_t::do_cut_pass( f_t& last_upper_bound, f_t& last_objective, f_t root_relax_objective, + reduced_cost_bounds_t& reduced_cost_bounds, i_t& cut_pool_size, - [[maybe_unused]] const std::vector& saved_solution) -> cut_pass_result_t + [[maybe_unused]] const std::vector& saved_solution) { #ifdef PRINT_FRACTIONAL_INFO settings_.log.printf("Found %d fractional variables on cut pass %d\n", num_fractional, cut_pass); @@ -3006,6 +3070,28 @@ auto branch_and_bound_t::do_cut_pass( if (cut_generation_time > 1.0) { settings_.log.debug("Cut generation time %.2f seconds\n", cut_generation_time); } + + num_fractional = fractional_variables(settings_, root_relax_soln_.x, var_types_, fractional); + + f_t pivot_out_integer_variables_start_time = tic(); + i_t num_integer_increased = pivot_out_integer_variables(original_lp_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional); + settings_.log.printf("Pivoted out %d integer variables in %e seconds\n", num_integer_increased, toc(pivot_out_integer_variables_start_time)); + dual_degenerate_feasibility_pump(original_lp_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional); + // Score the cuts f_t score_start_time = tic(); cut_pool.score_cuts(root_relax_soln_.x); @@ -3073,14 +3159,20 @@ auto branch_and_bound_t::do_cut_pass( if (settings_.reduced_cost_strengthening >= 1 && upper_bound_.load() < last_upper_bound) { mutex_upper_.lock(); last_upper_bound = upper_bound_.load(); - std::vector lower_bounds; - std::vector upper_bounds; - find_reduced_cost_fixings(upper_bound_.load(), lower_bounds, upper_bounds); + std::vector lower_bounds = original_lp_.lower; + std::vector upper_bounds = original_lp_.upper; + f_t previous_max_objective = reduced_cost_bounds.get_max_objective(); + i_t new_bounds = reduced_cost_bounds.update_bounds_from_new_incumbent( + upper_bound_.load(), var_types_, lower_bounds, upper_bounds); mutex_upper_.unlock(); mutex_original_lp_.lock(); original_lp_.lower = lower_bounds; original_lp_.upper = upper_bounds; mutex_original_lp_.unlock(); + if (1 || new_bounds > 0) { + settings_.log.printf( + "Updated %d bounds using reduced cost strengthening from new incumbent. Max objective %e Current objective %e Previous max objective %e\n", new_bounds, reduced_cost_bounds.get_max_objective(), upper_bound_.load(), previous_max_objective); + } } // Try to do bound strengthening @@ -3178,27 +3270,13 @@ auto branch_and_bound_t::do_cut_pass( } root_objective_ = compute_objective(original_lp_, root_relax_soln_.x); + if (settings_.reduced_cost_strengthening >= 1) { + update_reduced_cost_bounds(root_objective_, root_relax_soln_.z, root_vstatus_, reduced_cost_bounds); + } + // Refresh fractional info after re-solving with cuts; the pre-cut count is stale. num_fractional = fractional_variables(settings_, root_relax_soln_.x, var_types_, fractional); - pivot_out_integer_variables(original_lp_, - basic_list, - nonbasic_list, - root_vstatus_, - root_relax_soln_, - basis_update, - num_fractional, - fractional); - - dual_degenerate_feasibility_pump(original_lp_, - basic_list, - nonbasic_list, - root_vstatus_, - root_relax_soln_, - basis_update, - num_fractional, - fractional); - if (settings_.benchmark_info_ptr != nullptr) { settings_.benchmark_info_ptr->root_lp_with_cuts = compute_user_objective(original_lp_, root_objective_); @@ -3395,6 +3473,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( } simplex::basis_update_mpf_t reduced_basis_update = basis_update; + reduced_basis_update.clear_work_estimate(); for (i_t k = 0; k < m; k++) { reduced_basic_list[k] = original_col_to_reduced_col[basic_list[k]]; } @@ -3426,15 +3505,76 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( lp_reduced.objective[reduced_col] = -1; } } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_LOWER) { - lp_reduced.objective[reduced_col] = 1; + lp_reduced.objective[reduced_col] = 0.1; } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_UPPER) { - lp_reduced.objective[reduced_col] = -1; + lp_reduced.objective[reduced_col] = -0.1; } } reduced_col++; } } + // Check reduced costs before calling primal simplex. + // Compute y = B^{-T} * c_B (BTRAN with the pump objective on basic variables) + std::vector c_basic_pump(m, 0.0); + for (i_t k = 0; k < m; k++) { + c_basic_pump[k] = lp_reduced.objective[reduced_basic_list[k]]; + } + std::vector y_pump(m); + reduced_basis_update.b_transpose_solve(c_basic_pump, y_pump); + + // Check if any nonbasic has a violated reduced cost + i_t num_violated = 0; + f_t max_violation = 0.0; + const i_t num_nonbasics_reduced = reduced_nonbasic_list.size(); + for (i_t k = 0; k < num_nonbasics_reduced; k++) { + const i_t j = reduced_nonbasic_list[k]; + // z[j] = c[j] - y^T * A(:,j) + f_t zj = lp_reduced.objective[j]; + const i_t col_start = A_reduced.col_start[j]; + const i_t col_end = A_reduced.col_start[j + 1]; + for (i_t p = col_start; p < col_end; p++) { + zj -= y_pump[A_reduced.i[p]] * A_reduced.x[p]; + } + // Check pricing condition + bool violated = false; + if (reduced_vstatus[j] == variable_status_t::NONBASIC_LOWER || + reduced_vstatus[j] == variable_status_t::NONBASIC_FIXED) { + if (zj < -settings_.dual_tol) { violated = true; } + } else if (reduced_vstatus[j] == variable_status_t::NONBASIC_UPPER) { + if (zj > settings_.dual_tol) { violated = true; } + } + if (violated) { + num_violated++; + max_violation = std::max(max_violation, std::abs(zj)); + } + } + + if (num_violated == 0) { + settings_.log.printf( + "Degenerate feasibility pump (%d/%d): skipping primal simplex, no violated reduced costs " + "(%d nonbasics checked)\n", + pump_iter, + max_pump_iter, + num_nonbasics_reduced); + primal_work_estimate += reduced_basis_update.work_estimate(); + reduced_basis_update.clear_work_estimate(); + // Don't count this as a pump iteration, but break if we've skipped twice + // in a row (perturbation isn't helping) + if (stalled) { break; } + stalled = true; + pump_iter--; + continue; + } + settings_.log.printf( + "Degenerate feasibility pump (%d/%d): %d violated reduced costs (max %.2e) out of %d " + "nonbasics\n", + pump_iter, + max_pump_iter, + num_violated, + max_violation, + num_nonbasics_reduced); + bool recompute_basis = false; const i_t iter_before = iter; simplex_solver_settings_t primal_settings = settings_; @@ -3498,18 +3638,59 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( best_reduced_vstatus = reduced_vstatus; } } else { + settings_.log.printf( + "Degenerate feasibility pump: primal simplex returned non-optimal status %d at pump_iter " + "%d. Work estimate %.2e\n", + static_cast(lp_status), + pump_iter, + primal_work_estimate); + // Even if we hit work/time limit, the solution may have improved. + // Check fractional count before breaking. + if (lp_status == simplex::primal_status_t::WORK_LIMIT || + lp_status == simplex::primal_status_t::TIME_LIMIT) { + std::vector adjusted_solution(lp.num_cols, 0.0); + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || + std::abs(soln.z[j]) <= settings_.tight_tol) { + adjusted_solution[j] = reduced_solution.x[reduced_col++]; + } else { + adjusted_solution[j] = soln.x[j]; + } + } + std::vector residual = lp.rhs; + matrix_vector_multiply(lp.A, 1.0, adjusted_solution, -1.0, residual); + const f_t primal_residual = vector_norm_inf(residual); + if (primal_residual <= 1e-6) { + std::vector tmp_fractional; + i_t num_fractional_reduced = + fractional_variables(settings_, adjusted_solution, var_types_, tmp_fractional); + settings_.log.printf( + "Degenerate feasibility pump (%d/%d): after work/time limit, fractional " + "variables %d/%d\n", + pump_iter, + max_pump_iter, + num_fractional_reduced, + num_fractional); + if (num_fractional_reduced < best_num_fractional) { + best_num_fractional = num_fractional_reduced; + best_reduced_vstatus = reduced_vstatus; + } + } + } break; } } settings_.log.printf( "Degenerate feasibility pump: Simplex iterations %d, Best number of fractional variables " - "%d/%d. Work estimate %.2e, Time %.2f\n", + "%d/%d. Work estimate %.2e, Time %.2f, Basis updates %d\n", iter, best_num_fractional, num_fractional, primal_work_estimate, - toc(dual_degenerate_feasibility_pump_start_time)); + toc(dual_degenerate_feasibility_pump_start_time), + reduced_basis_update.num_updates()); if (best_num_fractional < num_fractional) { // Translate the vstatus from the reduced problem to the vstatus for the original problem i_t reduced_cols = 0; @@ -3690,7 +3871,7 @@ void branch_and_bound_t::apply_delta_x_for_integer_pivot( } template -void branch_and_bound_t::pivot_out_integer_variables( +i_t branch_and_bound_t::pivot_out_integer_variables( const simplex::lp_problem_t& lp, std::vector& basic_list, std::vector& nonbasic_list, @@ -3700,13 +3881,13 @@ void branch_and_bound_t::pivot_out_integer_variables( i_t& num_fractional, std::vector& fractional) { - if (num_fractional == 0) { return; } + if (num_fractional == 0) { return 0; } f_t pivot_out_integer_variables_start_time = tic(); std::vector zero_reduced_costs_vars; std::vector zero_reduced_costs_vars_nonbasic_index; bool dual_degenerate = check_for_dual_degeneracy( solution, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); - if (!dual_degenerate) { return; } + if (!dual_degenerate) { return 0; } lp_solution_t soln_copy = solution; std::vector basic_list_copy = basic_list; @@ -3937,11 +4118,13 @@ void branch_and_bound_t::pivot_out_integer_variables( if (num_new_fractional < start_num_fractional) { i_t num_integer_increased = start_num_fractional - num_new_fractional; integer_pivots_.fetch_add(num_integer_increased, std::memory_order_release); +#if 0 settings_.log.printf("Pivoted out %d integer variables: %d -> %d in %.2f\n", num_integer_increased, start_num_fractional, num_new_fractional, toc(pivot_out_integer_variables_start_time)); +#endif num_fractional = num_new_fractional; fractional = new_fractional; basic_list = basic_list_copy; @@ -3949,7 +4132,9 @@ void branch_and_bound_t::pivot_out_integer_variables( vstatus = vstatus_copy; basis_update = basis_update_copy; solution = soln_copy; + return num_integer_increased; } + return 0; } template @@ -4168,7 +4353,11 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut is_running_ = true; lower_bound_numerical_ = inf; - pivot_out_integer_variables(original_lp_, + reduced_cost_bounds_t reduced_cost_bounds(original_lp_.num_cols); + update_reduced_cost_bounds(root_objective_, root_relax_soln_.z, root_vstatus_, reduced_cost_bounds); + + f_t pivot_out_integer_variables_start_time = tic(); + i_t num_integer_increased = pivot_out_integer_variables(original_lp_, basic_list, nonbasic_list, root_vstatus_, @@ -4176,6 +4365,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut basis_update, num_fractional, fractional); + settings_.log.printf("Pivoted out %d integer variables in %e seconds\n", num_integer_increased, toc(pivot_out_integer_variables_start_time)); dual_degenerate_feasibility_pump(original_lp_, basic_list, @@ -4274,6 +4464,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut last_upper_bound, last_objective, root_relax_objective, + reduced_cost_bounds, cut_pool_size, saved_solution); root_fj_cpu_worker.stop(); @@ -4390,10 +4581,17 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut } if (settings_.reduced_cost_strengthening >= 2 && upper_bound_.load() < last_upper_bound) { - std::vector lower_bounds; - std::vector upper_bounds; - i_t num_fixed = find_reduced_cost_fixings(upper_bound_.load(), lower_bounds, upper_bounds); - if (num_fixed > 0) { + std::vector lower_bounds = original_lp_.lower; + std::vector upper_bounds = original_lp_.upper; + f_t previous_max_objective = reduced_cost_bounds.get_max_objective(); + i_t num_changed = reduced_cost_bounds.update_bounds_from_new_incumbent( + upper_bound_.load(), var_types_, lower_bounds, upper_bounds); + settings_.log.printf("Updated %d bounds using reduced cost strengthening from new incumbent. Max objective %e Current objective %e Previous max objective %e\n", num_changed, reduced_cost_bounds.get_max_objective(), upper_bound_.load(), previous_max_objective); + mutex_original_lp_.lock(); + original_lp_.lower = lower_bounds; + original_lp_.upper = upper_bounds; + mutex_original_lp_.unlock(); + if (num_changed > 0) { std::vector bounds_changed(original_lp_.num_cols, true); std::vector row_sense; diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index eaf622b1e3..185e3ce79e 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -40,6 +40,7 @@ #include #include #include +#include #include #include @@ -90,6 +91,152 @@ struct deterministic_bfs_policy_t; template struct deterministic_diving_policy_t; +template +struct objective_bound_pair_t { + objective_bound_pair_t() + : objective(std::numeric_limits::quiet_NaN()), + bound(std::numeric_limits::quiet_NaN()) + { + } + objective_bound_pair_t(f_t objective_in, f_t bound_in) + : objective(objective_in), bound(bound_in) + { + } + bool is_valid() { return objective == objective && bound == bound; } + f_t objective; + f_t bound; +}; + +template +class reduced_cost_bounds_t { + public: + reduced_cost_bounds_t(i_t original_cols) + : max_objective_(-std::numeric_limits::infinity()), lower_bounds_(original_cols), upper_bounds_(original_cols) + { + } + + i_t add_lower_bound(i_t col, f_t objective, f_t bound) + { + if (col < static_cast(lower_bounds_.size())) { + if (!lower_bounds_[col].is_valid()) { + lower_bounds_[col] = objective_bound_pair_t(objective, bound); + if (objective > max_objective_) { + max_objective_ = objective; + } + return 1; + } else { + if (bound > lower_bounds_[col].bound) { + lower_bounds_[col] = objective_bound_pair_t(objective, bound); + if (objective > max_objective_) { + max_objective_ = objective; + } + return 2; + } else if (bound == lower_bounds_[col].bound && objective > lower_bounds_[col].objective) { + lower_bounds_[col].objective = objective; + if (objective > max_objective_) { + max_objective_ = objective; + } + return 1; + } else { + return -2; + } + } + } else { + return -1; + } + } + + i_t add_upper_bound(i_t col, f_t objective, f_t bound) + { + if (col < static_cast(upper_bounds_.size())) { + if (!upper_bounds_[col].is_valid()) { + upper_bounds_[col] = objective_bound_pair_t(objective, bound); + if (objective > max_objective_) { + max_objective_ = objective; + } + return 1; + } else { + if (bound < upper_bounds_[col].bound) { + upper_bounds_[col] = objective_bound_pair_t(objective, bound); + if (objective > max_objective_) { + max_objective_ = objective; + } + return 2; + } else if (bound == upper_bounds_[col].bound && objective > upper_bounds_[col].objective) { + upper_bounds_[col].objective = objective; + if (objective > max_objective_) { + max_objective_ = objective; + } + return 1; + } else { + return -2; + } + } + } else { + return -1; + } + } + + i_t update_bounds_from_new_incumbent(f_t incumbent_objective, + const std::vector& var_types, + std::vector& lower_bounds, + std::vector& upper_bounds) + { + const i_t n = static_cast(lower_bounds_.size()); + f_t max_objective = -std::numeric_limits::infinity(); + i_t bounds_updated = 0; + for (i_t j = 0; j < n; ++j) { + if (lower_bounds_[j].is_valid()) { + if (incumbent_objective <= lower_bounds_[j].objective && + lower_bounds_[j].bound > lower_bounds[j]) { + //printf("RCF Variable %d (%d): lower %e -> %e\n", j, static_cast(var_types[j]), lower_bounds[j], lower_bounds_[j].bound); + lower_bounds[j] = lower_bounds_[j].bound; + bounds_updated++; + lower_bounds_[j].bound = lower_bounds_[j].objective = std::numeric_limits::quiet_NaN(); + } + if (lower_bounds_[j].objective > max_objective) { + max_objective = lower_bounds_[j].objective; + } + } + if (upper_bounds_[j].is_valid()) { + if (incumbent_objective <= upper_bounds_[j].objective && + upper_bounds_[j].bound < upper_bounds[j]) { + //printf("RCF Variable %d (%d): upper %e -> %e\n", j, static_cast(var_types[j]), upper_bounds[j], upper_bounds_[j].bound); + upper_bounds[j] = upper_bounds_[j].bound; + bounds_updated++; + upper_bounds_[j].bound = upper_bounds_[j].objective = std::numeric_limits::quiet_NaN(); + } + if (upper_bounds_[j].objective > max_objective) { + max_objective = upper_bounds_[j].objective; + } + } + } + max_objective_ = max_objective; + return bounds_updated; + } + + f_t get_current_lower_bound(i_t col) + { + if (col < static_cast(lower_bounds_.size())) { return lower_bounds_[col].bound; } + return std::numeric_limits::quiet_NaN(); + } + + f_t get_current_upper_bound(i_t col) + { + if (col < static_cast(upper_bounds_.size())) { return upper_bounds_[col].bound; } + return std::numeric_limits::quiet_NaN(); + } + + i_t num_cols() { return static_cast(lower_bounds_.size()); } + + f_t get_max_objective() { return max_objective_; } + + private: + f_t max_objective_; + std::vector> lower_bounds_; + std::vector> upper_bounds_; +}; + template class branch_and_bound_t { public: @@ -176,6 +323,10 @@ class branch_and_bound_t { std::vector& edge_norms, f_t& work_estimate); + void update_reduced_cost_bounds(f_t relaxation_objective, + const std::vector& reduced_costs, + const std::vector& var_status, + reduced_cost_bounds_t& reduced_cost_bounds); i_t find_reduced_cost_fixings(f_t upper_bound, std::vector& lower_bounds, std::vector& upper_bounds); @@ -324,6 +475,7 @@ class branch_and_bound_t { f_t& last_upper_bound, f_t& last_objective, f_t root_relax_objective, + reduced_cost_bounds_t& reduced_cost_bounds, i_t& cut_pool_size, const std::vector& saved_solution); @@ -348,7 +500,7 @@ class branch_and_bound_t { std::vector& zero_reduced_costs_vars, std::vector& zero_reduced_costs_vars_nonbasic_index); - void pivot_out_integer_variables(const simplex::lp_problem_t& lp, + i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, std::vector& basic_list, std::vector& nonbasic_list, std::vector& vstatus, From 7641ad3523d25c2ef498cc6a589845fe472ac243 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 12 Aug 2026 16:04:48 -0700 Subject: [PATCH 020/113] V3 of pump and pivots, better work estimates, also exploit primal degeneracy --- cpp/src/branch_and_bound/branch_and_bound.cpp | 727 +++++++++++++++--- cpp/src/branch_and_bound/branch_and_bound.hpp | 50 +- cpp/src/dual_simplex/basis_updates.cpp | 6 +- cpp/src/dual_simplex/phase2.cpp | 30 +- cpp/src/dual_simplex/right_looking_lu.cpp | 21 +- 5 files changed, 706 insertions(+), 128 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index a41bc1bd95..b842ddfb63 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3171,7 +3171,7 @@ typename branch_and_bound_t::cut_pass_result_t branch_and_bound_t 0) { settings_.log.printf( - "Updated %d bounds using reduced cost strengthening from new incumbent. Max objective %e Current objective %e Previous max objective %e\n", new_bounds, reduced_cost_bounds.get_max_objective(), upper_bound_.load(), previous_max_objective); + "Updated %d integer bounds using reduced cost strengthening from new incumbent. Max objective %e Current objective %e Previous max objective %e\n", new_bounds, reduced_cost_bounds.get_max_objective(), upper_bound_.load(), previous_max_objective); } } @@ -3272,6 +3272,16 @@ typename branch_and_bound_t::cut_pass_result_t branch_and_bound_t= 1) { update_reduced_cost_bounds(root_objective_, root_relax_soln_.z, root_vstatus_, reduced_cost_bounds); + settings_.log.printf("New reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); + pivot_to_improve_reduced_cost_strengthening(original_lp_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional, root_objective_, reduced_cost_bounds); + settings_.log.printf("After pivoting: new reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); } // Refresh fractional info after re-solving with cuts; the pre-cut count is stale. @@ -3577,10 +3587,17 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( bool recompute_basis = false; const i_t iter_before = iter; + f_t primal_work_before = primal_work_estimate; + f_t pump_call_start_time = tic(); simplex_solver_settings_t primal_settings = settings_; primal_settings.log.log = false; - primal_settings.time_limit = settings_.time_limit - toc(exploration_stats_.start_time); - primal_settings.work_limit = root_relax_work_estimate_; + primal_settings.time_limit = settings_.time_limit; + primal_settings.work_limit = root_relax_work_estimate_ / 10; + settings_.log.printf( + "Degenerate feasibility pump: calling primal simplex with %d rows, %d cols, %d nnz, " + "%d basis updates, work_limit %.2e, primal_work_estimate %.2e\n", + m, n, A_reduced.col_start[n], reduced_basis_update.num_updates(), + primal_settings.work_limit, primal_work_estimate); simplex::primal_status_t lp_status = simplex::primal_phase2_with_advanced_basis(2, exploration_stats_.start_time, @@ -3593,6 +3610,16 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( reduced_solution, iter, primal_work_estimate); + f_t pump_call_time = toc(pump_call_start_time); + f_t pump_call_work = primal_work_estimate - primal_work_before; + i_t pump_call_iters = iter - iter_before; + settings_.log.printf( + "Degenerate feasibility pump: primal simplex returned status %d, %d iters, " + "work %.2e (%.2e/iter), time %.2f (%.2e work/s)\n", + static_cast(lp_status), pump_call_iters, pump_call_work, + pump_call_iters > 0 ? pump_call_work / pump_call_iters : 0.0, + pump_call_time, + pump_call_time > 0 ? pump_call_work / pump_call_time : 0.0); // Detect a stall: the solve made no pivots, so the incumbent vertex was // already optimal for this objective and x did not move. Perturb next pass. stalled = (iter == iter_before); @@ -3768,7 +3795,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( } template -void branch_and_bound_t::apply_delta_x_for_integer_pivot( +i_t branch_and_bound_t::apply_delta_x_for_integer_pivot( const simplex::lp_problem_t& lp, std::vector& basic_list, std::vector& nonbasic_list, @@ -3799,7 +3826,15 @@ void branch_and_bound_t::apply_delta_x_for_integer_pivot( bool binding_integer = leaving_index != -1 && is_fractional(solution.x[leaving_index], var_types_[leaving_index], settings_.integer_tol); - if (!binding_integer) { return; } + if (!binding_integer) { + if (leaving_index == -1) { + return -4; // unbounded or entering hit its own bound + } else if (var_types_[leaving_index] != variable_type_t::INTEGER) { + return -5; // continuous variable won ratio test + } else { + return -6; // integer variable won but it's not fractional (already at integer value) + } + } std::vector test_x = solution.x; i_t integer_destroyed = 0; @@ -3815,7 +3850,7 @@ void branch_and_bound_t::apply_delta_x_for_integer_pivot( } } // Require a strict net decrease in fractional integers. - if (integer_destroyed >= 0) { return; } + if (integer_destroyed >= 0) { return -2; } solution.x = test_x; basic_list[basic_leaving] = entering_index; @@ -3863,51 +3898,29 @@ void branch_and_bound_t::apply_delta_x_for_integer_pivot( deficient, slacks_needed, factorize_work_estimate); - if (rank == CONCURRENT_HALT_RETURN || rank == TIME_LIMIT_RETURN) { return; } - if (rank < 0 || rank != lp.num_rows) { return; } + if (rank == CONCURRENT_HALT_RETURN || rank == TIME_LIMIT_RETURN) { return -3; } + if (rank < 0 || rank != lp.num_rows) { return -3; } simplex::reorder_basic_list(q, basic_list); basis_update.reset(L, U, p); } + + return 0; } template -i_t branch_and_bound_t::pivot_out_integer_variables( +void branch_and_bound_t::fast_slack_integer_pivot( const simplex::lp_problem_t& lp, + const std::vector& fractional, + const std::vector& row_to_slack, + const simplex::lp_solution_t& solution, std::vector& basic_list, std::vector& nonbasic_list, + std::vector& nonbasic_index, std::vector& vstatus, - simplex::lp_solution_t& solution, + simplex::lp_solution_t& soln, simplex::basis_update_mpf_t& basis_update, - i_t& num_fractional, - std::vector& fractional) + f_t& work_estimate) { - if (num_fractional == 0) { return 0; } - f_t pivot_out_integer_variables_start_time = tic(); - std::vector zero_reduced_costs_vars; - std::vector zero_reduced_costs_vars_nonbasic_index; - bool dual_degenerate = check_for_dual_degeneracy( - solution, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); - if (!dual_degenerate) { return 0; } - - lp_solution_t soln_copy = solution; - std::vector basic_list_copy = basic_list; - std::vector nonbasic_list_copy = nonbasic_list; - std::vector vstatus_copy = vstatus; - simplex::basis_update_mpf_t basis_update_copy = basis_update; - - const i_t start_num_fractional = num_fractional; - - const i_t num_zero_reduced_costs_vars = zero_reduced_costs_vars.size(); - - std::vector row_to_slack(lp.num_rows, -1); - for (i_t j : new_slacks_) { - if (lp.lower[j] != 0 || lp.upper[j] != inf) { continue; } - const i_t p = lp.A.col_start[j]; - row_to_slack[lp.A.i[p]] = j; - } - - f_t work_estimate = 0.0; - std::vector fast_candidates; std::vector fast_rows; std::vector fast_nonbasic_slacks; @@ -3923,7 +3936,7 @@ i_t branch_and_bound_t::pivot_out_integer_variables( const i_t i = lp.A.i[p]; const i_t slack = row_to_slack[i]; if (slack >= 0) { - if (vstatus_copy[slack] == variable_status_t::BASIC) { + if (vstatus[slack] == variable_status_t::BASIC) { num_basic_slacks++; } else if (std::abs(solution.z[slack]) <= 1e-10) { num_nonbasic_slacks_with_reduced_cost_zero++; @@ -3944,24 +3957,26 @@ i_t branch_and_bound_t::pivot_out_integer_variables( fast_candidates.size()); } - // Build a reverse index nonbasic_index[v] = position of v in nonbasic_list_copy, or -1 if not + // Build a reverse index nonbasic_index[v] = position of v in nonbasic_list, or -1 if not // present. Used to locate the entering variable's slot in the fast-candidate path. // apply_delta_x_for_integer_pivot keeps this index consistent by applying an O(1) fix-up // on each successful pivot; the two variables whose (non)basic status changes are the only // entries that need to be updated. - std::vector nonbasic_index(lp.num_cols, -1); - for (i_t p = 0; p < static_cast(nonbasic_list_copy.size()); ++p) { - nonbasic_index[nonbasic_list_copy[p]] = p; + nonbasic_index.assign(lp.num_cols, -1); + for (i_t p = 0; p < static_cast(nonbasic_list.size()); ++p) { + nonbasic_index[nonbasic_list[p]] = p; } const i_t num_candidates = fast_candidates.size(); + f_t last_log = tic(); + f_t loop_start = tic(); for (i_t k = 0; k < num_candidates; k++) { const i_t j = fast_candidates[k]; const i_t row = fast_rows[k]; const i_t nonbasic_slack = fast_nonbasic_slacks[k]; // Skip if state changed by a prior successful pivot. - if (vstatus_copy[j] != variable_status_t::BASIC) { continue; } - if (vstatus_copy[nonbasic_slack] == variable_status_t::BASIC) { continue; } + if (vstatus[j] != variable_status_t::BASIC) { continue; } + if (vstatus[nonbasic_slack] == variable_status_t::BASIC) { continue; } const i_t col_start = lp.A.col_start[j]; const i_t col_end = lp.A.col_start[j + 1]; f_t a_ij = 0.0; @@ -3975,7 +3990,7 @@ i_t branch_and_bound_t::pivot_out_integer_variables( f_t bound = a_ij > 0 ? lp.lower[j] : lp.upper[j]; if (std::abs(bound) == inf) { continue; } - const f_t delta_xj = bound - soln_copy.x[j]; + const f_t delta_xj = bound - soln.x[j]; const f_t scale = -delta_xj * a_ij; if (std::abs(scale) <= 1e-12) { continue; } @@ -4007,7 +4022,7 @@ i_t branch_and_bound_t::pivot_out_integer_variables( const i_t jj = delta_x_sparse.i[h]; if (jj == j) continue; const f_t val = delta_x_sparse.x[h]; - const f_t slack_value = soln_copy.x[jj]; + const f_t slack_value = soln.x[jj]; if (val < -slack_value) { ok = false; break; @@ -4038,80 +4053,377 @@ i_t branch_and_bound_t::pivot_out_integer_variables( // live in L), so u_multiply is a single sparse matvec against U0. std::vector b_inv_abar(lp.num_rows); for (i_t h = 0; h < lp.num_rows; ++h) { - b_inv_abar[h] = -direction * delta_x[basic_list_copy[h]]; + b_inv_abar[h] = -direction * delta_x[basic_list[h]]; } std::vector utilde_dense; - basis_update_copy.u_multiply(b_inv_abar, utilde_dense); + basis_update.u_multiply(b_inv_abar, utilde_dense); sparse_vector_t utilde_sparse; utilde_sparse.from_dense(utilde_dense); - apply_delta_x_for_integer_pivot(lp, - basic_list_copy, - nonbasic_list_copy, + i_t error = apply_delta_x_for_integer_pivot(lp, + basic_list, + nonbasic_list, nonbasic_index, - vstatus_copy, + vstatus, entering_index, nonbasic_entering, direction, delta_x, utilde_sparse, - soln_copy, - basis_update_copy, + soln, + basis_update, work_estimate); // apply_delta_x_for_integer_pivot only mutates vstatus when the pivot actually fires, // so entering_index transitioning to BASIC is a reliable success signal. - if (vstatus_copy[entering_index] == variable_status_t::BASIC) { + if (!error) { settings_.log.printf( "Fast candidate pivot succeeded: j=%d entering slack=%d row=%d\n", j, entering_index, row); } + + if (toc(last_log) > 1.0) { + settings_.log.printf("Fast candidates %d/%d processed in %.2f seconds\n", k + 1, num_candidates, toc(loop_start)); + last_log = tic(); + } } + settings_.log.printf("Fast candidates: %d/%d processed in %.2f seconds\n", num_candidates, num_candidates, toc(loop_start)); +} - for (i_t k = 0; k < num_zero_reduced_costs_vars; k++) { - const i_t j = zero_reduced_costs_vars[k]; - if (var_types_[j] == variable_type_t::INTEGER) { continue; } - if (vstatus_copy[j] == variable_status_t::BASIC) { continue; } +template +i_t branch_and_bound_t::pivot_out_integer_variables( + const simplex::lp_problem_t& lp, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + i_t& num_fractional, + std::vector& fractional) +{ + if (num_fractional == 0) { return 0; } + f_t pivot_out_integer_variables_start_time = tic(); + std::vector zero_reduced_costs_vars; + std::vector zero_reduced_costs_vars_nonbasic_index; + bool dual_degenerate = check_for_dual_degeneracy( + solution, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); + if (!dual_degenerate) { return 0; } - const i_t direction = (vstatus_copy[j] == variable_status_t::NONBASIC_LOWER || - vstatus_copy[j] == variable_status_t::NONBASIC_FIXED) - ? 1 - : -1; - const i_t entering_index = j; - const i_t nonbasic_entering = zero_reduced_costs_vars_nonbasic_index[k]; - if (nonbasic_entering < 0 || nonbasic_entering >= static_cast(nonbasic_list_copy.size()) || - nonbasic_list_copy[nonbasic_entering] != j) { - continue; + lp_solution_t soln_copy = solution; + std::vector basic_list_copy = basic_list; + std::vector nonbasic_list_copy = nonbasic_list; + std::vector vstatus_copy = vstatus; + simplex::basis_update_mpf_t basis_update_copy = basis_update; + + const i_t start_num_fractional = num_fractional; + + const i_t num_zero_reduced_costs_vars = zero_reduced_costs_vars.size(); + + std::vector row_to_slack(lp.num_rows, -1); + for (i_t j : new_slacks_) { + if (lp.lower[j] != 0 || lp.upper[j] != inf) { continue; } + const i_t p = lp.A.col_start[j]; + row_to_slack[lp.A.i[p]] = j; + } + + f_t work_estimate = 0.0; + + // Count primal degenerate basic variables + i_t num_degenerate = 0; + i_t num_degenerate_continuous = 0; + i_t num_degenerate_integer = 0; + for (i_t k = 0; k < lp.num_rows; k++) { + const i_t j = basic_list_copy[k]; + const f_t slack_to_lower = soln_copy.x[j] - lp.lower[j]; + const f_t slack_to_upper = lp.upper[j] - soln_copy.x[j]; + if (slack_to_lower <= settings_.primal_tol || slack_to_upper <= settings_.primal_tol) { + num_degenerate++; + if (var_types_[j] == variable_type_t::INTEGER) { + num_degenerate_integer++; + } else { + num_degenerate_continuous++; + } } + } + const f_t degeneracy_fraction = static_cast(num_degenerate) / lp.num_rows; + settings_.log.printf( + "Primal degeneracy: %d/%d basic variables are degenerate (%.1f%%), " + "continuous=%d, integer=%d\n", + num_degenerate, lp.num_rows, + 100.0 * degeneracy_fraction, + num_degenerate_continuous, num_degenerate_integer); + + // Skip pivot_out entirely if primal degeneracy is too high — the ratio test + // will almost always be won by a degenerate variable, making pivots hopeless. + if (degeneracy_fraction > 0.5) { + settings_.log.printf("Skipping pivot_out_integer_variables: degeneracy %.1f%% > 50%%\n", + 100.0 * degeneracy_fraction); + return 0; + } + + std::vector nonbasic_index; + fast_slack_integer_pivot(lp, + fractional, + row_to_slack, + solution, + basic_list_copy, + nonbasic_list_copy, + nonbasic_index, + vstatus_copy, + soln_copy, + basis_update_copy, + work_estimate); + + + std::vector work_list = fractional; + std::vector to_basic_position(lp.num_cols, -1); - // Solve B * dxB = A(:, j) so utilde is valid for the MPF update. - // Apply direction when forming delta_x (same convention as primal_phase2). - sparse_vector_t rhs(lp.A, j); - sparse_vector_t delta_xB; - sparse_vector_t utilde_sparse; - basis_update_copy.b_solve(rhs, delta_xB, utilde_sparse); + for (i_t k = 0; k < lp.num_rows; k++) { + to_basic_position[basic_list_copy[k]] = k; + } + + sparse_vector_t ep; + ep.n = lp.num_rows; + ep.i.resize(1); + ep.x.resize(1); + ep.x[0] = 1.0; + + std::vector delta_y_dense(lp.num_rows, 0.0); + + // Track which entering variables are actually tried (to detect duplication) + std::vector entering_tried_count(lp.num_cols, 0); + + i_t worklist_total_processed = 0; + i_t worklist_skipped = 0; + i_t worklist_btran_done = 0; + i_t worklist_ftran_done = 0; + i_t worklist_pivots_succeeded = 0; + i_t worklist_readded = 0; + f_t worklist_btran_time = 0.0; + f_t worklist_dot_time = 0.0; + f_t worklist_ftran_time = 0.0; + i_t worklist_no_candidates = 0; // target had no nonzero dot_q + i_t worklist_ratio_test_fail = 0; // ratio test didn't pick a fractional integer (error -1) + i_t worklist_net_increase_fail = 0; // pivot would net-increase fractionals (error -2) + i_t worklist_unbounded = 0; // entering hit its own bound or unbounded (error -4) + i_t worklist_continuous_won = 0; // continuous variable won ratio test (error -5) + i_t worklist_nonfrac_int_won = 0; // non-fractional integer won ratio test (error -6) + + f_t worklist_loop_start = tic(); + f_t worklist_last_log = tic(); + + while (!work_list.empty()) { + const i_t j = work_list.back(); + const i_t p = to_basic_position[j]; + work_list.pop_back(); + worklist_total_processed++; + + // Skip if j is no longer basic and fractional (may have been fixed by a prior pivot) + if (p < 0) { worklist_skipped++; continue; } + if (vstatus_copy[j] != variable_status_t::BASIC) { worklist_skipped++; continue; } + if (!is_fractional(soln_copy.x[j], var_types_[j], settings_.integer_tol)) { worklist_skipped++; continue; } + + // We want to pivot variable j out of the basis. + // We solve B^T * delta_y = e_p, where p is the position of j in the basis. + // Or delta_y = B^{-T} e_p, or delta_y^T = e_p^T B^{-T} + + ep.i[0] = p; + sparse_vector_t delta_y_sparse; + sparse_vector_t UTsol_sparse; + f_t btran_start = tic(); + basis_update_copy.b_transpose_solve(ep, delta_y_sparse, UTsol_sparse); + worklist_btran_time += toc(btran_start); + worklist_btran_done++; + + // Scatter delta_y_sparse into dense workspace for dot product computation + const i_t delta_y_nz = delta_y_sparse.i.size(); + for (i_t h = 0; h < delta_y_nz; h++) { + delta_y_dense[delta_y_sparse.i[h]] = delta_y_sparse.x[h]; + } + + // We also have that + // B*delta_xB + N*delta_xN = 0 + // So delta_xB = -B^{-1} N * delta_xN + // And delta_xB[p] = e_p^T * delta_xB = -e_p^T B^{-1} N * delta_xN + // = -delta_y^T N * delta_xN + // Recall that delta_xN = e_q where q is the entering variables + // So delta_xB[p] = -delta_y^T A(:, q) + // + // For p to be the leaving variable, we need it to be the binding + // member in the ratio test + // x_B + alpha * delta_xB >= l_B + // x_B + alpha * delta_xB <= u_B + // + // Or alpha <= (l_B[p] - x_B[p]) / delta_xB[p] when delta_xB[p] < 0 + // Or alpha <= (u_B[p] - x_B[p]) / delta_xB[p] when delta_xB[p] > 0 + // + // Thus, if we want to push x_B[p] up to u_B[p], we want + // alpha = (u_B[p] - x_B[p]) / delta_xB[p] to be small + // And if we want to push x_B[p] down to l_B[p], we want + // alpha = (l_B[p] - x_B[p]) / delta_xB[p] to be small + // + // Or equivalently, we want delta_xB[p] to be large + + // Find top 3 candidates by merit = |dot_q| / nnz(A(:,q)) + // Large |dot_q| means the target moves a lot (small step to hit bound). + // Small nnz means the FTRAN result is likely sparse, so fewer competing + // basic variables will have nonzero delta_xB components to block the target. + // Skip entering variables that have already been tried (and failed) by prior targets. + f_t values[3] = {0.0, 0.0, 0.0}; + i_t indices[3] = {-1, -1, -1}; + f_t dot_start = tic(); + for (i_t q : zero_reduced_costs_vars) { + if (var_types_[q] == variable_type_t::INTEGER) { continue; } + if (nonbasic_index[q] < 0) { continue; } + if (entering_tried_count[q] > 0) { continue; } + // Compute dot_q = delta_y^T * A(:, q) using dense delta_y + const i_t col_start = lp.A.col_start[q]; + const i_t col_end = lp.A.col_start[q + 1]; + const i_t col_nnz = col_end - col_start; + f_t dot_q = 0.0; + for (i_t pp = col_start; pp < col_end; pp++) { + dot_q += delta_y_dense[lp.A.i[pp]] * lp.A.x[pp]; + } + const f_t abs_dot_q = std::abs(dot_q); + if (abs_dot_q <= 1e-12) { continue; } + const f_t merit = abs_dot_q / static_cast(col_nnz); + + if (merit > values[0]) { + indices[2] = indices[1]; values[2] = values[1]; + indices[1] = indices[0]; values[1] = values[0]; + indices[0] = q; values[0] = merit; + } else if (merit > values[1]) { + indices[2] = indices[1]; values[2] = values[1]; + indices[1] = q; values[1] = merit; + } else if (merit > values[2]) { + indices[2] = q; values[2] = merit; + } + } + worklist_dot_time += toc(dot_start); + + if (indices[0] == -1) { worklist_no_candidates++; } + + // Try the top 3 candidates + for (i_t h = 0; h < 3; h++) { + if (indices[h] == -1) break; + + const i_t q = indices[h]; + const i_t entering_index = q; + const i_t nonbasic_entering = nonbasic_index[q]; + if (nonbasic_entering < 0) { continue; } + entering_tried_count[q]++; + + // Determine direction based on entering variable's status + const i_t direction = (vstatus_copy[q] == variable_status_t::NONBASIC_LOWER || + vstatus_copy[q] == variable_status_t::NONBASIC_FIXED) + ? 1 + : -1; + + // Solve B * delta_xB = A(:, q) so utilde is valid for the MPF update. + sparse_vector_t rhs(lp.A, q); + sparse_vector_t delta_xB; + sparse_vector_t utilde_sparse; + f_t ftran_start = tic(); + basis_update_copy.b_solve(rhs, delta_xB, utilde_sparse); + worklist_ftran_time += toc(ftran_start); + worklist_ftran_done++; + + std::vector delta_xB_dense; + delta_xB.to_dense(delta_xB_dense); + std::vector delta_x(lp.num_cols, 0.0); + for (i_t i = 0; i < lp.num_rows; i++) { + delta_x[basic_list_copy[i]] = -direction * delta_xB_dense[i]; + } + delta_x[q] = direction; + + i_t error = apply_delta_x_for_integer_pivot(lp, + basic_list_copy, + nonbasic_list_copy, + nonbasic_index, + vstatus_copy, + entering_index, + nonbasic_entering, + direction, + delta_x, + utilde_sparse, + soln_copy, + basis_update_copy, + work_estimate); + + if (error == -2) { worklist_net_increase_fail++; } + if (error == -4) { worklist_unbounded++; worklist_ratio_test_fail++; } + if (error == -5) { worklist_continuous_won++; worklist_ratio_test_fail++; } + if (error == -6) { worklist_nonfrac_int_won++; worklist_ratio_test_fail++; } + + if (!error) { + worklist_pivots_succeeded++; + // Update to_basic_position for the variables that changed status + // entering_index is now basic, leaving_index is now nonbasic + // Find the leaving variable: it's the one that took entering_index's slot in nonbasic_list + const i_t leaving_index = nonbasic_list_copy[nonbasic_entering]; + to_basic_position[entering_index] = to_basic_position[leaving_index]; + to_basic_position[leaving_index] = -1; + + // We did a successful pivot; add fractional variables whose values changed to work list + for (i_t k : fractional) { + if (vstatus_copy[k] != variable_status_t::BASIC) { continue; } + if (std::abs(delta_x[k]) > settings_.zero_tol) { + //work_list.push_back(k); + //worklist_readded++; + } + } + break; + } + } - std::vector delta_xB_dense; - delta_xB.to_dense(delta_xB_dense); - std::vector delta_x(lp.num_cols, 0.0); - for (i_t h = 0; h < static_cast(basic_list_copy.size()); h++) { - delta_x[basic_list_copy[h]] = -direction * delta_xB_dense[h]; + // Clear dense workspace for next target + for (i_t h = 0; h < delta_y_nz; h++) { + delta_y_dense[delta_y_sparse.i[h]] = 0.0; } - delta_x[j] = direction; - apply_delta_x_for_integer_pivot(lp, - basic_list_copy, - nonbasic_list_copy, - nonbasic_index, - vstatus_copy, - entering_index, - nonbasic_entering, - direction, - delta_x, - utilde_sparse, - soln_copy, - basis_update_copy, - work_estimate); + if (toc(worklist_last_log) > 1.0) { + settings_.log.printf( + "Worklist progress: %d/%d processed, %d pivots, %d ratio_fail (unb=%d cont=%d nfint=%d), %d net_inc_fail, %d no_cand, %.2f seconds\n", + worklist_total_processed, static_cast(fractional.size()), + worklist_pivots_succeeded, worklist_ratio_test_fail, + worklist_unbounded, worklist_continuous_won, worklist_nonfrac_int_won, + worklist_net_increase_fail, + worklist_no_candidates, toc(worklist_loop_start)); + worklist_last_log = tic(); + } } + // Count unique entering variables and duplication + i_t unique_entering = 0; + i_t max_entering_count = 0; + i_t entering_tried_once = 0; + i_t entering_tried_multiple = 0; + for (i_t q = 0; q < lp.num_cols; q++) { + if (entering_tried_count[q] > 0) { + unique_entering++; + max_entering_count = std::max(max_entering_count, entering_tried_count[q]); + if (entering_tried_count[q] == 1) { entering_tried_once++; } + else { entering_tried_multiple++; } + } + } + settings_.log.printf( + "Worklist entering stats: unique=%d, tried_once=%d, tried_multiple=%d, " + "max_count=%d, total_ftran=%d, duplication_ratio=%.1fx\n", + unique_entering, entering_tried_once, entering_tried_multiple, + max_entering_count, worklist_ftran_done, + worklist_ftran_done / std::max(1.0, static_cast(unique_entering))); + + settings_.log.printf( + "Worklist stats: processed=%d skipped=%d btran=%d ftran=%d pivots=%d readded=%d " + "btran_time=%.2f dot_time=%.2f ftran_time=%.2f zero_rc_vars=%d " + "no_candidates=%d ratio_test_fail=%d (unbounded=%d continuous_won=%d nonfrac_int_won=%d) " + "net_increase_fail=%d\n", + worklist_total_processed, worklist_skipped, worklist_btran_done, worklist_ftran_done, + worklist_pivots_succeeded, worklist_readded, + worklist_btran_time, worklist_dot_time, worklist_ftran_time, + num_zero_reduced_costs_vars, + worklist_no_candidates, worklist_ratio_test_fail, + worklist_unbounded, worklist_continuous_won, worklist_nonfrac_int_won, + worklist_net_increase_fail); + std::vector new_fractional; const i_t num_new_fractional = fractional_variables(settings_, soln_copy.x, var_types_, new_fractional); @@ -4137,6 +4449,218 @@ i_t branch_and_bound_t::pivot_out_integer_variables( return 0; } +template +void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( + const simplex::lp_problem_t& lp, + const std::vector& basic_list, + const std::vector& nonbasic_list, + const std::vector& vstatus, + const simplex::lp_solution_t& soln, + const simplex::basis_update_mpf_t& basis_update, + const i_t num_fractional, + const std::vector& fractional, + const f_t relaxation_objective, + reduced_cost_bounds_t& reduced_cost_bounds) +{ + // Count primal degenerate basic variables + i_t num_degenerate = 0; + i_t num_degenerate_continuous = 0; + i_t num_degenerate_integer = 0; + std::vector degenerate_integer_list; + degenerate_integer_list.reserve(lp.num_rows); + for (i_t k = 0; k < lp.num_rows; k++) { + const i_t j = basic_list[k]; + const f_t slack_to_lower = soln.x[j] - lp.lower[j]; + const f_t slack_to_upper = lp.upper[j] - soln.x[j]; + if (slack_to_lower <= settings_.primal_tol || slack_to_upper <= settings_.primal_tol) { + num_degenerate++; + if (var_types_[j] == variable_type_t::INTEGER) { + num_degenerate_integer++; + degenerate_integer_list.push_back(j); + } else { + num_degenerate_continuous++; + } + } + } + + if (num_degenerate_integer == 0) return; + + std::vector variable_to_basic_position(lp.num_cols, -1); + for (i_t k = 0; k < lp.num_rows; k++) { + variable_to_basic_position[basic_list[k]] = k; + } + std::vector delta_y(lp.num_rows, 0); + std::vector delta_z(lp.num_cols, 0); + std::vector delta_z_mark(lp.num_cols, 0); + std::vector delta_z_indices; + delta_z_indices.reserve(lp.num_cols); + + f_t work_estimate = 0; + const f_t threshold = 100.0 * settings_.integer_tol; + const f_t tol = 1e-2; + const f_t pivot_tol = settings_.pivot_tol; + const f_t dual_tol = settings_.dual_tol / 10; + + i_t num_bounds_added = 0; + for (i_t j : degenerate_integer_list) { + // x_j is a degenerate integer basic variable. + // We would like a dual-feasible point where x_j is nonbasic with a nonzero + // reduced cost that may be used for reduced cost strengthening. + // We do not need to take the pivot; a dual step along either ray is enough. + // + // One BTRAN: B^T * delta_y = e_p, which matches direction == -1 in + // compute_reduced_cost_update (B^T * delta_y = -direction * e_p). + // The opposite direction is the negated (delta_y, delta_z) ray. + const i_t leaving_index = j; + const i_t p = variable_to_basic_position[j]; + if (p == -1) continue; + + sparse_vector_t ep(lp.num_rows, 1); + ep.i[0] = p; + ep.x[0] = 1.0; + sparse_vector_t delta_y_sparse; + sparse_vector_t UTsol_sparse; + basis_update.b_transpose_solve(ep, delta_y_sparse, UTsol_sparse); + + // delta_zN = -N^T * delta_y, delta_z[leaving] = -1 + delta_y_sparse.to_dense(delta_y); + simplex::compute_reduced_cost_update(lp, + basic_list, + nonbasic_list, + delta_y, + leaving_index, + /*direction=*/-1, + delta_z_mark, + delta_z_indices, + delta_z, + work_estimate); + + const f_t lower_j = lp.lower[j]; + const f_t upper_j = lp.upper[j]; + + // Try both dual rays. scale == +1 uses the computed delta_z (direction -1); + // scale == -1 uses -delta_z (direction +1). Either or both may yield an RCS bound. + for (const f_t scale : {1.0, -1.0}) { + // Maximum dual step-length alpha that keeps dual feasibility on this ray. + // zl_j + alpha * delta_zN_j >= 0 for nonbasic j on lower bound + // zu_j + alpha * delta_zN_j <= 0 for nonbasic j on upper bound + f_t alpha = inf; + for (i_t jj : delta_z_indices) { + if (vstatus[jj] == variable_status_t::NONBASIC_FIXED) { continue; } + const f_t dz = scale * delta_z[jj]; + if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && dz < -pivot_tol) { + const f_t ratio = std::max((-dual_tol - soln.z[jj]) / dz, 0.0); + if (ratio < alpha) { alpha = ratio; } + } + if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && dz > pivot_tol) { + const f_t ratio = std::max((dual_tol - soln.z[jj]) / dz, 0.0); + if (ratio < alpha) { alpha = ratio; } + } + } + if (alpha == 0.0 || !std::isfinite(alpha)) { continue; } + + // Verify dual feasibility of the new point z_new = z + alpha * scale * delta_z + // For NONBASIC_LOWER: z_new[jj] >= -dual_tol + // For NONBASIC_UPPER: z_new[jj] <= dual_tol + { + f_t max_dual_infeas = 0.0; + i_t num_dual_infeas = 0; + i_t worst_j = -1; + for (i_t jj : delta_z_indices) { + if (vstatus[jj] == variable_status_t::NONBASIC_FIXED) { continue; } + const f_t new_zj = soln.z[jj] + alpha * scale * delta_z[jj]; + if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && new_zj < -settings_.dual_tol) { + num_dual_infeas++; + if (std::abs(new_zj) > max_dual_infeas) { + max_dual_infeas = std::abs(new_zj); + worst_j = jj; + } + } + if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && new_zj > settings_.dual_tol) { + num_dual_infeas++; + if (std::abs(new_zj) > max_dual_infeas) { + max_dual_infeas = std::abs(new_zj); + worst_j = jj; + } + } + } + // Also check the leaving variable itself + const f_t new_zj_leaving = soln.z[j] + alpha * scale * delta_z[j]; + if (num_dual_infeas > 0) { + settings_.log.printf( + "WARNING pivot_to_improve_rc: dual infeasibility after step! " + "var=%d alpha=%.6e scale=%.0f num_infeas=%d max_infeas=%.6e worst_j=%d " + "new_rc_leaving=%.6e\n", + j, alpha, scale, num_dual_infeas, max_dual_infeas, worst_j, new_zj_leaving); + } + } + + // Claim: We don't actually need to take a pivot if all we want to do is add a bound + // coming from reduced cost strengthening + const f_t new_reduced_cost = soln.z[j] + alpha * scale * delta_z[j]; + + // x_j <= l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] + // Let u_tilde_j = u_j - epsilon, so that floor(u_tilde_j) = u_j - 1 + // We want to solve for want the incumbent objective needs to be to make + // x_j <= u_tilde_j + // This means l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] <= + // u_tilde_j Or equivalently, incumbent_objective <= relaxation_objective + + // reduced_costs[j] * (u_tilde_j - l_j) when reduced_costs[j] > 0 + if (lower_j > -inf && new_reduced_cost > threshold) { + const f_t u_tilde_j = var_types_[j] == variable_type_t::INTEGER + ? upper_j - tol + : std::max(upper_j - 1.0, lower_j); + const f_t bound_j = + var_types_[j] == variable_type_t::INTEGER ? std::floor(u_tilde_j) : u_tilde_j; + const f_t diff = u_tilde_j - lower_j; + const f_t objective_j = relaxation_objective + diff * new_reduced_cost; + if (((var_types_[j] == variable_type_t::INTEGER && bound_j == upper_j - 1.0) || + var_types_[j] != variable_type_t::INTEGER) && + std::isfinite(objective_j) && std::isfinite(bound_j)) { + i_t info = reduced_cost_bounds.add_upper_bound(j, objective_j, bound_j); + if (info > 0) { num_bounds_added++; } + //settings_.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. Info %d\n", objective_j, bound_j, j, info); + } + } + + // x_j >= u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] when + // reduced_costs[j] < 0 Let l_tilde_j = l_j + epsilon, so that ceil(l_tilde_j) = l_j + 1 We want + // to solve for want the incumbent objective needs to be to make x_j >= l_tilde_j This means + // u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j Or + // equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * + // (l_tilde_j - u_j) when reduced_costs[j] < 0 + if (upper_j < inf && new_reduced_cost < -threshold) { + const f_t l_tilde_j = var_types_[j] == variable_type_t::INTEGER + ? lower_j + tol + : std::min(lower_j + 1.0, upper_j); + const f_t bound_j = + var_types_[j] == variable_type_t::INTEGER ? std::ceil(l_tilde_j) : l_tilde_j; + const f_t diff = l_tilde_j - upper_j; + const f_t objective_j = relaxation_objective + diff * new_reduced_cost; + if (((var_types_[j] == variable_type_t::INTEGER && bound_j == lower_j + 1.0) || + var_types_[j] != variable_type_t::INTEGER) && + std::isfinite(objective_j) && std::isfinite(bound_j)) { + i_t info = reduced_cost_bounds.add_lower_bound(j, objective_j, bound_j); + if (info > 0) { num_bounds_added++; } + //settings_.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. Info %d\n", objective_j, bound_j, j, info); + } + } + } + + // Clear arrays for next iteration + for (i_t k : delta_z_indices) { + delta_z_mark[k] = 0; + delta_z[k] = 0.0; + } + delta_z[leaving_index] = 0.0; + delta_z_indices.clear(); + for (i_t k : delta_y_sparse.i) { + delta_y[k] = 0.0; + } + } + settings_.log.printf("Added %d bounds for reduced cost strengthening\n", num_bounds_added); +} + template mip_status_t branch_and_bound_t::solve(mip_solution_t& solution) { @@ -4355,6 +4879,17 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut reduced_cost_bounds_t reduced_cost_bounds(original_lp_.num_cols); update_reduced_cost_bounds(root_objective_, root_relax_soln_.z, root_vstatus_, reduced_cost_bounds); + settings_.log.printf("New reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); + pivot_to_improve_reduced_cost_strengthening(original_lp_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional, + root_objective_, reduced_cost_bounds); + settings_.log.printf("After pivoting: new reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); f_t pivot_out_integer_variables_start_time = tic(); i_t num_integer_increased = pivot_out_integer_variables(original_lp_, @@ -4586,7 +5121,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut f_t previous_max_objective = reduced_cost_bounds.get_max_objective(); i_t num_changed = reduced_cost_bounds.update_bounds_from_new_incumbent( upper_bound_.load(), var_types_, lower_bounds, upper_bounds); - settings_.log.printf("Updated %d bounds using reduced cost strengthening from new incumbent. Max objective %e Current objective %e Previous max objective %e\n", num_changed, reduced_cost_bounds.get_max_objective(), upper_bound_.load(), previous_max_objective); + settings_.log.printf("Updated %d integer bounds using reduced cost strengthening from new incumbent. Max objective %e Current objective %e Previous max objective %e\n", num_changed, reduced_cost_bounds.get_max_objective(), upper_bound_.load(), previous_max_objective); mutex_original_lp_.lock(); original_lp_.lower = lower_bounds; original_lp_.upper = upper_bounds; diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 185e3ce79e..a6c1526d95 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -184,14 +184,14 @@ class reduced_cost_bounds_t { { const i_t n = static_cast(lower_bounds_.size()); f_t max_objective = -std::numeric_limits::infinity(); - i_t bounds_updated = 0; + i_t integer_bounds_updated = 0; for (i_t j = 0; j < n; ++j) { if (lower_bounds_[j].is_valid()) { if (incumbent_objective <= lower_bounds_[j].objective && lower_bounds_[j].bound > lower_bounds[j]) { //printf("RCF Variable %d (%d): lower %e -> %e\n", j, static_cast(var_types[j]), lower_bounds[j], lower_bounds_[j].bound); lower_bounds[j] = lower_bounds_[j].bound; - bounds_updated++; + if (var_types[j] == simplex::variable_type_t::INTEGER) { integer_bounds_updated++; } lower_bounds_[j].bound = lower_bounds_[j].objective = std::numeric_limits::quiet_NaN(); } if (lower_bounds_[j].objective > max_objective) { @@ -203,7 +203,7 @@ class reduced_cost_bounds_t { upper_bounds_[j].bound < upper_bounds[j]) { //printf("RCF Variable %d (%d): upper %e -> %e\n", j, static_cast(var_types[j]), upper_bounds[j], upper_bounds_[j].bound); upper_bounds[j] = upper_bounds_[j].bound; - bounds_updated++; + if (var_types[j] == simplex::variable_type_t::INTEGER) { integer_bounds_updated++; } upper_bounds_[j].bound = upper_bounds_[j].objective = std::numeric_limits::quiet_NaN(); } if (upper_bounds_[j].objective > max_objective) { @@ -212,7 +212,7 @@ class reduced_cost_bounds_t { } } max_objective_ = max_objective; - return bounds_updated; + return integer_bounds_updated; } f_t get_current_lower_bound(i_t col) @@ -500,16 +500,28 @@ class branch_and_bound_t { std::vector& zero_reduced_costs_vars, std::vector& zero_reduced_costs_vars_nonbasic_index); + void fast_slack_integer_pivot(const simplex::lp_problem_t& lp, + const std::vector& fractional, + const std::vector& row_to_slack, + const simplex::lp_solution_t& solution, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& nonbasic_index, + std::vector& vstatus, + simplex::lp_solution_t& soln, + simplex::basis_update_mpf_t& basis_update, + f_t& work_estimate); + i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& vstatus, - simplex::lp_solution_t& soln, - simplex::basis_update_mpf_t& basis_update, - i_t& num_fractional, - std::vector& fractional); - - void apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + simplex::lp_solution_t& soln, + simplex::basis_update_mpf_t& basis_update, + i_t& num_fractional, + std::vector& fractional); + + i_t apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, std::vector& basic_list, std::vector& nonbasic_list, std::vector& nonbasic_index, @@ -532,6 +544,18 @@ class branch_and_bound_t { i_t& num_fractional, std::vector& fractional); + void pivot_to_improve_reduced_cost_strengthening( + const simplex::lp_problem_t& lp, + const std::vector& basic_list, + const std::vector& nonbasic_list, + const std::vector& vstatus, + const simplex::lp_solution_t& soln, + const simplex::basis_update_mpf_t& basis_update, + i_t num_fractional, + const std::vector& fractional, + f_t relaxation_objective, + reduced_cost_bounds_t& reduced_cost_bounds); + // Repairs low-quality solutions from the heuristics, if it is applicable. void repair_heuristic_solutions(); diff --git a/cpp/src/dual_simplex/basis_updates.cpp b/cpp/src/dual_simplex/basis_updates.cpp index 84468ba097..c2a7027548 100644 --- a/cpp/src/dual_simplex/basis_updates.cpp +++ b/cpp/src/dual_simplex/basis_updates.cpp @@ -1507,7 +1507,7 @@ f_t basis_update_mpf_t::dot_product(i_t col, nz_mark++; } } - work_estimate_ += 2 * nz_mark + (col_end - col_start); + work_estimate_ += 2 * (col_end - col_start) + 2 * nz_mark; return dot; } @@ -1524,7 +1524,7 @@ void basis_update_mpf_t::add_sparse_column(const csc_matrix_t @@ -1549,7 +1549,7 @@ void basis_update_mpf_t::add_sparse_column(const csc_matrix_t diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 6e8ef4bbdd..8c59bd464f 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -1456,7 +1456,7 @@ i_t update_steepest_edge_norms(const simplex_solver_settings_t& settin work_estimate += 2 * v_sparse.i.size(); } v_sparse.scatter(v); - work_estimate += 2 * v_sparse.i.size(); + work_estimate += 4 * v_sparse.i.size(); const i_t leaving_index = basic_list[basic_leaving_index]; const f_t prev_dy_norm_squared = delta_y_steepest_edge[leaving_index]; @@ -1508,7 +1508,7 @@ i_t update_steepest_edge_norms(const simplex_solver_settings_t& settin delta_y_steepest_edge[j] = new_val; } } - work_estimate += 5 * scaled_delta_xB_nz; + work_estimate += 6 * scaled_delta_xB_nz; const i_t v_nz = v_sparse.i.size(); for (i_t k = 0; k < v_nz; ++k) { @@ -2904,7 +2904,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t num_refactors = 0; i_t total_bound_flips = 0; f_t delta_y_nz_percentage = 0.0; - phase2::phase2_timers_t timers(false); + phase2::phase2_timers_t timers(true); // Sparse vectors for main loop (declared outside loop for instrumentation) sparse_vector_t delta_y_sparse(m, 0); @@ -2922,10 +2922,11 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += ft.work_estimate(); ft.clear_work_estimate(); + f_t last_work_reported = 0.0; if (work_unit_context) { work_unit_context->record_work_sync_on_horizon((phase2_work_estimate) / 1e8); + last_work_reported = phase2_work_estimate; } - phase2_work_estimate = 0.0; if (phase == 2) { settings.log.printf("%5d %+.16e %7d %.8e %.2e %.2f\n", @@ -3080,6 +3081,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } } + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); phase2::prepare_optimality(0, primal_infeasibility, lp, @@ -3301,6 +3304,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, obj = phase2::compute_perturbed_objective(objective, x); phase2_work_estimate += 2 * n; if (dual_infeas <= settings.dual_tol && primal_infeasibility <= settings.primal_tol) { + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); phase2::prepare_optimality(1, primal_infeasibility, lp, @@ -3358,6 +3363,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (primal_infeasibility <= settings.primal_tol && orig_dual_infeas <= settings.dual_tol) { + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); phase2::prepare_optimality(2, primal_infeasibility, lp, @@ -3790,16 +3797,19 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += 3 * delta_z_indices.size(); phase2::clear_delta_z(entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); + // Flush basis update work into the total work estimate every iteration + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); + f_t now = toc(start_time); // Feature logging for regression training (every FEATURE_LOG_INTERVAL iterations) if ((iter % FEATURE_LOG_INTERVAL) == 0 && work_unit_context) { [[maybe_unused]] i_t iters_elapsed = iter - last_feature_log_iter; - phase2_work_estimate += ft.work_estimate(); - ft.clear_work_estimate(); - work_unit_context->record_work_sync_on_horizon(phase2_work_estimate / 1e8); - phase2_work_estimate = 0.0; + work_unit_context->record_work_sync_on_horizon( + (phase2_work_estimate - last_work_reported) / 1e8); + last_work_reported = phase2_work_estimate; last_feature_log_iter = iter; } @@ -3839,6 +3849,10 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } if (iter >= iter_limit) { status = dual_status_t::ITERATION_LIMIT; } + // Flush any remaining work from the basis update into the total work estimate + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); + if (phase == 2) { timers.print_timers(settings); constexpr bool print_stats = false; diff --git a/cpp/src/dual_simplex/right_looking_lu.cpp b/cpp/src/dual_simplex/right_looking_lu.cpp index 6a717cd257..63f5cb7c0f 100644 --- a/cpp/src/dual_simplex/right_looking_lu.cpp +++ b/cpp/src/dual_simplex/right_looking_lu.cpp @@ -209,7 +209,8 @@ class trailing_matrix_t { const f_t max_in_col = max_in_column_[j]; const i_t c_start = col_start_[j]; const i_t c_end = col_end_[j]; - for (i_t p = c_start; p < c_end; p++) { + i_t p; + for (p = c_start; p < c_end; p++) { const i_t i = c_i_[p]; const f_t val = c_x_[p]; const i_t rdeg = row_counts_.get_count(i); @@ -224,7 +225,7 @@ class trailing_matrix_t { if (markowitz <= markowitz_lower_bound) { break; } } } - work_estimate_ += 3 * (c_end - c_start); + work_estimate_ += 3 * (p - c_start); nsearch++; if (markowitz <= markowitz_lower_bound) { break; } } @@ -241,19 +242,21 @@ class trailing_matrix_t { assert(rdeg == nz); const i_t r_start = row_start_[i]; const i_t r_end = row_end_[i]; - for (i_t p = r_start; p < r_end; p++) { + i_t p; + for (p = r_start; p < r_end; p++) { const i_t j = r_j_[p]; // Look up the value from the column copy of j f_t val = 0; const i_t c_start = col_start_[j]; const i_t c_end = col_end_[j]; - for (i_t q = c_start; q < c_end; q++) { + i_t q; + for (q = c_start; q < c_end; q++) { if (c_i_[q] == i) { val = c_x_[q]; break; } } - work_estimate_ += 2 * (c_end - c_start); + work_estimate_ += 2 * (q - c_start); const f_t max_in_col = max_in_column_[j]; const i_t cdeg = col_counts_.get_count(j); assert(cdeg >= 0); @@ -267,7 +270,7 @@ class trailing_matrix_t { if (markowitz <= markowitz_lower_bound) { break; } } } - work_estimate_ += 5 * (r_end - r_start); + work_estimate_ += 5 * (p - r_start); nsearch++; if (markowitz <= markowitz_lower_bound) { break; } } @@ -334,7 +337,7 @@ class trailing_matrix_t { } } } - work_estimate_ += 2 * (c_end - c_start) + 6 * (pivot_col_count - n_fillin); + work_estimate_ += 2 * (c_end - c_start) + 5 * (pivot_col_count - n_fillin); // Step 2b: Remove cancellations (entries that became zero). if (n_cancel > 0) { @@ -1285,12 +1288,14 @@ class symmetric_trailing_matrix_t { const i_t j = r_j_[rp]; // Look up A(pivot_p, j) from column j f_t val = 0; - for (i_t q = col_start_[j]; q < col_end_[j]; q++) { + i_t q; + for (q = col_start_[j]; q < col_end_[j]; q++) { if (c_i_[q] == pivot_p) { val = c_x_[q]; break; } } + work_estimate_ += 2 * (q - col_start_[j]); const f_t lj = val / pivot_val; pivot_col_val_[j] = lj; pivot_col_mark_[j] = 1; From 39cb576e0f2c0ad7622fcaf5225c4754dbff7c89 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 14 Aug 2026 14:30:49 -0700 Subject: [PATCH 021/113] Devex pricing in primal; try to remove perturbations in dual --- .../mathematical_optimization/constants.h | 2 + .../pdlp/solver_settings.hpp | 2 + cpp/src/dual_simplex/phase2.cpp | 82 +++++++++++-- cpp/src/dual_simplex/primal.cpp | 116 ++++++++++++++++-- .../dual_simplex/simplex_solver_settings.hpp | 6 + cpp/src/dual_simplex/solve.cpp | 8 ++ cpp/src/math_optimization/solver_settings.cu | 2 + cpp/src/pdlp/solve.cu | 3 + 8 files changed, 202 insertions(+), 19 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/constants.h b/cpp/include/cuopt/mathematical_optimization/constants.h index 5a807de308..d21b293a32 100644 --- a/cpp/include/cuopt/mathematical_optimization/constants.h +++ b/cpp/include/cuopt/mathematical_optimization/constants.h @@ -54,6 +54,8 @@ #define CUOPT_CUDSS_DETERMINISTIC "cudss_deterministic" #define CUOPT_PRESOLVE "presolve" #define CUOPT_INITIAL_PERTURBATION "initial_perturbation" +#define CUOPT_REMOVE_PERTURBATION "remove_perturbation" +#define CUOPT_PRIMAL_PRICING "primal_pricing" #define CUOPT_MIP_PROBING "mip_probing" #define CUOPT_DUAL_POSTSOLVE "dual_postsolve" #define CUOPT_MIP_DETERMINISM_MODE "mip_determinism_mode" diff --git a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp index 80197e8a85..8d267c8be5 100644 --- a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp @@ -298,6 +298,8 @@ class pdlp_solver_settings_t { i_t dualize{-1}; i_t ordering{-1}; i_t initial_perturbation{-1}; + i_t remove_perturbation{-1}; + i_t primal_pricing{0}; i_t barrier_dual_initial_point{-1}; i_t postsolve_info{-1}; // Ruiz equilibration for QCQP (barrier) scaling: -1 automatic (row/column diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 8c59bd464f..e2a3c17d11 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -464,7 +464,30 @@ void initial_perturbation(const lp_problem_t& lp, max_abs_obj_coeff = std::max(max_abs_obj_coeff, std::abs(lp.objective[j])); } - const f_t dual_tol = settings.dual_tol; + // Dampen large costs + if (max_abs_obj_coeff > 100.0) { + max_abs_obj_coeff = std::sqrt(std::sqrt(max_abs_obj_coeff)); + } + // Ensure a minimum perturbation even for tiny-cost problems + if (max_abs_obj_coeff < 1.0) { + max_abs_obj_coeff = 1.0; + } + + // If few boxed variables, cap max_abs_obj_coeff at 1.0 + i_t num_boxed = 0; + for (i_t j = 0; j < n; ++j) { + if (lp.lower[j] > -inf && lp.upper[j] < inf && lp.lower[j] != lp.upper[j]) { + num_boxed++; + } + } + if (static_cast(num_boxed) / n < 0.01) { + max_abs_obj_coeff = std::min(max_abs_obj_coeff, f_t(1.0)); + } + + const f_t perturbation_base = 5e-7 * max_abs_obj_coeff; + + settings.log.printf("Perturbation debug: max_abs_obj_coeff=%e (dampened), perturbation_base=%e, n=%d, num_boxed=%d\n", + max_abs_obj_coeff, perturbation_base, n, num_boxed); objective.resize(n); f_t sum_perturb = 0.0; @@ -476,22 +499,26 @@ void initial_perturbation(const lp_problem_t& lp, const f_t lower = lp.lower[j]; const f_t upper = lp.upper[j]; - if (vstatus[j] == variable_status_t::NONBASIC_FIXED || - vstatus[j] == variable_status_t::NONBASIC_FREE || lower == upper || - lower == -inf && upper == inf) { + // Skip truly fixed variables and free variables + if (lower == upper || (lower == -inf && upper == inf)) { + continue; + } + // Skip basic variables + if (vstatus[j] == variable_status_t::BASIC) { continue; } const f_t rand_val = random.random(); - const f_t perturb = - (1e-5 * std::abs(obj) + 1e-7 * max_abs_obj_coeff + 10 * dual_tol) * (1.0 + rand_val); + const f_t cost_factor = std::min(std::abs(obj) + 1.0, max_abs_obj_coeff + 1.0); + const f_t perturb = (1.0 + rand_val) * cost_factor * perturbation_base; - if (vstatus[j] == variable_status_t::NONBASIC_LOWER || lower > -inf && upper < inf && obj > 0) { + if (vstatus[j] == variable_status_t::NONBASIC_LOWER || + vstatus[j] == variable_status_t::NONBASIC_FIXED) { + // NONBASIC_FIXED from phase 1 for boxed variables — treat as at lower bound objective[j] = obj + perturb; sum_perturb += perturb; num_perturb++; - } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER || - lower > -inf && upper < inf && obj < 0) { + } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER) { objective[j] = obj - perturb; sum_perturb += perturb; num_perturb++; @@ -1543,6 +1570,40 @@ i_t check_steepest_edge_norms(const simplex_solver_settings_t& setting return 0; } +// Remove the perturbation from a variable that is leaving the basis. Since it +// is nonbasic, its cost affects only its own reduced cost. If removing the +// perturbation would violate dual feasibility, apply just enough perturbation +// to maintain feasibility (for one-sided variables) or leave it unperturbed +// (for boxed variables, which can be flipped). +template +void remove_leaving_perturbation(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + i_t leaving_index, + std::vector& z, + std::vector& objective) +{ + const f_t perturb = objective[leaving_index] - lp.objective[leaving_index]; + if (perturb == 0.0) return; + + z[leaving_index] -= perturb; + objective[leaving_index] = lp.objective[leaving_index]; + + // Restore dual feasibility if needed + const f_t lower = lp.lower[leaving_index]; + const f_t upper = lp.upper[leaving_index]; + if (upper == inf && lower > -inf && z[leaving_index] < -settings.tight_tol) { + // At lower bound, needs z >= 0 + const f_t correction = -z[leaving_index]; + z[leaving_index] = 0.0; + objective[leaving_index] += correction; + } else if (lower == -inf && upper < inf && z[leaving_index] > settings.tight_tol) { + // At upper bound, needs z <= 0 + const f_t correction = z[leaving_index]; + z[leaving_index] = 0.0; + objective[leaving_index] -= correction; + } +} + template i_t compute_perturbation(const lp_problem_t& lp, const simplex_solver_settings_t& settings, @@ -3657,6 +3718,9 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, basic_list, entering_index, scaled_delta_xB_sparse, delta_x, phase2_work_estimate); timers.start_timer(phase2_work_estimate + ft.work_estimate()); + if (settings.remove_perturbation == 1) { + phase2::remove_leaving_perturbation(lp, settings, leaving_index, z, objective); + } f_t sum_perturb = 0.0; phase2::compute_perturbation( lp, settings, delta_z_indices, z, objective, sum_perturb, phase2_work_estimate); diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 8867c8c9b4..7548e2d28f 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -266,6 +266,54 @@ i_t phase2_pricing(const lp_problem_t& lp, return entering_index; } +template +i_t devex_pricing(const lp_problem_t& lp, + const std::vector& z, + const std::vector& devex_weight, + const std::vector& nonbasic_list, + const std::vector& vstatus, + f_t dual_tol, + i_t& direction, + i_t& basic_entering, + f_t& dual_inf, + f_t& work_estimate) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + i_t entering_index = -1; + f_t max_score = 0.0; + dual_inf = 0.0; + for (i_t k = 0; k < n - m; ++k) { + const i_t j = nonbasic_list[k]; + if (vstatus[j] == variable_status_t::NONBASIC_FIXED) { continue; } + f_t infeas = 0.0; + i_t dir = 0; + if ((vstatus[j] == variable_status_t::NONBASIC_LOWER || + vstatus[j] == variable_status_t::NONBASIC_FREE) && + z[j] < -dual_tol) { + infeas = -z[j]; + dir = 1; + } else if ((vstatus[j] == variable_status_t::NONBASIC_UPPER || + vstatus[j] == variable_status_t::NONBASIC_FREE) && + z[j] > dual_tol) { + infeas = z[j]; + dir = -1; + } + if (infeas > 0.0) { + dual_inf += infeas; + const f_t score = (infeas * infeas) / devex_weight[j]; + if (score > max_score) { + max_score = score; + basic_entering = k; + entering_index = j; + direction = dir; + } + } + } + work_estimate += 7 * (n - m); + return entering_index; +} + template f_t primal_infeasibility(const lp_problem_t& lp, const simplex_solver_settings_t& settings, @@ -811,6 +859,7 @@ primal_status_t primal_phase2_with_advanced_basis( std::vector incoming_vstatus = vstatus; work_estimate += 2.0 * n; settings.log.printf("Primal Simplex\n"); + settings.log.printf("Pricing: %s\n", settings.primal_pricing == 1 ? "Devex" : "Dantzig"); // Nonbasics must be on their bounds before forming B x_B = b - A_N x_N. // Setting them after the solve leaves ||A*x - b|| large whenever x_N != 0. set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); @@ -893,7 +942,9 @@ primal_status_t primal_phase2_with_advanced_basis( sparse_vector_t etilde(m, 0); std::vector delta_z(n); std::vector delta_x(n); - work_estimate += 2 * m + 2 * n; + std::vector devex_weight(n, 1.0); + i_t num_bad_devex_weight = 0; + work_estimate += 2 * m + 3 * n; f_t dual_inf = init_dual_inf; f_t obj = compute_objective(lp, x); @@ -922,15 +973,29 @@ primal_status_t primal_phase2_with_advanced_basis( timers.start_timer(work_estimate + basis_update.work_estimate()); i_t nonbasic_entering = -1; i_t direction; - i_t entering_index = phase2_pricing(lp, - z, - nonbasic_list, - vstatus, - pricing_dual_tol, - direction, - nonbasic_entering, - dual_inf, - work_estimate); + i_t entering_index; + if (settings.primal_pricing == 1) { + entering_index = devex_pricing(lp, + z, + devex_weight, + nonbasic_list, + vstatus, + pricing_dual_tol, + direction, + nonbasic_entering, + dual_inf, + work_estimate); + } else { + entering_index = phase2_pricing(lp, + z, + nonbasic_list, + vstatus, + pricing_dual_tol, + direction, + nonbasic_entering, + dual_inf, + work_estimate); + } timers.pricing_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); if (entering_index == -1) { if (phase == 2) { @@ -1249,6 +1314,37 @@ primal_status_t primal_phase2_with_advanced_basis( update_y(dual_step_length, delta_y, y, work_estimate); update_z(dual_step_length, nonbasic_list, entering_index, delta_z, z, work_estimate); timers.update_duals_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); + + // Devex weight update (only when using Devex pricing) + if (settings.primal_pricing == 1) { + const f_t pivot = scaled_delta_xB[basic_leaving]; + const f_t pivot_sq = pivot * pivot; + const f_t w_enter = devex_weight[entering_index]; + // Exact pivot weight for entering variable is 1/pivot_sq + // Check if stored weight was a bad approximation + const f_t exact_pivot_weight = 1.0 / pivot_sq; + if (w_enter > 3.0 * exact_pivot_weight) { num_bad_devex_weight++; } + // Update weights for all nonbasic columns using the pivot row (delta_z) + // After compute_delta_z and update_z, delta_z[j] still holds the raw + // pivot row entries (update_z multiplies by dual_step_length into z, not delta_z) + for (i_t k = 0; k < n - m; ++k) { + const i_t j = nonbasic_list[k]; + const f_t a_j = delta_z[j]; + const f_t candidate = (a_j * a_j / pivot_sq) * w_enter; + if (candidate > devex_weight[j]) { devex_weight[j] = candidate; } + } + // Weight for leaving variable (now nonbasic) + devex_weight[leaving_index] = std::max(1.0 / pivot_sq, f_t(1e-4)); + // Weight for entering variable (now basic) — reset + devex_weight[entering_index] = 1.0; + work_estimate += 5 * (n - m); + // Reset framework if too many bad weights + if (num_bad_devex_weight > 3) { + std::fill(devex_weight.begin(), devex_weight.end(), f_t(1.0)); + num_bad_devex_weight = 0; + } + } + timers.start_timer(work_estimate + basis_update.work_estimate()); should_refactor = basis_update.update(utilde_sparse, etilde, basic_leaving) == 1; timers.lu_update_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 50e8ca15a7..0189cd51de 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -78,11 +78,14 @@ struct simplex_solver_settings_t { dualize(-1), ordering(-1), initial_perturbation(-1), + remove_perturbation(-1), + primal_pricing(0), barrier_dual_initial_point(-1), postsolve_info(-1), qcqp_ruiz_equilibration(-1), check_Q(false), crossover(false), + unscaled_max_abs_obj_coeff(-1.0), refactor_frequency(100), iteration_log_frequency(1000), first_iteration_log(2), @@ -173,12 +176,15 @@ struct simplex_solver_settings_t { i_t dualize; // -1 automatic, 0 to not dualize, 1 to dualize i_t ordering; // -1 automatic, 0 to use nested dissection, 1 to use AMD i_t initial_perturbation; // -1 automatic, 0 to not perturb, 1 to perturb + i_t remove_perturbation; // -1 automatic, 0 disabled, 1 enabled + i_t primal_pricing; // 0 Dantzig (default), 1 Devex i_t barrier_dual_initial_point; // -1 automatic, 0 to use Lustig, Marsten, and Shanno initial // point, 1 to use initial point form dual least squares problem i_t postsolve_info; // -1 automatic (disabled), 0 disabled, 1 enabled i_t qcqp_ruiz_equilibration; // -1 automatic (imbalance heuristic), 0 disabled, 1 enabled bool check_Q; // true to check if Q is positive semidefinite bool crossover; // true to do crossover, false to not + f_t unscaled_max_abs_obj_coeff; // max |c_j| before scaling (-1 = not set, compute from lp) i_t refactor_frequency; // number of basis updates before refactorization i_t iteration_log_frequency; // number of iterations between log updates i_t first_iteration_log; // number of iterations to log at beginning of solve diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index fedb9de356..cd866ea1d2 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -221,6 +221,14 @@ lp_status_t solve_linear_program_with_advanced_basis( presolved_lp.A.col_start[presolved_lp.num_cols]); std::vector column_scales; std::vector row_scales_simplex; + // Compute max |c_j| before scaling for perturbation calibration + if (settings.unscaled_max_abs_obj_coeff < 0.0) { + f_t max_obj = 0.0; + for (i_t j = 0; j < presolved_lp.num_cols; ++j) { + max_obj = std::max(max_obj, std::abs(presolved_lp.objective[j])); + } + const_cast&>(settings).unscaled_max_abs_obj_coeff = max_obj; + } { raft::common::nvtx::range scope_scaling("DualSimplex::scaling"); scaling(presolved_lp, settings, lp, column_scales, row_scales_simplex); diff --git a/cpp/src/math_optimization/solver_settings.cu b/cpp/src/math_optimization/solver_settings.cu index ff39bbda22..8b7172d2f7 100644 --- a/cpp/src/math_optimization/solver_settings.cu +++ b/cpp/src/math_optimization/solver_settings.cu @@ -139,6 +139,8 @@ solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings {CUOPT_DUALIZE, &pdlp_settings.dualize, -1, 1, -1}, {CUOPT_ORDERING, &pdlp_settings.ordering, -1, 1, -1}, {CUOPT_INITIAL_PERTURBATION, &pdlp_settings.initial_perturbation, -1, 1, -1}, + {CUOPT_REMOVE_PERTURBATION, &pdlp_settings.remove_perturbation, -1, 1, -1}, + {CUOPT_PRIMAL_PRICING, &pdlp_settings.primal_pricing, 0, 1, 0}, {CUOPT_BARRIER_DUAL_INITIAL_POINT, &pdlp_settings.barrier_dual_initial_point, -1, 1, -1}, {CUOPT_POSTSOLVE_INFO, &pdlp_settings.postsolve_info, -1, 1, -1}, {CUOPT_MIP_CUT_PASSES, &mip_settings.max_cut_passes, -1, std::numeric_limits::max(), 10}, diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index beb53e8a17..7d8ce1e2d3 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -598,6 +598,8 @@ std::tuple, simplex::lp_status_t, f_t, f_t, f_t dual_simplex_settings.iteration_limit = settings.iteration_limit; dual_simplex_settings.concurrent_halt = settings.concurrent_halt; dual_simplex_settings.initial_perturbation = settings.initial_perturbation; + dual_simplex_settings.remove_perturbation = settings.remove_perturbation; + dual_simplex_settings.primal_pricing = settings.primal_pricing; if (dual_simplex_settings.concurrent_halt != nullptr) { // Don't show the dual simplex log in concurrent mode. Show the PDLP log instead dual_simplex_settings.log.log = false; @@ -653,6 +655,7 @@ std::tuple, simplex::lp_status_t, f_t, f_t, f_t primal_settings.time_limit = settings.time_limit; primal_settings.iteration_limit = settings.iteration_limit; primal_settings.concurrent_halt = settings.concurrent_halt; + primal_settings.primal_pricing = settings.primal_pricing; if (primal_settings.concurrent_halt != nullptr) { // Don't show the primal simplex log in concurrent mode. Show the PDLP log instead primal_settings.log.log = false; From f54d57fedf73ba7763d1f25729d111d9cc62a8c2 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 26 Aug 2026 13:04:26 -0700 Subject: [PATCH 022/113] 15% improvement in dual simplex; new BFRT, perturbations, initial point, etc. The bound-flipping ratio test is rewritten. Instead of a heap-based approach a coarse filter is used to increase the step-length by multiples of 10. This is followed by a bucket sort. Each bucket contains variables with the same Harris ratio. We start from the final bucket, and go backward, trying to find a variable that statisfies our pivot threshold (more than 1/10th the maximum pivot) and maximizes the step length. When Phase-I completes and there are many bounded variables with NONBASIC_FIXED status with a small reduced costs | z_j | < dual_tol. These variables can be put on either bound. So we try three different initial points: 1) Set these variables on their lower bounds 2) Use a heuristic that assumes we have a slack basis B and sets the variable to the bound that minimizes the column sum. 3) Use a heuristic that sets the variable on the bound with the smallest absolute value. We test each of these points and choose the one that improves over the default (lower bounds) with less primal infeasibilities and small sum of primal infeasibilites squared. We do pertubations differently: 1) We attempt to remove perturbations as a variable leaves the basis. 2) We add a perturbation to the cost of the entering variable when we take a degenerate step. This accumulates after many degenerate steps and when we have a refactorization results in a different y and thus different reduced costs z 3) We don't apply perturbations if we are close to optimal. 4) We call set_primal_variables_on_bound after applying perturbation. Below we show a table of the baseline cuOpt code, compared to v5 (this PR). The HiGHS times were taken from a faster machine. So HiGHS advantage is slightly exaggerated. But it still exists. Problem Baseline v5 HiGHS v5/HiGHS ------------------------------------------------------------------------- momentum1 0.69 0.73 300.00 0.00 var-smallemery-m6j6 0.66 0.74 300.00 0.00 supportcase42 0.52 0.77 36.20 0.02 neos-5114902-kasavu 87.42 23.31 300.00 0.08 neos-5049753-cuanza 7.69 3.23 29.85 0.11 supportcase12 4.68 4.30 37.11 0.12 proteindesign121hz512p9 0.91 0.48 2.04 0.24 proteindesign122trx11p8 0.64 0.31 1.26 0.25 roi5alpha10n8 1.26 3.29 11.90 0.28 supportcase22 2.04 1.11 3.97 0.28 supportcase18 0.06 0.04 0.12 0.33 neos-787933 0.07 0.07 0.18 0.39 supportcase7 1.29 1.46 3.52 0.41 rocII-5-11 0.09 0.09 0.21 0.43 30n20b8 0.08 0.05 0.11 0.45 neos-860300 0.10 0.07 0.15 0.47 neos-5093327-huahum 0.23 0.23 0.48 0.48 cryptanalysiskb128n5obj14 29.78 6.14 12.54 0.49 co-100 0.68 0.64 1.28 0.50 fhnw-binpack4-48 0.07 0.03 0.06 0.50 mzzv11 40.19 8.46 16.71 0.51 rd-rplusc-21 0.23 0.25 0.49 0.51 lectsched-5-obj 0.13 0.10 0.18 0.56 roi2alpha3n4 0.31 0.67 1.16 0.58 dws008-01 0.04 0.03 0.05 0.60 neos-5052403-cygnet 300.00 187.30 300.00 0.62 cvs16r128-89 0.94 1.08 1.72 0.63 cryptanalysiskb128n5obj16 29.53 6.53 10.27 0.64 neos-5195221-niemur 0.50 0.24 0.37 0.65 neos-5188808-nattai 0.30 0.19 0.29 0.66 neos-3004026-krka 0.07 0.08 0.12 0.67 thor50dday 0.27 0.25 0.37 0.68 neos-4300652-rahue 1.22 0.56 0.80 0.70 square47 79.88 89.39 126.92 0.70 n3div36 0.11 0.13 0.18 0.72 blp-ar98 0.10 0.08 0.11 0.73 neos-960392 8.93 1.97 2.66 0.74 tbfp-network 9.06 6.73 9.04 0.74 neos-4647030-tutaki 2.47 2.33 3.09 0.75 decomp2 0.19 0.10 0.12 0.83 wachplan 0.25 0.23 0.26 0.88 netdiversion 7.65 8.61 9.43 0.91 supportcase40 0.24 0.22 0.24 0.92 istanbul-no-cutoff 0.71 0.87 0.94 0.93 supportcase10 300.00 105.36 113.55 0.93 ns1760995 135.95 250.52 269.53 0.93 neos-5104907-jarama 124.70 88.17 89.64 0.98 h80x6320d 0.05 0.04 0.04 1.00 highschool1-aigio 300.00 300.00 300.00 1.00 neos-1456979 0.05 0.04 0.04 1.00 neos-3988577-wolgan 278.89 300.00 300.00 1.00 neos859080 0.01 0.01 0.01 1.00 physiciansched3-3 300.00 300.00 300.00 1.00 pk1 0.02 0.01 0.01 1.00 rail02 300.00 300.00 300.00 1.00 s100 300.00 300.00 300.00 1.00 savsched1 300.00 300.00 300.00 1.00 supportcase19 300.00 300.00 300.00 1.00 swath3 0.03 0.03 0.03 1.00 timtab1 0.01 0.01 0.01 1.00 traininstance2 0.09 0.04 0.04 1.00 traininstance6 0.04 0.03 0.03 1.00 neos-4532248-waihi 2.61 0.92 0.90 1.02 cod105 9.06 7.73 7.46 1.04 square41 28.57 42.48 39.86 1.07 comp21-2idx 1.55 0.38 0.35 1.09 neos-873061 1.46 1.51 1.38 1.09 blp-ic98 0.12 0.11 0.10 1.10 neos-3555904-turama 1.31 1.52 1.37 1.11 piperout-27 0.67 0.29 0.26 1.12 sp97ar 0.40 0.37 0.33 1.12 piperout-08 0.39 0.18 0.16 1.12 triptim1 71.42 58.41 51.43 1.14 bnatt500 0.29 0.16 0.14 1.14 mushroom-best 0.26 0.23 0.20 1.15 leo2 0.14 0.15 0.13 1.15 neos-848589 1.25 1.02 0.85 1.20 neos-1122047 2.09 1.96 1.61 1.22 germanrr 0.31 0.33 0.27 1.22 net12 0.55 0.43 0.35 1.23 drayage-100-23 0.07 0.05 0.04 1.25 neos-3381206-awhea 0.08 0.05 0.04 1.25 neos-4738912-atrato 0.05 0.05 0.04 1.25 uct-subprob 0.11 0.10 0.08 1.25 ns1644855 300.00 300.00 238.30 1.26 leo1 0.10 0.09 0.07 1.29 atlanta-ip 6.85 5.86 4.54 1.29 neos-5107597-kakapo 0.04 0.13 0.10 1.30 air05 0.28 0.24 0.18 1.33 swath1 0.04 0.04 0.03 1.33 sp98ar 0.39 0.42 0.31 1.35 radiationm18-12-05 0.23 0.19 0.14 1.36 bnatt400 0.16 0.11 0.08 1.38 sct2 0.22 0.20 0.14 1.43 mcsched 0.27 0.23 0.16 1.44 supportcase6 7.39 6.03 4.18 1.44 irp 0.11 0.13 0.09 1.44 neos-4722843-widden 1.17 2.25 1.54 1.46 ns1952667 8.95 1.18 0.80 1.47 hypothyroid-k1 4.36 4.47 3.01 1.49 rail507 7.17 4.22 2.84 1.49 neos-3402294-bobin 3.07 1.79 1.20 1.49 drayage-25-23 0.08 0.06 0.04 1.50 icir97_tension 0.03 0.03 0.02 1.50 n5-3 0.03 0.03 0.02 1.50 neos-1582420 0.12 0.09 0.06 1.50 rococoC10-001000 0.04 0.03 0.02 1.50 neos-3216931-puriri 7.33 4.89 3.23 1.51 fast0507 7.37 4.24 2.77 1.53 ns1116954 156.23 16.44 10.73 1.53 neos-4763324-toguru 8.32 7.78 5.00 1.56 neos-3402454-bohle 221.94 115.56 74.15 1.56 nexp-150-20-8-5 0.10 0.11 0.07 1.57 trento1 3.08 3.49 2.21 1.58 qap10 15.57 10.58 6.68 1.58 neos8 0.38 0.42 0.26 1.62 rmatr200-p5 7.50 7.64 4.61 1.66 cmflsp50-24-8-8 0.77 0.73 0.44 1.66 neos-3083819-nubu 0.06 0.05 0.03 1.67 ran14x18-disj-8 0.04 0.05 0.03 1.67 rocI-4-11 0.11 0.10 0.06 1.67 neos-2987310-joes 1.52 1.67 1.00 1.67 neos-662469 1.65 0.91 0.53 1.72 chromaticindex512-7 16.62 37.48 21.24 1.76 k1mushroom 31.00 30.56 16.80 1.82 rococoB10-011000 0.13 0.11 0.06 1.83 comp07-2idx 4.03 2.05 1.10 1.86 reblock115 0.16 0.17 0.09 1.89 opm2-z10-s4 89.23 85.57 44.75 1.91 neos-3024952-loue 0.41 0.39 0.20 1.95 neos-2746589-doon 7.53 5.95 3.02 1.97 50v-10 0.02 0.02 0.01 2.00 b1c1s1 0.05 0.04 0.02 2.00 bppc4-08 0.08 0.04 0.02 2.00 cost266-UUE 0.03 0.04 0.02 2.00 eil33-2 0.05 0.06 0.03 2.00 enlight_hard 0.02 0.02 0.01 2.00 exp-1-500-5-5 0.02 0.02 0.01 2.00 fhnw-binpack4-4 0.03 0.02 0.01 2.00 gen-ip002 0.02 0.02 0.01 2.00 gen-ip054 0.03 0.02 0.01 2.00 glass4 0.02 0.02 0.01 2.00 graphdraw-domain 0.03 0.02 0.01 2.00 mad 0.02 0.02 0.01 2.00 markshare2 0.02 0.02 0.01 2.00 markshare_4_0 0.02 0.02 0.01 2.00 mas74 0.03 0.02 0.01 2.00 mas76 0.02 0.02 0.01 2.00 neos-3046615-murg 0.02 0.02 0.01 2.00 neos-3754480-nidda 0.02 0.02 0.01 2.00 neos-4338804-snowy 0.03 0.02 0.01 2.00 neos-4954672-berkel 0.02 0.02 0.01 2.00 neos-911970 0.03 0.02 0.01 2.00 neos5 0.02 0.02 0.01 2.00 pg 0.03 0.02 0.01 2.00 sp150x300d 0.02 0.02 0.01 2.00 supportcase26 0.03 0.02 0.01 2.00 tr12-30 0.03 0.02 0.01 2.00 mzzv42z 10.10 1.91 0.95 2.01 radiationm40-10-02 1.58 1.44 0.71 2.03 supportcase33 0.99 1.12 0.55 2.04 eilA101-2 2.68 2.86 1.39 2.06 fiball 0.71 0.30 0.14 2.14 chromaticindex1024-7 45.86 201.94 93.65 2.16 roll3000 0.12 0.13 0.06 2.17 n2seq36q 0.46 0.53 0.24 2.21 nursesched-sprint02 0.38 0.51 0.23 2.22 nw04 0.44 1.38 0.61 2.26 physiciansched6-2 11.15 15.62 6.72 2.32 seymour 0.83 0.89 0.38 2.34 ns1830653 0.42 0.33 0.14 2.36 seymour1 0.83 0.91 0.38 2.39 rmatr100-p10 0.27 0.34 0.14 2.43 nursesched-medium-hint03 10.47 11.61 4.73 2.45 neos-1171448 2.35 2.08 0.84 2.48 splice1k1 21.62 22.71 9.17 2.48 neos-4387871-tavua 0.10 0.10 0.04 2.50 nu25-pr12 0.05 0.05 0.02 2.50 unitcal_7 0.96 1.00 0.39 2.56 glass-sc 0.31 0.31 0.12 2.58 buildingenergy 300.00 300.00 115.61 2.59 sing44 9.62 14.79 5.69 2.60 bab2 300.00 76.29 28.97 2.63 graph20-20-1rand 0.23 0.32 0.12 2.67 rail01 223.84 196.04 71.23 2.75 bab6 143.43 38.82 14.06 2.76 CMS750_4 0.38 0.39 0.14 2.79 map16715-04 13.26 19.01 6.80 2.80 neos-2978193-inde 0.16 0.14 0.05 2.80 sorrell3 1.07 1.13 0.40 2.82 assign1-5-8 0.03 0.03 0.01 3.00 binkar10_1 0.02 0.03 0.01 3.00 csched008 0.11 0.09 0.03 3.00 ic97_potential 0.02 0.03 0.01 3.00 lotsize 0.03 0.03 0.01 3.00 mik-250-20-75-4 0.02 0.03 0.01 3.00 neos-2657525-crna 0.03 0.03 0.01 3.00 neos-3627168-kasai 0.04 0.03 0.01 3.00 p200x1188c 0.03 0.03 0.01 3.00 pg5_34 0.03 0.03 0.01 3.00 sing326 9.55 14.16 4.69 3.02 cbs-cta 0.42 0.32 0.10 3.20 csched007 0.19 0.16 0.05 3.20 uccase9 11.54 17.21 5.20 3.31 map10 11.08 21.74 6.25 3.48 neos-1171737 0.70 0.70 0.20 3.50 fastxgemm-n2r6s0t2 0.18 0.29 0.08 3.62 neos-957323 300.00 28.38 7.74 3.67 neos-1445765 0.16 0.35 0.09 3.89 gmu-35-40 0.03 0.04 0.01 4.00 neos17 0.03 0.04 0.01 4.00 s250r10 300.00 300.00 71.74 4.18 neos-1354092 300.00 300.00 70.92 4.23 milo-v12-6-r2-40-1 0.25 0.22 0.05 4.40 ns1208400 3.96 1.60 0.36 4.44 app1-1 0.08 0.09 0.02 4.50 neos-950242 1.05 0.81 0.18 4.50 uccase12 72.32 6.47 1.38 4.69 peg-solitaire-a3 1.88 2.54 0.52 4.88 beasleyC3 0.05 0.05 0.01 5.00 gmu-35-50 0.05 0.05 0.01 5.00 irish-electricity 181.58 300.00 59.53 5.04 academictimetablesmall 14.89 4.15 0.82 5.06 ex10 300.00 300.00 59.25 5.06 dano3_3 46.88 96.86 19.06 5.08 dano3_5 46.79 96.97 19.03 5.10 gfd-schedulen180f7d50m30k18 80.00 39.02 6.81 5.73 neos-2075418-temuka 128.33 300.00 50.57 5.93 mc11 0.06 0.06 0.01 6.00 neos-933966 16.79 18.35 2.80 6.55 neos-827175 9.36 1.91 0.29 6.59 neos-3656078-kumeu 2.37 1.57 0.23 6.83 app1-2 5.03 5.33 0.70 7.61 snp-02-004-104 14.18 23.47 2.83 8.29 neos-4413714-turia 3.97 16.53 1.93 8.56 neos-631710 300.00 300.00 31.97 9.38 satellites2-40 29.70 125.33 10.51 11.92 brazil3 300.00 93.59 7.24 12.93 ex9 300.00 300.00 14.07 21.32 satellites2-60-fs 4.17 171.22 3.34 51.26 ------------------------------------------------------------------------- Geomean Baseline/v5: 1.1452 Shifted(+1s): 1.0759 Geomean v5/HiGHS: 1.5487 Shifted(+1s): 1.1823 (240 problems) --- .../bound_flipping_ratio_test.cpp | 465 +++++---- .../bound_flipping_ratio_test.hpp | 39 +- cpp/src/dual_simplex/phase2.cpp | 898 ++++++++++++------ 3 files changed, 908 insertions(+), 494 deletions(-) diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index 3fbfbd1f82..1086b23ddf 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -11,12 +11,14 @@ #include #include +#include namespace cuopt::mathematical_optimization::simplex { template i_t bound_flipping_ratio_test_t::compute_breakpoints(std::vector& indicies, - std::vector& ratios) + std::vector& ratios, + std::vector& harris_ratios) { i_t n = n_; i_t m = m_; @@ -34,19 +36,21 @@ i_t bound_flipping_ratio_test_t::compute_breakpoints(std::vector& if (vstatus_[j] == variable_status_t::NONBASIC_FIXED) { continue; } if (vstatus_[j] == variable_status_t::NONBASIC_LOWER && delta_z_[j] < -pivot_tol) { indicies[idx] = k; - ratios[idx] = std::max((-dual_tol - z_[j]) / delta_z_[j], 0.0); + ratios[idx] = std::max((-z_[j]) / delta_z_[j], 0.0); + harris_ratios[idx] = std::max((-dual_tol - z_[j]) / delta_z_[j], 0.0); if constexpr (verbose) { settings_.log.printf("ratios[%d] = %e\n", idx, ratios[idx]); } idx++; } if (vstatus_[j] == variable_status_t::NONBASIC_UPPER && delta_z_[j] > pivot_tol) { indicies[idx] = k; - ratios[idx] = std::max((dual_tol - z_[j]) / delta_z_[j], 0.0); + ratios[idx] = std::max((-z_[j]) / delta_z_[j], 0.0); + harris_ratios[idx] = std::max((dual_tol - z_[j]) / delta_z_[j], 0.0); if constexpr (verbose) { settings_.log.printf("ratios[%d] = %e\n", idx, ratios[idx]); } idx++; } } - work_estimate_ += 4 * nz; - work_estimate_ += 4 * idx; + work_estimate_ += 5 * nz; + work_estimate_ += 5 * idx; pivot_tol /= 10; } return idx; @@ -60,7 +64,8 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, f_t& slope, f_t& step_length, i_t& nonbasic_entering, - i_t& entering_index) + i_t& entering_index, + f_t& max_val) { // Find the minimum ratio f_t min_val = inf; @@ -68,27 +73,21 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, i_t candidate = -1; f_t zero_tol = settings_.zero_tol; i_t k_idx = -1; + max_val = 0.0; i_t min_found = 0; - i_t harris_found = 0; for (i_t k = start; k < end; ++k) { if (ratios[k] < min_val) { min_val = ratios[k]; candidate = indicies[k]; k_idx = k; min_found++; - } else if (ratios[k] < min_val + zero_tol) { - // Use Harris to select variables with larger pivots - const i_t j = nonbasic_list_[indicies[k]]; - if (std::abs(delta_z_[j]) > std::abs(delta_z_[candidate])) { - min_val = ratios[k]; - candidate = indicies[k]; - k_idx = k; - } - harris_found++; + } + if (ratios[k] > max_val) { + max_val = ratios[k]; } } - work_estimate_ += (end - start) + 2 * min_found + 6 * harris_found; + work_estimate_ += (end - start) + 2 * min_found; step_length = min_val; nonbasic_entering = candidate; @@ -125,8 +124,19 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // Compute the initial set of breakpoints std::vector indicies(nz); std::vector ratios(nz); - work_estimate_ += 2 * nz; - i_t num_breakpoints = compute_breakpoints(indicies, ratios); + std::vector harris_ratios(nz); + work_estimate_ += 3 * nz; + double t0 = tic(); + i_t num_breakpoints = compute_breakpoints(indicies, ratios, harris_ratios); + time_compute_breakpoints_ += toc(t0); + num_breakpoints_ = num_breakpoints; + // Count zero ratios + num_harris_zero_ = 0; + num_exact_zero_ = 0; + for (i_t k = 0; k < num_breakpoints; k++) { + if (harris_ratios[k] == 0.0) num_harris_zero_++; + if (ratios[k] == 0.0) num_exact_zero_++; + } if constexpr (verbose) { settings_.log.printf("Initial breakpoints %d\n", num_breakpoints); } if (num_breakpoints == 0) { nonbasic_entering = -1; @@ -136,9 +146,12 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, f_t slope = slope_; nonbasic_entering = -1; i_t entering_index = RATIO_TEST_NO_ENTERING_VARIABLE; + f_t max_step_length; + t0 = tic(); i_t k_idx = single_pass( - 0, num_breakpoints, indicies, ratios, slope, step_length, nonbasic_entering, entering_index); + 0, num_breakpoints, indicies, harris_ratios, slope, step_length, nonbasic_entering, entering_index, max_step_length); + time_single_pass_ += toc(t0); if (k_idx == RATIO_TEST_NUMERICAL_ISSUES) { return RATIO_TEST_NUMERICAL_ISSUES; } bool continue_search = k_idx >= 0 && num_breakpoints > 1 && slope > 0.0; if (!continue_search) { @@ -150,6 +163,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, entering_index, std::abs(delta_z_[entering_index])); } + num_buckets_used_ = 0; + step_length_result_ = step_length; return entering_index; } @@ -162,188 +177,270 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, slope); } - // Continue the search using a heap to order the breakpoints - ratios[k_idx] = ratios[num_breakpoints - 1]; - indicies[k_idx] = indicies[num_breakpoints - 1]; - - constexpr bool use_bucket_pass = false; - - if (use_bucket_pass) { - f_t max_ratio = 0.0; - for (i_t k = 0; k < num_breakpoints - 1; ++k) { - if (ratios[k] > max_ratio) { max_ratio = ratios[k]; } + // This code is complicated. There are several important concepts that are needed to understand it. + // + // We are trying to compute the maximum step length we can take while: + // 1) Staying mostly dual feasible (we allow ourselves to be infeasible by dual_tol amount) + // 2) Increasing the dual objective + // 3) Selecting a variable with a large pivot (| delta_z[j] |) + // + // Let alpha be the step length. For each nonbasic variable j, we have + // z_j(alpha) = z_j + alpha * delta_z_j + // + // To stay dual feasible, we either need to keep + // z_j(alpha) >= 0, if j is on it's lower bound, or + // z_j(alpha) <= 0, if j is on it's upper bound. + // + // Consider the equation z_j(alpha) = z_j + alpha * delta_z_j = 0. Each variable j puts a bound on alpha: + // + // alpha_j <= -z_j / delta_z_j, if x_j = l_j (z_j >= 0) and delta_z_j < 0 + // alpha_j <= -z_j / delta_z_j, if x_j = u_j (z_j <= 0) and delta_z_j > 0 + // + // The code refers to these alpha_j as ratios, since they are the ratio of z_j to delta_z_j. + // + // Now we could take alpha = min_j alpha_j, and remain dual feasible. However, we are allowed to increase the step-length + // if j is a variable such that l_j <= x_j <= u_j. To see why imagine that our variable was currenlty on it's lower bound, + // with z_j > 0 and delta_z_j < 0, if we push alpha past alpha_j, than z_j(alpha) < 0. This is fine as long as we flip + // the variable to be on it's upper bound. Thus, we can push alpha past alpha_j for *bounded* variables. + // + // Note that this does not work if we try to increase alpha past alpha_j for a variable with a single bound. We would just + // be making ourselves dual infeasible. So we need to check whether a variable is bounded. + // + // The dual objective as a function of the step-length alpha, is piecewise linear and concave. The breakpoints of this + // piecewise linear function occur at each of the alpha_j values. + // We can keep increasing the step-length as long as the slope remains nonnegative. After that + // we must stop, because we could decrease the dual objective. So the code tracks the cumulative slope of the dual objective. + // + // Now we don't need to exactly feasible: z_j >= 0 if x_j = l_j and z_j <= 0 if x_j = u_j. We can violate these bounds by + // the dual feasibility tolerance eps. We allow ourselves to be infeasible if it would help us get a larger pivot + // (delta_z_j). Small pivots can cause numerical issues, so we would like to avoid them. + // + // With this tolerance we get the equations: + // z_j(alpha) = z_j + alpha * delta_z_j >= -eps if x_j = l_j + // z_j(alpha) = z_j + alpha * delta_z_j <= eps if x_j = u_j + // + // This gives bounds on alpha. We call these alpha_harris_j, for Paula Harris, who proposed this method. + // + // alpha_harris_j <= (-eps - z_j) / delta_z_j, if x_j = l_j and delta_z_j < 0 + // alpha_harris_j <= (eps - z_j) / delta_z_j, if x_j = u_j and delta_z_j > 0 + // + // Let alpha_harris = min_j alpha_harris_j. We can select the variable with the largest | delta_z_j | from those + // candidates { j | alpha_j <= alpha_harris }. + // + // We combine these two ideas (increasing the step length for bounded variables) and allowing ourselves to be slightly dual infeasible + // to choose a larger pivot. + // + // We partition the variables into buckets. Let B_k be the set of variables in bucket k. B_0 is defined as { j | alpha_j <= alpha_harris }. + // We then compute alpha_harris_1 = min_{j not in B_0} alpha_j. And B_1 is defined as { j not in B_0 | alpha_j <= alpha_harris_1 }. And + // so on. + // + // We want to balance two different things: + // 1) Taking a larger step length to increase the dual objective as much as possible, + // 2) Choosing a large pivot for numerical stability. + // + // Let max_pivot = max_j | delta_z_j |. We start working our way backward from the largest bucket to the smallest bucket, + // we choose a variable j that satisfies | delta_z_j | >= 0.1 * max_pivot. Since we can always choose a smaller step length + // for the sake of numerical stability. + // + // Now the final thing to understand is that the ratios alpha_j are not sorted in any particular order. And we don't want to + // pay the O(num_breakpoints * log(num_breakpoints)) cost of sorting them. + // + // So we set a threshold on the step-length and check if all variables j with alpha_j <= threshold have already caused the + // slope to go negative. If so, we just need to consider those candidate variables with alpha_j <= threshold. If not, we + // multiply the threshold by 10. This cost us O(log10(max_step_length/min_step_length) * num_breakpoints) time. So we aren't + // totally linear. But the hope is we are better than a sort. + + // Use a coarse filter to find candidates + f_t minimum_harris_ratio = step_length; + f_t coarse_threshold = (minimum_harris_ratio > 0.0) + ? std::min(10.0 * minimum_harris_ratio, max_step_length) + : max_step_length; + f_t total_slope = slope; + bool found_unbounded = false; + std::vector candidates(num_breakpoints); + std::iota(candidates.begin(), candidates.end(), 0); + work_estimate_ += 2 * num_breakpoints; + i_t scan_start = 0; + i_t num_candidates = 0; + + // This is O( log10(max_step_length/min_step_length) * num_breakpoints) + t0 = tic(); + while (total_slope >= 0.0 && coarse_threshold <= max_step_length && !found_unbounded) { + for (i_t h = scan_start; h < num_breakpoints; ++h) { + const i_t k = candidates[h]; + if (ratios[k] <= coarse_threshold) { + // Candidate is less than coarse threshold, move it to the front of the candidate list + std::swap(candidates[h], candidates[num_candidates]); + num_candidates++; + const i_t j = nonbasic_list_[indicies[k]]; + if (!bounded_variables_[j]) { + found_unbounded = true; + } else { + total_slope -= std::abs(delta_z_[j]) * (upper_[j] - lower_[j]); + } + } } - work_estimate_ += 2 * num_breakpoints; - settings_.log.printf( - "Starting heap passes. %d breakpoints max ratio %e\n", num_breakpoints - 1, max_ratio); - bucket_pass( - indicies, ratios, num_breakpoints - 1, slope, step_length, nonbasic_entering, entering_index); + work_estimate_ += 3 * (num_breakpoints - scan_start); + work_estimate_ += 8 * (num_candidates - scan_start); + scan_start = num_candidates; + coarse_threshold *= 10.0; } + time_coarse_filter_ += toc(t0); - heap_passes( - indicies, ratios, num_breakpoints - 1, slope, step_length, nonbasic_entering, entering_index); - - if constexpr (verbose) { - settings_.log.printf("BFRT step length %e entering index %d non basic entering %d pivot %e\n", - step_length, - entering_index, - nonbasic_entering, - std::abs(delta_z_[entering_index])); - } - return entering_index; -} + candidates.resize(num_candidates); -template -void bound_flipping_ratio_test_t::heap_passes(const std::vector& current_indicies, - const std::vector& current_ratios, - i_t num_breakpoints, - f_t& slope, - f_t& step_length, - i_t& nonbasic_entering, - i_t& entering_index) -{ - std::vector bare_idx(num_breakpoints); - constexpr bool verbose = false; - const f_t dual_tol = settings_.dual_tol; - const f_t zero_tol = settings_.zero_tol; - const std::vector& delta_z = delta_z_; - const std::vector& nonbasic_list = nonbasic_list_; - const i_t N = num_breakpoints; - for (i_t k = 0; k < N; ++k) { - bare_idx[k] = k; - if constexpr (verbose) { - settings_.log.printf("Adding index %d ratio %e pivot %e to heap\n", - current_indicies[k], - current_ratios[k], - std::abs(delta_z[nonbasic_list[current_indicies[k]]])); + // Check for variables with one sided bounds. These define the maximum step length. + if (found_unbounded) { + for (i_t h = 0; h < num_candidates; h++) { + const i_t k = candidates[h]; + const i_t j = nonbasic_list_[indicies[k]]; + if (!bounded_variables_[j]) { + max_step_length = std::min(max_step_length, harris_ratios[k]); + } } - } - work_estimate_ += N; - - auto compare = [zero_tol, ¤t_ratios, ¤t_indicies, &delta_z, &nonbasic_list]( - const i_t& a, const i_t& b) { - return (current_ratios[a] > current_ratios[b]) || - (current_ratios[b] - current_ratios[a] < zero_tol && - std::abs(delta_z[nonbasic_list[current_indicies[a]]]) > - std::abs(delta_z[nonbasic_list[current_indicies[b]]])); - }; - - std::make_heap(bare_idx.begin(), bare_idx.end(), compare); - work_estimate_ += 10 * bare_idx.size(); - - while (bare_idx.size() > 0 && slope > 0) { - // Remove minimum ratio from the heap and rebalance - i_t heap_index = bare_idx.front(); - std::pop_heap(bare_idx.begin(), bare_idx.end(), compare); - bare_idx.pop_back(); - work_estimate_ += 7 * std::log2(bare_idx.size() + 1); - - nonbasic_entering = current_indicies[heap_index]; - const i_t j = entering_index = nonbasic_list_[nonbasic_entering]; - step_length = current_ratios[heap_index]; - - if (bounded_variables_[j]) { - // We have a bounded variable - const f_t interval = upper_[j] - lower_[j]; - const f_t delta_slope = std::abs(delta_z_[j]) * interval; - const f_t pivot = std::abs(delta_z[j]); - if constexpr (verbose) { - settings_.log.printf( - "heap %d step-length %.12e pivot %e nonbasic entering %d slope %e delta_slope %e new " - "slope %e\n", - bare_idx.size(), - current_ratios[heap_index], - pivot, - nonbasic_entering, - slope, - delta_slope, - slope - delta_slope); + work_estimate_ += 5 * num_candidates; + + // Remove candidates that are greater than the maximum step length + work_estimate_ += 2 * candidates.size(); + for (i_t h = static_cast(candidates.size()) - 1; h >= 0; h--) { + const i_t k = candidates[h]; + const f_t ratio = ratios[k]; + if (ratio > max_step_length) { + // Swap with the last candidate and remove + candidates[h] = candidates.back(); + candidates.pop_back(); } - slope -= delta_slope; - } else { - // The variable is not bounded. Stop the search. - break; } - work_estimate_ += 10; + num_candidates = candidates.size(); + } - if (toc(start_time_) > settings_.time_limit) { - entering_index = RATIO_TEST_TIME_LIMIT; - return; + // Use a bucket sort to partition candidates into buckets by successive Harris breakpoints + // bucket_start[k] = index in candidates[] where bucket k starts + // Bucket k contains candidates[bucket_start[k]] .. candidates[bucket_start[k+1] - 1] + f_t threshold = minimum_harris_ratio; + i_t num_buckets = 0; + std::vector bucket_start(num_candidates + 1, 0); + f_t cumulative_slope = slope; + scan_start = 0; + work_estimate_ += num_candidates + 1; + + // This is O(num_buckets * num_candidates) + i_t slope_breaker_k = -1; // the candidate k that made slope go negative + t0 = tic(); + while (cumulative_slope >= 0.0 && scan_start < num_candidates && threshold <= max_step_length) { + f_t next_threshold = inf; + i_t write = scan_start; + + work_estimate_ += 7 * (num_candidates - scan_start); + for (i_t h = scan_start; h < num_candidates; h++) { + const i_t k = candidates[h]; + const f_t ratio = ratios[k]; + + if (ratio <= threshold) { + const i_t j = nonbasic_list_[indicies[k]]; + if (bounded_variables_[j]) { + cumulative_slope -= std::abs(delta_z_[j]) * (upper_[j] - lower_[j]); + if (cumulative_slope < 0.0 && slope_breaker_k < 0) { + slope_breaker_k = k; + } + } + std::swap(candidates[h], candidates[write]); + write++; + } else { + const i_t j = nonbasic_list_[indicies[k]]; + const f_t harris_ratio = harris_ratios[k]; + next_threshold = std::min(next_threshold, harris_ratio); + } } - if (settings_.concurrent_halt != nullptr && *settings_.concurrent_halt == 1) { - entering_index = CONCURRENT_HALT_RETURN; - return; + + + bucket_start[++num_buckets] = write; + if (write == scan_start) break; // No progress — prevent infinite loop + scan_start = write; + threshold = next_threshold; + + if (cumulative_slope < 0.0) break; + } + time_bucket_sort_ += toc(t0); + bucket0_size_ = (num_buckets > 0) ? bucket_start[1] : 0; + + // Compute the maximum pivot + // This is O(num_candidates) + f_t max_pivot = 0.0; + for (i_t h = 0; h < bucket_start[num_buckets]; h++) { + const i_t k = candidates[h]; + const i_t j = nonbasic_list_[indicies[k]]; + const f_t pivot = std::abs(delta_z_[j]); + if (pivot > max_pivot) { + max_pivot = pivot; } } -} - -template -void bound_flipping_ratio_test_t::bucket_pass(const std::vector& current_indicies, - const std::vector& current_ratios, - i_t num_breakpoints, - f_t& slope, - f_t& step_length, - i_t& nonbasic_entering, - i_t& entering_index) -{ - const f_t dual_tol = settings_.dual_tol; - const f_t zero_tol = settings_.zero_tol; - const std::vector& delta_z = delta_z_; - const std::vector& nonbasic_list = nonbasic_list_; - const i_t N = num_breakpoints; - - const i_t K = 400; // 0, -16, -15, ...., 0, 1, ...., 400 - 18 = 382 - std::vector buckets(K, 0.0); - std::vector bucket_count(K, 0); - for (i_t k = 0; k < N; ++k) { - const i_t idx = current_indicies[k]; - const f_t ratio = current_ratios[k]; - const f_t min_exponent = -16.0; - const f_t max_exponent = 382.0; - const f_t exponent = std::max(min_exponent, std::min(max_exponent, std::log10(ratio))); - const i_t bucket_idx = ratio == 0.0 ? 0 : static_cast(exponent - min_exponent + 1); - // settings_.log.printf("Ratio %e exponent %e bucket_idx %d\n", ratio, exponent, bucket_idx); - const i_t j = nonbasic_list[idx]; - const f_t interval = upper_[j] - lower_[j]; - const f_t delta_slope = std::abs(delta_z_[j]) * interval; - buckets[bucket_idx] += delta_slope; - bucket_count[bucket_idx]++; + work_estimate_ += 4 * bucket_start[num_buckets]; + + // Select the entering variable + // Scan from last bucket to first. Within each bucket, pick the variable with + // the largest ratio (step length) that has |delta_z| > pivot_threshold + f_t pivot_threshold = std::max(settings_.pivot_tol, std::min(0.1 * max_pivot, 1.0)); + i_t entering_k = -1; + + // This is O(num_candidates) + for (i_t b = num_buckets - 1; b >= 0; b--) { + const i_t b_start = bucket_start[b]; + const i_t b_end = bucket_start[b + 1]; + f_t best_ratio = -1.0; + for (i_t h = b_start; h < b_end; h++) { + const i_t k = candidates[h]; + const i_t j = nonbasic_list_[indicies[k]]; + const f_t pivot = std::abs(delta_z_[j]); + if (pivot > pivot_threshold && ratios[k] > best_ratio) { + best_ratio = ratios[k]; + entering_k = k; + } + } + work_estimate_ += 4 * (bucket_start[b + 1] - bucket_start[b]); + if (entering_k >= 0) break; } - - std::vector cumulative_sum(K, 0.0); - cumulative_sum[0] = buckets[0]; - if (cumulative_sum[0] > slope) { - settings_.log.printf( - "Bucket 0. Count in bucket %d. Slope %e. Cumulative sum %e. Bucket value %e\n", - bucket_count[0], - slope, - cumulative_sum[0], - buckets[0]); - return; + work_estimate_ += 2 * num_buckets; + + // Step = entering variable's breakpoint ratio + num_buckets_used_ = num_buckets; + if (entering_k < 0) { + // Fallback to single_pass result + used_fallback_ = true; + bucket_selected_ = -1; + step_length_result_ = step_length; + selected_is_slope_breaker_ = false; + return entering_index; } - i_t k; - bool exceeded = false; - for (k = 1; k < K; ++k) { - cumulative_sum[k] = cumulative_sum[k - 1] + buckets[k]; - if (cumulative_sum[k] > slope) { - exceeded = true; - break; + step_length = ratios[entering_k]; + nonbasic_entering = indicies[entering_k]; + entering_index = nonbasic_list_[nonbasic_entering]; + + // Record whether we selected the slope breaker + selected_is_slope_breaker_ = (entering_k == slope_breaker_k); + + // Record which bucket was selected + used_fallback_ = false; + for (i_t b = 0; b < num_buckets; b++) { + if (entering_k >= 0) { + // Find which bucket entering_k is in based on its position in candidates + i_t pos = -1; + for (i_t h = 0; h < num_candidates; h++) { + if (candidates[h] == entering_k) { pos = h; break; } + } + if (pos >= bucket_start[b] && pos < bucket_start[b + 1]) { + bucket_selected_ = b; + break; + } } } + step_length_result_ = step_length; + + return entering_index; - if (exceeded) { - settings_.log.printf( - "Value in bucket %d. Count in buckets %d. Slope %e. Cumulative sum %e. Next sum %e Bucket " - "value %e\n", - k, - bucket_count[k], - slope, - cumulative_sum[k - 1], - cumulative_sum[k], - buckets[k - 1]); - } } + #ifdef DUAL_SIMPLEX_INSTANTIATE_DOUBLE template class bound_flipping_ratio_test_t; diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp index 2f73069451..6e9dc647c1 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp @@ -56,8 +56,26 @@ class bound_flipping_ratio_test_t { i_t compute_step_length(f_t& step_length, i_t& nonbasic_entering); f_t work_estimate() const { return work_estimate_; } + // Timing fields (filled by compute_step_length) + f_t time_compute_breakpoints_{0.0}; + f_t time_single_pass_{0.0}; + f_t time_coarse_filter_{0.0}; + f_t time_bucket_sort_{0.0}; + f_t time_pivot_selection_{0.0}; + + // Diagnostic fields + i_t num_buckets_used_{0}; // number of buckets in bucket sort + i_t bucket_selected_{-1}; // which bucket the entering variable came from (-1 = single_pass/fallback) + f_t step_length_result_{0.0}; // the step length chosen + bool used_fallback_{false}; // true if we fell back to single_pass result + i_t bucket0_size_{0}; // size of first bucket (candidates with ratio <= min_harris) + i_t num_breakpoints_{0}; // total breakpoints computed + bool selected_is_slope_breaker_{false}; // true if we selected the variable that made slope go negative + i_t num_harris_zero_{0}; // number of harris_ratios that are exactly 0 + i_t num_exact_zero_{0}; // number of exact ratios that are exactly 0 + private: - i_t compute_breakpoints(std::vector& indices, std::vector& ratios); + i_t compute_breakpoints(std::vector& indices, std::vector& ratios, std::vector& harris_ratios); i_t single_pass(i_t start, i_t end, const std::vector& indices, @@ -65,23 +83,8 @@ class bound_flipping_ratio_test_t { f_t& slope, f_t& step_length, i_t& nonbasic_entering, - i_t& enetering_index); - void heap_passes(const std::vector& current_indicies, - const std::vector& current_ratios, - i_t num_breakpoints, - f_t& slope, - f_t& step_lenght, - i_t& nonbasic_entering, - i_t& entering_index); - - void bucket_pass(const std::vector& current_indicies, - const std::vector& current_ratios, - i_t num_breakpoints, - f_t& slope, - f_t& step_length, - i_t& nonbasic_entering, - i_t& entering_index); - + i_t& entering_index, + f_t& max_val); const std::vector& lower_; const std::vector& upper_; const std::vector& bounded_variables_; diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index e2a3c17d11..0b94a67d28 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -464,7 +464,7 @@ void initial_perturbation(const lp_problem_t& lp, max_abs_obj_coeff = std::max(max_abs_obj_coeff, std::abs(lp.objective[j])); } - // Dampen large costs + // Dampen large costs if (max_abs_obj_coeff > 100.0) { max_abs_obj_coeff = std::sqrt(std::sqrt(max_abs_obj_coeff)); } @@ -503,16 +503,15 @@ void initial_perturbation(const lp_problem_t& lp, if (lower == upper || (lower == -inf && upper == inf)) { continue; } - // Skip basic variables - if (vstatus[j] == variable_status_t::BASIC) { - continue; - } const f_t rand_val = random.random(); const f_t cost_factor = std::min(std::abs(obj) + 1.0, max_abs_obj_coeff + 1.0); const f_t perturb = (1.0 + rand_val) * cost_factor * perturbation_base; - if (vstatus[j] == variable_status_t::NONBASIC_LOWER || + if (vstatus[j] == variable_status_t::BASIC) { + // Skip basic variables + continue; + } else if (vstatus[j] == variable_status_t::NONBASIC_LOWER || vstatus[j] == variable_status_t::NONBASIC_FIXED) { // NONBASIC_FIXED from phase 1 for boxed variables — treat as at lower bound objective[j] = obj + perturb; @@ -1572,35 +1571,49 @@ i_t check_steepest_edge_norms(const simplex_solver_settings_t& setting // Remove the perturbation from a variable that is leaving the basis. Since it // is nonbasic, its cost affects only its own reduced cost. If removing the -// perturbation would violate dual feasibility, apply just enough perturbation -// to maintain feasibility (for one-sided variables) or leave it unperturbed -// (for boxed variables, which can be flipped). +// perturbation would violate dual feasibility, the perturbation is left in +// place (for boxed variables) or reduced to the minimum needed (for one-sided +// variables). template void remove_leaving_perturbation(const lp_problem_t& lp, const simplex_solver_settings_t& settings, i_t leaving_index, + i_t direction, std::vector& z, std::vector& objective) { const f_t perturb = objective[leaving_index] - lp.objective[leaving_index]; if (perturb == 0.0) return; - z[leaving_index] -= perturb; - objective[leaving_index] = lp.objective[leaving_index]; - - // Restore dual feasibility if needed const f_t lower = lp.lower[leaving_index]; const f_t upper = lp.upper[leaving_index]; - if (upper == inf && lower > -inf && z[leaving_index] < -settings.tight_tol) { - // At lower bound, needs z >= 0 - const f_t correction = -z[leaving_index]; - z[leaving_index] = 0.0; - objective[leaving_index] += correction; - } else if (lower == -inf && upper < inf && z[leaving_index] > settings.tight_tol) { - // At upper bound, needs z <= 0 - const f_t correction = z[leaving_index]; - z[leaving_index] = 0.0; - objective[leaving_index] -= correction; + const bool boxed = (lower > -inf && upper < inf); + + if (boxed) { + // Only remove if it won't create dual infeasibility. + // direction=1 means going to lower bound (needs z >= 0 after removal) + // direction=-1 means going to upper bound (needs z <= 0 after removal) + const f_t new_z = z[leaving_index] - perturb; + if (direction == 1 && new_z < -settings.tight_tol) { return; } + if (direction == -1 && new_z > settings.tight_tol) { return; } + z[leaving_index] = new_z; + objective[leaving_index] = lp.objective[leaving_index]; + } else { + z[leaving_index] -= perturb; + objective[leaving_index] = lp.objective[leaving_index]; + + // Restore dual feasibility if needed for one-sided variables + if (upper == inf && lower > -inf && z[leaving_index] < -settings.tight_tol) { + // At lower bound, needs z >= 0 + const f_t correction = -z[leaving_index]; + z[leaving_index] = 0.0; + objective[leaving_index] += correction; + } else if (lower == -inf && upper < inf && z[leaving_index] > settings.tight_tol) { + // At upper bound, needs z <= 0 + const f_t correction = z[leaving_index]; + z[leaving_index] = 0.0; + objective[leaving_index] -= correction; + } } } @@ -1608,9 +1621,12 @@ template i_t compute_perturbation(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& delta_z_indices, + const std::vector& vstatus, std::vector& z, std::vector& objective, f_t& sum_perturb, + i_t entering_index, + f_t step_length, f_t& work_estimate) { const i_t n = lp.num_cols; @@ -1626,32 +1642,27 @@ i_t compute_perturbation(const lp_problem_t& lp, objective[j] += violation; num_perturb++; sum_perturb += violation; -#ifdef PERTURBATION_DEBUG - if (violation > 1e-1) { - settings.log.printf( - "perturbation: violation %e j %d lower %e\n", violation, j, lp.lower[j]); - } -#endif } else if (lp.lower[j] == -inf && lp.upper[j] < inf && z[j] > tight_tol) { const f_t violation = z[j]; z[j] -= violation; // z[j] <- 0 objective[j] -= violation; num_perturb++; sum_perturb += violation; -#ifdef PERTURBATION_DEWBUG - if (violation > 1e-1) { - settings.log.printf( - "perturbation: violation %e j %d upper %e\n", violation, j, lp.upper[j]); - } -#endif } } - work_estimate += 7 * delta_z_indices.size(); -#ifdef PERTURBATION_DEBUG - if (num_perturb > 0) { - settings.log.printf("Perturbed %d dual variables by %e\n", num_perturb, sum_perturb); + // On degenerate steps, shift the entering variable's cost (like HiGHS) + // This accumulates shifts that break degeneracy at the next refactorization + if (entering_index >= 0 && step_length == 0.0) { + assert(vstatus[entering_index] != variable_status_t::BASIC); + const f_t shift = -z[entering_index]; + if (shift != 0.0) { + objective[entering_index] += shift; + z[entering_index] = 0.0; + sum_perturb += std::abs(shift); + num_perturb++; + } } -#endif + work_estimate += 7 * delta_z_indices.size(); return 0; } @@ -2273,19 +2284,26 @@ void bound_info(const lp_problem_t& lp, } template -void set_primal_variables_on_bounds(const lp_problem_t& lp, +i_t set_primal_variables_on_bounds(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& z, std::vector& vstatus, - std::vector& x) + std::vector& x, + i_t degen_type = 0) { PHASE2_NVTX_RANGE("DualSimplex::set_primal_variables_on_bounds"); const i_t n = lp.num_cols; f_t tol = 1e-10; + i_t num_fixed_to_lower = 0; + i_t num_fixed_to_upper = 0; + i_t num_lower_to_upper = 0; + i_t num_upper_to_lower = 0; + i_t num_set_fixed = 0; for (i_t j = 0; j < n; ++j) { // We set z_j = 0 for basic variables // But we explicitally skip setting basic variables here if (vstatus[j] == variable_status_t::BASIC) { continue; } + const variable_status_t old_vstatus = vstatus[j]; // We will flip the status of variables between nonbasic lower and nonbasic // upper here to improve dual feasibility const f_t fixed_tolerance = settings.fixed_tol; @@ -2306,29 +2324,69 @@ void set_primal_variables_on_bounds(const lp_problem_t& lp, vstatus[j] == variable_status_t::NONBASIC_UPPER) { x[j] = lp.upper[j]; } else if (z[j] >= 0 && lp.lower[j] > -inf) { - if (vstatus[j] != variable_status_t::NONBASIC_LOWER) { - settings.log.debug( - "Setting nonbasic lower variable (zj %e) %d to %e (current %e). vstatus %d\n", - z[j], - j, - lp.lower[j], - x[j], - static_cast(vstatus[j])); + // For boxed variables with degenerate z, use heuristic based on degen_type + if (degen_type >= 1 && std::abs(z[j]) < settings.dual_tol && lp.upper[j] < inf) { + if (degen_type == 1) { + // Column-sum heuristic + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + f_t col_sum = 0.0; + for (i_t k = col_start; k < col_end; k++) { + col_sum += lp.A.x[k]; + } + if (col_sum < 0.0) { + x[j] = lp.upper[j]; + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } else { + x[j] = lp.lower[j]; + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } + } else { + // degen_type == 3: abs_bound (prefer bound closer to zero, like HiGHS) + if (std::abs(lp.upper[j]) < std::abs(lp.lower[j])) { + x[j] = lp.upper[j]; + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } else { + x[j] = lp.lower[j]; + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } + } + } else { + x[j] = lp.lower[j]; + vstatus[j] = variable_status_t::NONBASIC_LOWER; } - x[j] = lp.lower[j]; - vstatus[j] = variable_status_t::NONBASIC_LOWER; } else if (z[j] <= 0 && lp.upper[j] < inf) { - if (vstatus[j] != variable_status_t::NONBASIC_UPPER) { - settings.log.debug( - "Setting nonbasic upper variable (zj %e) %d to %e (current %e). vstatus %d\n", - z[j], - j, - lp.upper[j], - x[j], - static_cast(vstatus[j])); + // For boxed variables with degenerate z, use heuristic based on degen_type + if (degen_type >= 1 && std::abs(z[j]) < settings.dual_tol && lp.lower[j] > -inf) { + if (degen_type == 1) { + // Column-sum heuristic + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + f_t col_sum = 0.0; + for (i_t k = col_start; k < col_end; k++) { + col_sum += lp.A.x[k]; + } + if (col_sum > 0.0) { + x[j] = lp.lower[j]; + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } else { + x[j] = lp.upper[j]; + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } + } else { + // degen_type == 3: abs_bound (prefer bound closer to zero) + if (std::abs(lp.lower[j]) < std::abs(lp.upper[j])) { + x[j] = lp.lower[j]; + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } else { + x[j] = lp.upper[j]; + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } + } + } else { + x[j] = lp.upper[j]; + vstatus[j] = variable_status_t::NONBASIC_UPPER; } - x[j] = lp.upper[j]; - vstatus[j] = variable_status_t::NONBASIC_UPPER; } else if (lp.upper[j] == inf && lp.lower[j] > -inf && z[j] < 0) { // dual infeasible if (vstatus[j] != variable_status_t::NONBASIC_LOWER) { @@ -2362,7 +2420,21 @@ void set_primal_variables_on_bounds(const lp_problem_t& lp, } else { assert(1 == 0); } + // Track changes + if (old_vstatus != vstatus[j]) { + if (old_vstatus == variable_status_t::NONBASIC_FIXED && vstatus[j] == variable_status_t::NONBASIC_LOWER) num_fixed_to_lower++; + else if (old_vstatus == variable_status_t::NONBASIC_FIXED && vstatus[j] == variable_status_t::NONBASIC_UPPER) num_fixed_to_upper++; + else if (old_vstatus == variable_status_t::NONBASIC_LOWER && vstatus[j] == variable_status_t::NONBASIC_UPPER) num_lower_to_upper++; + else if (old_vstatus == variable_status_t::NONBASIC_UPPER && vstatus[j] == variable_status_t::NONBASIC_LOWER) num_upper_to_lower++; + else if (vstatus[j] == variable_status_t::NONBASIC_FIXED) num_set_fixed++; + } } + i_t total_changes = num_fixed_to_lower + num_fixed_to_upper + num_lower_to_upper + num_upper_to_lower + num_set_fixed; + if (total_changes > 0) { + settings.log.printf("set_primal_variables_on_bounds: %d changes (fixed->lower=%d, fixed->upper=%d, lower->upper=%d, upper->lower=%d, ->fixed=%d)\n", + total_changes, num_fixed_to_lower, num_fixed_to_upper, num_lower_to_upper, num_upper_to_lower, num_set_fixed); + } + return total_changes; } template @@ -2387,6 +2459,180 @@ f_t amount_of_perturbation(const lp_problem_t& lp, const std::vector +i_t attempt_to_remove_perturbations(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + basis_update_mpf_t& ft, + const std::vector& basic_list, + const std::vector& nonbasic_list, + std::vector& vstatus, + std::vector& objective, + std::vector& z, + std::vector& y, + std::vector& x, + std::vector& xB_workspace, + std::vector& squared_infeasibilities, + std::vector& infeasibility_indices, + f_t& primal_infeasibility, + f_t& primal_infeasibility_squared, + f_t& work_estimate) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + const i_t n_minus_m = n - m; + + // Check if there's any perturbation + const f_t perturbation = amount_of_perturbation(lp, objective); + if (perturbation <= 1e-6) return 0; // OPTIMAL + + // Count perturbations on basic vs nonbasic variables + i_t num_basic_perturbed = 0; + i_t num_nonbasic_boxed_perturbed = 0; + i_t num_nonbasic_other_perturbed = 0; + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + if (objective[j] != lp.objective[j]) num_basic_perturbed++; + } + for (i_t k = 0; k < n_minus_m; ++k) { + const i_t j = nonbasic_list[k]; + if (objective[j] != lp.objective[j]) { + const f_t lower = lp.lower[j]; + const f_t upper = lp.upper[j]; + if (lower > -inf && upper < inf && lower != upper) { + num_nonbasic_boxed_perturbed++; + } else { + num_nonbasic_other_perturbed++; + } + } + } + + if (num_basic_perturbed == 0 && num_nonbasic_other_perturbed == 0) { + // Safe path: perturbation only on nonbasic boxed variables. + // y is unaffected; z[j] - perturb gives exact unperturbed reduced cost. + i_t num_flipped = 0; + for (i_t k = 0; k < n_minus_m; ++k) { + const i_t j = nonbasic_list[k]; + const f_t perturb = objective[j] - lp.objective[j]; + if (perturb == 0.0) continue; + const f_t new_z = z[j] - perturb; + if (vstatus[j] == variable_status_t::NONBASIC_LOWER && new_z < -settings.dual_tol) { + vstatus[j] = variable_status_t::NONBASIC_UPPER; + z[j] = new_z; + objective[j] = lp.objective[j]; + num_flipped++; + } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER && new_z > settings.dual_tol) { + vstatus[j] = variable_status_t::NONBASIC_LOWER; + z[j] = new_z; + objective[j] = lp.objective[j]; + num_flipped++; + } else { + z[j] = new_z; + objective[j] = lp.objective[j]; + } + } + work_estimate += 5 * n_minus_m; + + // Recompute x_B with flipped statuses + compute_primal_solution_from_basis(lp, ft, basic_list, nonbasic_list, vstatus, x, xB_workspace, work_estimate); + work_estimate += 2 * n; + primal_infeasibility_squared = + compute_initial_primal_infeasibilities(lp, settings, basic_list, x, + squared_infeasibilities, infeasibility_indices, + primal_infeasibility); + work_estimate += 4 * m + 2 * n; + + if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL + settings.log.printf("Removed perturbation. Continuing dual simplex (primal_inf=%.2e)\n", primal_infeasibility); + return 1; // CONTINUE_DUAL + } + + // Perturbation on basic (or one-sided nonbasic) variables: need to recompute (y, z). + std::vector unperturbed_y(m); + std::vector unperturbed_z(n); + compute_dual_solution_from_basis(lp, ft, basic_list, nonbasic_list, + unperturbed_y, unperturbed_z, work_estimate); + + // Check if removal is clean (no dual infeasibility) + const f_t dual_infeas = dual_infeasibility(lp, settings, vstatus, unperturbed_z, + settings.tight_tol, settings.dual_tol); + work_estimate += 3 * n; + if (dual_infeas <= settings.dual_tol) { + settings.log.printf("Removed perturbation of %.2e.\n", perturbation); + z = unperturbed_z; + y = unperturbed_y; + objective = lp.objective; + work_estimate += 3 * n + 2 * m; + return 0; // OPTIMAL + } + + // Flip boxed nonbasics that are dual infeasible, and check for one-sided infeasibility + std::vector new_vstatus = vstatus; + i_t num_flipped = 0; + f_t residual_dual_infeas = 0.0; + for (i_t k = 0; k < n_minus_m; ++k) { + const i_t j = nonbasic_list[k]; + const f_t zj = unperturbed_z[j]; + const f_t lower = lp.lower[j]; + const f_t upper = lp.upper[j]; + const bool boxed = (lower > -inf && upper < inf && lower != upper); + + if (new_vstatus[j] == variable_status_t::NONBASIC_LOWER && zj < -settings.dual_tol) { + if (boxed) { + new_vstatus[j] = variable_status_t::NONBASIC_UPPER; + num_flipped++; + } else { + residual_dual_infeas = std::max(residual_dual_infeas, -zj); + } + } else if (new_vstatus[j] == variable_status_t::NONBASIC_UPPER && zj > settings.dual_tol) { + if (boxed) { + new_vstatus[j] = variable_status_t::NONBASIC_LOWER; + num_flipped++; + } else { + residual_dual_infeas = std::max(residual_dual_infeas, zj); + } + } + } + work_estimate += 5 * n_minus_m; + + if (residual_dual_infeas > settings.dual_tol) { + // One-sided infeasibility remains — can't continue with dual simplex. + // new_vstatus is discarded; vstatus unchanged. + settings.log.printf("Perturbation removal: %d flips, residual_dual_infeas=%.2e (PRIMAL_CLEANUP)\n", + num_flipped, residual_dual_infeas); + return 2; // PRIMAL_CLEANUP + } + + // All infeasibility was on boxed variables — accept unperturbed solution + vstatus = new_vstatus; + z = unperturbed_z; + y = unperturbed_y; + objective = lp.objective; + work_estimate += 3 * n + 2 * m; + + // Recompute x_B with flipped statuses + compute_primal_solution_from_basis(lp, ft, basic_list, nonbasic_list, vstatus, x, xB_workspace, work_estimate); + work_estimate += 2 * n; + primal_infeasibility_squared = + compute_initial_primal_infeasibilities(lp, settings, basic_list, x, + squared_infeasibilities, infeasibility_indices, + primal_infeasibility); + work_estimate += 4 * m + 2 * n; + + settings.log.printf("Perturbation removal: %d flips, primal_inf=%.2e\n", num_flipped, primal_infeasibility); + if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL + settings.log.printf("Continuing dual after flip (primal_inf=%.2e)\n", primal_infeasibility); + return 1; // CONTINUE_DUAL +} + template void prepare_optimality(i_t info, f_t orig_primal_infeas, @@ -2414,86 +2660,21 @@ void prepare_optimality(i_t info, sol.objective = compute_objective(lp, sol.x); sol.user_objective = compute_user_objective(lp, sol.objective); - f_t perturbation = amount_of_perturbation(lp, objective); - f_t orig_perturbation = perturbation; - if (perturbation > 1e-6 && phase == 2) { - // Try to remove perturbation - std::vector unperturbed_y(m); - std::vector unperturbed_z(n); - compute_dual_solution_from_basis( - lp, ft, basic_list, nonbasic_list, unperturbed_y, unperturbed_z, work_estimate); - { - const f_t dual_infeas = dual_infeasibility( - lp, settings, vstatus, unperturbed_z, settings.tight_tol, settings.dual_tol); - if (dual_infeas <= settings.dual_tol) { - settings.log.printf("Removed perturbation of %.2e.\n", perturbation); - z = unperturbed_z; - y = unperturbed_y; - perturbation = 0.0; - } else { - settings.log.printf("Failed to remove perturbation of %.2e.\n", perturbation); - settings.log.printf("Unperturbed dual infeasibility: %.2e\n", dual_infeas); - settings.log.printf("Objective: %+.16e\n", sol.user_objective); - settings.log.printf("Num updates: %d\n", ft.num_updates()); - settings.log.printf("Iterations: %d\n", iter); - - i_t dual_iter = iter; - - // Primal pivots in place, so keep the perturbed solution to fall back on. - // The factor is snapshot rather than refactorized on failure: the copy is - // exact, keeps ft consistent with the restored basis, and cannot itself - // fail the way a refactorization can. - const basis_update_mpf_t saved_ft = ft; - const std::vector saved_x = sol.x; - const std::vector saved_y = sol.y; - const std::vector saved_z = sol.z; - const std::vector saved_vstatus = vstatus; - const std::vector saved_basic_list = basic_list; - const std::vector saved_nonbasic_list = nonbasic_list; - - // Reoptimize the unperturbed objective from this basis. The point is - // primal feasible, so primal simplex stays in phase 2 and pivots only to - // restore dual feasibility. It writes through sol, so x, y and z here see - // the cleaned up solution. It prints no summary; the one below reports the - // final result. - primal_status_t primal_status = primal_phase2_with_advanced_basis(2, - start_time, - lp, - settings, - vstatus, - ft, - basic_list, - nonbasic_list, - sol, - iter, - work_estimate, - false); - if (primal_status == primal_status_t::OPTIMAL) { - // z now prices the original objective, so no perturbation remains. - settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); - perturbation = 0.0; - sol.objective = compute_objective(lp, sol.x); - sol.user_objective = compute_user_objective(lp, sol.objective); - } else { - // Restore the perturbed optimum; a partially pivoted basis is worse than - // the dual feasible point we started from. - settings.log.printf("Primal cleanup failed. Reporting the perturbed solution.\n"); - ft = saved_ft; - sol.x = saved_x; - sol.y = saved_y; - sol.z = saved_z; - vstatus = saved_vstatus; - basic_list = saved_basic_list; - nonbasic_list = saved_nonbasic_list; - } - } - } - } + const f_t perturbation = amount_of_perturbation(lp, objective); sol.l2_primal_residual = l2_primal_residual(lp, sol); sol.l2_dual_residual = l2_dual_residual(lp, sol); const f_t dual_infeas = dual_infeasibility(lp, settings, vstatus, z, 0.0, 0.0); - const f_t primal_infeas = primal_infeasibility(lp, settings, vstatus, x); + // Compute max primal infeasibility for reporting + f_t primal_infeas = 0.0; + for (i_t j = 0; j < n; ++j) { + if (x[j] < lp.lower[j]) { + primal_infeas = std::max(primal_infeas, lp.lower[j] - x[j]); + } + if (x[j] > lp.upper[j]) { + primal_infeas = std::max(primal_infeas, x[j] - lp.upper[j]); + } + } if (phase == 1 && iter > 0) { settings.log.printf("Dual phase I complete. Iterations %d. Time %.2f\n", iter, toc(start_time)); } @@ -2521,14 +2702,13 @@ void prepare_optimality(i_t info, primal_infeasibility_breakdown( lp, settings, vstatus, x, basic_infeas, nonbasic_infeas, basic_over); settings.log.printf( - "Primal infeasibility %e/%e (Basic %e, Nonbasic %e, Basic over %e). Perturbation %e/%e. Info " + "Primal infeasibility %e/%e (Basic %e, Nonbasic %e, Basic over %e). Perturbation %e. Info " "%d\n", primal_infeas, orig_primal_infeas, basic_infeas, nonbasic_infeas, basic_over, - orig_perturbation, perturbation, info); } @@ -2618,6 +2798,29 @@ class phase2_timers_t { update_infeasibility_time.work; // clang-format off print_one(settings, "BFRT time", bfrt_time, total_time, total_work); + if (bfrt_time.time > 0.1) { + settings.log.printf(" BFRT breakpoints: %.2fs\n", bfrt_breakpoints_time); + settings.log.printf(" BFRT single_pass: %.2fs\n", bfrt_single_pass_time); + settings.log.printf(" BFRT coarse: %.2fs\n", bfrt_coarse_time); + settings.log.printf(" BFRT bucket: %.2fs\n", bfrt_bucket_time); + settings.log.printf(" BFRT select: %.2fs\n", bfrt_select_time); + } + if (bfrt_calls > 0) { + settings.log.printf(" BFRT calls: %d, zero_steps: %d (%.1f%%), single_pass_only: %d, bucket_used: %d, not_last_bucket: %d, fallback: %d\n", + bfrt_calls, bfrt_zero_steps, 100.0 * bfrt_zero_steps / bfrt_calls, + bfrt_single_pass_only, bfrt_bucket_used, bfrt_not_last_bucket, bfrt_fallback); + settings.log.printf(" BFRT slope_breaker: %d, not_slope_breaker: %d (%.1f%%)\n", + bfrt_selected_slope_breaker, bfrt_not_slope_breaker, + bfrt_bucket_used > 0 ? 100.0 * bfrt_not_slope_breaker / bfrt_bucket_used : 0.0); + if (bfrt_zero_steps > 0) { + settings.log.printf(" BFRT zero-step avg: num_buckets=%.1f, bucket0_size=%.1f, num_breakpoints=%.1f, harris_zero=%.1f, exact_zero=%.1f\n", + 1.0 * bfrt_zero_step_num_buckets_sum / bfrt_zero_steps, + 1.0 * bfrt_zero_step_bucket0_sum / bfrt_zero_steps, + 1.0 * bfrt_zero_step_num_breakpoints_sum / bfrt_zero_steps, + 1.0 * bfrt_zero_step_harris_zero_sum / bfrt_zero_steps, + 1.0 * bfrt_zero_step_exact_zero_sum / bfrt_zero_steps); + } + } print_one(settings, "Pricing time", pricing_time, total_time, total_work); print_one(settings, "BTran time", btran_time, total_time, total_work); print_one(settings, "FTran time", ftran_time, total_time, total_work); @@ -2638,6 +2841,25 @@ class phase2_timers_t { // clang-format on } work_timer_t bfrt_time; + f_t bfrt_breakpoints_time{0.0}; + f_t bfrt_single_pass_time{0.0}; + f_t bfrt_coarse_time{0.0}; + f_t bfrt_bucket_time{0.0}; + f_t bfrt_select_time{0.0}; + // BFRT diagnostic counters + i_t bfrt_calls{0}; + i_t bfrt_zero_steps{0}; // step_length == 0 + i_t bfrt_single_pass_only{0}; // no bound flips (single_pass decided) + i_t bfrt_bucket_used{0}; // bucket sort was used + i_t bfrt_not_last_bucket{0}; // selected from a bucket other than the last + i_t bfrt_fallback{0}; // fell back to single_pass result after bucket sort + i_t bfrt_zero_step_num_buckets_sum{0}; // sum of num_buckets on zero-step iters + i_t bfrt_zero_step_bucket0_sum{0}; // sum of bucket0 size on zero-step iters + i_t bfrt_zero_step_num_breakpoints_sum{0}; // sum of num_breakpoints on zero-step iters + i_t bfrt_zero_step_harris_zero_sum{0}; // sum of harris_ratios==0 on zero-step iters + i_t bfrt_zero_step_exact_zero_sum{0}; // sum of exact ratios==0 on zero-step iters + i_t bfrt_selected_slope_breaker{0}; // times we selected the slope breaker + i_t bfrt_not_slope_breaker{0}; // times we selected something else work_timer_t pricing_time; work_timer_t btran_time; work_timer_t ftran_time; @@ -2777,10 +2999,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (toc(start_time) > settings.time_limit) { return dual_status_t::TIME_LIMIT; } } - if (settings.initial_perturbation == 1 && phase == 2) { - phase2::initial_perturbation(lp, settings, vstatus, objective); - } - // Populate c_basic after basis is initialized for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; @@ -2811,8 +3029,117 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, assert(dual_res_norm < 1e-3); #endif - phase2::set_primal_variables_on_bounds(lp, settings, z, vstatus, x); - phase2_work_estimate += 5 * (n - m); + // Count degenerate NONBASIC_FIXED variables before bound assignment + i_t num_degen = 0; + { + i_t num_fixed = 0; + for (i_t j = 0; j < n; ++j) { + if (vstatus[j] == variable_status_t::NONBASIC_FIXED) { + if (std::abs(lp.lower[j] - lp.upper[j]) >= settings.fixed_tol) { + num_fixed++; + if (std::abs(z[j]) < settings.dual_tol) num_degen++; + } + } + } + settings.log.printf("NONBASIC_FIXED boxed: %d, degenerate (|z_j| < dual_tol): %d\n", num_fixed, num_degen); + } + + // Try 3 strategies for degenerate bound assignment, pick best + f_t best_sum_infeas = inf; + i_t best_num_infeas = m; + i_t best_degen_type = 0; + std::vector best_vstatus; + std::vector best_x; + const char* degen_names[] = {"default", "column-sum", "abs-bound"}; + const i_t degen_types[] = {0, 1, 3}; + f_t all_sum_infeas[3]; + i_t all_num_infeas[3]; + + for (i_t di = 0; di < 3; di++) { + const i_t dt = degen_types[di]; + std::vector try_vstatus = vstatus; + std::vector try_x = x; + phase2::set_primal_variables_on_bounds(lp, settings, z, try_vstatus, try_x, dt); + phase2::compute_primal_variables(ft, lp.rhs, lp.A, basic_list, nonbasic_list, + settings.tight_tol, try_x, xB_workspace, phase2_work_estimate); + f_t sum_infeas = 0.0; + i_t num_infeas = 0; + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + f_t infeas = std::max(lp.lower[j] - try_x[j], try_x[j] - lp.upper[j]); + if (infeas > 0.0) { sum_infeas += infeas; num_infeas++; } + } + all_sum_infeas[di] = sum_infeas; + all_num_infeas[di] = num_infeas; + if (di == 0) { + // Default is the baseline + best_sum_infeas = sum_infeas; + best_num_infeas = num_infeas; + best_degen_type = 0; + best_vstatus = try_vstatus; + best_x = try_x; + } else { + // Only pick alternative if BOTH fewer infeasibilities AND lower sum + if (num_infeas <= best_num_infeas && sum_infeas < best_sum_infeas) { + best_sum_infeas = sum_infeas; + best_num_infeas = num_infeas; + best_degen_type = di; + best_vstatus = try_vstatus; + best_x = try_x; + } + } + if (phase == 1 || num_degen == 0) { + for (i_t t = 1; t < 3; t++) { + all_sum_infeas[t] = sum_infeas; + all_num_infeas[t] = num_infeas; + } + break; + } + } + vstatus = best_vstatus; + x = best_x; + settings.log.printf("Bound assignment: default(%d/%.2e) colsum(%d/%.2e) abs-bound(%d/%.2e) -> %s\n", + all_num_infeas[0], all_sum_infeas[0], + all_num_infeas[1], all_sum_infeas[1], + all_num_infeas[2], all_sum_infeas[2], + degen_names[best_degen_type]); + phase2_work_estimate += 15 * (n - m); + + // Near-optimality check: decide whether to apply initial perturbation + if (settings.initial_perturbation != 0 && phase == 2) { + i_t num_primal_infeas = 0; + f_t max_primal_infeas = 0.0; + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + f_t infeas = std::max(lp.lower[j] - x[j], x[j] - lp.upper[j]); + if (infeas > settings.primal_tol) { + num_primal_infeas++; + max_primal_infeas = std::max(max_primal_infeas, infeas); + } + } + bool near_optimal = (num_primal_infeas < 1000 && max_primal_infeas < 1e-3); + bool apply_perturbation = (settings.initial_perturbation == 1) || !near_optimal; + settings.log.printf("Near-optimal check: num_primal_infeas=%d, max_primal_infeas=%.2e, near_optimal=%d, apply_perturbation=%d\n", + num_primal_infeas, max_primal_infeas, near_optimal, apply_perturbation); + if (apply_perturbation) { + phase2::initial_perturbation(lp, settings, vstatus, objective); + // Recompute y, z with perturbed objective + for (i_t k = 0; k < m; ++k) { + c_basic[k] = objective[basic_list[k]]; + } + phase2_work_estimate += 3 * m; + ft.b_transpose_solve(c_basic, y); + phase2::compute_reduced_costs( + objective, lp.A, y, basic_list, nonbasic_list, z, phase2_work_estimate); + // Reassign bounds based on perturbed z (breaks degeneracy) + i_t num_bound_changes2 = phase2::set_primal_variables_on_bounds(lp, settings, z, vstatus, x); + phase2_work_estimate += 5 * (n - m); + if (num_bound_changes2 > 0) { + phase2::compute_primal_variables(ft, lp.rhs, lp.A, basic_list, nonbasic_list, + settings.tight_tol, x, xB_workspace, phase2_work_estimate); + } + } + } #ifdef PRINT_VSTATUS_CHANGES i_t num_vstatus_changes; @@ -2836,16 +3163,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } phase2_work_estimate += 3 * n; - phase2::compute_primal_variables(ft, - lp.rhs, - lp.A, - basic_list, - nonbasic_list, - settings.tight_tol, - x, - xB_workspace, - phase2_work_estimate); - if (toc(start_time) > settings.time_limit) { return dual_status_t::TIME_LIMIT; } if (print_norms) { settings.log.printf("|| x || %e\n", vector_norm2(x)); } @@ -2964,6 +3281,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t dense_delta_z = 0; i_t num_refactors = 0; i_t total_bound_flips = 0; + i_t max_bound_flips = 0; f_t delta_y_nz_percentage = 0.0; phase2::phase2_timers_t timers(true); @@ -3144,6 +3462,51 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += ft.work_estimate(); ft.clear_work_estimate(); + + // Before declaring optimal, attempt to remove perturbation. + if (phase == 2) { + i_t removal_status = phase2::attempt_to_remove_perturbations( + lp, settings, ft, basic_list, nonbasic_list, vstatus, objective, + z, y, x, xB_workspace, squared_infeasibilities, infeasibility_indices, + primal_infeasibility, primal_infeasibility_squared, phase2_work_estimate); + if (removal_status == 1) { // CONTINUE_DUAL + obj = phase2::compute_perturbed_objective(objective, x); + phase2_work_estimate += 2 * n; + continue; + } + if (removal_status == 2) { // PRIMAL_CLEANUP + const f_t perturbation = phase2::amount_of_perturbation(lp, objective); + settings.log.printf("Failed to remove perturbation of %.2e.\n", perturbation); + settings.log.printf("Num updates: %d\n", ft.num_updates()); + settings.log.printf("Iterations: %d\n", iter); + i_t dual_iter = iter; + primal_status_t primal_status = primal_phase2_with_advanced_basis(2, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + phase2_work_estimate, + false); + if (primal_status == primal_status_t::OPTIMAL) { + settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); + objective = lp.objective; + } else { + settings.log.printf("Primal cleanup failed.\n"); + const f_t dual_infeas = phase2::dual_infeasibility( + lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); + if (dual_infeas > 10.0 * settings.dual_tol) { + return dual_status_t::NUMERICAL; + } + } + } + // removal_status == 0 (OPTIMAL) or primal cleanup done: fall through to prepare_optimality + } + phase2::prepare_optimality(0, primal_infeasibility, lp, @@ -3160,8 +3523,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, iter, x, y, - z, - sol); + z, + sol); status = dual_status_t::OPTIMAL; break; } @@ -3305,6 +3668,36 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, return dual_status_t::NUMERICAL; } timers.bfrt_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); + timers.bfrt_breakpoints_time += bfrt.time_compute_breakpoints_; + timers.bfrt_single_pass_time += bfrt.time_single_pass_; + timers.bfrt_coarse_time += bfrt.time_coarse_filter_; + timers.bfrt_bucket_time += bfrt.time_bucket_sort_; + timers.bfrt_select_time += bfrt.time_pivot_selection_; + // BFRT diagnostics + timers.bfrt_calls++; + if (step_length == 0.0) { + timers.bfrt_zero_steps++; + timers.bfrt_zero_step_num_buckets_sum += bfrt.num_buckets_used_; + timers.bfrt_zero_step_bucket0_sum += bfrt.bucket0_size_; + timers.bfrt_zero_step_num_breakpoints_sum += bfrt.num_breakpoints_; + timers.bfrt_zero_step_harris_zero_sum += bfrt.num_harris_zero_; + timers.bfrt_zero_step_exact_zero_sum += bfrt.num_exact_zero_; + } + if (bfrt.num_buckets_used_ == 0) { + timers.bfrt_single_pass_only++; + } else { + timers.bfrt_bucket_used++; + if (bfrt.used_fallback_) { + timers.bfrt_fallback++; + } else if (bfrt.bucket_selected_ < bfrt.num_buckets_used_ - 1) { + timers.bfrt_not_last_bucket++; + } + if (bfrt.selected_is_slope_breaker_) { + timers.bfrt_selected_slope_breaker++; + } else { + timers.bfrt_not_slope_breaker++; + } + } } else { entering_index = phase2::phase2_ratio_test( lp, settings, vstatus, nonbasic_list, z, delta_z, step_length, nonbasic_entering_index); @@ -3319,137 +3712,51 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += 2 * n; if (perturbation > 0.0 && phase == 2) { - // Try to remove perturbation - std::vector unperturbed_y(m); - std::vector unperturbed_z(n); - phase2_work_estimate += m + n; - phase2::compute_dual_solution_from_basis( - lp, ft, basic_list, nonbasic_list, unperturbed_y, unperturbed_z, phase2_work_estimate); - { - const f_t dual_infeas = phase2::dual_infeasibility( - lp, settings, vstatus, unperturbed_z, settings.tight_tol, settings.dual_tol); - phase2_work_estimate += 3 * n; - settings.log.printf("Dual infeasibility after removing perturbation %e\n", dual_infeas); - if (dual_infeas <= settings.dual_tol) { - settings.log.printf("Removed perturbation of %.2e.\n", perturbation); - z = unperturbed_z; - y = unperturbed_y; - phase2_work_estimate += 2 * n + 2 * m; - perturbation = 0.0; - - std::vector unperturbed_x(n); - phase2_work_estimate += n; - phase2::compute_primal_solution_from_basis(lp, - ft, - basic_list, - nonbasic_list, - vstatus, - unperturbed_x, - xB_workspace, - phase2_work_estimate); - x = unperturbed_x; - primal_infeasibility_squared = - phase2::compute_initial_primal_infeasibilities(lp, - settings, - basic_list, - x, - squared_infeasibilities, - infeasibility_indices, - primal_infeasibility); - phase2_work_estimate += 4 * m + 2 * n; - settings.log.printf("Updated primal infeasibility: %e\n", primal_infeasibility); - - objective = lp.objective; - phase2_work_estimate += 2 * n; - // Need to reset the objective value, since we have recomputed x - obj = phase2::compute_perturbed_objective(objective, x); - phase2_work_estimate += 2 * n; - if (dual_infeas <= settings.dual_tol && primal_infeasibility <= settings.primal_tol) { - phase2_work_estimate += ft.work_estimate(); - ft.clear_work_estimate(); - phase2::prepare_optimality(1, - primal_infeasibility, - lp, - settings, - ft, - objective, - basic_list, - nonbasic_list, - vstatus, - phase, - start_time, - max_val, - phase2_work_estimate, - iter, - x, - y, - z, - sol); - status = dual_status_t::OPTIMAL; - break; - } - settings.log.printf( - "Continuing with perturbation removed and steepest edge norms reset\n"); - // Clear delta_z before restarting the iteration - phase2_work_estimate += 3 * delta_z_indices.size(); - phase2::clear_delta_z( - entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); - continue; - } else { - std::vector unperturbed_x(n); - phase2_work_estimate += n; - phase2::compute_primal_solution_from_basis(lp, - ft, - basic_list, - nonbasic_list, - vstatus, - unperturbed_x, - xB_workspace, - phase2_work_estimate); - x = unperturbed_x; - phase2_work_estimate += 2 * n; - primal_infeasibility_squared = - phase2::compute_initial_primal_infeasibilities(lp, - settings, - basic_list, - x, - squared_infeasibilities, - infeasibility_indices, - primal_infeasibility); - phase2_work_estimate += 4 * m + 2 * n; - - const f_t orig_dual_infeas = phase2::dual_infeasibility( - lp, settings, vstatus, z, settings.tight_tol, settings.dual_tol); - phase2_work_estimate += 3 * n; - - if (primal_infeasibility <= settings.primal_tol && - orig_dual_infeas <= settings.dual_tol) { - phase2_work_estimate += ft.work_estimate(); - ft.clear_work_estimate(); - phase2::prepare_optimality(2, - primal_infeasibility, - lp, - settings, - ft, - objective, - basic_list, - nonbasic_list, - vstatus, - phase, - start_time, - max_val, - phase2_work_estimate, - iter, - x, - y, - z, - sol); - status = dual_status_t::OPTIMAL; - break; - } - settings.log.printf("Failed to remove perturbation of %.2e.\n", perturbation); + i_t removal_status = phase2::attempt_to_remove_perturbations( + lp, settings, ft, basic_list, nonbasic_list, vstatus, objective, + z, y, x, xB_workspace, squared_infeasibilities, infeasibility_indices, + primal_infeasibility, primal_infeasibility_squared, phase2_work_estimate); + if (removal_status == 0) { // OPTIMAL + obj = phase2::compute_perturbed_objective(objective, x); + phase2_work_estimate += 2 * n; + if (primal_infeasibility <= settings.primal_tol) { + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); + phase2::prepare_optimality(1, + primal_infeasibility, + lp, + settings, + ft, + objective, + basic_list, + nonbasic_list, + vstatus, + phase, + start_time, + max_val, + phase2_work_estimate, + iter, + x, + y, + z, + sol); + status = dual_status_t::OPTIMAL; + break; } + settings.log.printf("Continuing with perturbation removed\n"); + phase2_work_estimate += 3 * delta_z_indices.size(); + phase2::clear_delta_z( + entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); + continue; + } else if (removal_status == 1) { // CONTINUE_DUAL + obj = phase2::compute_perturbed_objective(objective, x); + phase2_work_estimate += 2 * n; + phase2_work_estimate += 3 * delta_z_indices.size(); + phase2::clear_delta_z( + entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); + continue; } + // removal_status == 2 (PRIMAL_CLEANUP): fall through to existing logic below } if (perturbation == 0.0 && phase == 2) { @@ -3545,6 +3852,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, timers.flip_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); total_bound_flips += num_flipped; + if (num_flipped > max_bound_flips) max_bound_flips = num_flipped; delta_xB_0_sparse.clear(); if (num_flipped > 0) { @@ -3718,12 +4026,13 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, basic_list, entering_index, scaled_delta_xB_sparse, delta_x, phase2_work_estimate); timers.start_timer(phase2_work_estimate + ft.work_estimate()); - if (settings.remove_perturbation == 1) { - phase2::remove_leaving_perturbation(lp, settings, leaving_index, z, objective); + if (settings.remove_perturbation != 0) { + phase2::remove_leaving_perturbation(lp, settings, leaving_index, direction, z, objective); } f_t sum_perturb = 0.0; phase2::compute_perturbation( - lp, settings, delta_z_indices, z, objective, sum_perturb, phase2_work_estimate); + lp, settings, delta_z_indices, vstatus, z, objective, sum_perturb, + entering_index, step_length, phase2_work_estimate); timers.perturb_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); // Update basis information @@ -3905,13 +4214,13 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, return dual_status_t::WORK_LIMIT; } - if (now > settings.time_limit) { return dual_status_t::TIME_LIMIT; } + if (now > settings.time_limit) { status = dual_status_t::TIME_LIMIT; break; } if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { return dual_status_t::CONCURRENT_LIMIT; } } - if (iter >= iter_limit) { status = dual_status_t::ITERATION_LIMIT; } + if (status != dual_status_t::TIME_LIMIT && iter >= iter_limit) { status = dual_status_t::ITERATION_LIMIT; } // Flush any remaining work from the basis update into the total work estimate phase2_work_estimate += ft.work_estimate(); @@ -3919,6 +4228,11 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (phase == 2) { timers.print_timers(settings); + i_t num_iters = iter - start_iter; + if (num_iters > 0) { + settings.log.printf("Bound flips: total=%d, avg=%.1f, max=%d\n", + total_bound_flips, 1.0 * total_bound_flips / num_iters, max_bound_flips); + } constexpr bool print_stats = false; if constexpr (print_stats) { settings.log.printf("Sparse delta_z %8d %8.2f%\n", From 917ccaffdcca84f05a9c90ea74527865c7aa7a69 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 26 Aug 2026 15:58:38 -0700 Subject: [PATCH 023/113] Avoid double counting slope contribution from variable selected in single_pass --- .../bound_flipping_ratio_test.cpp | 23 +++++-------------- .../bound_flipping_ratio_test.hpp | 1 - 2 files changed, 6 insertions(+), 18 deletions(-) diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index 1086b23ddf..d0e6e2c23b 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -61,7 +61,6 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, i_t end, const std::vector& indicies, const std::vector& ratios, - f_t& slope, f_t& step_length, i_t& nonbasic_entering, i_t& entering_index, @@ -95,20 +94,7 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, if (nonbasic_entering == -1) { return RATIO_TEST_NUMERICAL_ISSUES; } const i_t j = entering_index = nonbasic_list_[nonbasic_entering]; - constexpr bool verbose = false; - if (bounded_variables_[j]) { - const f_t interval = upper_[j] - lower_[j]; - const f_t delta_slope = std::abs(delta_z_[j]) * interval; - if constexpr (verbose) { - settings_.log.printf("single pass delta slope %e slope %e after slope %e step length %e\n", - delta_slope, - slope, - slope - delta_slope, - step_length); - } - slope -= delta_slope; - return k_idx; // we should see if we can continue to increase the step-length - } + if (bounded_variables_[j]) { return k_idx; } return -1; // we are done. do not increase the step-length further } @@ -150,10 +136,13 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, t0 = tic(); i_t k_idx = single_pass( - 0, num_breakpoints, indicies, harris_ratios, slope, step_length, nonbasic_entering, entering_index, max_step_length); + 0, num_breakpoints, indicies, harris_ratios, step_length, nonbasic_entering, entering_index, max_step_length); time_single_pass_ += toc(t0); if (k_idx == RATIO_TEST_NUMERICAL_ISSUES) { return RATIO_TEST_NUMERICAL_ISSUES; } - bool continue_search = k_idx >= 0 && num_breakpoints > 1 && slope > 0.0; + // The variable selected by single_pass is guaranteed to be in the first bucket: it + // defines the minimum Harris ratio, and its exact ratio is no greater than its Harris + // ratio. Its slope contribution is therefore applied by the bucket pass below. + bool continue_search = k_idx >= 0 && num_breakpoints > 1; if (!continue_search) { if constexpr (verbose) { settings_.log.printf( diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp index 6e9dc647c1..328f91be0e 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp @@ -80,7 +80,6 @@ class bound_flipping_ratio_test_t { i_t end, const std::vector& indices, const std::vector& ratios, - f_t& slope, f_t& step_length, i_t& nonbasic_entering, i_t& entering_index, From 36ca4d337fb661437b33e3e54e8a6cc615cd65f8 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 27 Aug 2026 16:24:39 -0700 Subject: [PATCH 024/113] Have BFRT decide on bound flips Have the bound-flipping ratio test return the set of variables that must flip at its selected step length. Flip exactly those variables in flip_bounds. The BFRT piecewise-linear objective model changes a bounded variable's bound when its reduced cost crosses zero. Flipping at zero is numerically unstable: small reduced-cost changes between iterations can move a variable repeatedly across zero, reverse its bound, and cause cycling. Use dual_tol / 10 to decide which variables to flip, leaving a small dead zone around zero. flip_bounds flips exactly these variables (previously it decided which bounds to flip using a separate mismatched tolerance). Within that dead zone, the objective represented by the BFRT model may differ from the objective computed using NONBASIC_UPPER/NONBASIC_LOWER. For a bounded variable this discrepancy is at most (upper_j - lower_j) * dual_tol / 10 and the total discrepancy is bounded by the sum of this quantity over the unflipped variables. Also stop the BFRT coarse-filter search after it has scanned all breakpoints. Without the scan_start < num_breakpoints condition, physiciansched3-3 loops forever when coarse_threshold remains zero after all candidates have already been processed. On the 240 MIPLIB LP relaxations, this change prevents five 300-second timeouts (neos-5052403-cygnet, physiciansched3-3, supportcase10, bab2, and brazil3), but introduces two new 300-second timeouts (neos-3988577-wolgan and neos-2075418-temuka). Unsolved runs are capped at 300 seconds. The presolve-only neos-787933 uses its presolve time. The HiGHS times were taken from a faster machine. So the HiGHS advantage is slightly exaggerated. But it still exists. Problem Baseline v5 v8 HiGHS v8/HiGHS ------------------------------------------------------------------------------------ momentum1 0.69 0.73 0.70 300.00 0.00 var-smallemery-m6j6 0.66 0.74 0.72 300.00 0.00 neos-5114902-kasavu 87.40 23.29 4.91 300.00 0.02 supportcase42 0.52 0.77 0.78 36.20 0.02 neos-5049753-cuanza 7.68 3.22 1.91 29.85 0.06 supportcase12 4.68 4.30 4.72 37.11 0.13 proteindesign121hz512p9 0.91 0.48 0.37 2.04 0.18 supportcase22 2.03 1.11 0.73 3.97 0.18 roi2alpha3n4 0.31 0.67 0.22 1.16 0.19 proteindesign122trx11p8 0.64 0.31 0.26 1.26 0.21 mzzv11 40.19 8.46 3.45 16.71 0.21 supportcase18 0.06 0.04 0.03 0.12 0.25 roi5alpha10n8 1.25 3.29 3.42 11.90 0.29 neos-5052403-cygnet 300.00 187.30 110.25 300.00 0.37 supportcase7 1.29 1.46 1.36 3.52 0.39 rd-rplusc-21 0.23 0.25 0.19 0.49 0.39 neos-787933 0.06 0.07 0.07 0.18 0.39 ns1760995 135.94 250.52 114.63 269.53 0.43 rocII-5-11 0.09 0.09 0.10 0.21 0.48 fhnw-binpack4-48 0.07 0.03 0.03 0.06 0.50 neos-5093327-huahum 0.23 0.23 0.25 0.48 0.52 30n20b8 0.08 0.05 0.06 0.11 0.55 lectsched-5-obj 0.13 0.10 0.10 0.18 0.56 neos-4647030-tutaki 2.47 2.33 1.72 3.09 0.56 physiciansched6-2 11.15 15.62 3.84 6.72 0.57 neos-3004026-krka 0.06 0.08 0.07 0.12 0.58 co-100 0.68 0.64 0.75 1.28 0.59 neos-5104907-jarama 124.69 88.16 52.83 89.64 0.59 neos-3402454-bohle 221.86 115.49 44.09 74.15 0.59 dws008-01 0.04 0.03 0.03 0.05 0.60 n3div36 0.11 0.13 0.11 0.18 0.61 neos-5188808-nattai 0.30 0.19 0.18 0.29 0.62 cvs16r128-89 0.93 1.08 1.13 1.72 0.66 neos-860300 0.10 0.07 0.10 0.15 0.67 neos-4300652-rahue 1.22 0.56 0.54 0.80 0.68 cryptanalysiskb128n5obj14 29.78 6.14 9.28 12.54 0.74 square47 79.87 89.37 95.31 126.92 0.75 istanbul-no-cutoff 0.71 0.87 0.71 0.94 0.76 thor50dday 0.27 0.25 0.28 0.37 0.76 ns1952667 8.95 1.18 0.62 0.80 0.77 neos-5195221-niemur 0.50 0.24 0.30 0.37 0.81 tbfp-network 9.06 6.73 7.80 9.04 0.86 cryptanalysiskb128n5obj16 29.53 6.53 8.93 10.27 0.87 blp-ic98 0.12 0.11 0.09 0.10 0.90 neos-3555904-turama 1.31 1.52 1.29 1.37 0.94 physiciansched3-3 300.00 300.00 288.13 300.00 0.96 comp21-2idx 1.55 0.38 0.34 0.35 0.97 square41 28.57 42.48 39.17 39.86 0.98 blp-ar98 0.10 0.08 0.11 0.11 1.00 decomp2 0.19 0.10 0.12 0.12 1.00 highschool1-aigio 300.00 300.00 300.00 300.00 1.00 leo2 0.13 0.15 0.13 0.13 1.00 neos-3988577-wolgan 278.89 300.00 300.00 300.00 1.00 neos-4738912-atrato 0.05 0.05 0.04 0.04 1.00 neos8 0.38 0.42 0.26 0.26 1.00 rail02 300.00 300.00 300.00 300.00 1.00 s100 300.00 300.00 300.00 300.00 1.00 savsched1 300.00 300.00 300.00 300.00 1.00 supportcase19 300.00 300.00 300.00 300.00 1.00 swath3 0.03 0.03 0.03 0.03 1.00 traininstance2 0.09 0.04 0.04 0.04 1.00 traininstance6 0.04 0.03 0.03 0.03 1.00 piperout-08 0.39 0.18 0.17 0.16 1.06 neos-873061 1.46 1.51 1.47 1.38 1.07 sct2 0.22 0.20 0.15 0.14 1.07 piperout-27 0.67 0.29 0.28 0.26 1.08 neos-4532248-waihi 2.61 0.92 0.97 0.90 1.08 neos-3216931-puriri 7.33 4.89 3.50 3.23 1.08 academictimetablesmall 14.89 4.15 0.90 0.82 1.10 germanrr 0.31 0.33 0.30 0.27 1.11 nursesched-sprint02 0.38 0.51 0.26 0.23 1.13 triptim1 71.42 58.41 58.50 51.43 1.14 neos-848589 1.24 1.01 0.97 0.85 1.14 cod105 9.06 7.73 8.56 7.46 1.15 supportcase10 300.00 105.36 133.84 113.55 1.18 sp97ar 0.40 0.37 0.39 0.33 1.18 wachplan 0.25 0.23 0.31 0.26 1.19 sp98ar 0.39 0.41 0.37 0.31 1.19 supportcase40 0.24 0.22 0.29 0.24 1.21 neos-960392 8.93 1.97 3.30 2.66 1.24 drayage-25-23 0.08 0.06 0.05 0.04 1.25 h80x6320d 0.05 0.04 0.05 0.04 1.25 neos-3381206-awhea 0.08 0.05 0.05 0.04 1.25 ns1644855 300.00 300.00 300.00 238.30 1.26 nursesched-medium-hint03 10.47 11.61 6.00 4.73 1.27 radiationm18-12-05 0.23 0.19 0.18 0.14 1.29 mushroom-best 0.26 0.23 0.26 0.20 1.30 supportcase6 7.39 6.03 5.52 4.18 1.32 swath1 0.04 0.04 0.04 0.03 1.33 comp07-2idx 4.03 2.05 1.47 1.10 1.34 neos-2746589-doon 7.53 5.95 4.11 3.02 1.36 neos-5107597-kakapo 0.04 0.13 0.14 0.10 1.40 net12 0.55 0.43 0.50 0.35 1.43 neos-3402294-bobin 3.07 1.79 1.72 1.20 1.43 neos-4722843-widden 1.17 2.25 2.21 1.54 1.44 irp 0.11 0.13 0.13 0.09 1.44 supportcase33 0.99 1.12 0.80 0.55 1.45 fast0507 7.36 4.24 4.05 2.77 1.46 mzzv42z 10.10 1.91 1.42 0.95 1.49 bnatt500 0.29 0.16 0.21 0.14 1.50 drayage-100-23 0.07 0.05 0.06 0.04 1.50 icir97_tension 0.03 0.03 0.03 0.02 1.50 neos-1582420 0.12 0.09 0.09 0.06 1.50 rococoC10-001000 0.04 0.03 0.03 0.02 1.50 air05 0.28 0.24 0.27 0.18 1.50 neos-1122047 2.09 1.96 2.43 1.61 1.51 nexp-150-20-8-5 0.10 0.11 0.11 0.07 1.57 neos-2978193-inde 0.16 0.14 0.08 0.05 1.60 rail507 7.17 4.22 4.64 2.84 1.63 hypothyroid-k1 4.36 4.47 4.93 3.01 1.64 neos-3083819-nubu 0.06 0.05 0.05 0.03 1.67 ran14x18-disj-8 0.04 0.05 0.05 0.03 1.67 cmflsp50-24-8-8 0.77 0.73 0.74 0.44 1.68 leo1 0.10 0.09 0.12 0.07 1.71 trento1 3.08 3.49 3.80 2.21 1.72 neos-662469 1.65 0.91 0.93 0.53 1.75 rmatr200-p5 7.50 7.63 8.13 4.61 1.76 atlanta-ip 6.85 5.86 8.14 4.54 1.79 rocI-4-11 0.11 0.10 0.11 0.06 1.83 mcsched 0.27 0.23 0.30 0.16 1.88 chromaticindex512-7 16.62 37.48 40.10 21.24 1.89 neos-3024952-loue 0.41 0.39 0.38 0.20 1.90 neos-2987310-joes 1.52 1.67 1.91 1.00 1.91 netdiversion 7.64 8.61 18.16 9.43 1.93 qap10 15.57 10.58 12.87 6.68 1.93 sing44 9.62 14.79 11.01 5.69 1.93 k1mushroom 31.00 30.55 32.65 16.80 1.94 opm2-z10-s4 89.23 85.57 87.53 44.75 1.96 neos-4763324-toguru 8.32 7.78 9.86 5.00 1.97 50v-10 0.02 0.02 0.02 0.01 2.00 bnatt400 0.16 0.11 0.16 0.08 2.00 eil33-2 0.05 0.06 0.06 0.03 2.00 enlight_hard 0.02 0.02 0.02 0.01 2.00 exp-1-500-5-5 0.02 0.02 0.02 0.01 2.00 fhnw-binpack4-4 0.03 0.02 0.02 0.01 2.00 gen-ip002 0.02 0.02 0.02 0.01 2.00 gen-ip054 0.03 0.02 0.02 0.01 2.00 glass4 0.02 0.02 0.02 0.01 2.00 mad 0.02 0.02 0.02 0.01 2.00 markshare2 0.02 0.02 0.02 0.01 2.00 markshare_4_0 0.02 0.02 0.02 0.01 2.00 mas74 0.03 0.02 0.02 0.01 2.00 mas76 0.02 0.02 0.02 0.01 2.00 mik-250-20-75-4 0.02 0.03 0.02 0.01 2.00 neos-1456979 0.05 0.04 0.08 0.04 2.00 neos-3046615-murg 0.02 0.02 0.02 0.01 2.00 neos-3754480-nidda 0.02 0.02 0.02 0.01 2.00 neos-4338804-snowy 0.03 0.02 0.02 0.01 2.00 neos-911970 0.03 0.02 0.02 0.01 2.00 neos5 0.02 0.02 0.02 0.01 2.00 neos859080 0.01 0.01 0.02 0.01 2.00 nu25-pr12 0.05 0.05 0.04 0.02 2.00 pg 0.03 0.02 0.02 0.01 2.00 pg5_34 0.03 0.03 0.02 0.01 2.00 pk1 0.02 0.01 0.02 0.01 2.00 sp150x300d 0.02 0.02 0.02 0.01 2.00 supportcase26 0.03 0.02 0.02 0.01 2.00 timtab1 0.01 0.01 0.02 0.01 2.00 radiationm40-10-02 1.58 1.44 1.45 0.71 2.04 ns1830653 0.42 0.33 0.29 0.14 2.07 uct-subprob 0.11 0.10 0.17 0.08 2.12 neos-957323 300.00 28.38 16.74 7.74 2.16 roll3000 0.12 0.13 0.13 0.06 2.17 dano3_3 46.88 96.86 42.60 19.06 2.24 dano3_5 46.79 96.97 42.62 19.03 2.24 neos-4387871-tavua 0.10 0.10 0.09 0.04 2.25 rmatr100-p10 0.27 0.34 0.32 0.14 2.29 unitcal_7 0.96 1.00 0.91 0.39 2.33 ns1116954 156.23 16.44 25.16 10.73 2.34 rail01 223.84 196.04 167.23 71.23 2.35 chromaticindex1024-7 45.86 201.94 220.63 93.65 2.36 csched007 0.19 0.16 0.12 0.05 2.40 eilA101-2 2.68 2.86 3.35 1.39 2.41 bab2 300.00 76.29 70.29 28.97 2.43 reblock115 0.16 0.17 0.22 0.09 2.44 seymour1 0.83 0.91 0.94 0.38 2.47 b1c1s1 0.05 0.04 0.05 0.02 2.50 cost266-UUE 0.03 0.04 0.05 0.02 2.50 n5-3 0.03 0.03 0.05 0.02 2.50 rococoB10-011000 0.13 0.11 0.15 0.06 2.50 uccase9 11.54 17.20 13.22 5.20 2.54 seymour 0.83 0.89 0.98 0.38 2.58 buildingenergy 300.00 300.00 300.00 115.61 2.59 splice1k1 21.62 22.71 23.93 9.17 2.61 sing326 9.54 14.16 12.30 4.69 2.62 neos-1171737 0.70 0.70 0.54 0.20 2.70 satellites2-60-fs 4.17 171.22 9.13 3.34 2.73 glass-sc 0.31 0.31 0.34 0.12 2.83 gfd-schedulen180f7d50m30k18 80.00 39.02 19.47 6.81 2.86 nw04 0.44 1.38 1.75 0.61 2.87 neos-1171448 2.35 2.08 2.41 0.84 2.87 binkar10_1 0.02 0.03 0.03 0.01 3.00 bppc4-08 0.08 0.04 0.06 0.02 3.00 csched008 0.11 0.09 0.09 0.03 3.00 graph20-20-1rand 0.23 0.32 0.36 0.12 3.00 graphdraw-domain 0.02 0.02 0.03 0.01 3.00 ic97_potential 0.02 0.03 0.03 0.01 3.00 neos-2657525-crna 0.03 0.03 0.03 0.01 3.00 neos-4954672-berkel 0.02 0.02 0.03 0.01 3.00 neos17 0.03 0.04 0.03 0.01 3.00 p200x1188c 0.03 0.03 0.03 0.01 3.00 tr12-30 0.03 0.02 0.03 0.01 3.00 map16715-04 13.26 19.01 20.42 6.80 3.00 CMS750_4 0.38 0.39 0.43 0.14 3.07 irish-electricity 181.58 300.00 183.12 59.53 3.08 n2seq36q 0.46 0.53 0.74 0.24 3.08 sorrell3 1.07 1.13 1.25 0.40 3.12 map10 11.08 21.74 19.82 6.25 3.17 bab6 143.43 38.82 44.66 14.06 3.18 ns1208400 3.96 1.60 1.17 0.36 3.25 neos-933966 16.79 18.35 9.31 2.80 3.33 neos-1445765 0.16 0.35 0.30 0.09 3.33 uccase12 72.32 6.47 4.77 1.38 3.46 neos-950242 1.05 0.81 0.66 0.18 3.67 assign1-5-8 0.03 0.03 0.04 0.01 4.00 gmu-35-40 0.03 0.04 0.04 0.01 4.00 gmu-35-50 0.05 0.05 0.04 0.01 4.00 lotsize 0.03 0.03 0.04 0.01 4.00 neos-3627168-kasai 0.04 0.03 0.04 0.01 4.00 s250r10 300.00 300.00 300.00 71.74 4.18 neos-1354092 300.00 300.00 300.00 70.92 4.23 fastxgemm-n2r6s0t2 0.18 0.29 0.34 0.08 4.25 milo-v12-6-r2-40-1 0.25 0.22 0.22 0.05 4.40 app1-1 0.08 0.09 0.09 0.02 4.50 fiball 0.71 0.30 0.66 0.14 4.71 cbs-cta 0.42 0.32 0.48 0.10 4.80 beasleyC3 0.05 0.05 0.05 0.01 5.00 mc11 0.06 0.06 0.05 0.01 5.00 ex10 300.00 300.00 300.00 59.25 5.06 peg-solitaire-a3 1.88 2.54 2.81 0.52 5.40 neos-3656078-kumeu 2.37 1.57 1.27 0.23 5.52 neos-827175 9.36 1.91 1.66 0.29 5.72 neos-2075418-temuka 128.30 300.00 300.00 50.57 5.93 snp-02-004-104 14.17 23.47 19.21 2.83 6.79 app1-2 5.03 5.33 5.26 0.70 7.51 neos-631710 300.00 300.00 300.00 31.97 9.38 neos-4413714-turia 3.97 16.53 20.16 1.93 10.45 brazil3 300.00 93.59 79.55 7.24 10.99 satellites2-40 29.70 125.33 186.43 10.51 17.74 ex9 300.00 300.00 300.00 14.07 21.32 Geomean Baseline/v8: 1.19 Shifted(+1s): 1.14 Geomean v5/v8: 1.05 Shifted(+1s): 1.06 Geomean v8/HiGHS: 1.48 Shifted(+1s): 1.12 (240 problems) --- .../bound_flipping_ratio_test.cpp | 35 +++++++++- .../bound_flipping_ratio_test.hpp | 7 +- cpp/src/dual_simplex/phase2.cpp | 70 +++++++------------ 3 files changed, 64 insertions(+), 48 deletions(-) diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index d0e6e2c23b..6fe4ec149f 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -98,14 +98,41 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, return -1; // we are done. do not increase the step-length further } +template +void bound_flipping_ratio_test_t::determine_flips( + f_t step_length, i_t entering_index, std::vector& flip_indices) const +{ + // The piecewise-linear model below assumes that a variable flips bounds as soon as + // its reduced cost crosses zero. In practice, small changes between iterations can + // make a reduced cost oscillate around zero, causing excessive bound flips and + // cycling. We therefore flip only after the violation exceeds dual_tol / 10. + // A bounded variable l_j <= x_j <= u_j contributes l_j*z_j to the dual objective + // when z_j >= 0 and u_j*z_j when z_j < 0. If x_j = l_j and + // -dual_tol/10 <= z_j < 0, the model uses u_j*z_j while the unflipped state uses + // l_j*z_j. Their difference is (u_j - l_j)*|z_j|, bounded by + // (u_j - l_j)*dual_tol/10. For multiple unflipped variables, the discrepancy is + // bounded by sum_j (u_j - l_j)*dual_tol/10. + const f_t flip_tol = settings_.dual_tol / 10; + for (const i_t j : delta_z_indices_) { + if (j == entering_index || !bounded_variables_[j]) { continue; } + const f_t new_z = z_[j] + step_length * delta_z_[j]; + if ((vstatus_[j] == variable_status_t::NONBASIC_LOWER && new_z < -flip_tol) || + (vstatus_[j] == variable_status_t::NONBASIC_UPPER && new_z > flip_tol)) { + flip_indices.push_back(j); + } + } +} + template i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, - i_t& nonbasic_entering) + i_t& nonbasic_entering, + std::vector& flip_indices) { const i_t m = m_; const i_t n = n_; const i_t nz = delta_z_indices_.size(); constexpr bool verbose = false; + flip_indices.clear(); // Compute the initial set of breakpoints std::vector indicies(nz); @@ -154,6 +181,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, } num_buckets_used_ = 0; step_length_result_ = step_length; + determine_flips(step_length, entering_index, flip_indices); return entering_index; } @@ -254,7 +282,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // This is O( log10(max_step_length/min_step_length) * num_breakpoints) t0 = tic(); - while (total_slope >= 0.0 && coarse_threshold <= max_step_length && !found_unbounded) { + while (total_slope >= 0.0 && coarse_threshold <= max_step_length && + scan_start < num_breakpoints && !found_unbounded) { for (i_t h = scan_start; h < num_breakpoints; ++h) { const i_t k = candidates[h]; if (ratios[k] <= coarse_threshold) { @@ -399,6 +428,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, bucket_selected_ = -1; step_length_result_ = step_length; selected_is_slope_breaker_ = false; + determine_flips(step_length, entering_index, flip_indices); return entering_index; } step_length = ratios[entering_k]; @@ -424,6 +454,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, } } step_length_result_ = step_length; + determine_flips(step_length, entering_index, flip_indices); return entering_index; diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp index 328f91be0e..3a87e923db 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp @@ -53,7 +53,9 @@ class bound_flipping_ratio_test_t { { } - i_t compute_step_length(f_t& step_length, i_t& nonbasic_entering); + i_t compute_step_length(f_t& step_length, + i_t& nonbasic_entering, + std::vector& flip_indices); f_t work_estimate() const { return work_estimate_; } // Timing fields (filled by compute_step_length) @@ -84,6 +86,9 @@ class bound_flipping_ratio_test_t { i_t& nonbasic_entering, i_t& entering_index, f_t& max_val); + void determine_flips(f_t step_length, + i_t entering_index, + std::vector& flip_indices) const; const std::vector& lower_; const std::vector& upper_; const std::vector& bounded_variables_; diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 0b94a67d28..123d6354f8 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -1232,30 +1232,19 @@ i_t phase2_ratio_test(const lp_problem_t& lp, template i_t flip_bounds(const lp_problem_t& lp, - const simplex_solver_settings_t& settings, - const std::vector& bounded_variables, - const std::vector& objective, - const std::vector& z, - const std::vector& delta_z_indices, - const std::vector& nonbasic_list, - i_t entering_index, - std::vector& vstatus, + const std::vector& bounded_variables, + const std::vector& flip_indices, + std::vector& vstatus, std::vector& delta_x, std::vector& mark, - std::vector& atilde, - std::vector& atilde_index, - f_t& work_estimate) + std::vector& atilde, + std::vector& atilde_index, + f_t& work_estimate) { i_t num_flipped = 0; - for (i_t k = 0; k < delta_z_indices.size(); ++k) { - const i_t j = delta_z_indices[k]; - if (j == entering_index) { continue; } - if (!bounded_variables[j]) { continue; } - // x_j is now a nonbasic bounded variable that will not enter the basis this - // iteration - const f_t dual_tol = - settings.dual_tol; // lower to 1e-7 or less will cause 25fv47 and d2q06c to cycle - if (vstatus[j] == variable_status_t::NONBASIC_LOWER && z[j] < -dual_tol) { + for (const i_t j : flip_indices) { + assert(bounded_variables[j]); + if (vstatus[j] == variable_status_t::NONBASIC_LOWER) { const f_t delta = lp.upper[j] - lp.lower[j]; const size_t atilde_start_size = atilde_index.size(); scatter_dense(lp.A, j, -delta, atilde, mark, atilde_index); @@ -1263,12 +1252,9 @@ i_t flip_bounds(const lp_problem_t& lp, 4 * (lp.A.col_start[j + 1] - lp.A.col_start[j]) + 10; delta_x[j] += delta; vstatus[j] = variable_status_t::NONBASIC_UPPER; -#ifdef BOUND_FLIP_DEBUG - settings.log.printf( - "Flipping nonbasic %d from lo %e to up %e. z %e\n", j, lp.lower[j], lp.upper[j], z[j]); -#endif num_flipped++; - } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER && z[j] > dual_tol) { + } else { + assert(vstatus[j] == variable_status_t::NONBASIC_UPPER); const f_t delta = lp.lower[j] - lp.upper[j]; const size_t atilde_start_size = atilde_index.size(); scatter_dense(lp.A, j, -delta, atilde, mark, atilde_index); @@ -1276,14 +1262,10 @@ i_t flip_bounds(const lp_problem_t& lp, 4 * (lp.A.col_start[j + 1] - lp.A.col_start[j]) + 10; delta_x[j] += delta; vstatus[j] = variable_status_t::NONBASIC_LOWER; -#ifdef BOUND_FLIP_DEBUG - settings.log.printf( - "Flipping nonbasic %d from up %e to lo %e. z %e\n", j, lp.upper[j], lp.lower[j], z[j]); -#endif num_flipped++; } } - work_estimate += 4 * delta_z_indices.size(); + work_estimate += 2 * flip_indices.size(); return num_flipped; } @@ -3629,6 +3611,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, f_t step_length; i_t entering_index = -1; i_t nonbasic_entering_index = -1; + std::vector flip_indices; const bool harris_ratio = settings.use_harris_ratio; const bool bound_flip_ratio = settings.use_bound_flip_ratio; { @@ -3661,7 +3644,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, delta_z, delta_z_indices, nonbasic_mark); - entering_index = bfrt.compute_step_length(step_length, nonbasic_entering_index); + entering_index = bfrt.compute_step_length( + step_length, nonbasic_entering_index, flip_indices); phase2_work_estimate += bfrt.work_estimate(); if (entering_index == RATIO_TEST_NUMERICAL_ISSUES) { settings.log.printf("Numerical issues encountered in ratio test.\n"); @@ -3835,20 +3819,16 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, timers.start_timer(phase2_work_estimate + ft.work_estimate()); // Update primal variable - const i_t num_flipped = phase2::flip_bounds(lp, - settings, - bounded_variables, - objective, - z, - delta_z_indices, - nonbasic_list, - entering_index, - vstatus, - delta_x_flip, - atilde_mark, - atilde, - atilde_index, - phase2_work_estimate); + const i_t num_flipped = bound_flip_ratio ? phase2::flip_bounds(lp, + bounded_variables, + flip_indices, + vstatus, + delta_x_flip, + atilde_mark, + atilde, + atilde_index, + phase2_work_estimate) + : 0; timers.flip_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); total_bound_flips += num_flipped; From 75a1e97a6c73281d0b3a3210aff4bf9b4d44f1d8 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 27 Aug 2026 17:10:36 -0700 Subject: [PATCH 025/113] Improve BFRT work estimates --- .../bound_flipping_ratio_test.cpp | 28 ++++++++++--------- .../bound_flipping_ratio_test.hpp | 2 +- 2 files changed, 16 insertions(+), 14 deletions(-) diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index 6fe4ec149f..35ef45c0ab 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -49,8 +49,7 @@ i_t bound_flipping_ratio_test_t::compute_breakpoints(std::vector& idx++; } } - work_estimate_ += 5 * nz; - work_estimate_ += 5 * idx; + work_estimate_ += 5 * nz + 5 * idx; pivot_tol /= 10; } return idx; @@ -99,8 +98,9 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, } template -void bound_flipping_ratio_test_t::determine_flips( - f_t step_length, i_t entering_index, std::vector& flip_indices) const +void bound_flipping_ratio_test_t::determine_flips(f_t step_length, + i_t entering_index, + std::vector& flip_indices) { // The piecewise-linear model below assumes that a variable flips bounds as soon as // its reduced cost crosses zero. In practice, small changes between iterations can @@ -121,6 +121,7 @@ void bound_flipping_ratio_test_t::determine_flips( flip_indices.push_back(j); } } + work_estimate_ += 5 * delta_z_indices_.size() + flip_indices.size(); } template @@ -150,6 +151,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, if (harris_ratios[k] == 0.0) num_harris_zero_++; if (ratios[k] == 0.0) num_exact_zero_++; } + work_estimate_ += 2 * num_breakpoints; if constexpr (verbose) { settings_.log.printf("Initial breakpoints %d\n", num_breakpoints); } if (num_breakpoints == 0) { nonbasic_entering = -1; @@ -298,8 +300,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, } } } - work_estimate_ += 3 * (num_breakpoints - scan_start); - work_estimate_ += 8 * (num_candidates - scan_start); + work_estimate_ += 2 * (num_breakpoints - scan_start) + 10 * (num_candidates - scan_start); scan_start = num_candidates; coarse_threshold *= 10.0; } @@ -319,8 +320,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, work_estimate_ += 5 * num_candidates; // Remove candidates that are greater than the maximum step length - work_estimate_ += 2 * candidates.size(); - for (i_t h = static_cast(candidates.size()) - 1; h >= 0; h--) { + const i_t candidates_before_removal = candidates.size(); + for (i_t h = candidates_before_removal - 1; h >= 0; h--) { const i_t k = candidates[h]; const f_t ratio = ratios[k]; if (ratio > max_step_length) { @@ -329,6 +330,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, candidates.pop_back(); } } + work_estimate_ += 2 * candidates_before_removal + 2 * (candidates_before_removal - candidates.size()); num_candidates = candidates.size(); } @@ -349,7 +351,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, f_t next_threshold = inf; i_t write = scan_start; - work_estimate_ += 7 * (num_candidates - scan_start); for (i_t h = scan_start; h < num_candidates; h++) { const i_t k = candidates[h]; const f_t ratio = ratios[k]; @@ -370,7 +371,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, next_threshold = std::min(next_threshold, harris_ratio); } } - + work_estimate_ += 3 * (num_candidates - scan_start) + 9 * (write - scan_start); bucket_start[++num_buckets] = write; if (write == scan_start) break; // No progress — prevent infinite loop @@ -415,10 +416,9 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, entering_k = k; } } - work_estimate_ += 4 * (bucket_start[b + 1] - bucket_start[b]); + work_estimate_ += 2 + 5 * (b_end - b_start); if (entering_k >= 0) break; } - work_estimate_ += 2 * num_buckets; // Step = entering variable's breakpoint ratio num_buckets_used_ = num_buckets; @@ -440,10 +440,11 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // Record which bucket was selected used_fallback_ = false; + i_t pos = -1; for (i_t b = 0; b < num_buckets; b++) { if (entering_k >= 0) { // Find which bucket entering_k is in based on its position in candidates - i_t pos = -1; + pos = -1; for (i_t h = 0; h < num_candidates; h++) { if (candidates[h] == entering_k) { pos = h; break; } } @@ -453,6 +454,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, } } } + work_estimate_ += (bucket_selected_ + 1) * (pos + 3); step_length_result_ = step_length; determine_flips(step_length, entering_index, flip_indices); diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp index 3a87e923db..0d78e68ee0 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp @@ -88,7 +88,7 @@ class bound_flipping_ratio_test_t { f_t& max_val); void determine_flips(f_t step_length, i_t entering_index, - std::vector& flip_indices) const; + std::vector& flip_indices); const std::vector& lower_; const std::vector& upper_; const std::vector& bounded_variables_; From d983159ea1dbe0aee14320783f4093209dfee0ae Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 28 Aug 2026 17:46:23 -0700 Subject: [PATCH 026/113] Fix bug in not handling concurrent halt. Use zero_tol in step-length computation to try to remain dual feasible --- cpp/src/branch_and_bound/branch_and_bound.cpp | 91 +++++++++++++++---- 1 file changed, 75 insertions(+), 16 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 42d1453833..9b0c083635 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3354,6 +3354,16 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t= settings_.time_limit) { + solver_status_ = mip_status_t::TIME_LIMIT; + set_final_solution(solution, root_objective_); + return cut_pass_action_t::RETURN; + } // Score the cuts f_t score_start_time = tic(); @@ -4032,11 +4042,11 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( nonbasic_list, vstatus); if (refactor_status == CONCURRENT_HALT_RETURN || refactor_status == TIME_LIMIT_RETURN) { - // TODO: On failure vstatus, basic_list, and nonbasic_list are in a bad state. - // We should save copies before the failure and restore them after the failure. return; } if (refactor_status != 0) { + // TODO: On failure vstatus, basic_list, and nonbasic_list are in a bad state. + // We should save copies before the failure and restore them after the failure. settings_.log.printf( "Failed to refactor basis after dual degenerate feasibility pump. " "%d deficient columns.\n", @@ -4790,8 +4800,8 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( f_t work_estimate = 0; const f_t threshold = 100.0 * settings_.integer_tol; const f_t tol = 1e-2; - const f_t pivot_tol = settings_.pivot_tol; - const f_t dual_tol = settings_.dual_tol / 10; + const f_t zero_tol = settings_.zero_tol; + const f_t harris_tol = settings_.dual_tol / 10; i_t num_bounds_added = 0; for (i_t j : degenerate_integer_list) { @@ -4840,12 +4850,12 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( for (i_t jj : delta_z_indices) { if (vstatus[jj] == variable_status_t::NONBASIC_FIXED) { continue; } const f_t dz = scale * delta_z[jj]; - if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && dz < -pivot_tol) { - const f_t ratio = std::max((-dual_tol - soln.z[jj]) / dz, 0.0); + if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && dz < -zero_tol) { + const f_t ratio = std::max((-harris_tol - soln.z[jj]) / dz, 0.0); if (ratio < alpha) { alpha = ratio; } } - if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && dz > pivot_tol) { - const f_t ratio = std::max((dual_tol - soln.z[jj]) / dz, 0.0); + if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && dz > zero_tol) { + const f_t ratio = std::max((harris_tol - soln.z[jj]) / dz, 0.0); if (ratio < alpha) { alpha = ratio; } } } @@ -4855,24 +4865,48 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( // For NONBASIC_LOWER: z_new[jj] >= -dual_tol // For NONBASIC_UPPER: z_new[jj] <= dual_tol { - f_t max_dual_infeas = 0.0; - i_t num_dual_infeas = 0; - i_t worst_j = -1; + f_t max_initial_dual_infeas = 0.0; + f_t max_dual_infeas = 0.0; + f_t worst_old_z = 0.0; + f_t worst_delta_z = 0.0; + f_t worst_step = 0.0; + f_t worst_new_z = 0.0; + i_t num_initial_dual_infeas = 0; + i_t num_dual_infeas = 0; + i_t worst_j = -1; for (i_t jj : delta_z_indices) { if (vstatus[jj] == variable_status_t::NONBASIC_FIXED) { continue; } - const f_t new_zj = soln.z[jj] + alpha * scale * delta_z[jj]; + const f_t old_zj = soln.z[jj]; + const f_t step = alpha * scale * delta_z[jj]; + const f_t new_zj = old_zj + step; + const bool initially_infeasible = + (vstatus[jj] == variable_status_t::NONBASIC_LOWER && + old_zj < -settings_.dual_tol) || + (vstatus[jj] == variable_status_t::NONBASIC_UPPER && old_zj > settings_.dual_tol); + if (initially_infeasible) { + num_initial_dual_infeas++; + max_initial_dual_infeas = std::max(max_initial_dual_infeas, std::abs(old_zj)); + } if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && new_zj < -settings_.dual_tol) { num_dual_infeas++; if (std::abs(new_zj) > max_dual_infeas) { max_dual_infeas = std::abs(new_zj); - worst_j = jj; + worst_j = jj; + worst_old_z = old_zj; + worst_delta_z = scale * delta_z[jj]; + worst_step = step; + worst_new_z = new_zj; } } if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && new_zj > settings_.dual_tol) { num_dual_infeas++; if (std::abs(new_zj) > max_dual_infeas) { max_dual_infeas = std::abs(new_zj); - worst_j = jj; + worst_j = jj; + worst_old_z = old_zj; + worst_delta_z = scale * delta_z[jj]; + worst_step = step; + worst_new_z = new_zj; } } } @@ -4881,9 +4915,24 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( if (num_dual_infeas > 0) { settings_.log.printf( "WARNING pivot_to_improve_rc: dual infeasibility after step! " - "var=%d alpha=%.6e scale=%.0f num_infeas=%d max_infeas=%.6e worst_j=%d " + "var=%d alpha=%.6e scale=%.0f initial_num_infeas=%d " + "initial_max_infeas=%.6e num_infeas=%d max_infeas=%.6e worst_j=%d " + "worst_status=%d old_z=%.16e delta_z=%.16e step=%.16e new_z=%.16e " "new_rc_leaving=%.6e\n", - j, alpha, scale, num_dual_infeas, max_dual_infeas, worst_j, new_zj_leaving); + j, + alpha, + scale, + num_initial_dual_infeas, + max_initial_dual_infeas, + num_dual_infeas, + max_dual_infeas, + worst_j, + static_cast(vstatus[worst_j]), + worst_old_z, + worst_delta_z, + worst_step, + worst_new_z, + new_zj_leaving); } } @@ -5203,6 +5252,16 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut basis_update, num_fractional, fractional); + if (received_halt_signal()) { + solver_status_ = mip_status_t::HALT; + set_final_solution(solution, root_objective_); + return solver_status_; + } + if (toc(exploration_stats_.start_time) >= settings_.time_limit) { + solver_status_ = mip_status_t::TIME_LIMIT; + set_final_solution(solution, root_objective_); + return solver_status_; + } if (num_fractional != 0 && settings_.max_cut_passes > 0) { print_table_header(); } From a55f471d1050e520fea513c962edcdaa9abe2a24 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 2 Sep 2026 11:27:00 -0700 Subject: [PATCH 027/113] Add option to turn of dual degenerate feasibility pump. Silence pivot_out_integer_variables in the tree --- .../mathematical_optimization/constants.h | 1 + .../mip/solver_settings.hpp | 1 + cpp/src/branch_and_bound/branch_and_bound.cpp | 199 +++++++++++------- cpp/src/branch_and_bound/branch_and_bound.hpp | 24 ++- .../dual_simplex/simplex_solver_settings.hpp | 4 +- cpp/src/math_optimization/solver_settings.cu | 1 + cpp/src/mip_heuristics/solver.cu | 4 + 7 files changed, 143 insertions(+), 91 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/constants.h b/cpp/include/cuopt/mathematical_optimization/constants.h index 534accfe94..5339a58a24 100644 --- a/cpp/include/cuopt/mathematical_optimization/constants.h +++ b/cpp/include/cuopt/mathematical_optimization/constants.h @@ -81,6 +81,7 @@ #define CUOPT_MIP_ZERO_HALF_CUTS "mip_zero_half_cuts" #define CUOPT_MIP_STRONG_CHVATAL_GOMORY_CUTS "mip_strong_chvatal_gomory_cuts" #define CUOPT_MIP_REDUCED_COST_STRENGTHENING "mip_reduced_cost_strengthening" +#define CUOPT_MIP_DUAL_DEGENERATE_FEASIBILITY_PUMP "mip_dual_degenerate_feasibility_pump" #define CUOPT_MIP_RINS "mip_rins" #define CUOPT_MIP_RENS "mip_rens" #define CUOPT_MIP_OBJECTIVE_STEP "mip_objective_step" diff --git a/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp index 4a363b1dbc..b177e38500 100644 --- a/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp @@ -134,6 +134,7 @@ class mip_solver_settings_t { i_t implied_bound_cuts = -1; i_t strong_chvatal_gomory_cuts = -1; i_t reduced_cost_strengthening = -1; + i_t dual_degenerate_feasibility_pump = -1; // -1 = automatic (on), 0 = off, 1 = on i_t objective_step = 1; // 0 = disable objective step tightening, 1 = enable f_t cut_change_threshold = -1.0; f_t cut_min_orthogonality = 0.5; diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 9b0c083635..7369d8586e 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -1748,6 +1748,7 @@ dual_status_t branch_and_bound_t::solve_node_lp( i_t num_fractional = fractional_variables(settings_, worker->leaf_solution.x, var_types_, fractional); pivot_out_integer_variables(worker->leaf_problem, + lp_settings, worker->basic_list, worker->nonbasic_list, worker->leaf_vstatus, @@ -3337,23 +3338,26 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t::apply_delta_x_for_integer_pivot( } template -void branch_and_bound_t::fast_slack_integer_pivot( +void branch_and_bound_t::fast_slack_integer_pivots( const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, const std::vector& fractional, const std::vector& row_to_slack, const simplex::lp_solution_t& solution, @@ -4254,8 +4259,8 @@ void branch_and_bound_t::fast_slack_integer_pivot( } } - if (fast_candidates.size() > 0) { - settings_.log.printf("Found %ld fast candidates for pivot out integer variables\n", + if (fast_candidates.size() > 0 && settings_.inside_mip < 2) { + settings.log.printf("Found %ld fast candidates for pivot out integer variables\n", fast_candidates.size()); } @@ -4377,22 +4382,26 @@ void branch_and_bound_t::fast_slack_integer_pivot( work_estimate); // apply_delta_x_for_integer_pivot only mutates vstatus when the pivot actually fires, // so entering_index transitioning to BASIC is a reliable success signal. - if (!error) { - settings_.log.printf( + if (!error && settings.inside_mip < 2) { + settings.log.printf( "Fast candidate pivot succeeded: j=%d entering slack=%d row=%d\n", j, entering_index, row); } - if (toc(last_log) > 1.0) { - settings_.log.printf("Fast candidates %d/%d processed in %.2f seconds\n", k + 1, num_candidates, toc(loop_start)); + if (settings.inside_mip < 2 && toc(last_log) > 1.0) { + settings.log.printf("Fast candidates %d/%d processed in %.2f seconds\n", k + 1, num_candidates, toc(loop_start)); last_log = tic(); } } - settings_.log.printf("Fast candidates: %d/%d processed in %.2f seconds\n", num_candidates, num_candidates, toc(loop_start)); + if (settings.inside_mip < 2) + { + settings.log.printf("Fast candidates: %d/%d processed in %.2f seconds\n", num_candidates, num_candidates, toc(loop_start)); + } } template i_t branch_and_bound_t::pivot_out_integer_variables( const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, std::vector& basic_list, std::vector& nonbasic_list, std::vector& vstatus, @@ -4446,34 +4455,40 @@ i_t branch_and_bound_t::pivot_out_integer_variables( } } const f_t degeneracy_fraction = static_cast(num_degenerate) / lp.num_rows; - settings_.log.printf( - "Primal degeneracy: %d/%d basic variables are degenerate (%.1f%%), " - "continuous=%d, integer=%d\n", - num_degenerate, lp.num_rows, - 100.0 * degeneracy_fraction, - num_degenerate_continuous, num_degenerate_integer); + if (settings.inside_mip < 2 && settings.inside_submip == 0) { + settings.log.printf( + "Primal degeneracy: %d/%d basic variables are degenerate (%.1f%%), " + "continuous=%d, integer=%d\n", + num_degenerate, + lp.num_rows, + 100.0 * degeneracy_fraction, + num_degenerate_continuous, + num_degenerate_integer); + } // Skip pivot_out entirely if primal degeneracy is too high — the ratio test // will almost always be won by a degenerate variable, making pivots hopeless. if (degeneracy_fraction > 0.5) { - settings_.log.printf("Skipping pivot_out_integer_variables: degeneracy %.1f%% > 50%%\n", - 100.0 * degeneracy_fraction); + if (settings.inside_mip < 2) { + settings.log.printf("Skipping pivot_out_integer_variables: degeneracy %.1f%% > 50%%\n", + 100.0 * degeneracy_fraction); + } return 0; } std::vector nonbasic_index; - fast_slack_integer_pivot(lp, - fractional, - row_to_slack, - solution, - basic_list_copy, - nonbasic_list_copy, - nonbasic_index, - vstatus_copy, - soln_copy, - basis_update_copy, - work_estimate); - + fast_slack_integer_pivots(lp, + settings, + fractional, + row_to_slack, + solution, + basic_list_copy, + nonbasic_list_copy, + nonbasic_index, + vstatus_copy, + soln_copy, + basis_update_copy, + work_estimate); std::vector work_list = fractional; std::vector to_basic_position(lp.num_cols, -1); @@ -4682,13 +4697,21 @@ i_t branch_and_bound_t::pivot_out_integer_variables( } if (toc(worklist_last_log) > 1.0) { - settings_.log.printf( - "Worklist progress: %d/%d processed, %d pivots, %d ratio_fail (unb=%d cont=%d nfint=%d), %d net_inc_fail, %d no_cand, %.2f seconds\n", - worklist_total_processed, static_cast(fractional.size()), - worklist_pivots_succeeded, worklist_ratio_test_fail, - worklist_unbounded, worklist_continuous_won, worklist_nonfrac_int_won, - worklist_net_increase_fail, - worklist_no_candidates, toc(worklist_loop_start)); + if (settings.inside_mip < 2) { + settings.log.printf( + "Worklist progress: %d/%d processed, %d pivots, %d ratio_fail (unb=%d cont=%d nfint=%d), " + "%d net_inc_fail, %d no_cand, %.2f seconds\n", + worklist_total_processed, + static_cast(fractional.size()), + worklist_pivots_succeeded, + worklist_ratio_test_fail, + worklist_unbounded, + worklist_continuous_won, + worklist_nonfrac_int_won, + worklist_net_increase_fail, + worklist_no_candidates, + toc(worklist_loop_start)); + } worklist_last_log = tic(); } } @@ -4706,25 +4729,39 @@ i_t branch_and_bound_t::pivot_out_integer_variables( else { entering_tried_multiple++; } } } - settings_.log.printf( - "Worklist entering stats: unique=%d, tried_once=%d, tried_multiple=%d, " - "max_count=%d, total_ftran=%d, duplication_ratio=%.1fx\n", - unique_entering, entering_tried_once, entering_tried_multiple, - max_entering_count, worklist_ftran_done, - worklist_ftran_done / std::max(1.0, static_cast(unique_entering))); - - settings_.log.printf( - "Worklist stats: processed=%d skipped=%d btran=%d ftran=%d pivots=%d readded=%d " - "btran_time=%.2f dot_time=%.2f ftran_time=%.2f zero_rc_vars=%d " - "no_candidates=%d ratio_test_fail=%d (unbounded=%d continuous_won=%d nonfrac_int_won=%d) " - "net_increase_fail=%d\n", - worklist_total_processed, worklist_skipped, worklist_btran_done, worklist_ftran_done, - worklist_pivots_succeeded, worklist_readded, - worklist_btran_time, worklist_dot_time, worklist_ftran_time, - num_zero_reduced_costs_vars, - worklist_no_candidates, worklist_ratio_test_fail, - worklist_unbounded, worklist_continuous_won, worklist_nonfrac_int_won, - worklist_net_increase_fail); + if (settings.inside_mip < 2) { + settings.log.printf( + "Worklist entering stats: unique=%d, tried_once=%d, tried_multiple=%d, " + "max_count=%d, total_ftran=%d, duplication_ratio=%.1fx\n", + unique_entering, + entering_tried_once, + entering_tried_multiple, + max_entering_count, + worklist_ftran_done, + worklist_ftran_done / std::max(1.0, static_cast(unique_entering))); + + settings.log.printf( + "Worklist stats: processed=%d skipped=%d btran=%d ftran=%d pivots=%d readded=%d " + "btran_time=%.2f dot_time=%.2f ftran_time=%.2f zero_rc_vars=%d " + "no_candidates=%d ratio_test_fail=%d (unbounded=%d continuous_won=%d nonfrac_int_won=%d) " + "net_increase_fail=%d\n", + worklist_total_processed, + worklist_skipped, + worklist_btran_done, + worklist_ftran_done, + worklist_pivots_succeeded, + worklist_readded, + worklist_btran_time, + worklist_dot_time, + worklist_ftran_time, + num_zero_reduced_costs_vars, + worklist_no_candidates, + worklist_ratio_test_fail, + worklist_unbounded, + worklist_continuous_won, + worklist_nonfrac_int_won, + worklist_net_increase_fail); + } std::vector new_fractional; const i_t num_new_fractional = @@ -4733,7 +4770,7 @@ i_t branch_and_bound_t::pivot_out_integer_variables( i_t num_integer_increased = start_num_fractional - num_new_fractional; integer_pivots_.fetch_add(num_integer_increased, std::memory_order_release); #if 0 - settings_.log.printf("Pivoted out %d integer variables: %d -> %d in %.2f\n", + settings.log.printf("Pivoted out %d integer variables: %d -> %d in %.2f\n", num_integer_increased, start_num_fractional, num_new_fractional, @@ -5235,6 +5272,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut f_t pivot_out_integer_variables_start_time = tic(); i_t num_integer_increased = pivot_out_integer_variables(original_lp_, + settings_, basic_list, nonbasic_list, root_vstatus_, @@ -5244,14 +5282,17 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut fractional); settings_.log.printf("Pivoted out %d integer variables in %e seconds\n", num_integer_increased, toc(pivot_out_integer_variables_start_time)); - dual_degenerate_feasibility_pump(original_lp_, - basic_list, - nonbasic_list, - root_vstatus_, - root_relax_soln_, - basis_update, - num_fractional, - fractional); + if (settings_.dual_degenerate_feasibility_pump != 0) { + dual_degenerate_feasibility_pump(original_lp_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional); + } + if (received_halt_signal()) { solver_status_ = mip_status_t::HALT; set_final_solution(solution, root_objective_); diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index a20e3d7551..52fc57e0de 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -505,19 +505,21 @@ class branch_and_bound_t { std::vector& zero_reduced_costs_vars, std::vector& zero_reduced_costs_vars_nonbasic_index); - void fast_slack_integer_pivot(const simplex::lp_problem_t& lp, - const std::vector& fractional, - const std::vector& row_to_slack, - const simplex::lp_solution_t& solution, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& nonbasic_index, - std::vector& vstatus, - simplex::lp_solution_t& soln, - simplex::basis_update_mpf_t& basis_update, - f_t& work_estimate); + void fast_slack_integer_pivots(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const std::vector& fractional, + const std::vector& row_to_slack, + const simplex::lp_solution_t& solution, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& nonbasic_index, + std::vector& vstatus, + simplex::lp_solution_t& soln, + simplex::basis_update_mpf_t& basis_update, + f_t& work_estimate); i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, std::vector& basic_list, std::vector& nonbasic_list, std::vector& vstatus, diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 4acfd97107..1f85a1baf3 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -103,6 +103,7 @@ struct simplex_solver_settings_t { strong_chvatal_gomory_cuts(-1), symmetry(-1), reduced_cost_strengthening(-1), + dual_degenerate_feasibility_pump(1), cut_change_threshold(1e-3), cut_min_orthogonality(0.5), mip_batch_pdlp_strong_branching(0), @@ -208,7 +209,8 @@ struct simplex_solver_settings_t { // cuts i_t symmetry; // -1 automatic, 0 to disable, >0 to enable different symmetry methods i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost - // strengthening + // strengthening + i_t dual_degenerate_feasibility_pump; // 0 to disable, 1 to enable f_t cut_change_threshold; // threshold for cut change f_t cut_min_orthogonality; // minimum orthogonality for cuts i_t diff --git a/cpp/src/math_optimization/solver_settings.cu b/cpp/src/math_optimization/solver_settings.cu index 8288db92ff..8dd5266a73 100644 --- a/cpp/src/math_optimization/solver_settings.cu +++ b/cpp/src/math_optimization/solver_settings.cu @@ -152,6 +152,7 @@ solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings {CUOPT_MIP_IMPLIED_BOUND_CUTS, &mip_settings.implied_bound_cuts, -1, 1, -1}, {CUOPT_MIP_STRONG_CHVATAL_GOMORY_CUTS, &mip_settings.strong_chvatal_gomory_cuts, -1, 1, -1}, {CUOPT_MIP_REDUCED_COST_STRENGTHENING, &mip_settings.reduced_cost_strengthening, -1, std::numeric_limits::max(), -1}, + {CUOPT_MIP_DUAL_DEGENERATE_FEASIBILITY_PUMP, &mip_settings.dual_degenerate_feasibility_pump, -1, 1, -1}, {CUOPT_MIP_RINS, &mip_settings.submip_params.rins, -1, 1, -1}, {CUOPT_MIP_RENS, &mip_settings.submip_params.rens, -1, 1, -1}, {CUOPT_MIP_OBJECTIVE_STEP, &mip_settings.objective_step, 0, 1, 1}, diff --git a/cpp/src/mip_heuristics/solver.cu b/cpp/src/mip_heuristics/solver.cu index f8eac0c4d8..05cbf95841 100644 --- a/cpp/src/mip_heuristics/solver.cu +++ b/cpp/src/mip_heuristics/solver.cu @@ -388,6 +388,10 @@ solution_t mip_solver_t::run_solver() context.settings.reduced_cost_strengthening == -1 ? 2 : context.settings.reduced_cost_strengthening; + branch_and_bound_settings.dual_degenerate_feasibility_pump = + context.settings.dual_degenerate_feasibility_pump == -1 + ? 1 + : context.settings.dual_degenerate_feasibility_pump; branch_and_bound_settings.symmetry = context.settings.symmetry; branch_and_bound_settings.diving_settings = context.settings.diving_params; From 435689335ffa9fd7ea43f000222f926a7a903c23 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 2 Sep 2026 14:03:00 -0700 Subject: [PATCH 028/113] Fix reduced-cost bounds for degenerate variables Signed-off-by: Christopher Maes --- cpp/src/branch_and_bound/branch_and_bound.cpp | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 7369d8586e..f6d045bda6 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -4876,6 +4876,8 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( const f_t lower_j = lp.lower[j]; const f_t upper_j = lp.upper[j]; + const bool at_lower = soln.x[j] - lower_j <= settings_.primal_tol; + const bool at_upper = upper_j - soln.x[j] <= settings_.primal_tol; // Try both dual rays. scale == +1 uses the computed delta_z (direction -1); // scale == -1 uses -delta_z (direction +1). Either or both may yield an RCS bound. @@ -4984,7 +4986,7 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( // This means l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] <= // u_tilde_j Or equivalently, incumbent_objective <= relaxation_objective + // reduced_costs[j] * (u_tilde_j - l_j) when reduced_costs[j] > 0 - if (lower_j > -inf && new_reduced_cost > threshold) { + if (at_lower && lower_j > -inf && new_reduced_cost > threshold) { const f_t u_tilde_j = var_types_[j] == variable_type_t::INTEGER ? upper_j - tol : std::max(upper_j - 1.0, lower_j); @@ -5007,7 +5009,7 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( // u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j Or // equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * // (l_tilde_j - u_j) when reduced_costs[j] < 0 - if (upper_j < inf && new_reduced_cost < -threshold) { + if (at_upper && upper_j < inf && new_reduced_cost < -threshold) { const f_t l_tilde_j = var_types_[j] == variable_type_t::INTEGER ? lower_j + tol : std::min(lower_j + 1.0, upper_j); From d2fb7fbd4ea7ee715c769989803c66326bd9b67a Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 2 Sep 2026 15:57:12 -0700 Subject: [PATCH 029/113] Equilibrate LP rows and strengthen degenerate perturbations These improvements were discovered through Hiverge's automated exploration of changes to cuOpt's dual simplex solver and then isolated and validated independently against the v8 baseline. Apply row equilibration to imbalanced linear programs before the existing column normalization. For each row, divide the matrix coefficients and right-hand side by the row infinity norm when the maximum-to-minimum row norm ratio exceeds 10. Restrict this equilibration to standalone linear programs. MIP already performs integer-aware row scaling before presolve, and QP and SOCP problems use the existing iterative Ruiz equilibration path. Increase the initial objective perturbation from 5e-7 * max_abs_objective to 1e-5 * max_abs_objective when more than 5% of the nonbasic variables are dual degenerate. Retain the original perturbation strength for other problems. The stronger perturbation separates coincident and nearly coincident reduced costs. Across the benchmark it reduces the BFRT zero-step rate from 34.0% with row scaling alone to 15.4%. Row equilibration improves the numerical conditioning of models with imbalanced constraint rows and prevents several expensive or failed cleanup trajectories. On 240 MIPLIB LP relaxations, v9 improves the raw runtime geometric mean by 1.1865x relative to v8 and the one-second-shifted geometric mean by 1.1037x. It solves 229 problems within 300 seconds, compared with 226 for v8. The HiGHS times were taken from a faster machine, so the apparent HiGHS advantage is slightly exaggerated. Problem v8 v9 Baseline HiGHS v9/HiGHS ------------------------------------------------------------------------------------ momentum1 0.70 0.69 0.69 300.00 0.00 var-smallemery-m6j6 0.72 0.69 0.66 300.00 0.00 neos-5114902-kasavu 4.91 2.74 87.40 300.00 0.01 supportcase42 0.78 0.80 0.52 36.20 0.02 neos-5049753-cuanza 1.91 1.02 7.68 29.85 0.03 roi5alpha10n8 3.42 1.35 1.25 11.90 0.11 mzzv11 3.45 2.40 40.19 16.71 0.14 supportcase12 4.72 6.35 4.68 37.11 0.17 roi2alpha3n4 0.22 0.22 0.31 1.16 0.19 co-100 0.75 0.28 0.68 1.28 0.22 proteindesign121hz512p9 0.37 0.46 0.91 2.04 0.23 neos-5104907-jarama 52.83 21.53 124.69 89.64 0.24 supportcase18 0.03 0.03 0.06 0.12 0.25 ns1644855 300.00 63.39 300.00 238.30 0.27 proteindesign122trx11p8 0.26 0.35 0.64 1.26 0.28 neos-1354092 300.00 20.05 300.00 70.92 0.28 ns1952667 0.62 0.26 8.95 0.80 0.33 rd-rplusc-21 0.19 0.16 0.23 0.49 0.33 neos-787933 0.07 0.06 0.06 0.18 0.33 neos-5052403-cygnet 110.25 109.39 300.00 300.00 0.36 neos-860300 0.10 0.06 0.10 0.15 0.40 ns1760995 114.63 114.33 135.94 269.53 0.42 sct2 0.15 0.06 0.22 0.14 0.43 rocII-5-11 0.10 0.10 0.09 0.21 0.48 supportcase7 1.36 1.68 1.29 3.52 0.48 neos-5093327-huahum 0.25 0.23 0.23 0.48 0.48 satellites2-40 186.43 5.40 29.70 10.51 0.51 physiciansched6-2 3.84 3.60 11.15 6.72 0.54 neos-4647030-tutaki 1.72 1.68 2.47 3.09 0.54 n3div36 0.11 0.10 0.11 0.18 0.56 supportcase22 0.73 2.28 2.03 3.97 0.57 neos-4300652-rahue 0.54 0.46 1.22 0.80 0.57 neos-3004026-krka 0.07 0.07 0.06 0.12 0.58 neos-5107597-kakapo 0.14 0.06 0.04 0.10 0.60 lectsched-5-obj 0.10 0.11 0.13 0.18 0.61 neos-5188808-nattai 0.18 0.18 0.30 0.29 0.62 cvs16r128-89 1.13 1.07 0.93 1.72 0.62 30n20b8 0.06 0.07 0.08 0.11 0.64 blp-ar98 0.11 0.07 0.10 0.11 0.64 decomp2 0.12 0.08 0.19 0.12 0.67 swath3 0.03 0.02 0.03 0.03 0.67 neos-1171448 2.41 0.58 2.35 0.84 0.69 wachplan 0.31 0.18 0.25 0.26 0.69 neos-5195221-niemur 0.30 0.26 0.50 0.37 0.70 thor50dday 0.28 0.26 0.27 0.37 0.70 supportcase10 133.84 79.85 300.00 113.55 0.70 supportcase33 0.80 0.40 0.99 0.55 0.73 square47 95.31 93.64 79.87 126.92 0.74 neos-3381206-awhea 0.05 0.03 0.08 0.04 0.75 buildingenergy 300.00 87.82 300.00 115.61 0.76 cryptanalysiskb128n5obj14 9.28 9.55 29.78 12.54 0.76 neos-2746589-doon 4.11 2.36 7.53 3.02 0.78 nursesched-medium-hint03 6.00 3.71 10.47 4.73 0.78 blp-ic98 0.09 0.08 0.12 0.10 0.80 neos-848589 0.97 0.70 1.24 0.85 0.82 dano3_3 42.60 16.02 46.88 19.06 0.84 dano3_5 42.62 16.04 46.79 19.03 0.84 neos-1171737 0.54 0.17 0.70 0.20 0.85 academictimetablesmall 0.90 0.70 14.89 0.82 0.85 tbfp-network 7.80 7.72 9.06 9.04 0.85 ns1116954 25.16 9.20 156.23 10.73 0.86 neos-3402454-bohle 44.09 64.41 221.86 74.15 0.87 comp21-2idx 0.34 0.31 1.55 0.35 0.89 leo2 0.13 0.12 0.13 0.13 0.92 radiationm18-12-05 0.18 0.13 0.23 0.14 0.93 neos-3555904-turama 1.29 1.31 1.31 1.37 0.96 mzzv42z 1.42 0.91 10.10 0.95 0.96 square41 39.17 38.25 28.57 39.86 0.96 cryptanalysiskb128n5obj16 8.93 10.05 29.53 10.27 0.98 drayage-100-23 0.06 0.04 0.07 0.04 1.00 dws008-01 0.03 0.05 0.04 0.05 1.00 h80x6320d 0.05 0.04 0.05 0.04 1.00 highschool1-aigio 300.00 300.00 300.00 300.00 1.00 icir97_tension 0.03 0.02 0.03 0.02 1.00 leo1 0.12 0.07 0.10 0.07 1.00 neos-3988577-wolgan 300.00 300.00 278.89 300.00 1.00 neos-4738912-atrato 0.04 0.04 0.05 0.04 1.00 neos8 0.26 0.26 0.38 0.26 1.00 nursesched-sprint02 0.26 0.23 0.38 0.23 1.00 physiciansched3-3 288.13 300.00 300.00 300.00 1.00 rail02 300.00 300.00 300.00 300.00 1.00 s100 300.00 300.00 300.00 300.00 1.00 savsched1 300.00 300.00 300.00 300.00 1.00 supportcase19 300.00 300.00 300.00 300.00 1.00 swath1 0.04 0.03 0.04 0.03 1.00 traininstance2 0.04 0.04 0.09 0.04 1.00 s250r10 300.00 71.91 300.00 71.74 1.00 neos-873061 1.47 1.46 1.46 1.38 1.06 fiball 0.66 0.15 0.71 0.14 1.07 neos-3402294-bobin 1.72 1.29 3.07 1.20 1.08 neos-4413714-turia 20.16 2.10 3.97 1.93 1.09 sp98ar 0.37 0.34 0.39 0.31 1.10 germanrr 0.30 0.30 0.31 0.27 1.11 cod105 8.56 8.39 9.06 7.46 1.12 comp07-2idx 1.47 1.26 4.03 1.10 1.15 neos-1122047 2.43 1.87 2.09 1.61 1.16 neos-4763324-toguru 9.86 5.82 8.32 5.00 1.16 supportcase6 5.52 4.93 7.39 4.18 1.18 supportcase40 0.29 0.29 0.24 0.24 1.21 neos-960392 3.30 3.29 8.93 2.66 1.24 drayage-25-23 0.05 0.05 0.08 0.04 1.25 neos-1456979 0.08 0.05 0.05 0.04 1.25 neos-3216931-puriri 3.50 4.20 7.33 3.23 1.30 cmflsp50-24-8-8 0.74 0.58 0.77 0.44 1.32 irp 0.13 0.12 0.11 0.09 1.33 sp97ar 0.39 0.44 0.40 0.33 1.33 neos-1582420 0.09 0.08 0.12 0.06 1.33 traininstance6 0.03 0.04 0.04 0.03 1.33 radiationm40-10-02 1.45 0.97 1.58 0.71 1.37 map10 19.82 8.54 11.08 6.25 1.37 cbs-cta 0.48 0.14 0.42 0.10 1.40 mushroom-best 0.26 0.28 0.26 0.20 1.40 fast0507 4.05 3.95 7.36 2.77 1.43 bnatt500 0.21 0.20 0.29 0.14 1.43 nexp-150-20-8-5 0.11 0.10 0.10 0.07 1.43 map16715-04 20.42 9.87 13.26 6.80 1.45 qap10 12.87 9.80 15.57 6.68 1.47 hypothyroid-k1 4.93 4.42 4.36 3.01 1.47 fhnw-binpack4-48 0.03 0.09 0.07 0.06 1.50 rocI-4-11 0.11 0.09 0.11 0.06 1.50 uct-subprob 0.17 0.12 0.11 0.08 1.50 neos-827175 1.66 0.44 9.36 0.29 1.52 rail507 4.64 4.37 7.17 2.84 1.54 uccase9 13.22 8.04 11.54 5.20 1.55 neos-4532248-waihi 0.97 1.40 2.61 0.90 1.56 reblock115 0.22 0.14 0.16 0.09 1.56 neos-957323 16.74 12.14 300.00 7.74 1.57 trento1 3.80 3.50 3.08 2.21 1.58 neos-2987310-joes 1.91 1.59 1.52 1.00 1.59 neos-3024952-loue 0.38 0.32 0.41 0.20 1.60 air05 0.27 0.29 0.28 0.18 1.61 neos-662469 0.93 0.86 1.65 0.53 1.62 bnatt400 0.16 0.13 0.16 0.08 1.62 neos-3083819-nubu 0.05 0.05 0.06 0.03 1.67 ran14x18-disj-8 0.05 0.05 0.04 0.03 1.67 roll3000 0.13 0.10 0.12 0.06 1.67 k1mushroom 32.65 28.33 31.00 16.80 1.69 opm2-z10-s4 87.53 75.72 89.23 44.75 1.69 istanbul-no-cutoff 0.71 1.63 0.71 0.94 1.73 neos-3656078-kumeu 1.27 0.40 2.37 0.23 1.74 rmatr200-p5 8.13 8.10 7.50 4.61 1.76 ns1830653 0.29 0.25 0.42 0.14 1.79 neos-2978193-inde 0.08 0.09 0.16 0.05 1.80 irish-electricity 183.12 109.17 181.58 59.53 1.83 mcsched 0.30 0.30 0.27 0.16 1.88 sing326 12.30 9.33 9.54 4.69 1.99 assign1-5-8 0.04 0.02 0.03 0.01 2.00 bppc4-08 0.06 0.04 0.08 0.02 2.00 eil33-2 0.06 0.06 0.05 0.03 2.00 ic97_potential 0.03 0.02 0.02 0.01 2.00 mik-250-20-75-4 0.02 0.02 0.02 0.01 2.00 neos-4338804-snowy 0.02 0.02 0.03 0.01 2.00 neos-4387871-tavua 0.09 0.08 0.10 0.04 2.00 neos-4954672-berkel 0.03 0.02 0.02 0.01 2.00 nu25-pr12 0.04 0.04 0.05 0.02 2.00 pg 0.02 0.02 0.03 0.01 2.00 tr12-30 0.03 0.02 0.03 0.01 2.00 satellites2-60-fs 9.13 6.89 4.17 3.34 2.06 ns1208400 1.17 0.76 3.96 0.36 2.11 splice1k1 23.93 19.92 21.62 9.17 2.17 sing44 11.01 12.54 9.62 5.69 2.20 gfd-schedulen180f7d50m30k18 19.47 15.81 80.00 6.81 2.32 rmatr100-p10 0.32 0.33 0.27 0.14 2.36 eilA101-2 3.35 3.29 2.68 1.39 2.37 netdiversion 18.16 22.67 7.64 9.43 2.40 bab6 44.66 34.02 143.43 14.06 2.42 chromaticindex512-7 40.10 53.03 16.62 21.24 2.50 graph20-20-1rand 0.36 0.30 0.23 0.12 2.50 n5-3 0.05 0.05 0.03 0.02 2.50 rococoB10-011000 0.15 0.15 0.13 0.06 2.50 seymour 0.98 0.95 0.83 0.38 2.50 seymour1 0.94 0.97 0.83 0.38 2.55 neos-950242 0.66 0.46 1.05 0.18 2.56 piperout-08 0.17 0.41 0.39 0.16 2.56 triptim1 58.50 132.50 71.42 51.43 2.58 bab2 70.29 75.19 300.00 28.97 2.60 atlanta-ip 8.14 12.20 6.85 4.54 2.69 unitcal_7 0.91 1.05 0.96 0.39 2.69 net12 0.50 0.96 0.55 0.35 2.74 glass-sc 0.34 0.33 0.31 0.12 2.75 neos-1445765 0.30 0.25 0.16 0.09 2.78 chromaticindex1024-7 220.63 262.17 45.86 93.65 2.80 nw04 1.75 1.71 0.44 0.61 2.80 sorrell3 1.25 1.14 1.07 0.40 2.85 n2seq36q 0.74 0.71 0.46 0.24 2.96 50v-10 0.02 0.03 0.02 0.01 3.00 b1c1s1 0.05 0.06 0.05 0.02 3.00 binkar10_1 0.03 0.03 0.02 0.01 3.00 cost266-UUE 0.05 0.06 0.03 0.02 3.00 fhnw-binpack4-4 0.02 0.03 0.03 0.01 3.00 gmu-35-40 0.04 0.03 0.03 0.01 3.00 lotsize 0.04 0.03 0.03 0.01 3.00 neos17 0.03 0.03 0.03 0.01 3.00 pg5_34 0.02 0.03 0.03 0.01 3.00 rococoC10-001000 0.03 0.06 0.04 0.02 3.00 csched007 0.12 0.16 0.19 0.05 3.20 neos-933966 9.31 9.15 16.79 2.80 3.27 milo-v12-6-r2-40-1 0.22 0.18 0.25 0.05 3.60 gmu-35-50 0.04 0.04 0.05 0.01 4.00 neos-3627168-kasai 0.04 0.04 0.04 0.01 4.00 p200x1188c 0.03 0.04 0.03 0.01 4.00 peg-solitaire-a3 2.81 2.14 1.88 0.52 4.12 uccase12 4.77 5.76 72.32 1.38 4.17 rail01 167.23 300.00 223.84 71.23 4.21 csched008 0.09 0.13 0.11 0.03 4.33 CMS750_4 0.43 0.63 0.38 0.14 4.50 piperout-27 0.28 1.23 0.67 0.26 4.73 neos-4722843-widden 2.21 7.67 1.17 1.54 4.98 app1-1 0.09 0.10 0.08 0.02 5.00 fastxgemm-n2r6s0t2 0.34 0.40 0.18 0.08 5.00 ex10 300.00 300.00 300.00 59.25 5.06 neos-2075418-temuka 300.00 300.00 128.30 50.57 5.93 beasleyC3 0.05 0.06 0.05 0.01 6.00 neos-631710 300.00 215.36 300.00 31.97 6.74 mc11 0.05 0.08 0.06 0.01 8.00 app1-2 5.26 5.77 5.03 0.70 8.24 snp-02-004-104 19.21 24.10 14.17 2.83 8.52 brazil3 79.55 61.89 300.00 7.24 8.55 gen-ip002 0.02 0.01 0.02 0.00 10.00 markshare2 0.02 0.01 0.02 0.00 10.00 markshare_4_0 0.02 0.01 0.02 0.00 10.00 mas74 0.02 0.01 0.03 0.00 10.00 neos859080 0.02 0.01 0.01 0.00 10.00 pk1 0.02 0.01 0.02 0.00 10.00 enlight_hard 0.02 0.02 0.02 0.00 20.00 exp-1-500-5-5 0.02 0.02 0.02 0.00 20.00 gen-ip054 0.02 0.02 0.03 0.00 20.00 glass4 0.02 0.02 0.02 0.00 20.00 graphdraw-domain 0.03 0.02 0.02 0.00 20.00 mad 0.02 0.02 0.02 0.00 20.00 mas76 0.02 0.02 0.02 0.00 20.00 neos-3046615-murg 0.02 0.02 0.02 0.00 20.00 neos-3754480-nidda 0.02 0.02 0.02 0.00 20.00 neos5 0.02 0.02 0.02 0.00 20.00 sp150x300d 0.02 0.02 0.02 0.00 20.00 supportcase26 0.02 0.02 0.03 0.00 20.00 timtab1 0.02 0.02 0.01 0.00 20.00 ex9 300.00 300.00 300.00 14.07 21.32 neos-2657525-crna 0.03 0.03 0.03 0.00 30.00 neos-911970 0.02 0.03 0.03 0.00 30.00 Geomean v8/v9: 1.1865 Shifted(+1s): 1.1037 Geomean Baseline/v9: 1.4170 Shifted(+1s): 1.2561 Geomean v9/HiGHS: 1.5257 Shifted(+1s): 1.0133 (240 problems) --- cpp/src/dual_simplex/phase2.cpp | 11 ++++++++--- cpp/src/dual_simplex/scaling.cpp | 33 ++++++++++++++++++++++++++++++++ 2 files changed, 41 insertions(+), 3 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 123d6354f8..f7e5d6c7f3 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -455,9 +455,9 @@ template void initial_perturbation(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& vstatus, + bool strongly_degenerate, std::vector& objective) { - const i_t m = lp.num_rows; const i_t n = lp.num_cols; f_t max_abs_obj_coeff = 0.0; for (i_t j = 0; j < n; ++j) { @@ -484,7 +484,11 @@ void initial_perturbation(const lp_problem_t& lp, max_abs_obj_coeff = std::min(max_abs_obj_coeff, f_t(1.0)); } - const f_t perturbation_base = 5e-7 * max_abs_obj_coeff; + // Sub-tolerance perturbations are less disruptive on ordinary problems, but + // are too small to separate reduced costs when a substantial part of the + // nonbasic set is dual degenerate. Use a stronger, still temporary shift in + // that case. The original costs are restored before declaring optimality. + const f_t perturbation_base = (strongly_degenerate ? 1e-5 : 5e-7) * max_abs_obj_coeff; settings.log.printf("Perturbation debug: max_abs_obj_coeff=%e (dampened), perturbation_base=%e, n=%d, num_boxed=%d\n", max_abs_obj_coeff, perturbation_base, n, num_boxed); @@ -3104,7 +3108,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, settings.log.printf("Near-optimal check: num_primal_infeas=%d, max_primal_infeas=%.2e, near_optimal=%d, apply_perturbation=%d\n", num_primal_infeas, max_primal_infeas, near_optimal, apply_perturbation); if (apply_perturbation) { - phase2::initial_perturbation(lp, settings, vstatus, objective); + const bool strongly_degenerate = num_degen > n / 20; + phase2::initial_perturbation(lp, settings, vstatus, strongly_degenerate, objective); // Recompute y, z with perturbed objective for (i_t k = 0; k < m; ++k) { c_basic[k] = objective[basic_list[k]]; diff --git a/cpp/src/dual_simplex/scaling.cpp b/cpp/src/dual_simplex/scaling.cpp index 98c409a630..8baaa0a8c7 100644 --- a/cpp/src/dual_simplex/scaling.cpp +++ b/cpp/src/dual_simplex/scaling.cpp @@ -253,6 +253,39 @@ i_t scaling(const lp_problem_t& unscaled, return 0; } + // MIP performs integer-aware row scaling before presolve, while QP and SOCP + // use the Ruiz path above. Apply this simpler equilibration only to LPs. + const bool use_lp_row_scaling = !settings.inside_mip && unscaled.second_order_cone_dims.empty() && + unscaled.Q.n == 0; + if (use_lp_row_scaling) { + csr_matrix_t Arow(0, 0, 0); + scaled.A.to_compressed_row(Arow); + std::vector row_norm(m, 1.0); + f_t max_row_norm = 0.0; + f_t min_row_norm = inf; + for (i_t i = 0; i < m; ++i) { + for (i_t p = Arow.row_start[i]; p < Arow.row_start[i + 1]; ++p) { + row_norm[i] = std::max(row_norm[i], std::abs(Arow.x[p])); + } + max_row_norm = std::max(max_row_norm, row_norm[i]); + min_row_norm = std::min(min_row_norm, row_norm[i]); + } + if (min_row_norm > 0.0 && max_row_norm / min_row_norm > 10.0) { + settings.log.printf("Applying row scaling. Maximum row norm %e, minimum row norm %e\n", + max_row_norm, + min_row_norm); + for (i_t j = 0; j < n; ++j) { + for (i_t p = scaled.A.col_start[j]; p < scaled.A.col_start[j + 1]; ++p) { + scaled.A.x[p] /= row_norm[scaled.A.i[p]]; + } + } + for (i_t i = 0; i < m; ++i) { + scaled.rhs[i] /= row_norm[i]; + row_scaling[i] = row_norm[i]; + } + } + } + column_scaling.resize(n); f_t max = 0; f_t min = std::numeric_limits::max(); From bfcbb158ecaa3c0d835fc7f309a105ca85f3895a Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 2 Sep 2026 17:42:18 -0700 Subject: [PATCH 030/113] Style fixes --- .../mip/solver_settings.hpp | 10 +- cpp/src/branch_and_bound/branch_and_bound.cpp | 350 ++++++++------ cpp/src/branch_and_bound/branch_and_bound.hpp | 74 ++- .../bound_flipping_ratio_test.cpp | 163 +++---- .../bound_flipping_ratio_test.hpp | 32 +- cpp/src/dual_simplex/phase2.cpp | 426 +++++++++++------- cpp/src/dual_simplex/primal.cpp | 16 +- cpp/src/dual_simplex/scaling.cpp | 4 +- .../dual_simplex/simplex_solver_settings.hpp | 14 +- 9 files changed, 622 insertions(+), 467 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp index c296185192..71438bfe8c 100644 --- a/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp @@ -133,12 +133,12 @@ class mip_solver_settings_t { i_t clique_cuts = -1; i_t zero_half_cuts = -1; i_t implied_bound_cuts = -1; - i_t strong_chvatal_gomory_cuts = -1; - i_t reduced_cost_strengthening = -1; + i_t strong_chvatal_gomory_cuts = -1; + i_t reduced_cost_strengthening = -1; i_t dual_degenerate_feasibility_pump = -1; // -1 = automatic (on), 0 = off, 1 = on - i_t objective_step = 1; // 0 = disable objective step tightening, 1 = enable - f_t cut_change_threshold = -1.0; - f_t cut_min_orthogonality = 0.5; + i_t objective_step = 1; // 0 = disable objective step tightening, 1 = enable + f_t cut_change_threshold = -1.0; + f_t cut_min_orthogonality = 0.5; i_t mip_batch_pdlp_strong_branching{ 0}; // 0 = DS only, 1 = cooperative DS + PDLP, 2 = batch PDLP only i_t mip_batch_pdlp_reliability_branching{ diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 83d0feb134..3763667c99 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -463,7 +463,6 @@ void branch_and_bound_t::report(const lp_problem_t& lp, settings_.log.printf("%s\n", log_line.c_str()); } - template void branch_and_bound_t::update_reduced_cost_bounds( f_t relaxation_objective, @@ -484,8 +483,9 @@ void branch_and_bound_t::update_reduced_cost_bounds( // Let u_tilde_j = u_j - epsilon, so that floor(u_tilde_j) = u_j - 1 // We want to solve for want the incumbent objective needs to be to make // x_j <= u_tilde_j - // This means l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] <= u_tilde_j - // Or equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * (u_tilde_j - l_j) when reduced_costs[j] > 0 + // This means l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] <= + // u_tilde_j Or equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * + // (u_tilde_j - l_j) when reduced_costs[j] > 0 if (lower_j > -inf && reduced_costs[j] > 0) { const f_t u_tilde_j = var_types_[j] == variable_type_t::INTEGER ? upper_j - tol @@ -495,18 +495,20 @@ void branch_and_bound_t::update_reduced_cost_bounds( const f_t diff = u_tilde_j - lower_j; const f_t objective_j = relaxation_objective + diff * reduced_costs[j]; if (((var_types_[j] == variable_type_t::INTEGER && bound_j == upper_j - 1.0) || - var_types_[j] != variable_type_t::INTEGER) && std::isfinite(objective_j) && std::isfinite(bound_j)) { + var_types_[j] != variable_type_t::INTEGER) && + std::isfinite(objective_j) && std::isfinite(bound_j)) { i_t info = reduced_cost_bounds.add_upper_bound(j, objective_j, bound_j); - //settings_.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. Info %d\n", objective_j, bound_j, j, info); + // settings_.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. + // Info %d\n", objective_j, bound_j, j, info); } } - // x_j >= u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] when reduced_costs[j] < 0 - // Let l_tilde_j = l_j + epsilon, so that ceil(l_tilde_j) = l_j + 1 - // We want to solve for want the incumbent objective needs to be to make - // x_j >= l_tilde_j - // This means u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j - // Or equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * (l_tilde_j - u_j) when reduced_costs[j] < 0 + // x_j >= u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] when + // reduced_costs[j] < 0 Let l_tilde_j = l_j + epsilon, so that ceil(l_tilde_j) = l_j + 1 We + // want to solve for want the incumbent objective needs to be to make x_j >= l_tilde_j This + // means u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j Or + // equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * (l_tilde_j + // - u_j) when reduced_costs[j] < 0 if (upper_j < inf && reduced_costs[j] < 0) { const f_t l_tilde_j = var_types_[j] == variable_type_t::INTEGER ? lower_j + tol @@ -516,16 +518,17 @@ void branch_and_bound_t::update_reduced_cost_bounds( const f_t diff = l_tilde_j - upper_j; const f_t objective_j = relaxation_objective + diff * reduced_costs[j]; if (((var_types_[j] == variable_type_t::INTEGER && bound_j == lower_j + 1.0) || - var_types_[j] != variable_type_t::INTEGER) && std::isfinite(objective_j) && std::isfinite(bound_j)) { + var_types_[j] != variable_type_t::INTEGER) && + std::isfinite(objective_j) && std::isfinite(bound_j)) { i_t info = reduced_cost_bounds.add_lower_bound(j, objective_j, bound_j); - //settings_.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. Info %d\n", objective_j, bound_j, j, info); + // settings_.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. + // Info %d\n", objective_j, bound_j, j, info); } } } } } - template i_t branch_and_bound_t::find_reduced_cost_fixings(f_t upper_bound, std::vector& lower_bounds, @@ -3470,7 +3473,9 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t::cut_pass_action_t branch_and_bound_t= 1 && upper_bound_.load() < last_upper_bound) { mutex_upper_.lock(); - last_upper_bound = upper_bound_.load(); + last_upper_bound = upper_bound_.load(); std::vector lower_bounds = original_lp_.lower; std::vector upper_bounds = original_lp_.upper; - f_t previous_max_objective = reduced_cost_bounds.get_max_objective(); - i_t new_bounds = reduced_cost_bounds.update_bounds_from_new_incumbent( + f_t previous_max_objective = reduced_cost_bounds.get_max_objective(); + i_t new_bounds = reduced_cost_bounds.update_bounds_from_new_incumbent( upper_bound_.load(), var_types_, lower_bounds, upper_bounds); mutex_upper_.unlock(); mutex_original_lp_.lock(); @@ -3573,7 +3578,12 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t 0) { settings_.log.printf( - "Updated %d integer bounds using reduced cost strengthening from new incumbent. Max objective %e Current objective %e Previous max objective %e\n", new_bounds, reduced_cost_bounds.get_max_objective(), upper_bound_.load(), previous_max_objective); + "Updated %d integer bounds using reduced cost strengthening from new incumbent. Max " + "objective %e Current objective %e Previous max objective %e\n", + new_bounds, + reduced_cost_bounds.get_max_objective(), + upper_bound_.load(), + previous_max_objective); } } @@ -3686,8 +3696,11 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t= 1) { - update_reduced_cost_bounds(root_objective_, root_relax_soln_.z, root_vstatus_, reduced_cost_bounds); - settings_.log.printf("New reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); + update_reduced_cost_bounds( + root_objective_, root_relax_soln_.z, root_vstatus_, reduced_cost_bounds); + settings_.log.printf("New reduced cost objective %e (current %e)\n", + reduced_cost_bounds.get_max_objective(), + upper_bound_.load()); pivot_to_improve_reduced_cost_strengthening(original_lp_, basic_list, nonbasic_list, @@ -3695,8 +3708,12 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t::dual_degenerate_feasibility_pump( reduced_basis_update.b_transpose_solve(c_basic_pump, y_pump); // Check if any nonbasic has a violated reduced cost - i_t num_violated = 0; - f_t max_violation = 0.0; + i_t num_violated = 0; + f_t max_violation = 0.0; const i_t num_nonbasics_reduced = reduced_nonbasic_list.size(); for (i_t k = 0; k < num_nonbasics_reduced; k++) { const i_t j = reduced_nonbasic_list[k]; // z[j] = c[j] - y^T * A(:,j) - f_t zj = lp_reduced.objective[j]; + f_t zj = lp_reduced.objective[j]; const i_t col_start = A_reduced.col_start[j]; - const i_t col_end = A_reduced.col_start[j + 1]; + const i_t col_end = A_reduced.col_start[j + 1]; for (i_t p = col_start; p < col_end; p++) { zj -= y_pump[A_reduced.i[p]] * A_reduced.x[p]; } @@ -4020,13 +4037,17 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( f_t pump_call_start_time = tic(); simplex_solver_settings_t primal_settings = settings_; primal_settings.log.log = false; - primal_settings.time_limit = settings_.time_limit; - primal_settings.work_limit = root_relax_work_estimate_ / 10; + primal_settings.time_limit = settings_.time_limit; + primal_settings.work_limit = root_relax_work_estimate_ / 10; settings_.log.printf( "Degenerate feasibility pump: calling primal simplex with %d rows, %d cols, %d nnz, " "%d basis updates, work_limit %.2e, primal_work_estimate %.2e\n", - m, n, A_reduced.col_start[n], reduced_basis_update.num_updates(), - primal_settings.work_limit, primal_work_estimate); + m, + n, + A_reduced.col_start[n], + reduced_basis_update.num_updates(), + primal_settings.work_limit, + primal_work_estimate); simplex::primal_status_t lp_status = simplex::primal_phase2_with_advanced_basis(2, exploration_stats_.start_time, @@ -4039,13 +4060,15 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( reduced_solution, iter, primal_work_estimate); - f_t pump_call_time = toc(pump_call_start_time); - f_t pump_call_work = primal_work_estimate - primal_work_before; + f_t pump_call_time = toc(pump_call_start_time); + f_t pump_call_work = primal_work_estimate - primal_work_before; i_t pump_call_iters = iter - iter_before; settings_.log.printf( "Degenerate feasibility pump: primal simplex returned status %d, %d iters, " "work %.2e (%.2e/iter), time %.2f (%.2e work/s)\n", - static_cast(lp_status), pump_call_iters, pump_call_work, + static_cast(lp_status), + pump_call_iters, + pump_call_work, pump_call_iters > 0 ? pump_call_work / pump_call_iters : 0.0, pump_call_time, pump_call_time > 0 ? pump_call_work / pump_call_time : 0.0); @@ -4384,7 +4407,7 @@ void branch_and_bound_t::fast_slack_integer_pivots( if (fast_candidates.size() > 0 && settings_.inside_mip < 2) { settings.log.printf("Found %ld fast candidates for pivot out integer variables\n", - fast_candidates.size()); + fast_candidates.size()); } // Build a reverse index nonbasic_index[v] = position of v in nonbasic_list, or -1 if not @@ -4398,8 +4421,8 @@ void branch_and_bound_t::fast_slack_integer_pivots( } const i_t num_candidates = fast_candidates.size(); - f_t last_log = tic(); - f_t loop_start = tic(); + f_t last_log = tic(); + f_t loop_start = tic(); for (i_t k = 0; k < num_candidates; k++) { const i_t j = fast_candidates[k]; const i_t row = fast_rows[k]; @@ -4491,18 +4514,18 @@ void branch_and_bound_t::fast_slack_integer_pivots( utilde_sparse.from_dense(utilde_dense); i_t error = apply_delta_x_for_integer_pivot(lp, - basic_list, - nonbasic_list, - nonbasic_index, - vstatus, - entering_index, - nonbasic_entering, - direction, - delta_x, - utilde_sparse, - soln, - basis_update, - work_estimate); + basic_list, + nonbasic_list, + nonbasic_index, + vstatus, + entering_index, + nonbasic_entering, + direction, + delta_x, + utilde_sparse, + soln, + basis_update, + work_estimate); // apply_delta_x_for_integer_pivot only mutates vstatus when the pivot actually fires, // so entering_index transitioning to BASIC is a reliable success signal. if (!error && settings.inside_mip < 2) { @@ -4511,13 +4534,18 @@ void branch_and_bound_t::fast_slack_integer_pivots( } if (settings.inside_mip < 2 && toc(last_log) > 1.0) { - settings.log.printf("Fast candidates %d/%d processed in %.2f seconds\n", k + 1, num_candidates, toc(loop_start)); + settings.log.printf("Fast candidates %d/%d processed in %.2f seconds\n", + k + 1, + num_candidates, + toc(loop_start)); last_log = tic(); } } - if (settings.inside_mip < 2) - { - settings.log.printf("Fast candidates: %d/%d processed in %.2f seconds\n", num_candidates, num_candidates, toc(loop_start)); + if (settings.inside_mip < 2) { + settings.log.printf("Fast candidates: %d/%d processed in %.2f seconds\n", + num_candidates, + num_candidates, + toc(loop_start)); } } @@ -4561,11 +4589,11 @@ i_t branch_and_bound_t::pivot_out_integer_variables( f_t work_estimate = 0.0; // Count primal degenerate basic variables - i_t num_degenerate = 0; + i_t num_degenerate = 0; i_t num_degenerate_continuous = 0; - i_t num_degenerate_integer = 0; + i_t num_degenerate_integer = 0; for (i_t k = 0; k < lp.num_rows; k++) { - const i_t j = basic_list_copy[k]; + const i_t j = basic_list_copy[k]; const f_t slack_to_lower = soln_copy.x[j] - lp.lower[j]; const f_t slack_to_upper = lp.upper[j] - soln_copy.x[j]; if (slack_to_lower <= settings_.primal_tol || slack_to_upper <= settings_.primal_tol) { @@ -4594,7 +4622,7 @@ i_t branch_and_bound_t::pivot_out_integer_variables( if (degeneracy_fraction > 0.5) { if (settings.inside_mip < 2) { settings.log.printf("Skipping pivot_out_integer_variables: degeneracy %.1f%% > 50%%\n", - 100.0 * degeneracy_fraction); + 100.0 * degeneracy_fraction); } return 0; } @@ -4631,24 +4659,24 @@ i_t branch_and_bound_t::pivot_out_integer_variables( // Track which entering variables are actually tried (to detect duplication) std::vector entering_tried_count(lp.num_cols, 0); - i_t worklist_total_processed = 0; - i_t worklist_skipped = 0; - i_t worklist_btran_done = 0; - i_t worklist_ftran_done = 0; - i_t worklist_pivots_succeeded = 0; - i_t worklist_readded = 0; - f_t worklist_btran_time = 0.0; - f_t worklist_dot_time = 0.0; - f_t worklist_ftran_time = 0.0; - i_t worklist_no_candidates = 0; // target had no nonzero dot_q - i_t worklist_ratio_test_fail = 0; // ratio test didn't pick a fractional integer (error -1) - i_t worklist_net_increase_fail = 0; // pivot would net-increase fractionals (error -2) - i_t worklist_unbounded = 0; // entering hit its own bound or unbounded (error -4) - i_t worklist_continuous_won = 0; // continuous variable won ratio test (error -5) - i_t worklist_nonfrac_int_won = 0; // non-fractional integer won ratio test (error -6) + i_t worklist_total_processed = 0; + i_t worklist_skipped = 0; + i_t worklist_btran_done = 0; + i_t worklist_ftran_done = 0; + i_t worklist_pivots_succeeded = 0; + i_t worklist_readded = 0; + f_t worklist_btran_time = 0.0; + f_t worklist_dot_time = 0.0; + f_t worklist_ftran_time = 0.0; + i_t worklist_no_candidates = 0; // target had no nonzero dot_q + i_t worklist_ratio_test_fail = 0; // ratio test didn't pick a fractional integer (error -1) + i_t worklist_net_increase_fail = 0; // pivot would net-increase fractionals (error -2) + i_t worklist_unbounded = 0; // entering hit its own bound or unbounded (error -4) + i_t worklist_continuous_won = 0; // continuous variable won ratio test (error -5) + i_t worklist_nonfrac_int_won = 0; // non-fractional integer won ratio test (error -6) f_t worklist_loop_start = tic(); - f_t worklist_last_log = tic(); + f_t worklist_last_log = tic(); while (!work_list.empty()) { const i_t j = work_list.back(); @@ -4657,9 +4685,18 @@ i_t branch_and_bound_t::pivot_out_integer_variables( worklist_total_processed++; // Skip if j is no longer basic and fractional (may have been fixed by a prior pivot) - if (p < 0) { worklist_skipped++; continue; } - if (vstatus_copy[j] != variable_status_t::BASIC) { worklist_skipped++; continue; } - if (!is_fractional(soln_copy.x[j], var_types_[j], settings_.integer_tol)) { worklist_skipped++; continue; } + if (p < 0) { + worklist_skipped++; + continue; + } + if (vstatus_copy[j] != variable_status_t::BASIC) { + worklist_skipped++; + continue; + } + if (!is_fractional(soln_copy.x[j], var_types_[j], settings_.integer_tol)) { + worklist_skipped++; + continue; + } // We want to pivot variable j out of the basis. // We solve B^T * delta_y = e_p, where p is the position of j in the basis. @@ -4707,9 +4744,9 @@ i_t branch_and_bound_t::pivot_out_integer_variables( // Small nnz means the FTRAN result is likely sparse, so fewer competing // basic variables will have nonzero delta_xB components to block the target. // Skip entering variables that have already been tried (and failed) by prior targets. - f_t values[3] = {0.0, 0.0, 0.0}; + f_t values[3] = {0.0, 0.0, 0.0}; i_t indices[3] = {-1, -1, -1}; - f_t dot_start = tic(); + f_t dot_start = tic(); for (i_t q : zero_reduced_costs_vars) { if (var_types_[q] == variable_type_t::INTEGER) { continue; } if (nonbasic_index[q] < 0) { continue; } @@ -4718,7 +4755,7 @@ i_t branch_and_bound_t::pivot_out_integer_variables( const i_t col_start = lp.A.col_start[q]; const i_t col_end = lp.A.col_start[q + 1]; const i_t col_nnz = col_end - col_start; - f_t dot_q = 0.0; + f_t dot_q = 0.0; for (i_t pp = col_start; pp < col_end; pp++) { dot_q += delta_y_dense[lp.A.i[pp]] * lp.A.x[pp]; } @@ -4727,14 +4764,20 @@ i_t branch_and_bound_t::pivot_out_integer_variables( const f_t merit = abs_dot_q / static_cast(col_nnz); if (merit > values[0]) { - indices[2] = indices[1]; values[2] = values[1]; - indices[1] = indices[0]; values[1] = values[0]; - indices[0] = q; values[0] = merit; + indices[2] = indices[1]; + values[2] = values[1]; + indices[1] = indices[0]; + values[1] = values[0]; + indices[0] = q; + values[0] = merit; } else if (merit > values[1]) { - indices[2] = indices[1]; values[2] = values[1]; - indices[1] = q; values[1] = merit; + indices[2] = indices[1]; + values[2] = values[1]; + indices[1] = q; + values[1] = merit; } else if (merit > values[2]) { - indices[2] = q; values[2] = merit; + indices[2] = q; + values[2] = merit; } } worklist_dot_time += toc(dot_start); @@ -4745,7 +4788,7 @@ i_t branch_and_bound_t::pivot_out_integer_variables( for (i_t h = 0; h < 3; h++) { if (indices[h] == -1) break; - const i_t q = indices[h]; + const i_t q = indices[h]; const i_t entering_index = q; const i_t nonbasic_entering = nonbasic_index[q]; if (nonbasic_entering < 0) { continue; } @@ -4775,39 +4818,48 @@ i_t branch_and_bound_t::pivot_out_integer_variables( delta_x[q] = direction; i_t error = apply_delta_x_for_integer_pivot(lp, - basic_list_copy, - nonbasic_list_copy, - nonbasic_index, - vstatus_copy, - entering_index, - nonbasic_entering, - direction, - delta_x, - utilde_sparse, - soln_copy, - basis_update_copy, - work_estimate); + basic_list_copy, + nonbasic_list_copy, + nonbasic_index, + vstatus_copy, + entering_index, + nonbasic_entering, + direction, + delta_x, + utilde_sparse, + soln_copy, + basis_update_copy, + work_estimate); if (error == -2) { worklist_net_increase_fail++; } - if (error == -4) { worklist_unbounded++; worklist_ratio_test_fail++; } - if (error == -5) { worklist_continuous_won++; worklist_ratio_test_fail++; } - if (error == -6) { worklist_nonfrac_int_won++; worklist_ratio_test_fail++; } + if (error == -4) { + worklist_unbounded++; + worklist_ratio_test_fail++; + } + if (error == -5) { + worklist_continuous_won++; + worklist_ratio_test_fail++; + } + if (error == -6) { + worklist_nonfrac_int_won++; + worklist_ratio_test_fail++; + } if (!error) { worklist_pivots_succeeded++; // Update to_basic_position for the variables that changed status // entering_index is now basic, leaving_index is now nonbasic // Find the leaving variable: it's the one that took entering_index's slot in nonbasic_list - const i_t leaving_index = nonbasic_list_copy[nonbasic_entering]; + const i_t leaving_index = nonbasic_list_copy[nonbasic_entering]; to_basic_position[entering_index] = to_basic_position[leaving_index]; - to_basic_position[leaving_index] = -1; + to_basic_position[leaving_index] = -1; // We did a successful pivot; add fractional variables whose values changed to work list for (i_t k : fractional) { if (vstatus_copy[k] != variable_status_t::BASIC) { continue; } if (std::abs(delta_x[k]) > settings_.zero_tol) { - //work_list.push_back(k); - //worklist_readded++; + // work_list.push_back(k); + // worklist_readded++; } } break; @@ -4840,16 +4892,19 @@ i_t branch_and_bound_t::pivot_out_integer_variables( } // Count unique entering variables and duplication - i_t unique_entering = 0; - i_t max_entering_count = 0; - i_t entering_tried_once = 0; + i_t unique_entering = 0; + i_t max_entering_count = 0; + i_t entering_tried_once = 0; i_t entering_tried_multiple = 0; for (i_t q = 0; q < lp.num_cols; q++) { if (entering_tried_count[q] > 0) { unique_entering++; max_entering_count = std::max(max_entering_count, entering_tried_count[q]); - if (entering_tried_count[q] == 1) { entering_tried_once++; } - else { entering_tried_multiple++; } + if (entering_tried_count[q] == 1) { + entering_tried_once++; + } else { + entering_tried_multiple++; + } } } if (settings.inside_mip < 2) { @@ -4957,9 +5012,9 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( std::vector delta_z_indices; delta_z_indices.reserve(lp.num_cols); - f_t work_estimate = 0; - const f_t threshold = 100.0 * settings_.integer_tol; - const f_t tol = 1e-2; + f_t work_estimate = 0; + const f_t threshold = 100.0 * settings_.integer_tol; + const f_t tol = 1e-2; const f_t zero_tol = settings_.zero_tol; const f_t harris_tol = settings_.dual_tol / 10; @@ -4997,8 +5052,8 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( delta_z, work_estimate); - const f_t lower_j = lp.lower[j]; - const f_t upper_j = lp.upper[j]; + const f_t lower_j = lp.lower[j]; + const f_t upper_j = lp.upper[j]; const bool at_lower = soln.x[j] - lower_j <= settings_.primal_tol; const bool at_upper = upper_j - soln.x[j] <= settings_.primal_tol; @@ -5042,8 +5097,7 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( const f_t step = alpha * scale * delta_z[jj]; const f_t new_zj = old_zj + step; const bool initially_infeasible = - (vstatus[jj] == variable_status_t::NONBASIC_LOWER && - old_zj < -settings_.dual_tol) || + (vstatus[jj] == variable_status_t::NONBASIC_LOWER && old_zj < -settings_.dual_tol) || (vstatus[jj] == variable_status_t::NONBASIC_UPPER && old_zj > settings_.dual_tol); if (initially_infeasible) { num_initial_dual_infeas++; @@ -5122,14 +5176,15 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( std::isfinite(objective_j) && std::isfinite(bound_j)) { i_t info = reduced_cost_bounds.add_upper_bound(j, objective_j, bound_j); if (info > 0) { num_bounds_added++; } - //settings_.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. Info %d\n", objective_j, bound_j, j, info); + // settings_.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. + // Info %d\n", objective_j, bound_j, j, info); } } // x_j >= u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] when - // reduced_costs[j] < 0 Let l_tilde_j = l_j + epsilon, so that ceil(l_tilde_j) = l_j + 1 We want - // to solve for want the incumbent objective needs to be to make x_j >= l_tilde_j This means - // u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j Or + // reduced_costs[j] < 0 Let l_tilde_j = l_j + epsilon, so that ceil(l_tilde_j) = l_j + 1 We + // want to solve for want the incumbent objective needs to be to make x_j >= l_tilde_j This + // means u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j Or // equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * // (l_tilde_j - u_j) when reduced_costs[j] < 0 if (at_upper && upper_j < inf && new_reduced_cost < -threshold) { @@ -5145,7 +5200,8 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( std::isfinite(objective_j) && std::isfinite(bound_j)) { i_t info = reduced_cost_bounds.add_lower_bound(j, objective_j, bound_j); if (info > 0) { num_bounds_added++; } - //settings_.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. Info %d\n", objective_j, bound_j, j, info); + // settings_.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. + // Info %d\n", objective_j, bound_j, j, info); } } } @@ -5247,7 +5303,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut lp_status_t root_status = lp_status_t::UNSET; solving_root_relaxation_ = true; - f_t root_relax_start_time = tic(); + f_t root_relax_start_time = tic(); root_relax_work_estimate_ = 0.0; if (!enable_concurrent_lp_root_solve()) { // RINS/SUBMIP path @@ -5383,8 +5439,11 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut lower_bound_numerical_ = inf; reduced_cost_bounds_t reduced_cost_bounds(original_lp_.num_cols); - update_reduced_cost_bounds(root_objective_, root_relax_soln_.z, root_vstatus_, reduced_cost_bounds); - settings_.log.printf("New reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); + update_reduced_cost_bounds( + root_objective_, root_relax_soln_.z, root_vstatus_, reduced_cost_bounds); + settings_.log.printf("New reduced cost objective %e (current %e)\n", + reduced_cost_bounds.get_max_objective(), + upper_bound_.load()); pivot_to_improve_reduced_cost_strengthening(original_lp_, basic_list, nonbasic_list, @@ -5393,20 +5452,25 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut basis_update, num_fractional, fractional, - root_objective_, reduced_cost_bounds); - settings_.log.printf("After pivoting: new reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); + root_objective_, + reduced_cost_bounds); + settings_.log.printf("After pivoting: new reduced cost objective %e (current %e)\n", + reduced_cost_bounds.get_max_objective(), + upper_bound_.load()); f_t pivot_out_integer_variables_start_time = tic(); - i_t num_integer_increased = pivot_out_integer_variables(original_lp_, - settings_, - basic_list, - nonbasic_list, - root_vstatus_, - root_relax_soln_, - basis_update, - num_fractional, - fractional); - settings_.log.printf("Pivoted out %d integer variables in %e seconds\n", num_integer_increased, toc(pivot_out_integer_variables_start_time)); + i_t num_integer_increased = pivot_out_integer_variables(original_lp_, + settings_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional); + settings_.log.printf("Pivoted out %d integer variables in %e seconds\n", + num_integer_increased, + toc(pivot_out_integer_variables_start_time)); if (settings_.dual_degenerate_feasibility_pump != 0) { dual_degenerate_feasibility_pump(original_lp_, @@ -5606,10 +5670,16 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut if (settings_.reduced_cost_strengthening >= 2 && upper_bound_.load() < last_upper_bound) { std::vector lower_bounds = original_lp_.lower; std::vector upper_bounds = original_lp_.upper; - f_t previous_max_objective = reduced_cost_bounds.get_max_objective(); - i_t num_changed = reduced_cost_bounds.update_bounds_from_new_incumbent( + f_t previous_max_objective = reduced_cost_bounds.get_max_objective(); + i_t num_changed = reduced_cost_bounds.update_bounds_from_new_incumbent( upper_bound_.load(), var_types_, lower_bounds, upper_bounds); - settings_.log.printf("Updated %d integer bounds using reduced cost strengthening from new incumbent. Max objective %e Current objective %e Previous max objective %e\n", num_changed, reduced_cost_bounds.get_max_objective(), upper_bound_.load(), previous_max_objective); + settings_.log.printf( + "Updated %d integer bounds using reduced cost strengthening from new incumbent. Max " + "objective %e Current objective %e Previous max objective %e\n", + num_changed, + reduced_cost_bounds.get_max_objective(), + upper_bound_.load(), + previous_max_objective); mutex_original_lp_.lock(); original_lp_.lower = lower_bounds; original_lp_.upper = upper_bounds; diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index a7d1baf32a..9ba0b94128 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -96,12 +96,10 @@ struct deterministic_diving_policy_t; template struct objective_bound_pair_t { objective_bound_pair_t() - : objective(std::numeric_limits::quiet_NaN()), - bound(std::numeric_limits::quiet_NaN()) + : objective(std::numeric_limits::quiet_NaN()), bound(std::numeric_limits::quiet_NaN()) { } - objective_bound_pair_t(f_t objective_in, f_t bound_in) - : objective(objective_in), bound(bound_in) + objective_bound_pair_t(f_t objective_in, f_t bound_in) : objective(objective_in), bound(bound_in) { } bool is_valid() { return objective == objective && bound == bound; } @@ -113,7 +111,9 @@ template class reduced_cost_bounds_t { public: reduced_cost_bounds_t(i_t original_cols) - : max_objective_(-std::numeric_limits::infinity()), lower_bounds_(original_cols), upper_bounds_(original_cols) + : max_objective_(-std::numeric_limits::infinity()), + lower_bounds_(original_cols), + upper_bounds_(original_cols) { } @@ -122,22 +122,16 @@ class reduced_cost_bounds_t { if (col < static_cast(lower_bounds_.size())) { if (!lower_bounds_[col].is_valid()) { lower_bounds_[col] = objective_bound_pair_t(objective, bound); - if (objective > max_objective_) { - max_objective_ = objective; - } + if (objective > max_objective_) { max_objective_ = objective; } return 1; } else { if (bound > lower_bounds_[col].bound) { lower_bounds_[col] = objective_bound_pair_t(objective, bound); - if (objective > max_objective_) { - max_objective_ = objective; - } + if (objective > max_objective_) { max_objective_ = objective; } return 2; } else if (bound == lower_bounds_[col].bound && objective > lower_bounds_[col].objective) { lower_bounds_[col].objective = objective; - if (objective > max_objective_) { - max_objective_ = objective; - } + if (objective > max_objective_) { max_objective_ = objective; } return 1; } else { return -2; @@ -153,22 +147,16 @@ class reduced_cost_bounds_t { if (col < static_cast(upper_bounds_.size())) { if (!upper_bounds_[col].is_valid()) { upper_bounds_[col] = objective_bound_pair_t(objective, bound); - if (objective > max_objective_) { - max_objective_ = objective; - } + if (objective > max_objective_) { max_objective_ = objective; } return 1; } else { if (bound < upper_bounds_[col].bound) { upper_bounds_[col] = objective_bound_pair_t(objective, bound); - if (objective > max_objective_) { - max_objective_ = objective; - } + if (objective > max_objective_) { max_objective_ = objective; } return 2; } else if (bound == upper_bounds_[col].bound && objective > upper_bounds_[col].objective) { upper_bounds_[col].objective = objective; - if (objective > max_objective_) { - max_objective_ = objective; - } + if (objective > max_objective_) { max_objective_ = objective; } return 1; } else { return -2; @@ -184,17 +172,19 @@ class reduced_cost_bounds_t { std::vector& lower_bounds, std::vector& upper_bounds) { - const i_t n = static_cast(lower_bounds_.size()); - f_t max_objective = -std::numeric_limits::infinity(); + const i_t n = static_cast(lower_bounds_.size()); + f_t max_objective = -std::numeric_limits::infinity(); i_t integer_bounds_updated = 0; for (i_t j = 0; j < n; ++j) { if (lower_bounds_[j].is_valid()) { if (incumbent_objective <= lower_bounds_[j].objective && lower_bounds_[j].bound > lower_bounds[j]) { - //printf("RCF Variable %d (%d): lower %e -> %e\n", j, static_cast(var_types[j]), lower_bounds[j], lower_bounds_[j].bound); + // printf("RCF Variable %d (%d): lower %e -> %e\n", j, static_cast(var_types[j]), + // lower_bounds[j], lower_bounds_[j].bound); lower_bounds[j] = lower_bounds_[j].bound; if (var_types[j] == simplex::variable_type_t::INTEGER) { integer_bounds_updated++; } - lower_bounds_[j].bound = lower_bounds_[j].objective = std::numeric_limits::quiet_NaN(); + lower_bounds_[j].bound = lower_bounds_[j].objective = + std::numeric_limits::quiet_NaN(); } if (lower_bounds_[j].objective > max_objective) { max_objective = lower_bounds_[j].objective; @@ -203,10 +193,12 @@ class reduced_cost_bounds_t { if (upper_bounds_[j].is_valid()) { if (incumbent_objective <= upper_bounds_[j].objective && upper_bounds_[j].bound < upper_bounds[j]) { - //printf("RCF Variable %d (%d): upper %e -> %e\n", j, static_cast(var_types[j]), upper_bounds[j], upper_bounds_[j].bound); + // printf("RCF Variable %d (%d): upper %e -> %e\n", j, static_cast(var_types[j]), + // upper_bounds[j], upper_bounds_[j].bound); upper_bounds[j] = upper_bounds_[j].bound; if (var_types[j] == simplex::variable_type_t::INTEGER) { integer_bounds_updated++; } - upper_bounds_[j].bound = upper_bounds_[j].objective = std::numeric_limits::quiet_NaN(); + upper_bounds_[j].bound = upper_bounds_[j].objective = + std::numeric_limits::quiet_NaN(); } if (upper_bounds_[j].objective > max_objective) { max_objective = upper_bounds_[j].objective; @@ -525,18 +517,18 @@ class branch_and_bound_t { std::vector& fractional); i_t apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& nonbasic_index, - std::vector& vstatus, - i_t entering_index, - i_t nonbasic_entering, - i_t direction, - std::vector& delta_x, - const sparse_vector_t& utilde_sparse, - simplex::lp_solution_t& solution, - simplex::basis_update_mpf_t& basis_update, - f_t& work_estimate); + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& nonbasic_index, + std::vector& vstatus, + i_t entering_index, + i_t nonbasic_entering, + i_t direction, + std::vector& delta_x, + const sparse_vector_t& utilde_sparse, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + f_t& work_estimate); void dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, std::vector& basic_list, diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index 35ef45c0ab..8b9688e021 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -35,16 +35,16 @@ i_t bound_flipping_ratio_test_t::compute_breakpoints(std::vector& const i_t k = nonbasic_mark_[j]; if (vstatus_[j] == variable_status_t::NONBASIC_FIXED) { continue; } if (vstatus_[j] == variable_status_t::NONBASIC_LOWER && delta_z_[j] < -pivot_tol) { - indicies[idx] = k; - ratios[idx] = std::max((-z_[j]) / delta_z_[j], 0.0); - harris_ratios[idx] = std::max((-dual_tol - z_[j]) / delta_z_[j], 0.0); + indicies[idx] = k; + ratios[idx] = std::max((-z_[j]) / delta_z_[j], 0.0); + harris_ratios[idx] = std::max((-dual_tol - z_[j]) / delta_z_[j], 0.0); if constexpr (verbose) { settings_.log.printf("ratios[%d] = %e\n", idx, ratios[idx]); } idx++; } if (vstatus_[j] == variable_status_t::NONBASIC_UPPER && delta_z_[j] > pivot_tol) { - indicies[idx] = k; - ratios[idx] = std::max((-z_[j]) / delta_z_[j], 0.0); - harris_ratios[idx] = std::max((dual_tol - z_[j]) / delta_z_[j], 0.0); + indicies[idx] = k; + ratios[idx] = std::max((-z_[j]) / delta_z_[j], 0.0); + harris_ratios[idx] = std::max((dual_tol - z_[j]) / delta_z_[j], 0.0); if constexpr (verbose) { settings_.log.printf("ratios[%d] = %e\n", idx, ratios[idx]); } idx++; } @@ -71,9 +71,9 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, i_t candidate = -1; f_t zero_tol = settings_.zero_tol; i_t k_idx = -1; - max_val = 0.0; + max_val = 0.0; - i_t min_found = 0; + i_t min_found = 0; for (i_t k = start; k < end; ++k) { if (ratios[k] < min_val) { min_val = ratios[k]; @@ -81,9 +81,7 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, k_idx = k; min_found++; } - if (ratios[k] > max_val) { - max_val = ratios[k]; - } + if (ratios[k] > max_val) { max_val = ratios[k]; } } work_estimate_ += (end - start) + 2 * min_found; @@ -140,13 +138,13 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, std::vector ratios(nz); std::vector harris_ratios(nz); work_estimate_ += 3 * nz; - double t0 = tic(); + double t0 = tic(); i_t num_breakpoints = compute_breakpoints(indicies, ratios, harris_ratios); time_compute_breakpoints_ += toc(t0); num_breakpoints_ = num_breakpoints; // Count zero ratios num_harris_zero_ = 0; - num_exact_zero_ = 0; + num_exact_zero_ = 0; for (i_t k = 0; k < num_breakpoints; k++) { if (harris_ratios[k] == 0.0) num_harris_zero_++; if (ratios[k] == 0.0) num_exact_zero_++; @@ -163,9 +161,15 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, i_t entering_index = RATIO_TEST_NO_ENTERING_VARIABLE; f_t max_step_length; - t0 = tic(); - i_t k_idx = single_pass( - 0, num_breakpoints, indicies, harris_ratios, step_length, nonbasic_entering, entering_index, max_step_length); + t0 = tic(); + i_t k_idx = single_pass(0, + num_breakpoints, + indicies, + harris_ratios, + step_length, + nonbasic_entering, + entering_index, + max_step_length); time_single_pass_ += toc(t0); if (k_idx == RATIO_TEST_NUMERICAL_ISSUES) { return RATIO_TEST_NUMERICAL_ISSUES; } // The variable selected by single_pass is guaranteed to be in the first bucket: it @@ -196,7 +200,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, slope); } - // This code is complicated. There are several important concepts that are needed to understand it. + // This code is complicated. There are several important concepts that are needed to understand + // it. // // We are trying to compute the maximum step length we can take while: // 1) Staying mostly dual feasible (we allow ourselves to be infeasible by dual_tol amount) @@ -210,64 +215,72 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // z_j(alpha) >= 0, if j is on it's lower bound, or // z_j(alpha) <= 0, if j is on it's upper bound. // - // Consider the equation z_j(alpha) = z_j + alpha * delta_z_j = 0. Each variable j puts a bound on alpha: + // Consider the equation z_j(alpha) = z_j + alpha * delta_z_j = 0. Each variable j puts a bound on + // alpha: // // alpha_j <= -z_j / delta_z_j, if x_j = l_j (z_j >= 0) and delta_z_j < 0 // alpha_j <= -z_j / delta_z_j, if x_j = u_j (z_j <= 0) and delta_z_j > 0 // // The code refers to these alpha_j as ratios, since they are the ratio of z_j to delta_z_j. // - // Now we could take alpha = min_j alpha_j, and remain dual feasible. However, we are allowed to increase the step-length - // if j is a variable such that l_j <= x_j <= u_j. To see why imagine that our variable was currenlty on it's lower bound, - // with z_j > 0 and delta_z_j < 0, if we push alpha past alpha_j, than z_j(alpha) < 0. This is fine as long as we flip - // the variable to be on it's upper bound. Thus, we can push alpha past alpha_j for *bounded* variables. + // Now we could take alpha = min_j alpha_j, and remain dual feasible. However, we are allowed to + // increase the step-length if j is a variable such that l_j <= x_j <= u_j. To see why imagine + // that our variable was currenlty on it's lower bound, with z_j > 0 and delta_z_j < 0, if we push + // alpha past alpha_j, than z_j(alpha) < 0. This is fine as long as we flip the variable to be on + // it's upper bound. Thus, we can push alpha past alpha_j for *bounded* variables. // - // Note that this does not work if we try to increase alpha past alpha_j for a variable with a single bound. We would just - // be making ourselves dual infeasible. So we need to check whether a variable is bounded. + // Note that this does not work if we try to increase alpha past alpha_j for a variable with a + // single bound. We would just be making ourselves dual infeasible. So we need to check whether a + // variable is bounded. // - // The dual objective as a function of the step-length alpha, is piecewise linear and concave. The breakpoints of this - // piecewise linear function occur at each of the alpha_j values. - // We can keep increasing the step-length as long as the slope remains nonnegative. After that - // we must stop, because we could decrease the dual objective. So the code tracks the cumulative slope of the dual objective. + // The dual objective as a function of the step-length alpha, is piecewise linear and concave. The + // breakpoints of this piecewise linear function occur at each of the alpha_j values. We can keep + // increasing the step-length as long as the slope remains nonnegative. After that we must stop, + // because we could decrease the dual objective. So the code tracks the cumulative slope of the + // dual objective. // - // Now we don't need to exactly feasible: z_j >= 0 if x_j = l_j and z_j <= 0 if x_j = u_j. We can violate these bounds by - // the dual feasibility tolerance eps. We allow ourselves to be infeasible if it would help us get a larger pivot - // (delta_z_j). Small pivots can cause numerical issues, so we would like to avoid them. + // Now we don't need to exactly feasible: z_j >= 0 if x_j = l_j and z_j <= 0 if x_j = u_j. We can + // violate these bounds by the dual feasibility tolerance eps. We allow ourselves to be infeasible + // if it would help us get a larger pivot (delta_z_j). Small pivots can cause numerical issues, so + // we would like to avoid them. // // With this tolerance we get the equations: // z_j(alpha) = z_j + alpha * delta_z_j >= -eps if x_j = l_j // z_j(alpha) = z_j + alpha * delta_z_j <= eps if x_j = u_j // - // This gives bounds on alpha. We call these alpha_harris_j, for Paula Harris, who proposed this method. + // This gives bounds on alpha. We call these alpha_harris_j, for Paula Harris, who proposed this + // method. // // alpha_harris_j <= (-eps - z_j) / delta_z_j, if x_j = l_j and delta_z_j < 0 // alpha_harris_j <= (eps - z_j) / delta_z_j, if x_j = u_j and delta_z_j > 0 // - // Let alpha_harris = min_j alpha_harris_j. We can select the variable with the largest | delta_z_j | from those - // candidates { j | alpha_j <= alpha_harris }. + // Let alpha_harris = min_j alpha_harris_j. We can select the variable with the largest | + // delta_z_j | from those candidates { j | alpha_j <= alpha_harris }. // - // We combine these two ideas (increasing the step length for bounded variables) and allowing ourselves to be slightly dual infeasible - // to choose a larger pivot. + // We combine these two ideas (increasing the step length for bounded variables) and allowing + // ourselves to be slightly dual infeasible to choose a larger pivot. // - // We partition the variables into buckets. Let B_k be the set of variables in bucket k. B_0 is defined as { j | alpha_j <= alpha_harris }. - // We then compute alpha_harris_1 = min_{j not in B_0} alpha_j. And B_1 is defined as { j not in B_0 | alpha_j <= alpha_harris_1 }. And - // so on. + // We partition the variables into buckets. Let B_k be the set of variables in bucket k. B_0 is + // defined as { j | alpha_j <= alpha_harris }. We then compute alpha_harris_1 = min_{j not in B_0} + // alpha_j. And B_1 is defined as { j not in B_0 | alpha_j <= alpha_harris_1 }. And so on. // // We want to balance two different things: // 1) Taking a larger step length to increase the dual objective as much as possible, // 2) Choosing a large pivot for numerical stability. // - // Let max_pivot = max_j | delta_z_j |. We start working our way backward from the largest bucket to the smallest bucket, - // we choose a variable j that satisfies | delta_z_j | >= 0.1 * max_pivot. Since we can always choose a smaller step length - // for the sake of numerical stability. + // Let max_pivot = max_j | delta_z_j |. We start working our way backward from the largest bucket + // to the smallest bucket, we choose a variable j that satisfies | delta_z_j | >= 0.1 * max_pivot. + // Since we can always choose a smaller step length for the sake of numerical stability. // - // Now the final thing to understand is that the ratios alpha_j are not sorted in any particular order. And we don't want to - // pay the O(num_breakpoints * log(num_breakpoints)) cost of sorting them. + // Now the final thing to understand is that the ratios alpha_j are not sorted in any particular + // order. And we don't want to pay the O(num_breakpoints * log(num_breakpoints)) cost of sorting + // them. // - // So we set a threshold on the step-length and check if all variables j with alpha_j <= threshold have already caused the - // slope to go negative. If so, we just need to consider those candidate variables with alpha_j <= threshold. If not, we - // multiply the threshold by 10. This cost us O(log10(max_step_length/min_step_length) * num_breakpoints) time. So we aren't - // totally linear. But the hope is we are better than a sort. + // So we set a threshold on the step-length and check if all variables j with alpha_j <= threshold + // have already caused the slope to go negative. If so, we just need to consider those candidate + // variables with alpha_j <= threshold. If not, we multiply the threshold by 10. This cost us + // O(log10(max_step_length/min_step_length) * num_breakpoints) time. So we aren't totally linear. + // But the hope is we are better than a sort. // Use a coarse filter to find candidates f_t minimum_harris_ratio = step_length; @@ -279,8 +292,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, std::vector candidates(num_breakpoints); std::iota(candidates.begin(), candidates.end(), 0); work_estimate_ += 2 * num_breakpoints; - i_t scan_start = 0; - i_t num_candidates = 0; + i_t scan_start = 0; + i_t num_candidates = 0; // This is O( log10(max_step_length/min_step_length) * num_breakpoints) t0 = tic(); @@ -292,7 +305,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // Candidate is less than coarse threshold, move it to the front of the candidate list std::swap(candidates[h], candidates[num_candidates]); num_candidates++; - const i_t j = nonbasic_list_[indicies[k]]; + const i_t j = nonbasic_list_[indicies[k]]; if (!bounded_variables_[j]) { found_unbounded = true; } else { @@ -313,16 +326,14 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, for (i_t h = 0; h < num_candidates; h++) { const i_t k = candidates[h]; const i_t j = nonbasic_list_[indicies[k]]; - if (!bounded_variables_[j]) { - max_step_length = std::min(max_step_length, harris_ratios[k]); - } + if (!bounded_variables_[j]) { max_step_length = std::min(max_step_length, harris_ratios[k]); } } work_estimate_ += 5 * num_candidates; // Remove candidates that are greater than the maximum step length const i_t candidates_before_removal = candidates.size(); for (i_t h = candidates_before_removal - 1; h >= 0; h--) { - const i_t k = candidates[h]; + const i_t k = candidates[h]; const f_t ratio = ratios[k]; if (ratio > max_step_length) { // Swap with the last candidate and remove @@ -330,7 +341,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, candidates.pop_back(); } } - work_estimate_ += 2 * candidates_before_removal + 2 * (candidates_before_removal - candidates.size()); + work_estimate_ += + 2 * candidates_before_removal + 2 * (candidates_before_removal - candidates.size()); num_candidates = candidates.size(); } @@ -341,12 +353,12 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, i_t num_buckets = 0; std::vector bucket_start(num_candidates + 1, 0); f_t cumulative_slope = slope; - scan_start = 0; + scan_start = 0; work_estimate_ += num_candidates + 1; // This is O(num_buckets * num_candidates) i_t slope_breaker_k = -1; // the candidate k that made slope go negative - t0 = tic(); + t0 = tic(); while (cumulative_slope >= 0.0 && scan_start < num_candidates && threshold <= max_step_length) { f_t next_threshold = inf; i_t write = scan_start; @@ -359,9 +371,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, const i_t j = nonbasic_list_[indicies[k]]; if (bounded_variables_[j]) { cumulative_slope -= std::abs(delta_z_[j]) * (upper_[j] - lower_[j]); - if (cumulative_slope < 0.0 && slope_breaker_k < 0) { - slope_breaker_k = k; - } + if (cumulative_slope < 0.0 && slope_breaker_k < 0) { slope_breaker_k = k; } } std::swap(candidates[h], candidates[write]); write++; @@ -375,8 +385,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, bucket_start[++num_buckets] = write; if (write == scan_start) break; // No progress — prevent infinite loop - scan_start = write; - threshold = next_threshold; + scan_start = write; + threshold = next_threshold; if (cumulative_slope < 0.0) break; } @@ -387,12 +397,10 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // This is O(num_candidates) f_t max_pivot = 0.0; for (i_t h = 0; h < bucket_start[num_buckets]; h++) { - const i_t k = candidates[h]; - const i_t j = nonbasic_list_[indicies[k]]; + const i_t k = candidates[h]; + const i_t j = nonbasic_list_[indicies[k]]; const f_t pivot = std::abs(delta_z_[j]); - if (pivot > max_pivot) { - max_pivot = pivot; - } + if (pivot > max_pivot) { max_pivot = pivot; } } work_estimate_ += 4 * bucket_start[num_buckets]; @@ -405,8 +413,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // This is O(num_candidates) for (i_t b = num_buckets - 1; b >= 0; b--) { const i_t b_start = bucket_start[b]; - const i_t b_end = bucket_start[b + 1]; - f_t best_ratio = -1.0; + const i_t b_end = bucket_start[b + 1]; + f_t best_ratio = -1.0; for (i_t h = b_start; h < b_end; h++) { const i_t k = candidates[h]; const i_t j = nonbasic_list_[indicies[k]]; @@ -424,9 +432,9 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, num_buckets_used_ = num_buckets; if (entering_k < 0) { // Fallback to single_pass result - used_fallback_ = true; - bucket_selected_ = -1; - step_length_result_ = step_length; + used_fallback_ = true; + bucket_selected_ = -1; + step_length_result_ = step_length; selected_is_slope_breaker_ = false; determine_flips(step_length, entering_index, flip_indices); return entering_index; @@ -440,13 +448,16 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // Record which bucket was selected used_fallback_ = false; - i_t pos = -1; + i_t pos = -1; for (i_t b = 0; b < num_buckets; b++) { if (entering_k >= 0) { // Find which bucket entering_k is in based on its position in candidates pos = -1; for (i_t h = 0; h < num_candidates; h++) { - if (candidates[h] == entering_k) { pos = h; break; } + if (candidates[h] == entering_k) { + pos = h; + break; + } } if (pos >= bucket_start[b] && pos < bucket_start[b + 1]) { bucket_selected_ = b; @@ -459,10 +470,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, determine_flips(step_length, entering_index, flip_indices); return entering_index; - } - #ifdef DUAL_SIMPLEX_INSTANTIATE_DOUBLE template class bound_flipping_ratio_test_t; diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp index 0d78e68ee0..4587037889 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp @@ -53,9 +53,7 @@ class bound_flipping_ratio_test_t { { } - i_t compute_step_length(f_t& step_length, - i_t& nonbasic_entering, - std::vector& flip_indices); + i_t compute_step_length(f_t& step_length, i_t& nonbasic_entering, std::vector& flip_indices); f_t work_estimate() const { return work_estimate_; } // Timing fields (filled by compute_step_length) @@ -66,18 +64,22 @@ class bound_flipping_ratio_test_t { f_t time_pivot_selection_{0.0}; // Diagnostic fields - i_t num_buckets_used_{0}; // number of buckets in bucket sort - i_t bucket_selected_{-1}; // which bucket the entering variable came from (-1 = single_pass/fallback) - f_t step_length_result_{0.0}; // the step length chosen - bool used_fallback_{false}; // true if we fell back to single_pass result - i_t bucket0_size_{0}; // size of first bucket (candidates with ratio <= min_harris) - i_t num_breakpoints_{0}; // total breakpoints computed - bool selected_is_slope_breaker_{false}; // true if we selected the variable that made slope go negative - i_t num_harris_zero_{0}; // number of harris_ratios that are exactly 0 - i_t num_exact_zero_{0}; // number of exact ratios that are exactly 0 + i_t num_buckets_used_{0}; // number of buckets in bucket sort + i_t bucket_selected_{ + -1}; // which bucket the entering variable came from (-1 = single_pass/fallback) + f_t step_length_result_{0.0}; // the step length chosen + bool used_fallback_{false}; // true if we fell back to single_pass result + i_t bucket0_size_{0}; // size of first bucket (candidates with ratio <= min_harris) + i_t num_breakpoints_{0}; // total breakpoints computed + bool selected_is_slope_breaker_{ + false}; // true if we selected the variable that made slope go negative + i_t num_harris_zero_{0}; // number of harris_ratios that are exactly 0 + i_t num_exact_zero_{0}; // number of exact ratios that are exactly 0 private: - i_t compute_breakpoints(std::vector& indices, std::vector& ratios, std::vector& harris_ratios); + i_t compute_breakpoints(std::vector& indices, + std::vector& ratios, + std::vector& harris_ratios); i_t single_pass(i_t start, i_t end, const std::vector& indices, @@ -86,9 +88,7 @@ class bound_flipping_ratio_test_t { i_t& nonbasic_entering, i_t& entering_index, f_t& max_val); - void determine_flips(f_t step_length, - i_t entering_index, - std::vector& flip_indices); + void determine_flips(f_t step_length, i_t entering_index, std::vector& flip_indices); const std::vector& lower_; const std::vector& upper_; const std::vector& bounded_variables_; diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index f7e5d6c7f3..9eb3224817 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -465,20 +465,14 @@ void initial_perturbation(const lp_problem_t& lp, } // Dampen large costs - if (max_abs_obj_coeff > 100.0) { - max_abs_obj_coeff = std::sqrt(std::sqrt(max_abs_obj_coeff)); - } + if (max_abs_obj_coeff > 100.0) { max_abs_obj_coeff = std::sqrt(std::sqrt(max_abs_obj_coeff)); } // Ensure a minimum perturbation even for tiny-cost problems - if (max_abs_obj_coeff < 1.0) { - max_abs_obj_coeff = 1.0; - } + if (max_abs_obj_coeff < 1.0) { max_abs_obj_coeff = 1.0; } // If few boxed variables, cap max_abs_obj_coeff at 1.0 i_t num_boxed = 0; for (i_t j = 0; j < n; ++j) { - if (lp.lower[j] > -inf && lp.upper[j] < inf && lp.lower[j] != lp.upper[j]) { - num_boxed++; - } + if (lp.lower[j] > -inf && lp.upper[j] < inf && lp.lower[j] != lp.upper[j]) { num_boxed++; } } if (static_cast(num_boxed) / n < 0.01) { max_abs_obj_coeff = std::min(max_abs_obj_coeff, f_t(1.0)); @@ -490,8 +484,13 @@ void initial_perturbation(const lp_problem_t& lp, // that case. The original costs are restored before declaring optimality. const f_t perturbation_base = (strongly_degenerate ? 1e-5 : 5e-7) * max_abs_obj_coeff; - settings.log.printf("Perturbation debug: max_abs_obj_coeff=%e (dampened), perturbation_base=%e, n=%d, num_boxed=%d\n", - max_abs_obj_coeff, perturbation_base, n, num_boxed); + settings.log.printf( + "Perturbation debug: max_abs_obj_coeff=%e (dampened), perturbation_base=%e, n=%d, " + "num_boxed=%d\n", + max_abs_obj_coeff, + perturbation_base, + n, + num_boxed); objective.resize(n); f_t sum_perturb = 0.0; @@ -504,19 +503,17 @@ void initial_perturbation(const lp_problem_t& lp, const f_t lower = lp.lower[j]; const f_t upper = lp.upper[j]; // Skip truly fixed variables and free variables - if (lower == upper || (lower == -inf && upper == inf)) { - continue; - } + if (lower == upper || (lower == -inf && upper == inf)) { continue; } - const f_t rand_val = random.random(); + const f_t rand_val = random.random(); const f_t cost_factor = std::min(std::abs(obj) + 1.0, max_abs_obj_coeff + 1.0); - const f_t perturb = (1.0 + rand_val) * cost_factor * perturbation_base; + const f_t perturb = (1.0 + rand_val) * cost_factor * perturbation_base; if (vstatus[j] == variable_status_t::BASIC) { // Skip basic variables continue; } else if (vstatus[j] == variable_status_t::NONBASIC_LOWER || - vstatus[j] == variable_status_t::NONBASIC_FIXED) { + vstatus[j] == variable_status_t::NONBASIC_FIXED) { // NONBASIC_FIXED from phase 1 for boxed variables — treat as at lower bound objective[j] = obj + perturb; sum_perturb += perturb; @@ -1236,14 +1233,14 @@ i_t phase2_ratio_test(const lp_problem_t& lp, template i_t flip_bounds(const lp_problem_t& lp, - const std::vector& bounded_variables, - const std::vector& flip_indices, - std::vector& vstatus, + const std::vector& bounded_variables, + const std::vector& flip_indices, + std::vector& vstatus, std::vector& delta_x, std::vector& mark, - std::vector& atilde, - std::vector& atilde_index, - f_t& work_estimate) + std::vector& atilde, + std::vector& atilde_index, + f_t& work_estimate) { i_t num_flipped = 0; for (const i_t j : flip_indices) { @@ -1571,8 +1568,8 @@ void remove_leaving_perturbation(const lp_problem_t& lp, const f_t perturb = objective[leaving_index] - lp.objective[leaving_index]; if (perturb == 0.0) return; - const f_t lower = lp.lower[leaving_index]; - const f_t upper = lp.upper[leaving_index]; + const f_t lower = lp.lower[leaving_index]; + const f_t upper = lp.upper[leaving_index]; const bool boxed = (lower > -inf && upper < inf); if (boxed) { @@ -1582,7 +1579,7 @@ void remove_leaving_perturbation(const lp_problem_t& lp, const f_t new_z = z[leaving_index] - perturb; if (direction == 1 && new_z < -settings.tight_tol) { return; } if (direction == -1 && new_z > settings.tight_tol) { return; } - z[leaving_index] = new_z; + z[leaving_index] = new_z; objective[leaving_index] = lp.objective[leaving_index]; } else { z[leaving_index] -= perturb; @@ -1592,12 +1589,12 @@ void remove_leaving_perturbation(const lp_problem_t& lp, if (upper == inf && lower > -inf && z[leaving_index] < -settings.tight_tol) { // At lower bound, needs z >= 0 const f_t correction = -z[leaving_index]; - z[leaving_index] = 0.0; + z[leaving_index] = 0.0; objective[leaving_index] += correction; } else if (lower == -inf && upper < inf && z[leaving_index] > settings.tight_tol) { // At upper bound, needs z <= 0 const f_t correction = z[leaving_index]; - z[leaving_index] = 0.0; + z[leaving_index] = 0.0; objective[leaving_index] -= correction; } } @@ -2271,20 +2268,20 @@ void bound_info(const lp_problem_t& lp, template i_t set_primal_variables_on_bounds(const lp_problem_t& lp, - const simplex_solver_settings_t& settings, - const std::vector& z, - std::vector& vstatus, - std::vector& x, - i_t degen_type = 0) + const simplex_solver_settings_t& settings, + const std::vector& z, + std::vector& vstatus, + std::vector& x, + i_t degen_type = 0) { PHASE2_NVTX_RANGE("DualSimplex::set_primal_variables_on_bounds"); - const i_t n = lp.num_cols; - f_t tol = 1e-10; + const i_t n = lp.num_cols; + f_t tol = 1e-10; i_t num_fixed_to_lower = 0; i_t num_fixed_to_upper = 0; i_t num_lower_to_upper = 0; i_t num_upper_to_lower = 0; - i_t num_set_fixed = 0; + i_t num_set_fixed = 0; for (i_t j = 0; j < n; ++j) { // We set z_j = 0 for basic variables // But we explicitally skip setting basic variables here @@ -2315,8 +2312,8 @@ i_t set_primal_variables_on_bounds(const lp_problem_t& lp, if (degen_type == 1) { // Column-sum heuristic const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - f_t col_sum = 0.0; + const i_t col_end = lp.A.col_start[j + 1]; + f_t col_sum = 0.0; for (i_t k = col_start; k < col_end; k++) { col_sum += lp.A.x[k]; } @@ -2347,8 +2344,8 @@ i_t set_primal_variables_on_bounds(const lp_problem_t& lp, if (degen_type == 1) { // Column-sum heuristic const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - f_t col_sum = 0.0; + const i_t col_end = lp.A.col_start[j + 1]; + f_t col_sum = 0.0; for (i_t k = col_start; k < col_end; k++) { col_sum += lp.A.x[k]; } @@ -2408,17 +2405,34 @@ i_t set_primal_variables_on_bounds(const lp_problem_t& lp, } // Track changes if (old_vstatus != vstatus[j]) { - if (old_vstatus == variable_status_t::NONBASIC_FIXED && vstatus[j] == variable_status_t::NONBASIC_LOWER) num_fixed_to_lower++; - else if (old_vstatus == variable_status_t::NONBASIC_FIXED && vstatus[j] == variable_status_t::NONBASIC_UPPER) num_fixed_to_upper++; - else if (old_vstatus == variable_status_t::NONBASIC_LOWER && vstatus[j] == variable_status_t::NONBASIC_UPPER) num_lower_to_upper++; - else if (old_vstatus == variable_status_t::NONBASIC_UPPER && vstatus[j] == variable_status_t::NONBASIC_LOWER) num_upper_to_lower++; - else if (vstatus[j] == variable_status_t::NONBASIC_FIXED) num_set_fixed++; + if (old_vstatus == variable_status_t::NONBASIC_FIXED && + vstatus[j] == variable_status_t::NONBASIC_LOWER) + num_fixed_to_lower++; + else if (old_vstatus == variable_status_t::NONBASIC_FIXED && + vstatus[j] == variable_status_t::NONBASIC_UPPER) + num_fixed_to_upper++; + else if (old_vstatus == variable_status_t::NONBASIC_LOWER && + vstatus[j] == variable_status_t::NONBASIC_UPPER) + num_lower_to_upper++; + else if (old_vstatus == variable_status_t::NONBASIC_UPPER && + vstatus[j] == variable_status_t::NONBASIC_LOWER) + num_upper_to_lower++; + else if (vstatus[j] == variable_status_t::NONBASIC_FIXED) + num_set_fixed++; } } - i_t total_changes = num_fixed_to_lower + num_fixed_to_upper + num_lower_to_upper + num_upper_to_lower + num_set_fixed; + i_t total_changes = num_fixed_to_lower + num_fixed_to_upper + num_lower_to_upper + + num_upper_to_lower + num_set_fixed; if (total_changes > 0) { - settings.log.printf("set_primal_variables_on_bounds: %d changes (fixed->lower=%d, fixed->upper=%d, lower->upper=%d, upper->lower=%d, ->fixed=%d)\n", - total_changes, num_fixed_to_lower, num_fixed_to_upper, num_lower_to_upper, num_upper_to_lower, num_set_fixed); + settings.log.printf( + "set_primal_variables_on_bounds: %d changes (fixed->lower=%d, fixed->upper=%d, " + "lower->upper=%d, upper->lower=%d, ->fixed=%d)\n", + total_changes, + num_fixed_to_lower, + num_fixed_to_upper, + num_lower_to_upper, + num_upper_to_lower, + num_set_fixed); } return total_changes; } @@ -2472,8 +2486,8 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, f_t& primal_infeasibility_squared, f_t& work_estimate) { - const i_t m = lp.num_rows; - const i_t n = lp.num_cols; + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; const i_t n_minus_m = n - m; // Check if there's any perturbation @@ -2481,7 +2495,7 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, if (perturbation <= 1e-6) return 0; // OPTIMAL // Count perturbations on basic vs nonbasic variables - i_t num_basic_perturbed = 0; + i_t num_basic_perturbed = 0; i_t num_nonbasic_boxed_perturbed = 0; i_t num_nonbasic_other_perturbed = 0; for (i_t k = 0; k < m; ++k) { @@ -2506,55 +2520,60 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, // y is unaffected; z[j] - perturb gives exact unperturbed reduced cost. i_t num_flipped = 0; for (i_t k = 0; k < n_minus_m; ++k) { - const i_t j = nonbasic_list[k]; + const i_t j = nonbasic_list[k]; const f_t perturb = objective[j] - lp.objective[j]; if (perturb == 0.0) continue; const f_t new_z = z[j] - perturb; if (vstatus[j] == variable_status_t::NONBASIC_LOWER && new_z < -settings.dual_tol) { - vstatus[j] = variable_status_t::NONBASIC_UPPER; - z[j] = new_z; + vstatus[j] = variable_status_t::NONBASIC_UPPER; + z[j] = new_z; objective[j] = lp.objective[j]; num_flipped++; } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER && new_z > settings.dual_tol) { - vstatus[j] = variable_status_t::NONBASIC_LOWER; - z[j] = new_z; + vstatus[j] = variable_status_t::NONBASIC_LOWER; + z[j] = new_z; objective[j] = lp.objective[j]; num_flipped++; } else { - z[j] = new_z; + z[j] = new_z; objective[j] = lp.objective[j]; } } work_estimate += 5 * n_minus_m; // Recompute x_B with flipped statuses - compute_primal_solution_from_basis(lp, ft, basic_list, nonbasic_list, vstatus, x, xB_workspace, work_estimate); + compute_primal_solution_from_basis( + lp, ft, basic_list, nonbasic_list, vstatus, x, xB_workspace, work_estimate); work_estimate += 2 * n; - primal_infeasibility_squared = - compute_initial_primal_infeasibilities(lp, settings, basic_list, x, - squared_infeasibilities, infeasibility_indices, - primal_infeasibility); + primal_infeasibility_squared = compute_initial_primal_infeasibilities(lp, + settings, + basic_list, + x, + squared_infeasibilities, + infeasibility_indices, + primal_infeasibility); work_estimate += 4 * m + 2 * n; if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL - settings.log.printf("Removed perturbation. Continuing dual simplex (primal_inf=%.2e)\n", primal_infeasibility); + settings.log.printf("Removed perturbation. Continuing dual simplex (primal_inf=%.2e)\n", + primal_infeasibility); return 1; // CONTINUE_DUAL } // Perturbation on basic (or one-sided nonbasic) variables: need to recompute (y, z). std::vector unperturbed_y(m); std::vector unperturbed_z(n); - compute_dual_solution_from_basis(lp, ft, basic_list, nonbasic_list, - unperturbed_y, unperturbed_z, work_estimate); + compute_dual_solution_from_basis( + lp, ft, basic_list, nonbasic_list, unperturbed_y, unperturbed_z, work_estimate); // Check if removal is clean (no dual infeasibility) - const f_t dual_infeas = dual_infeasibility(lp, settings, vstatus, unperturbed_z, - settings.tight_tol, settings.dual_tol); + const f_t dual_infeas = + dual_infeasibility(lp, settings, vstatus, unperturbed_z, settings.tight_tol, settings.dual_tol); work_estimate += 3 * n; if (dual_infeas <= settings.dual_tol) { settings.log.printf("Removed perturbation of %.2e.\n", perturbation); - z = unperturbed_z; - y = unperturbed_y; + z = unperturbed_z; + y = unperturbed_y; objective = lp.objective; work_estimate += 3 * n + 2 * m; return 0; // OPTIMAL @@ -2562,13 +2581,13 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, // Flip boxed nonbasics that are dual infeasible, and check for one-sided infeasibility std::vector new_vstatus = vstatus; - i_t num_flipped = 0; - f_t residual_dual_infeas = 0.0; + i_t num_flipped = 0; + f_t residual_dual_infeas = 0.0; for (i_t k = 0; k < n_minus_m; ++k) { - const i_t j = nonbasic_list[k]; - const f_t zj = unperturbed_z[j]; - const f_t lower = lp.lower[j]; - const f_t upper = lp.upper[j]; + const i_t j = nonbasic_list[k]; + const f_t zj = unperturbed_z[j]; + const f_t lower = lp.lower[j]; + const f_t upper = lp.upper[j]; const bool boxed = (lower > -inf && upper < inf && lower != upper); if (new_vstatus[j] == variable_status_t::NONBASIC_LOWER && zj < -settings.dual_tol) { @@ -2592,28 +2611,35 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, if (residual_dual_infeas > settings.dual_tol) { // One-sided infeasibility remains — can't continue with dual simplex. // new_vstatus is discarded; vstatus unchanged. - settings.log.printf("Perturbation removal: %d flips, residual_dual_infeas=%.2e (PRIMAL_CLEANUP)\n", - num_flipped, residual_dual_infeas); + settings.log.printf( + "Perturbation removal: %d flips, residual_dual_infeas=%.2e (PRIMAL_CLEANUP)\n", + num_flipped, + residual_dual_infeas); return 2; // PRIMAL_CLEANUP } // All infeasibility was on boxed variables — accept unperturbed solution - vstatus = new_vstatus; - z = unperturbed_z; - y = unperturbed_y; + vstatus = new_vstatus; + z = unperturbed_z; + y = unperturbed_y; objective = lp.objective; work_estimate += 3 * n + 2 * m; // Recompute x_B with flipped statuses - compute_primal_solution_from_basis(lp, ft, basic_list, nonbasic_list, vstatus, x, xB_workspace, work_estimate); + compute_primal_solution_from_basis( + lp, ft, basic_list, nonbasic_list, vstatus, x, xB_workspace, work_estimate); work_estimate += 2 * n; - primal_infeasibility_squared = - compute_initial_primal_infeasibilities(lp, settings, basic_list, x, - squared_infeasibilities, infeasibility_indices, - primal_infeasibility); + primal_infeasibility_squared = compute_initial_primal_infeasibilities(lp, + settings, + basic_list, + x, + squared_infeasibilities, + infeasibility_indices, + primal_infeasibility); work_estimate += 4 * m + 2 * n; - settings.log.printf("Perturbation removal: %d flips, primal_inf=%.2e\n", num_flipped, primal_infeasibility); + settings.log.printf( + "Perturbation removal: %d flips, primal_inf=%.2e\n", num_flipped, primal_infeasibility); if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL settings.log.printf("Continuing dual after flip (primal_inf=%.2e)\n", primal_infeasibility); return 1; // CONTINUE_DUAL @@ -2644,22 +2670,18 @@ void prepare_optimality(i_t info, const i_t m = lp.num_rows; const i_t n = lp.num_cols; - sol.objective = compute_objective(lp, sol.x); - sol.user_objective = compute_user_objective(lp, sol.objective); + sol.objective = compute_objective(lp, sol.x); + sol.user_objective = compute_user_objective(lp, sol.objective); const f_t perturbation = amount_of_perturbation(lp, objective); - sol.l2_primal_residual = l2_primal_residual(lp, sol); - sol.l2_dual_residual = l2_dual_residual(lp, sol); - const f_t dual_infeas = dual_infeasibility(lp, settings, vstatus, z, 0.0, 0.0); + sol.l2_primal_residual = l2_primal_residual(lp, sol); + sol.l2_dual_residual = l2_dual_residual(lp, sol); + const f_t dual_infeas = dual_infeasibility(lp, settings, vstatus, z, 0.0, 0.0); // Compute max primal infeasibility for reporting f_t primal_infeas = 0.0; for (i_t j = 0; j < n; ++j) { - if (x[j] < lp.lower[j]) { - primal_infeas = std::max(primal_infeas, lp.lower[j] - x[j]); - } - if (x[j] > lp.upper[j]) { - primal_infeas = std::max(primal_infeas, x[j] - lp.upper[j]); - } + if (x[j] < lp.lower[j]) { primal_infeas = std::max(primal_infeas, lp.lower[j] - x[j]); } + if (x[j] > lp.upper[j]) { primal_infeas = std::max(primal_infeas, x[j] - lp.upper[j]); } } if (phase == 1 && iter > 0) { settings.log.printf("Dual phase I complete. Iterations %d. Time %.2f\n", iter, toc(start_time)); @@ -2834,18 +2856,18 @@ class phase2_timers_t { f_t bfrt_select_time{0.0}; // BFRT diagnostic counters i_t bfrt_calls{0}; - i_t bfrt_zero_steps{0}; // step_length == 0 - i_t bfrt_single_pass_only{0}; // no bound flips (single_pass decided) - i_t bfrt_bucket_used{0}; // bucket sort was used - i_t bfrt_not_last_bucket{0}; // selected from a bucket other than the last - i_t bfrt_fallback{0}; // fell back to single_pass result after bucket sort - i_t bfrt_zero_step_num_buckets_sum{0}; // sum of num_buckets on zero-step iters - i_t bfrt_zero_step_bucket0_sum{0}; // sum of bucket0 size on zero-step iters - i_t bfrt_zero_step_num_breakpoints_sum{0}; // sum of num_breakpoints on zero-step iters - i_t bfrt_zero_step_harris_zero_sum{0}; // sum of harris_ratios==0 on zero-step iters - i_t bfrt_zero_step_exact_zero_sum{0}; // sum of exact ratios==0 on zero-step iters - i_t bfrt_selected_slope_breaker{0}; // times we selected the slope breaker - i_t bfrt_not_slope_breaker{0}; // times we selected something else + i_t bfrt_zero_steps{0}; // step_length == 0 + i_t bfrt_single_pass_only{0}; // no bound flips (single_pass decided) + i_t bfrt_bucket_used{0}; // bucket sort was used + i_t bfrt_not_last_bucket{0}; // selected from a bucket other than the last + i_t bfrt_fallback{0}; // fell back to single_pass result after bucket sort + i_t bfrt_zero_step_num_buckets_sum{0}; // sum of num_buckets on zero-step iters + i_t bfrt_zero_step_bucket0_sum{0}; // sum of bucket0 size on zero-step iters + i_t bfrt_zero_step_num_breakpoints_sum{0}; // sum of num_breakpoints on zero-step iters + i_t bfrt_zero_step_harris_zero_sum{0}; // sum of harris_ratios==0 on zero-step iters + i_t bfrt_zero_step_exact_zero_sum{0}; // sum of exact ratios==0 on zero-step iters + i_t bfrt_selected_slope_breaker{0}; // times we selected the slope breaker + i_t bfrt_not_slope_breaker{0}; // times we selected something else work_timer_t pricing_time; work_timer_t btran_time; work_timer_t ftran_time; @@ -3027,7 +3049,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } } } - settings.log.printf("NONBASIC_FIXED boxed: %d, degenerate (|z_j| < dual_tol): %d\n", num_fixed, num_degen); + settings.log.printf( + "NONBASIC_FIXED boxed: %d, degenerate (|z_j| < dual_tol): %d\n", num_fixed, num_degen); } // Try 3 strategies for degenerate bound assignment, pick best @@ -3037,23 +3060,33 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, std::vector best_vstatus; std::vector best_x; const char* degen_names[] = {"default", "column-sum", "abs-bound"}; - const i_t degen_types[] = {0, 1, 3}; + const i_t degen_types[] = {0, 1, 3}; f_t all_sum_infeas[3]; i_t all_num_infeas[3]; for (i_t di = 0; di < 3; di++) { - const i_t dt = degen_types[di]; + const i_t dt = degen_types[di]; std::vector try_vstatus = vstatus; - std::vector try_x = x; + std::vector try_x = x; phase2::set_primal_variables_on_bounds(lp, settings, z, try_vstatus, try_x, dt); - phase2::compute_primal_variables(ft, lp.rhs, lp.A, basic_list, nonbasic_list, - settings.tight_tol, try_x, xB_workspace, phase2_work_estimate); + phase2::compute_primal_variables(ft, + lp.rhs, + lp.A, + basic_list, + nonbasic_list, + settings.tight_tol, + try_x, + xB_workspace, + phase2_work_estimate); f_t sum_infeas = 0.0; i_t num_infeas = 0; for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; - f_t infeas = std::max(lp.lower[j] - try_x[j], try_x[j] - lp.upper[j]); - if (infeas > 0.0) { sum_infeas += infeas; num_infeas++; } + f_t infeas = std::max(lp.lower[j] - try_x[j], try_x[j] - lp.upper[j]); + if (infeas > 0.0) { + sum_infeas += infeas; + num_infeas++; + } } all_sum_infeas[di] = sum_infeas; all_num_infeas[di] = num_infeas; @@ -3062,16 +3095,16 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, best_sum_infeas = sum_infeas; best_num_infeas = num_infeas; best_degen_type = 0; - best_vstatus = try_vstatus; - best_x = try_x; + best_vstatus = try_vstatus; + best_x = try_x; } else { // Only pick alternative if BOTH fewer infeasibilities AND lower sum if (num_infeas <= best_num_infeas && sum_infeas < best_sum_infeas) { best_sum_infeas = sum_infeas; best_num_infeas = num_infeas; best_degen_type = di; - best_vstatus = try_vstatus; - best_x = try_x; + best_vstatus = try_vstatus; + best_x = try_x; } } if (phase == 1 || num_degen == 0) { @@ -3083,12 +3116,16 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } } vstatus = best_vstatus; - x = best_x; - settings.log.printf("Bound assignment: default(%d/%.2e) colsum(%d/%.2e) abs-bound(%d/%.2e) -> %s\n", - all_num_infeas[0], all_sum_infeas[0], - all_num_infeas[1], all_sum_infeas[1], - all_num_infeas[2], all_sum_infeas[2], - degen_names[best_degen_type]); + x = best_x; + settings.log.printf( + "Bound assignment: default(%d/%.2e) colsum(%d/%.2e) abs-bound(%d/%.2e) -> %s\n", + all_num_infeas[0], + all_sum_infeas[0], + all_num_infeas[1], + all_sum_infeas[1], + all_num_infeas[2], + all_sum_infeas[2], + degen_names[best_degen_type]); phase2_work_estimate += 15 * (n - m); // Near-optimality check: decide whether to apply initial perturbation @@ -3097,16 +3134,21 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, f_t max_primal_infeas = 0.0; for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; - f_t infeas = std::max(lp.lower[j] - x[j], x[j] - lp.upper[j]); + f_t infeas = std::max(lp.lower[j] - x[j], x[j] - lp.upper[j]); if (infeas > settings.primal_tol) { num_primal_infeas++; max_primal_infeas = std::max(max_primal_infeas, infeas); } } - bool near_optimal = (num_primal_infeas < 1000 && max_primal_infeas < 1e-3); + bool near_optimal = (num_primal_infeas < 1000 && max_primal_infeas < 1e-3); bool apply_perturbation = (settings.initial_perturbation == 1) || !near_optimal; - settings.log.printf("Near-optimal check: num_primal_infeas=%d, max_primal_infeas=%.2e, near_optimal=%d, apply_perturbation=%d\n", - num_primal_infeas, max_primal_infeas, near_optimal, apply_perturbation); + settings.log.printf( + "Near-optimal check: num_primal_infeas=%d, max_primal_infeas=%.2e, near_optimal=%d, " + "apply_perturbation=%d\n", + num_primal_infeas, + max_primal_infeas, + near_optimal, + apply_perturbation); if (apply_perturbation) { const bool strongly_degenerate = num_degen > n / 20; phase2::initial_perturbation(lp, settings, vstatus, strongly_degenerate, objective); @@ -3122,8 +3164,15 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t num_bound_changes2 = phase2::set_primal_variables_on_bounds(lp, settings, z, vstatus, x); phase2_work_estimate += 5 * (n - m); if (num_bound_changes2 > 0) { - phase2::compute_primal_variables(ft, lp.rhs, lp.A, basic_list, nonbasic_list, - settings.tight_tol, x, xB_workspace, phase2_work_estimate); + phase2::compute_primal_variables(ft, + lp.rhs, + lp.A, + basic_list, + nonbasic_list, + settings.tight_tol, + x, + xB_workspace, + phase2_work_estimate); } } } @@ -3452,10 +3501,22 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, // Before declaring optimal, attempt to remove perturbation. if (phase == 2) { - i_t removal_status = phase2::attempt_to_remove_perturbations( - lp, settings, ft, basic_list, nonbasic_list, vstatus, objective, - z, y, x, xB_workspace, squared_infeasibilities, infeasibility_indices, - primal_infeasibility, primal_infeasibility_squared, phase2_work_estimate); + i_t removal_status = phase2::attempt_to_remove_perturbations(lp, + settings, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + z, + y, + x, + xB_workspace, + squared_infeasibilities, + infeasibility_indices, + primal_infeasibility, + primal_infeasibility_squared, + phase2_work_estimate); if (removal_status == 1) { // CONTINUE_DUAL obj = phase2::compute_perturbed_objective(objective, x); phase2_work_estimate += 2 * n; @@ -3466,19 +3527,19 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, settings.log.printf("Failed to remove perturbation of %.2e.\n", perturbation); settings.log.printf("Num updates: %d\n", ft.num_updates()); settings.log.printf("Iterations: %d\n", iter); - i_t dual_iter = iter; + i_t dual_iter = iter; primal_status_t primal_status = primal_phase2_with_advanced_basis(2, - start_time, - lp, - settings, - vstatus, - ft, - basic_list, - nonbasic_list, - sol, - iter, - phase2_work_estimate, - false); + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + phase2_work_estimate, + false); if (primal_status == primal_status_t::OPTIMAL) { settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); objective = lp.objective; @@ -3486,9 +3547,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, settings.log.printf("Primal cleanup failed.\n"); const f_t dual_infeas = phase2::dual_infeasibility( lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); - if (dual_infeas > 10.0 * settings.dual_tol) { - return dual_status_t::NUMERICAL; - } + if (dual_infeas > 10.0 * settings.dual_tol) { return dual_status_t::NUMERICAL; } } } // removal_status == 0 (OPTIMAL) or primal cleanup done: fall through to prepare_optimality @@ -3510,8 +3569,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, iter, x, y, - z, - sol); + z, + sol); status = dual_status_t::OPTIMAL; break; } @@ -3649,8 +3708,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, delta_z, delta_z_indices, nonbasic_mark); - entering_index = bfrt.compute_step_length( - step_length, nonbasic_entering_index, flip_indices); + entering_index = + bfrt.compute_step_length(step_length, nonbasic_entering_index, flip_indices); phase2_work_estimate += bfrt.work_estimate(); if (entering_index == RATIO_TEST_NUMERICAL_ISSUES) { settings.log.printf("Numerical issues encountered in ratio test.\n"); @@ -3701,10 +3760,22 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += 2 * n; if (perturbation > 0.0 && phase == 2) { - i_t removal_status = phase2::attempt_to_remove_perturbations( - lp, settings, ft, basic_list, nonbasic_list, vstatus, objective, - z, y, x, xB_workspace, squared_infeasibilities, infeasibility_indices, - primal_infeasibility, primal_infeasibility_squared, phase2_work_estimate); + i_t removal_status = phase2::attempt_to_remove_perturbations(lp, + settings, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + z, + y, + x, + xB_workspace, + squared_infeasibilities, + infeasibility_indices, + primal_infeasibility, + primal_infeasibility_squared, + phase2_work_estimate); if (removal_status == 0) { // OPTIMAL obj = phase2::compute_perturbed_objective(objective, x); phase2_work_estimate += 2 * n; @@ -3833,7 +3904,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, atilde, atilde_index, phase2_work_estimate) - : 0; + : 0; timers.flip_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); total_bound_flips += num_flipped; @@ -4004,7 +4075,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::check_primal_infeasibilities( lp, settings, basic_list, x, squared_infeasibilities, infeasibility_indices); #endif - timers.update_infeasibility_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); + timers.update_infeasibility_time += + timers.stop_timer(phase2_work_estimate + ft.work_estimate()); // Clear delta_x phase2::clear_delta_x( @@ -4015,9 +4087,16 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::remove_leaving_perturbation(lp, settings, leaving_index, direction, z, objective); } f_t sum_perturb = 0.0; - phase2::compute_perturbation( - lp, settings, delta_z_indices, vstatus, z, objective, sum_perturb, - entering_index, step_length, phase2_work_estimate); + phase2::compute_perturbation(lp, + settings, + delta_z_indices, + vstatus, + z, + objective, + sum_perturb, + entering_index, + step_length, + phase2_work_estimate); timers.perturb_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); // Update basis information @@ -4165,8 +4244,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if ((iter % FEATURE_LOG_INTERVAL) == 0 && work_unit_context) { [[maybe_unused]] i_t iters_elapsed = iter - last_feature_log_iter; - work_unit_context->record_work_sync_on_horizon( - (phase2_work_estimate - last_work_reported) / 1e8); + work_unit_context->record_work_sync_on_horizon((phase2_work_estimate - last_work_reported) / + 1e8); last_work_reported = phase2_work_estimate; last_feature_log_iter = iter; @@ -4199,13 +4278,18 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, return dual_status_t::WORK_LIMIT; } - if (now > settings.time_limit) { status = dual_status_t::TIME_LIMIT; break; } + if (now > settings.time_limit) { + status = dual_status_t::TIME_LIMIT; + break; + } if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { return dual_status_t::CONCURRENT_LIMIT; } } - if (status != dual_status_t::TIME_LIMIT && iter >= iter_limit) { status = dual_status_t::ITERATION_LIMIT; } + if (status != dual_status_t::TIME_LIMIT && iter >= iter_limit) { + status = dual_status_t::ITERATION_LIMIT; + } // Flush any remaining work from the basis update into the total work estimate phase2_work_estimate += ft.work_estimate(); @@ -4216,7 +4300,9 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t num_iters = iter - start_iter; if (num_iters > 0) { settings.log.printf("Bound flips: total=%d, avg=%.1f, max=%d\n", - total_bound_flips, 1.0 * total_bound_flips / num_iters, max_bound_flips); + total_bound_flips, + 1.0 * total_bound_flips / num_iters, + max_bound_flips); } constexpr bool print_stats = false; if constexpr (print_stats) { diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 7548e2d28f..1a67956e47 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -575,10 +575,10 @@ i_t primal_ratio_test(const lp_problem_t& lp, i_t direction, f_t& work_estimate) { - const i_t m = lp.num_rows; - basic_leaving = -1; - i_t leaving_index = -1; - constexpr f_t pivot_tol = 1e-8; + const i_t m = lp.num_rows; + basic_leaving = -1; + i_t leaving_index = -1; + constexpr f_t pivot_tol = 1e-8; constexpr f_t harris_tol = 1e-8; // Harris ratio test: two passes. @@ -963,9 +963,7 @@ primal_status_t primal_phase2_with_advanced_basis( work_estimate += basis_update.work_estimate(); basis_update.clear_work_estimate(); - if (work_estimate > settings.work_limit) { - return primal_status_t::WORK_LIMIT; - } + if (work_estimate > settings.work_limit) { return primal_status_t::WORK_LIMIT; } primal_timers_t timers(false); @@ -1328,8 +1326,8 @@ primal_status_t primal_phase2_with_advanced_basis( // After compute_delta_z and update_z, delta_z[j] still holds the raw // pivot row entries (update_z multiplies by dual_step_length into z, not delta_z) for (i_t k = 0; k < n - m; ++k) { - const i_t j = nonbasic_list[k]; - const f_t a_j = delta_z[j]; + const i_t j = nonbasic_list[k]; + const f_t a_j = delta_z[j]; const f_t candidate = (a_j * a_j / pivot_sq) * w_enter; if (candidate > devex_weight[j]) { devex_weight[j] = candidate; } } diff --git a/cpp/src/dual_simplex/scaling.cpp b/cpp/src/dual_simplex/scaling.cpp index 8baaa0a8c7..102036e635 100644 --- a/cpp/src/dual_simplex/scaling.cpp +++ b/cpp/src/dual_simplex/scaling.cpp @@ -255,8 +255,8 @@ i_t scaling(const lp_problem_t& unscaled, // MIP performs integer-aware row scaling before presolve, while QP and SOCP // use the Ruiz path above. Apply this simpler equilibration only to LPs. - const bool use_lp_row_scaling = !settings.inside_mip && unscaled.second_order_cone_dims.empty() && - unscaled.Q.n == 0; + const bool use_lp_row_scaling = + !settings.inside_mip && unscaled.second_order_cone_dims.empty() && unscaled.Q.n == 0; if (use_lp_row_scaling) { csr_matrix_t Arow(0, 0, 0); scaled.A.to_compressed_row(Arow); diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index ef9804e46c..5324b7f0e6 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -193,9 +193,9 @@ struct simplex_solver_settings_t { i_t augmented; // -1 automatic, 0 to solve with ADAT, 1 to solve with augmented system i_t dualize; // -1 automatic, 0 to not dualize, 1 to dualize i_t ordering; // -1 automatic, 0 to use nested dissection, 1 to use AMD - i_t initial_perturbation; // -1 automatic, 0 to not perturb, 1 to perturb - i_t remove_perturbation; // -1 automatic, 0 disabled, 1 enabled - i_t primal_pricing; // 0 Dantzig (default), 1 Devex + i_t initial_perturbation; // -1 automatic, 0 to not perturb, 1 to perturb + i_t remove_perturbation; // -1 automatic, 0 disabled, 1 enabled + i_t primal_pricing; // 0 Dantzig (default), 1 Devex barrier_dual_initial_point_t barrier_dual_initial_point; // -1 automatic, 0 Lustig-Marsten-Shanno, // 1 dual least squares, 2 SeDuMi mu-based @@ -224,11 +224,11 @@ struct simplex_solver_settings_t { i_t strong_chvatal_gomory_cuts; // -1 automatic, 0 to disable, >0 to enable strong Chvatal Gomory // cuts i_t symmetry; // -1 automatic, 0 to disable, >0 to enable different symmetry methods - i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost - // strengthening + i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost + // strengthening i_t dual_degenerate_feasibility_pump; // 0 to disable, 1 to enable - f_t cut_change_threshold; // threshold for cut change - f_t cut_min_orthogonality; // minimum orthogonality for cuts + f_t cut_change_threshold; // threshold for cut change + f_t cut_min_orthogonality; // minimum orthogonality for cuts i_t mip_batch_pdlp_strong_branching; // 0 = DS only, 1 = cooperative DS + PDLP, 2 = batch PDLP only i_t mip_batch_pdlp_reliability_branching; // 0 = DS only, 1 = cooperative DS + PDLP, 2 = batch From c3a025139a51c23b0a2a24885da8be724dfc9e62 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 2 Sep 2026 21:29:35 -0700 Subject: [PATCH 031/113] Prefer large pivots with Harris ratio buckets This improvement was discovered through Hiverge's automated exploration of changes to cuOpt's dual simplex solver and then isolated on top of the v9 row-equilibration and perturbation changes. The bound-flipping ratio test searches Harris buckets from the latest to the earliest so that it favors a longer dual step. Within the selected bucket, choose the candidate with the largest absolute pivot instead of the largest exact breakpoint ratio. Use the exact ratio only to break ties between equal pivots. The change recovers ex9 and neos-3988577-wolgan, which time out in v9, but introduces a timeout on irish-electricity after failed primal cleanup. The larger pivots reduce total BFRT zero steps by 22.8% and improve the aggregate benchmark despite that cleanup regression. Problem v9 v10 Baseline HiGHS v10/HiGHS ------------------------------------------------------------------------------------- var-smallemery-m6j6 0.69 0.69 0.66 300.00 0.00 momentum1 0.69 0.70 0.69 300.00 0.00 neos-5114902-kasavu 2.74 2.73 87.40 300.00 0.01 supportcase42 0.80 0.76 0.52 36.20 0.02 neos-5049753-cuanza 1.02 1.05 7.68 29.85 0.04 supportcase12 6.35 4.11 4.68 37.11 0.11 roi5alpha10n8 1.35 1.32 1.25 11.90 0.11 mzzv11 2.40 2.12 40.19 16.71 0.13 ns1760995 114.33 44.10 135.94 269.53 0.16 ns1952667 0.26 0.15 8.95 0.80 0.19 proteindesign121hz512p9 0.46 0.40 0.91 2.04 0.20 roi2alpha3n4 0.22 0.24 0.31 1.16 0.21 co-100 0.28 0.27 0.68 1.28 0.21 neos-5104907-jarama 21.53 19.35 124.69 89.64 0.22 neos-5052403-cygnet 109.39 72.96 300.00 300.00 0.24 proteindesign122trx11p8 0.35 0.33 0.64 1.26 0.26 neos-1354092 20.05 20.09 300.00 70.92 0.28 supportcase18 0.03 0.04 0.06 0.12 0.33 rd-rplusc-21 0.16 0.17 0.23 0.49 0.35 30n20b8 0.07 0.04 0.08 0.11 0.36 ns1644855 63.39 88.00 300.00 238.30 0.37 neos-787933 0.06 0.07 0.06 0.18 0.39 sct2 0.06 0.06 0.22 0.14 0.43 supportcase7 1.68 1.53 1.29 3.52 0.43 wachplan 0.18 0.12 0.25 0.26 0.46 neos-860300 0.06 0.07 0.10 0.15 0.47 rocII-5-11 0.10 0.10 0.09 0.21 0.48 neos-5093327-huahum 0.23 0.23 0.23 0.48 0.48 supportcase22 2.28 1.92 2.03 3.97 0.48 physiciansched6-2 3.60 3.25 11.15 6.72 0.48 neos-5107597-kakapo 0.06 0.05 0.04 0.10 0.50 cvs16r128-89 1.07 0.88 0.93 1.72 0.51 satellites2-40 5.40 5.56 29.70 10.51 0.53 neos-4647030-tutaki 1.68 1.69 2.47 3.09 0.55 lectsched-5-obj 0.11 0.10 0.13 0.18 0.56 supportcase10 79.85 66.38 300.00 113.55 0.58 buildingenergy 87.82 68.31 300.00 115.61 0.59 n3div36 0.10 0.11 0.11 0.18 0.61 neos-5188808-nattai 0.18 0.19 0.30 0.29 0.66 tbfp-network 7.72 6.01 9.06 9.04 0.66 neos-5195221-niemur 0.26 0.25 0.50 0.37 0.68 thor50dday 0.26 0.25 0.27 0.37 0.68 nursesched-medium-hint03 3.71 3.23 10.47 4.73 0.68 square47 93.64 87.67 79.87 126.92 0.69 cryptanalysiskb128n5obj14 9.55 8.73 29.78 12.54 0.70 ns1116954 9.20 7.52 156.23 10.73 0.70 neos-1171448 0.58 0.60 2.35 0.84 0.71 academictimetablesmall 0.70 0.59 14.89 0.82 0.72 neos-3402454-bohle 64.41 55.07 221.86 74.15 0.74 neos-2746589-doon 2.36 2.25 7.53 3.02 0.75 neos-4300652-rahue 0.46 0.60 1.22 0.80 0.75 neos-3004026-krka 0.07 0.09 0.06 0.12 0.75 neos-3381206-awhea 0.03 0.03 0.08 0.04 0.75 square41 38.25 31.75 28.57 39.86 0.80 blp-ic98 0.08 0.08 0.12 0.10 0.80 dws008-01 0.05 0.04 0.04 0.05 0.80 supportcase33 0.40 0.44 0.99 0.55 0.80 neos-848589 0.70 0.68 1.24 0.85 0.80 comp21-2idx 0.31 0.28 1.55 0.35 0.80 cod105 8.39 6.13 9.06 7.46 0.82 decomp2 0.08 0.10 0.19 0.12 0.83 cryptanalysiskb128n5obj16 10.05 8.70 29.53 10.27 0.85 fiball 0.15 0.12 0.71 0.14 0.86 blp-ar98 0.07 0.10 0.10 0.11 0.91 dano3_3 16.02 17.37 46.88 19.06 0.91 dano3_5 16.04 17.42 46.79 19.03 0.92 neos-3555904-turama 1.31 1.27 1.31 1.37 0.93 neos-3988577-wolgan 300.00 281.89 278.89 300.00 0.94 neos-1171737 0.17 0.19 0.70 0.20 0.95 drayage-25-23 0.05 0.04 0.08 0.04 1.00 h80x6320d 0.04 0.04 0.05 0.04 1.00 highschool1-aigio 300.00 300.00 300.00 300.00 1.00 icir97_tension 0.02 0.02 0.03 0.02 1.00 leo1 0.07 0.07 0.10 0.07 1.00 leo2 0.12 0.13 0.13 0.13 1.00 neos-1456979 0.05 0.04 0.05 0.04 1.00 neos8 0.26 0.26 0.38 0.26 1.00 nursesched-sprint02 0.23 0.23 0.38 0.23 1.00 physiciansched3-3 300.00 300.00 300.00 300.00 1.00 radiationm18-12-05 0.13 0.14 0.23 0.14 1.00 rail02 300.00 300.00 300.00 300.00 1.00 s100 300.00 300.00 300.00 300.00 1.00 savsched1 300.00 300.00 300.00 300.00 1.00 supportcase19 300.00 300.00 300.00 300.00 1.00 swath3 0.02 0.03 0.03 0.03 1.00 traininstance6 0.04 0.03 0.04 0.03 1.00 neos-873061 1.46 1.43 1.46 1.38 1.04 neos-957323 12.14 8.37 300.00 7.74 1.08 mzzv42z 0.91 1.03 10.10 0.95 1.08 neos-3402294-bobin 1.29 1.33 3.07 1.20 1.11 germanrr 0.30 0.30 0.31 0.27 1.11 s250r10 71.91 81.59 300.00 71.74 1.14 supportcase6 4.93 4.83 7.39 4.18 1.16 neos-1582420 0.08 0.07 0.12 0.06 1.17 supportcase40 0.29 0.28 0.24 0.24 1.17 neos-4763324-toguru 5.82 5.85 8.32 5.00 1.17 neos-1122047 1.87 1.90 2.09 1.61 1.18 sp98ar 0.34 0.37 0.39 0.31 1.19 neos-4413714-turia 2.10 2.33 3.97 1.93 1.21 rail507 4.37 3.52 7.17 2.84 1.24 drayage-100-23 0.04 0.05 0.07 0.04 1.25 neos-4738912-atrato 0.04 0.05 0.05 0.04 1.25 traininstance2 0.04 0.05 0.09 0.04 1.25 nexp-150-20-8-5 0.10 0.09 0.10 0.07 1.29 fast0507 3.95 3.58 7.36 2.77 1.29 sp97ar 0.44 0.43 0.40 0.33 1.30 irp 0.12 0.12 0.11 0.09 1.33 swath1 0.03 0.04 0.04 0.03 1.33 radiationm40-10-02 0.97 0.95 1.58 0.71 1.34 cmflsp50-24-8-8 0.58 0.59 0.77 0.44 1.34 map16715-04 9.87 9.21 13.26 6.80 1.35 comp07-2idx 1.26 1.51 4.03 1.10 1.37 map10 8.54 8.67 11.08 6.25 1.39 cbs-cta 0.14 0.14 0.42 0.10 1.40 neos-2978193-inde 0.09 0.07 0.16 0.05 1.40 trento1 3.50 3.10 3.08 2.21 1.40 hypothyroid-k1 4.42 4.40 4.36 3.01 1.46 uccase9 8.04 7.66 11.54 5.20 1.47 qap10 9.80 9.87 15.57 6.68 1.48 neos-827175 0.44 0.43 9.36 0.29 1.48 neos-2987310-joes 1.59 1.49 1.52 1.00 1.49 bnatt500 0.20 0.21 0.29 0.14 1.50 mushroom-best 0.28 0.30 0.26 0.20 1.50 neos-960392 3.29 3.99 8.93 2.66 1.50 air05 0.29 0.27 0.28 0.18 1.50 reblock115 0.14 0.14 0.16 0.09 1.56 neos-4532248-waihi 1.40 1.43 2.61 0.90 1.59 uct-subprob 0.12 0.13 0.11 0.08 1.62 ns1830653 0.25 0.23 0.42 0.14 1.64 fhnw-binpack4-48 0.09 0.10 0.07 0.06 1.67 neos-3083819-nubu 0.05 0.05 0.06 0.03 1.67 ran14x18-disj-8 0.05 0.05 0.04 0.03 1.67 rocI-4-11 0.09 0.10 0.11 0.06 1.67 roll3000 0.10 0.10 0.12 0.06 1.67 opm2-z10-s4 75.72 75.53 89.23 44.75 1.69 istanbul-no-cutoff 1.63 1.62 0.71 0.94 1.72 neos-3656078-kumeu 0.40 0.40 2.37 0.23 1.74 neos-662469 0.86 0.93 1.65 0.53 1.75 k1mushroom 28.33 29.61 31.00 16.80 1.76 rmatr200-p5 8.10 8.20 7.50 4.61 1.78 sing326 9.33 8.37 9.54 4.69 1.78 neos-933966 9.15 5.17 16.79 2.80 1.85 rmatr100-p10 0.33 0.26 0.27 0.14 1.86 bnatt400 0.13 0.15 0.16 0.08 1.88 mcsched 0.30 0.30 0.27 0.16 1.88 atlanta-ip 12.20 9.04 6.85 4.54 1.99 assign1-5-8 0.02 0.02 0.03 0.01 2.00 b1c1s1 0.06 0.04 0.05 0.02 2.00 bppc4-08 0.04 0.04 0.08 0.02 2.00 eil33-2 0.06 0.06 0.05 0.03 2.00 fhnw-binpack4-4 0.03 0.02 0.03 0.01 2.00 ic97_potential 0.02 0.02 0.02 0.01 2.00 mik-250-20-75-4 0.02 0.02 0.02 0.01 2.00 neos-3024952-loue 0.32 0.40 0.41 0.20 2.00 neos-4954672-berkel 0.02 0.02 0.02 0.01 2.00 pg5_34 0.03 0.02 0.03 0.01 2.00 tr12-30 0.02 0.02 0.03 0.01 2.00 graph20-20-1rand 0.30 0.25 0.23 0.12 2.08 neos-3216931-puriri 4.20 6.83 7.33 3.23 2.11 ns1208400 0.76 0.78 3.96 0.36 2.17 chromaticindex512-7 53.03 46.15 16.62 21.24 2.17 splice1k1 19.92 20.18 21.62 9.17 2.20 eilA101-2 3.29 3.08 2.68 1.39 2.22 nw04 1.71 1.36 0.44 0.61 2.23 sing44 12.54 12.80 9.62 5.69 2.25 gfd-schedulen180f7d50m30k18 15.81 15.35 80.00 6.81 2.25 n2seq36q 0.71 0.56 0.46 0.24 2.33 triptim1 132.50 121.77 71.42 51.43 2.37 netdiversion 22.67 22.84 7.64 9.43 2.42 piperout-08 0.41 0.39 0.39 0.16 2.44 chromaticindex1024-7 262.17 232.72 45.86 93.65 2.48 satellites2-60-fs 6.89 8.34 4.17 3.34 2.50 neos-4387871-tavua 0.08 0.10 0.10 0.04 2.50 neos-950242 0.46 0.45 1.05 0.18 2.50 nu25-pr12 0.04 0.05 0.05 0.02 2.50 rococoB10-011000 0.15 0.15 0.13 0.06 2.50 rococoC10-001000 0.06 0.05 0.04 0.02 2.50 glass-sc 0.33 0.31 0.31 0.12 2.58 seymour1 0.97 1.00 0.83 0.38 2.63 seymour 0.95 1.01 0.83 0.38 2.66 unitcal_7 1.05 1.04 0.96 0.39 2.67 sorrell3 1.14 1.10 1.07 0.40 2.75 bab2 75.19 80.63 300.00 28.97 2.78 bab6 34.02 40.19 143.43 14.06 2.86 neos-1445765 0.25 0.26 0.16 0.09 2.89 net12 0.96 1.03 0.55 0.35 2.94 50v-10 0.03 0.03 0.02 0.01 3.00 binkar10_1 0.03 0.03 0.02 0.01 3.00 cost266-UUE 0.06 0.06 0.03 0.02 3.00 gmu-35-40 0.03 0.03 0.03 0.01 3.00 gmu-35-50 0.04 0.03 0.05 0.01 3.00 lotsize 0.03 0.03 0.03 0.01 3.00 n5-3 0.05 0.06 0.03 0.02 3.00 neos-4338804-snowy 0.02 0.03 0.03 0.01 3.00 pg 0.02 0.03 0.03 0.01 3.00 csched007 0.16 0.17 0.19 0.05 3.40 peg-solitaire-a3 2.14 1.78 1.88 0.52 3.42 milo-v12-6-r2-40-1 0.18 0.18 0.25 0.05 3.60 uccase12 5.76 5.36 72.32 1.38 3.88 app1-1 0.10 0.08 0.08 0.02 4.00 csched008 0.13 0.12 0.11 0.03 4.00 neos-3627168-kasai 0.04 0.04 0.04 0.01 4.00 neos17 0.03 0.04 0.03 0.01 4.00 p200x1188c 0.04 0.04 0.03 0.01 4.00 rail01 300.00 300.00 223.84 71.23 4.21 CMS750_4 0.63 0.63 0.38 0.14 4.50 piperout-27 1.23 1.26 0.67 0.26 4.85 neos-631710 215.36 155.73 300.00 31.97 4.87 neos-4722843-widden 7.67 7.75 1.17 1.54 5.03 irish-electricity 109.17 300.00 181.58 59.53 5.04 ex10 300.00 300.00 300.00 59.25 5.06 fastxgemm-n2r6s0t2 0.40 0.47 0.18 0.08 5.87 neos-2075418-temuka 300.00 300.00 128.30 50.57 5.93 beasleyC3 0.06 0.06 0.05 0.01 6.00 snp-02-004-104 24.10 21.83 14.17 2.83 7.71 app1-2 5.77 5.58 5.03 0.70 7.97 mc11 0.08 0.08 0.06 0.01 8.00 brazil3 61.89 63.40 300.00 7.24 8.76 ex9 300.00 132.66 300.00 14.07 9.43 enlight_hard 0.02 0.01 0.02 0.00 10.00 gen-ip054 0.02 0.01 0.03 0.00 10.00 pk1 0.01 0.01 0.02 0.00 10.00 exp-1-500-5-5 0.02 0.02 0.02 0.00 20.00 gen-ip002 0.01 0.02 0.02 0.00 20.00 glass4 0.02 0.02 0.02 0.00 20.00 graphdraw-domain 0.02 0.02 0.02 0.00 20.00 mad 0.02 0.02 0.02 0.00 20.00 markshare2 0.01 0.02 0.02 0.00 20.00 markshare_4_0 0.01 0.02 0.02 0.00 20.00 mas74 0.01 0.02 0.03 0.00 20.00 mas76 0.02 0.02 0.02 0.00 20.00 neos-2657525-crna 0.03 0.02 0.03 0.00 20.00 neos-3046615-murg 0.02 0.02 0.02 0.00 20.00 neos-911970 0.03 0.02 0.03 0.00 20.00 neos5 0.02 0.02 0.02 0.00 20.00 neos859080 0.01 0.02 0.01 0.00 20.00 supportcase26 0.02 0.02 0.03 0.00 20.00 timtab1 0.02 0.02 0.01 0.00 20.00 neos-3754480-nidda 0.02 0.03 0.02 0.00 30.00 sp150x300d 0.02 0.03 0.02 0.00 30.00 Geomean v9/v10: 1.0157 Shifted(+1s): 1.0209 Geomean Baseline/v10: 1.4393 Shifted(+1s): 1.2823 Geomean v10/HiGHS: 1.5021 Shifted(+1s): 0.9925 (240 problems) --- cpp/src/dual_simplex/bound_flipping_ratio_test.cpp | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index 8b9688e021..d18ed95e90 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -406,7 +406,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // Select the entering variable // Scan from last bucket to first. Within each bucket, pick the variable with - // the largest ratio (step length) that has |delta_z| > pivot_threshold + // the largest |delta_z|, provided |delta_z| > pivot_threshold, breaking ties + // by preferring the larger step length. f_t pivot_threshold = std::max(settings_.pivot_tol, std::min(0.1 * max_pivot, 1.0)); i_t entering_k = -1; @@ -414,12 +415,15 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, for (i_t b = num_buckets - 1; b >= 0; b--) { const i_t b_start = bucket_start[b]; const i_t b_end = bucket_start[b + 1]; + f_t best_pivot = -1.0; f_t best_ratio = -1.0; for (i_t h = b_start; h < b_end; h++) { const i_t k = candidates[h]; const i_t j = nonbasic_list_[indicies[k]]; const f_t pivot = std::abs(delta_z_[j]); - if (pivot > pivot_threshold && ratios[k] > best_ratio) { + if (pivot > pivot_threshold && + (pivot > best_pivot || (pivot == best_pivot && ratios[k] > best_ratio))) { + best_pivot = pivot; best_ratio = ratios[k]; entering_k = k; } From ef5ff666e288491ece37e31eb251091c54de09f0 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 2 Sep 2026 21:31:33 -0700 Subject: [PATCH 032/113] Add parameters to control degenerate pivots --- .../mathematical_optimization/constants.h | 2 + .../mip/solver_settings.hpp | 2 + cpp/src/branch_and_bound/branch_and_bound.cpp | 106 ++++++++++-------- .../dual_simplex/simplex_solver_settings.hpp | 4 + cpp/src/math_optimization/solver_settings.cpp | 2 + cpp/src/mip_heuristics/solver.cu | 4 + 6 files changed, 73 insertions(+), 47 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/constants.h b/cpp/include/cuopt/mathematical_optimization/constants.h index 4868c01dc4..9079e2a23d 100644 --- a/cpp/include/cuopt/mathematical_optimization/constants.h +++ b/cpp/include/cuopt/mathematical_optimization/constants.h @@ -84,6 +84,8 @@ #define CUOPT_MIP_STRONG_CHVATAL_GOMORY_CUTS "mip_strong_chvatal_gomory_cuts" #define CUOPT_MIP_REDUCED_COST_STRENGTHENING "mip_reduced_cost_strengthening" #define CUOPT_MIP_DUAL_DEGENERATE_FEASIBILITY_PUMP "mip_dual_degenerate_feasibility_pump" +#define CUOPT_MIP_PRIMAL_DEGENERATE_PIVOTS "mip_primal_degenerate_pivots" +#define CUOPT_MIP_DUAL_DEGENERATE_PIVOTS "mip_dual_degenerate_pivots" #define CUOPT_MIP_RINS "mip_rins" #define CUOPT_MIP_RENS "mip_rens" #define CUOPT_MIP_OBJECTIVE_STEP "mip_objective_step" diff --git a/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp index 71438bfe8c..f12c33e818 100644 --- a/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp @@ -136,6 +136,8 @@ class mip_solver_settings_t { i_t strong_chvatal_gomory_cuts = -1; i_t reduced_cost_strengthening = -1; i_t dual_degenerate_feasibility_pump = -1; // -1 = automatic (on), 0 = off, 1 = on + i_t primal_degenerate_pivots = -1; // -1 = automatic (on), 0 = off, 1 = on + i_t dual_degenerate_pivots = -1; // -1 = automatic (on), 0 = off, 1 = on i_t objective_step = 1; // 0 = disable objective step tightening, 1 = enable f_t cut_change_threshold = -1.0; f_t cut_min_orthogonality = 0.5; diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 3763667c99..e9bf0cbd2e 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -1785,15 +1785,17 @@ dual_status_t branch_and_bound_t::solve_node_lp( std::vector fractional; i_t num_fractional = fractional_variables(settings_, worker->leaf_solution.x, var_types_, fractional); - pivot_out_integer_variables(worker->leaf_problem, - lp_settings, - worker->basic_list, - worker->nonbasic_list, - worker->leaf_vstatus, - worker->leaf_solution, - worker->basis_factors, - num_fractional, - fractional); + if (settings_.dual_degenerate_pivots != 0) { + pivot_out_integer_variables(worker->leaf_problem, + lp_settings, + worker->basic_list, + worker->nonbasic_list, + worker->leaf_vstatus, + worker->leaf_solution, + worker->basis_factors, + num_fractional, + fractional); + } } } } @@ -3464,15 +3466,18 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t::cut_pass_action_t branch_and_bound_t::solve(mip_solution_t& solut settings_.log.printf("New reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); - pivot_to_improve_reduced_cost_strengthening(original_lp_, - basic_list, - nonbasic_list, - root_vstatus_, - root_relax_soln_, - basis_update, - num_fractional, - fractional, - root_objective_, - reduced_cost_bounds); + if (settings_.primal_degenerate_pivots != 0) { + pivot_to_improve_reduced_cost_strengthening(original_lp_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional, + root_objective_, + reduced_cost_bounds); + } settings_.log.printf("After pivoting: new reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); f_t pivot_out_integer_variables_start_time = tic(); - i_t num_integer_increased = pivot_out_integer_variables(original_lp_, - settings_, - basic_list, - nonbasic_list, - root_vstatus_, - root_relax_soln_, - basis_update, - num_fractional, - fractional); + i_t num_integer_increased = 0; + if (settings_.dual_degenerate_pivots != 0) { + num_integer_increased = pivot_out_integer_variables(original_lp_, + settings_, + basic_list, + nonbasic_list, + root_vstatus_, + root_relax_soln_, + basis_update, + num_fractional, + fractional); + } settings_.log.printf("Pivoted out %d integer variables in %e seconds\n", num_integer_increased, toc(pivot_out_integer_variables_start_time)); diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 5324b7f0e6..3d9c439e5a 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -108,6 +108,8 @@ struct simplex_solver_settings_t { symmetry(-1), reduced_cost_strengthening(-1), dual_degenerate_feasibility_pump(1), + primal_degenerate_pivots(1), + dual_degenerate_pivots(1), cut_change_threshold(1e-3), cut_min_orthogonality(0.5), mip_batch_pdlp_strong_branching(0), @@ -227,6 +229,8 @@ struct simplex_solver_settings_t { i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost // strengthening i_t dual_degenerate_feasibility_pump; // 0 to disable, 1 to enable + i_t primal_degenerate_pivots; // 0 to disable, 1 to enable + i_t dual_degenerate_pivots; // 0 to disable, 1 to enable f_t cut_change_threshold; // threshold for cut change f_t cut_min_orthogonality; // minimum orthogonality for cuts i_t diff --git a/cpp/src/math_optimization/solver_settings.cpp b/cpp/src/math_optimization/solver_settings.cpp index bff005a90d..6646e09bbf 100644 --- a/cpp/src/math_optimization/solver_settings.cpp +++ b/cpp/src/math_optimization/solver_settings.cpp @@ -157,6 +157,8 @@ solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings {CUOPT_MIP_STRONG_CHVATAL_GOMORY_CUTS, &mip_settings.strong_chvatal_gomory_cuts, -1, 1, -1}, {CUOPT_MIP_REDUCED_COST_STRENGTHENING, &mip_settings.reduced_cost_strengthening, -1, std::numeric_limits::max(), -1}, {CUOPT_MIP_DUAL_DEGENERATE_FEASIBILITY_PUMP, &mip_settings.dual_degenerate_feasibility_pump, -1, 1, -1}, + {CUOPT_MIP_PRIMAL_DEGENERATE_PIVOTS, &mip_settings.primal_degenerate_pivots, -1, 1, -1}, + {CUOPT_MIP_DUAL_DEGENERATE_PIVOTS, &mip_settings.dual_degenerate_pivots, -1, 1, -1}, {CUOPT_MIP_RINS, &mip_settings.submip_params.rins, -1, 1, -1}, {CUOPT_MIP_RENS, &mip_settings.submip_params.rens, -1, 1, -1}, {CUOPT_MIP_OBJECTIVE_STEP, &mip_settings.objective_step, 0, 1, 1}, diff --git a/cpp/src/mip_heuristics/solver.cu b/cpp/src/mip_heuristics/solver.cu index 6173aa6354..e27a94c64a 100644 --- a/cpp/src/mip_heuristics/solver.cu +++ b/cpp/src/mip_heuristics/solver.cu @@ -403,6 +403,10 @@ solution_t mip_solver_t::run_solver() context.settings.dual_degenerate_feasibility_pump == -1 ? 1 : context.settings.dual_degenerate_feasibility_pump; + branch_and_bound_settings.primal_degenerate_pivots = + context.settings.primal_degenerate_pivots == -1 ? 1 : context.settings.primal_degenerate_pivots; + branch_and_bound_settings.dual_degenerate_pivots = + context.settings.dual_degenerate_pivots == -1 ? 1 : context.settings.dual_degenerate_pivots; branch_and_bound_settings.symmetry = context.settings.symmetry; branch_and_bound_settings.diving_settings = context.settings.diving_params; From 2e02e232000f2c7568eaec10ca001bc84bf6be40 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 2 Sep 2026 21:32:08 -0700 Subject: [PATCH 033/113] Style fixes --- cpp/include/cuopt/mathematical_optimization/constants.h | 4 ++-- cpp/src/branch_and_bound/branch_and_bound.cpp | 4 ++-- cpp/src/dual_simplex/simplex_solver_settings.hpp | 4 ++-- cpp/src/mip_heuristics/solver.cu | 3 ++- 4 files changed, 8 insertions(+), 7 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/constants.h b/cpp/include/cuopt/mathematical_optimization/constants.h index 9079e2a23d..aae5c1ce6f 100644 --- a/cpp/include/cuopt/mathematical_optimization/constants.h +++ b/cpp/include/cuopt/mathematical_optimization/constants.h @@ -84,8 +84,8 @@ #define CUOPT_MIP_STRONG_CHVATAL_GOMORY_CUTS "mip_strong_chvatal_gomory_cuts" #define CUOPT_MIP_REDUCED_COST_STRENGTHENING "mip_reduced_cost_strengthening" #define CUOPT_MIP_DUAL_DEGENERATE_FEASIBILITY_PUMP "mip_dual_degenerate_feasibility_pump" -#define CUOPT_MIP_PRIMAL_DEGENERATE_PIVOTS "mip_primal_degenerate_pivots" -#define CUOPT_MIP_DUAL_DEGENERATE_PIVOTS "mip_dual_degenerate_pivots" +#define CUOPT_MIP_PRIMAL_DEGENERATE_PIVOTS "mip_primal_degenerate_pivots" +#define CUOPT_MIP_DUAL_DEGENERATE_PIVOTS "mip_dual_degenerate_pivots" #define CUOPT_MIP_RINS "mip_rins" #define CUOPT_MIP_RENS "mip_rens" #define CUOPT_MIP_OBJECTIVE_STEP "mip_objective_step" diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index e9bf0cbd2e..5534360fb6 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3466,7 +3466,7 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t::solve(mip_solution_t& solut upper_bound_.load()); f_t pivot_out_integer_variables_start_time = tic(); - i_t num_integer_increased = 0; + i_t num_integer_increased = 0; if (settings_.dual_degenerate_pivots != 0) { num_integer_increased = pivot_out_integer_variables(original_lp_, settings_, diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 3d9c439e5a..bed2c4eee5 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -229,8 +229,8 @@ struct simplex_solver_settings_t { i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost // strengthening i_t dual_degenerate_feasibility_pump; // 0 to disable, 1 to enable - i_t primal_degenerate_pivots; // 0 to disable, 1 to enable - i_t dual_degenerate_pivots; // 0 to disable, 1 to enable + i_t primal_degenerate_pivots; // 0 to disable, 1 to enable + i_t dual_degenerate_pivots; // 0 to disable, 1 to enable f_t cut_change_threshold; // threshold for cut change f_t cut_min_orthogonality; // minimum orthogonality for cuts i_t diff --git a/cpp/src/mip_heuristics/solver.cu b/cpp/src/mip_heuristics/solver.cu index e27a94c64a..645de93cd6 100644 --- a/cpp/src/mip_heuristics/solver.cu +++ b/cpp/src/mip_heuristics/solver.cu @@ -404,7 +404,8 @@ solution_t mip_solver_t::run_solver() ? 1 : context.settings.dual_degenerate_feasibility_pump; branch_and_bound_settings.primal_degenerate_pivots = - context.settings.primal_degenerate_pivots == -1 ? 1 : context.settings.primal_degenerate_pivots; + context.settings.primal_degenerate_pivots == -1 ? 1 + : context.settings.primal_degenerate_pivots; branch_and_bound_settings.dual_degenerate_pivots = context.settings.dual_degenerate_pivots == -1 ? 1 : context.settings.dual_degenerate_pivots; branch_and_bound_settings.symmetry = context.settings.symmetry; From 0f95053204b3ddef63ac0d5bf784ddca41d63f6f Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 4 Sep 2026 11:49:54 -0700 Subject: [PATCH 034/113] Snapshot slack columns for branch-and-bound workers Signed-off-by: Christopher Maes --- cpp/src/branch_and_bound/branch_and_bound.cpp | 35 +++++++---- cpp/src/branch_and_bound/branch_and_bound.hpp | 1 + .../deterministic_workers.hpp | 59 ++++++++++++------- cpp/src/branch_and_bound/worker.hpp | 7 +++ cpp/src/branch_and_bound/worker_pool.hpp | 20 +++++-- cpp/src/mip_heuristics/root_heuristics.hpp | 19 +++++- 6 files changed, 99 insertions(+), 42 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 5534360fb6..83bc52b581 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -1788,6 +1788,7 @@ dual_status_t branch_and_bound_t::solve_node_lp( if (settings_.dual_degenerate_pivots != 0) { pivot_out_integer_variables(worker->leaf_problem, lp_settings, + worker->new_slacks, worker->basic_list, worker->nonbasic_list, worker->leaf_vstatus, @@ -3117,7 +3118,7 @@ void branch_and_bound_t::launch_root_heuristics( // Using shared_ptr here, so the lifetime of the object is tied to the related task. This allows // the solver to send the stop signal and immediately continue the execution. auto current_heuristic = root_heuristics.create_new_cut_pass_heuristic( - Arow_, var_types_, lp_solution.x, edge_norms_, settings_); + Arow_, var_types_, lp_solution.x, edge_norms_, new_slacks_, settings_); auto worker_count = root_heuristics.worker_count_; current_heuristic->initialize_pseudocost( @@ -3470,6 +3471,7 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t i_t branch_and_bound_t::pivot_out_integer_variables( const simplex::lp_problem_t& lp, const simplex::simplex_solver_settings_t& settings, + const std::vector& new_slacks, std::vector& basic_list, std::vector& nonbasic_list, std::vector& vstatus, @@ -4587,7 +4590,7 @@ i_t branch_and_bound_t::pivot_out_integer_variables( const i_t num_zero_reduced_costs_vars = zero_reduced_costs_vars.size(); std::vector row_to_slack(lp.num_rows, -1); - for (i_t j : new_slacks_) { + for (i_t j : new_slacks) { if (lp.lower[j] != 0 || lp.upper[j] != inf) { continue; } const i_t p = lp.A.col_start[j]; row_to_slack[lp.A.i[p]] = j; @@ -5472,6 +5475,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut if (settings_.dual_degenerate_pivots != 0) { num_integer_increased = pivot_out_integer_variables(original_lp_, settings_, + new_slacks_, basic_list, nonbasic_list, root_vstatus_, @@ -5788,19 +5792,21 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut var_types_, symmetry_, settings_, - pc_, - root_relax_soln_.x, - edge_norms_); + pc_, + root_relax_soln_.x, + edge_norms_, + new_slacks_); submip_worker_pool_.init(num_submip_workers, original_lp_, Arow_, var_types_, symmetry_, settings_, - pc_, - root_relax_soln_.x, - edge_norms_, - num_bfs_workers); + pc_, + root_relax_soln_.x, + edge_norms_, + new_slacks_, + num_bfs_workers); if (num_diving_workers > 0) { diving_worker_pool_.init(num_diving_workers, @@ -5812,6 +5818,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut pc_, root_relax_soln_.x, edge_norms_, + new_slacks_, num_bfs_workers + num_submip_workers); } @@ -6011,9 +6018,10 @@ void branch_and_bound_t::run_deterministic_coordinator(const csr_matri Arow, var_types_, settings_, - pc_, - root_relax_soln_.x, - edge_norms_); + pc_, + root_relax_soln_.x, + edge_norms_, + new_slacks_); if (num_diving_workers > 0) { // Extract diving types from search_strategies (skip BEST_FIRST at index 0) @@ -6030,7 +6038,8 @@ void branch_and_bound_t::run_deterministic_coordinator(const csr_matri settings_, pc_, root_relax_soln_.x, - edge_norms_); + edge_norms_, + new_slacks_); } } diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 9ba0b94128..fdc0cf41d5 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -508,6 +508,7 @@ class branch_and_bound_t { i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, const simplex::simplex_solver_settings_t& settings, + const std::vector& new_slacks, std::vector& basic_list, std::vector& nonbasic_list, std::vector& vstatus, diff --git a/cpp/src/branch_and_bound/deterministic_workers.hpp b/cpp/src/branch_and_bound/deterministic_workers.hpp index fae259ac3f..7c31d023cc 100644 --- a/cpp/src/branch_and_bound/deterministic_workers.hpp +++ b/cpp/src/branch_and_bound/deterministic_workers.hpp @@ -89,11 +89,13 @@ class deterministic_worker_base_t : public branch_and_bound_worker_t { const csr_matrix_t& Arow, const std::vector& var_types, const simplex::simplex_solver_settings_t& settings, - pseudo_costs_t& pc, - const std::vector& root_solution, - const std::vector& root_edge_norm, - const std::string& context_name) - : base_t(id, original_lp, Arow, var_types, settings, pc, root_solution, root_edge_norm), + pseudo_costs_t& pc, + const std::vector& root_solution, + const std::vector& root_edge_norm, + const std::vector& new_slacks, + const std::string& context_name) + : base_t( + id, original_lp, Arow, var_types, settings, pc, root_solution, root_edge_norm, new_slacks), work_context(context_name), pc_snapshot(1, settings) { @@ -144,18 +146,20 @@ class deterministic_bfs_worker_t const csr_matrix_t& Arow, const std::vector& var_types, const simplex::simplex_solver_settings_t& settings, - pseudo_costs_t& pc, - const std::vector& root_solution, - const std::vector& root_edge_norm) + pseudo_costs_t& pc, + const std::vector& root_solution, + const std::vector& root_edge_norm, + const std::vector& new_slacks) : base_t(id, original_lp, Arow, var_types, settings, - pc, - root_solution, - root_edge_norm, - "BB_Worker_" + std::to_string(id)) + pc, + root_solution, + root_edge_norm, + new_slacks, + "BB_Worker_" + std::to_string(id)) { } @@ -313,16 +317,18 @@ class deterministic_diving_worker_t const simplex::simplex_solver_settings_t& settings, pseudo_costs_t& pc, const std::vector& root_solution, - const std::vector& root_edge_norm) + const std::vector& root_edge_norm, + const std::vector& new_slacks) : base_t(id, original_lp, Arow, var_types, settings, - pc, - root_solution, - root_edge_norm, - "Diving_Worker_" + std::to_string(id)), + pc, + root_solution, + root_edge_norm, + new_slacks, + "Diving_Worker_" + std::to_string(id)), diving_type(type) { dive_lower = original_lp.lower; @@ -430,12 +436,13 @@ class deterministic_bfs_worker_pool_t const simplex::simplex_solver_settings_t& settings, pseudo_costs_t& pc, const std::vector& root_solution, - const std::vector& root_edge_norm) + const std::vector& root_edge_norm, + const std::vector& new_slacks) { this->workers_.reserve(num_workers); for (int i = 0; i < num_workers; ++i) { this->workers_.emplace_back( - i, original_lp, Arow, var_types, settings, pc, root_solution, root_edge_norm); + i, original_lp, Arow, var_types, settings, pc, root_solution, root_edge_norm, new_slacks); } } @@ -469,13 +476,23 @@ class deterministic_diving_worker_pool_t const simplex::simplex_solver_settings_t& settings, pseudo_costs_t& pc, const std::vector& root_solution, - const std::vector& root_edge_norm) + const std::vector& root_edge_norm, + const std::vector& new_slacks) { this->workers_.reserve(num_workers); for (int i = 0; i < num_workers; ++i) { search_strategy_t type = diving_types[i % diving_types.size()]; this->workers_.emplace_back( - i, type, original_lp, Arow, var_types, settings, pc, root_solution, root_edge_norm); + i, + type, + original_lp, + Arow, + var_types, + settings, + pc, + root_solution, + root_edge_norm, + new_slacks); } } diff --git a/cpp/src/branch_and_bound/worker.hpp b/cpp/src/branch_and_bound/worker.hpp index 0ec0f74bf9..5c705087d2 100644 --- a/cpp/src/branch_and_bound/worker.hpp +++ b/cpp/src/branch_and_bound/worker.hpp @@ -98,6 +98,7 @@ class branch_and_bound_worker_t { const std::vector& root_solution; const std::vector& root_edge_norm; const std::vector& var_types; + const std::vector& new_slacks; pseudo_costs_t& pseudo_costs; @@ -120,6 +121,7 @@ class branch_and_bound_worker_t { pseudo_costs_t& pc, const std::vector& root_solution, const std::vector& root_edge_norm, + const std::vector& new_slacks, uint64_t rng_offset = 0) : worker_id(worker_id), search_strategy(search_strategy_t::BEST_FIRST), @@ -138,6 +140,7 @@ class branch_and_bound_worker_t { root_solution(root_solution), root_edge_norm(root_edge_norm), var_types(var_type), + new_slacks(new_slacks), pseudo_costs(pc) { } @@ -179,6 +182,7 @@ class bfs_worker_t : public branch_and_bound_worker_t { pseudo_costs_t& pc, const std::vector& root_solution, const std::vector& root_edge_norm, + const std::vector& new_slacks, uint64_t rng_offset = 0) : Base(worker_id, original_lp, @@ -188,6 +192,7 @@ class bfs_worker_t : public branch_and_bound_worker_t { pc, root_solution, root_edge_norm, + new_slacks, rng_offset) { this->start_lower = original_lp.lower; @@ -265,6 +270,7 @@ class diving_worker_t : public branch_and_bound_worker_t { pseudo_costs_t& pc, const std::vector& root_solution, const std::vector& root_edge_norm, + const std::vector& new_slacks, uint64_t rng_offset = 0) : Base(worker_id, original_lp, @@ -274,6 +280,7 @@ class diving_worker_t : public branch_and_bound_worker_t { pc, root_solution, root_edge_norm, + new_slacks, rng_offset) { this->start_lower = original_lp.lower; diff --git a/cpp/src/branch_and_bound/worker_pool.hpp b/cpp/src/branch_and_bound/worker_pool.hpp index c4e54a61f1..bdb420a405 100644 --- a/cpp/src/branch_and_bound/worker_pool.hpp +++ b/cpp/src/branch_and_bound/worker_pool.hpp @@ -24,10 +24,11 @@ class worker_pool_t { const std::vector& var_type, mip_symmetry_t* symmetry, const simplex::simplex_solver_settings_t& settings, - pseudo_costs_t& pc, - const std::vector& root_solution, - const std::vector& root_edge_norm, - const uint64_t rng_offset = 0) + pseudo_costs_t& pc, + const std::vector& root_solution, + const std::vector& root_edge_norm, + const std::vector& new_slacks, + const uint64_t rng_offset = 0) { assert(!is_initialized_); assert(num_workers > 0); @@ -37,7 +38,16 @@ class worker_pool_t { idle_workers_.clear_resize(num_workers); for (i_t i = 0; i < num_workers; ++i) { workers_[i] = std::make_unique( - i, original_lp, Arow, var_type, settings, pc, root_solution, root_edge_norm, rng_offset); + i, + original_lp, + Arow, + var_type, + settings, + pc, + root_solution, + root_edge_norm, + new_slacks, + rng_offset); idle_workers_.push_back(i); // Propagate the (possibly null) symmetry pointer; workers lazily build // their orbital_fixing/lexical_reduction state via ensure_orbital_fixing(). diff --git a/cpp/src/mip_heuristics/root_heuristics.hpp b/cpp/src/mip_heuristics/root_heuristics.hpp index b9645579b5..2a893bcc19 100644 --- a/cpp/src/mip_heuristics/root_heuristics.hpp +++ b/cpp/src/mip_heuristics/root_heuristics.hpp @@ -19,6 +19,7 @@ struct cut_pass_heuristics_t { csr_matrix_t Arow_; std::vector root_solution_; std::vector root_edge_norm_; + std::vector new_slacks_; pseudo_costs_t pseudo_costs_; omp_atomic_t active_workers_; std::atomic halt_; @@ -31,11 +32,13 @@ struct cut_pass_heuristics_t { const std::vector& var_types, const std::vector& root_solution, const std::vector& root_edge_norm, + const std::vector& new_slacks, const simplex::simplex_solver_settings_t& settings) : var_types_(var_types), Arow_(Arow), root_solution_(root_solution), root_edge_norm_(root_edge_norm), + new_slacks_(new_slacks), pseudo_costs_(root_solution.size(), settings), active_workers_(0), halt_(false), @@ -82,7 +85,15 @@ struct cut_pass_heuristics_t { search_strategy_t type) { submip_worker_ = std::make_unique>( - id, lp, Arow_, var_types_, settings, pseudo_costs_, root_solution_, root_edge_norm_); + id, + lp, + Arow_, + var_types_, + settings, + pseudo_costs_, + root_solution_, + root_edge_norm_, + new_slacks_); submip_worker_->start_node = mip_node_t(root_obj, root_vstatus); submip_worker_->leaf_vstatus = root_vstatus; submip_worker_->leaf_solution.x = sol; @@ -121,7 +132,8 @@ struct cut_pass_heuristics_t { settings, pseudo_costs_, root_solution_, - root_edge_norm_)); + root_edge_norm_, + new_slacks_)); worker->start_node = root_node.detach_copy(); worker->start_lower = lp.lower; worker->start_upper = lp.upper; @@ -201,10 +213,11 @@ struct root_heuristics_t { const std::vector& var_types, const std::vector& root_solution, const std::vector& root_edge_norm, + const std::vector& new_slacks, const simplex::simplex_solver_settings_t& settings) { return cut_passes_heuristics_.emplace_back(std::make_shared>( - Arow, var_types, root_solution, root_edge_norm, settings)); + Arow, var_types, root_solution, root_edge_norm, new_slacks, settings)); } }; From 9acaebc5c2a89c7719398330f53e885a85c368bc Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 9 Sep 2026 12:28:23 -0700 Subject: [PATCH 035/113] Restrict concurrent halt to root LP solves --- cpp/src/dual_simplex/solve.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index 607a4f798f..867225d79e 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -362,7 +362,7 @@ lp_status_t solve_linear_program_with_advanced_basis( primal_phase2(2, start_time, lp, settings, vstatus, solution, iter); // TODO: We need to update ft if the basis changed } - if (settings.inside_mip && settings.concurrent_halt != nullptr) { + if (settings.inside_mip == 1 && settings.concurrent_halt != nullptr) { settings.log.debug("Setting concurrent halt to 1 inside_mip\n"); *settings.concurrent_halt = 1; } From 4da434909048723848d1f14c211951e9ac12efa5 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 9 Sep 2026 12:52:36 -0700 Subject: [PATCH 036/113] Preserve caller cancellation in node LP solves --- cpp/src/branch_and_bound/branch_and_bound.cpp | 13 ++++++++----- cpp/src/branch_and_bound/branch_and_bound.hpp | 12 +++++++----- 2 files changed, 15 insertions(+), 10 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 83bc52b581..cf0008d80d 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -1643,6 +1643,7 @@ bool branch_and_bound_t::apply_symmetry_reductions( template dual_status_t branch_and_bound_t::solve_node_lp( mip_node_t* node_ptr, + const simplex_solver_settings_t& settings, branch_and_bound_worker_t* worker, branch_and_bound_stats_t& stats, logger_t& log, @@ -1684,8 +1685,7 @@ dual_status_t branch_and_bound_t::solve_node_lp( } #endif - simplex_solver_settings_t lp_settings = settings_; - lp_settings.concurrent_halt = &node_concurrent_halt_; + simplex_solver_settings_t lp_settings = settings; lp_settings.set_log(false); f_t cutoff = upper_bound_.load(); if (worker->leaf_problem.objective_step.has_step()) { @@ -1925,7 +1925,10 @@ void branch_and_bound_t::plunge_with(bfs_worker_t* worker, node_ptr->packed_vstatus, worker->leaf_problem.num_cols, worker->leaf_vstatus); assert(worker->leaf_vstatus.size() == worker->leaf_problem.num_cols); - dual_status_t lp_status = solve_node_lp(node_ptr, worker, exploration_stats_, settings_.log); + simplex_solver_settings_t lp_settings = settings_; + lp_settings.concurrent_halt = &node_concurrent_halt_; + dual_status_t lp_status = + solve_node_lp(node_ptr, lp_settings, worker, exploration_stats_, settings_.log); ++exploration_stats_.nodes_since_last_log; ++exploration_stats_.nodes_explored; --exploration_stats_.nodes_unexplored; @@ -2260,7 +2263,7 @@ void branch_and_bound_t::dive_with(diving_worker_t* worker, node_ptr->packed_vstatus, worker->leaf_problem.num_cols, worker->leaf_vstatus); assert(worker->leaf_vstatus.size() == worker->leaf_problem.num_cols); - dual_status_t lp_status = solve_node_lp(node_ptr, worker, dive_stats, log, max_iter); + dual_status_t lp_status = solve_node_lp(node_ptr, settings, worker, dive_stats, log, max_iter); ++dive_stats.nodes_explored; if (lp_status == dual_status_t::TIME_LIMIT) { @@ -2985,7 +2988,7 @@ void branch_and_bound_t::recursive_submip( break; } - dual_status_t lp_status = solve_node_lp(&node, worker, stats, log, max_iter); + dual_status_t lp_status = solve_node_lp(&node, submip_settings, worker, stats, log, max_iter); if (lp_status != dual_status_t::OPTIMAL) { DEBUG_SUBMIP("{}Round {}: simplex returned {}", submip_settings.log.log_prefix, diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index fdc0cf41d5..9d5df5a842 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -611,11 +611,13 @@ class branch_and_bound_t { root_heuristics_t& root_heuristics); // Solve the LP relaxation of a leaf node - simplex::dual_status_t solve_node_lp(mip_node_t* node_ptr, - branch_and_bound_worker_t* worker, - branch_and_bound_stats_t& stats, - simplex::logger_t& log, - int64_t iter_limit = std::numeric_limits::max()); + simplex::dual_status_t solve_node_lp( + mip_node_t* node_ptr, + const simplex::simplex_solver_settings_t& settings, + branch_and_bound_worker_t* worker, + branch_and_bound_stats_t& stats, + simplex::logger_t& log, + int64_t iter_limit = std::numeric_limits::max()); // Apply symmetry-based bound reductions (orbital fixing and, when // settings_.symmetry == 2, lexical reduction) to the current node. From 8c939857afba0163de2f7dbb0a7854dd6ab93328 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Wed, 9 Sep 2026 16:24:52 -0700 Subject: [PATCH 037/113] Prefer large pivots with Harris ratio buckets This improvement was discovered through Hiverge's automated exploration of changes to cuOpt's dual simplex solver and then isolated on top of the v9 row-equilibration and perturbation changes. The bound-flipping ratio test searches Harris buckets from the latest to the earliest so that it favors a longer dual step. Within the selected bucket, choose the candidate with the largest absolute pivot instead of the largest exact breakpoint ratio. Use the exact ratio only to break ties between equal pivots. The change recovers ex9 and neos-3988577-wolgan, which time out in v9, but introduces a timeout on irish-electricity after failed primal cleanup. The larger pivots reduce total BFRT zero steps by 22.8% and improve the aggregate benchmark despite that cleanup regression. Problem v9 v10 Baseline HiGHS v10/HiGHS ------------------------------------------------------------------------------------- var-smallemery-m6j6 0.69 0.69 0.66 300.00 0.00 momentum1 0.69 0.70 0.69 300.00 0.00 neos-5114902-kasavu 2.74 2.73 87.40 300.00 0.01 supportcase42 0.80 0.76 0.52 36.20 0.02 neos-5049753-cuanza 1.02 1.05 7.68 29.85 0.04 supportcase12 6.35 4.11 4.68 37.11 0.11 roi5alpha10n8 1.35 1.32 1.25 11.90 0.11 mzzv11 2.40 2.12 40.19 16.71 0.13 ns1760995 114.33 44.10 135.94 269.53 0.16 ns1952667 0.26 0.15 8.95 0.80 0.19 proteindesign121hz512p9 0.46 0.40 0.91 2.04 0.20 roi2alpha3n4 0.22 0.24 0.31 1.16 0.21 co-100 0.28 0.27 0.68 1.28 0.21 neos-5104907-jarama 21.53 19.35 124.69 89.64 0.22 neos-5052403-cygnet 109.39 72.96 300.00 300.00 0.24 proteindesign122trx11p8 0.35 0.33 0.64 1.26 0.26 neos-1354092 20.05 20.09 300.00 70.92 0.28 supportcase18 0.03 0.04 0.06 0.12 0.33 rd-rplusc-21 0.16 0.17 0.23 0.49 0.35 30n20b8 0.07 0.04 0.08 0.11 0.36 ns1644855 63.39 88.00 300.00 238.30 0.37 neos-787933 0.06 0.07 0.06 0.18 0.39 sct2 0.06 0.06 0.22 0.14 0.43 supportcase7 1.68 1.53 1.29 3.52 0.43 wachplan 0.18 0.12 0.25 0.26 0.46 neos-860300 0.06 0.07 0.10 0.15 0.47 rocII-5-11 0.10 0.10 0.09 0.21 0.48 neos-5093327-huahum 0.23 0.23 0.23 0.48 0.48 supportcase22 2.28 1.92 2.03 3.97 0.48 physiciansched6-2 3.60 3.25 11.15 6.72 0.48 neos-5107597-kakapo 0.06 0.05 0.04 0.10 0.50 cvs16r128-89 1.07 0.88 0.93 1.72 0.51 satellites2-40 5.40 5.56 29.70 10.51 0.53 neos-4647030-tutaki 1.68 1.69 2.47 3.09 0.55 lectsched-5-obj 0.11 0.10 0.13 0.18 0.56 supportcase10 79.85 66.38 300.00 113.55 0.58 buildingenergy 87.82 68.31 300.00 115.61 0.59 n3div36 0.10 0.11 0.11 0.18 0.61 neos-5188808-nattai 0.18 0.19 0.30 0.29 0.66 tbfp-network 7.72 6.01 9.06 9.04 0.66 neos-5195221-niemur 0.26 0.25 0.50 0.37 0.68 thor50dday 0.26 0.25 0.27 0.37 0.68 nursesched-medium-hint03 3.71 3.23 10.47 4.73 0.68 square47 93.64 87.67 79.87 126.92 0.69 cryptanalysiskb128n5obj14 9.55 8.73 29.78 12.54 0.70 ns1116954 9.20 7.52 156.23 10.73 0.70 neos-1171448 0.58 0.60 2.35 0.84 0.71 academictimetablesmall 0.70 0.59 14.89 0.82 0.72 neos-3402454-bohle 64.41 55.07 221.86 74.15 0.74 neos-2746589-doon 2.36 2.25 7.53 3.02 0.75 neos-4300652-rahue 0.46 0.60 1.22 0.80 0.75 neos-3004026-krka 0.07 0.09 0.06 0.12 0.75 neos-3381206-awhea 0.03 0.03 0.08 0.04 0.75 square41 38.25 31.75 28.57 39.86 0.80 blp-ic98 0.08 0.08 0.12 0.10 0.80 dws008-01 0.05 0.04 0.04 0.05 0.80 supportcase33 0.40 0.44 0.99 0.55 0.80 neos-848589 0.70 0.68 1.24 0.85 0.80 comp21-2idx 0.31 0.28 1.55 0.35 0.80 cod105 8.39 6.13 9.06 7.46 0.82 decomp2 0.08 0.10 0.19 0.12 0.83 cryptanalysiskb128n5obj16 10.05 8.70 29.53 10.27 0.85 fiball 0.15 0.12 0.71 0.14 0.86 blp-ar98 0.07 0.10 0.10 0.11 0.91 dano3_3 16.02 17.37 46.88 19.06 0.91 dano3_5 16.04 17.42 46.79 19.03 0.92 neos-3555904-turama 1.31 1.27 1.31 1.37 0.93 neos-3988577-wolgan 300.00 281.89 278.89 300.00 0.94 neos-1171737 0.17 0.19 0.70 0.20 0.95 drayage-25-23 0.05 0.04 0.08 0.04 1.00 h80x6320d 0.04 0.04 0.05 0.04 1.00 highschool1-aigio 300.00 300.00 300.00 300.00 1.00 icir97_tension 0.02 0.02 0.03 0.02 1.00 leo1 0.07 0.07 0.10 0.07 1.00 leo2 0.12 0.13 0.13 0.13 1.00 neos-1456979 0.05 0.04 0.05 0.04 1.00 neos8 0.26 0.26 0.38 0.26 1.00 nursesched-sprint02 0.23 0.23 0.38 0.23 1.00 physiciansched3-3 300.00 300.00 300.00 300.00 1.00 radiationm18-12-05 0.13 0.14 0.23 0.14 1.00 rail02 300.00 300.00 300.00 300.00 1.00 s100 300.00 300.00 300.00 300.00 1.00 savsched1 300.00 300.00 300.00 300.00 1.00 supportcase19 300.00 300.00 300.00 300.00 1.00 swath3 0.02 0.03 0.03 0.03 1.00 traininstance6 0.04 0.03 0.04 0.03 1.00 neos-873061 1.46 1.43 1.46 1.38 1.04 neos-957323 12.14 8.37 300.00 7.74 1.08 mzzv42z 0.91 1.03 10.10 0.95 1.08 neos-3402294-bobin 1.29 1.33 3.07 1.20 1.11 germanrr 0.30 0.30 0.31 0.27 1.11 s250r10 71.91 81.59 300.00 71.74 1.14 supportcase6 4.93 4.83 7.39 4.18 1.16 neos-1582420 0.08 0.07 0.12 0.06 1.17 supportcase40 0.29 0.28 0.24 0.24 1.17 neos-4763324-toguru 5.82 5.85 8.32 5.00 1.17 neos-1122047 1.87 1.90 2.09 1.61 1.18 sp98ar 0.34 0.37 0.39 0.31 1.19 neos-4413714-turia 2.10 2.33 3.97 1.93 1.21 rail507 4.37 3.52 7.17 2.84 1.24 drayage-100-23 0.04 0.05 0.07 0.04 1.25 neos-4738912-atrato 0.04 0.05 0.05 0.04 1.25 traininstance2 0.04 0.05 0.09 0.04 1.25 nexp-150-20-8-5 0.10 0.09 0.10 0.07 1.29 fast0507 3.95 3.58 7.36 2.77 1.29 sp97ar 0.44 0.43 0.40 0.33 1.30 irp 0.12 0.12 0.11 0.09 1.33 swath1 0.03 0.04 0.04 0.03 1.33 radiationm40-10-02 0.97 0.95 1.58 0.71 1.34 cmflsp50-24-8-8 0.58 0.59 0.77 0.44 1.34 map16715-04 9.87 9.21 13.26 6.80 1.35 comp07-2idx 1.26 1.51 4.03 1.10 1.37 map10 8.54 8.67 11.08 6.25 1.39 cbs-cta 0.14 0.14 0.42 0.10 1.40 neos-2978193-inde 0.09 0.07 0.16 0.05 1.40 trento1 3.50 3.10 3.08 2.21 1.40 hypothyroid-k1 4.42 4.40 4.36 3.01 1.46 uccase9 8.04 7.66 11.54 5.20 1.47 qap10 9.80 9.87 15.57 6.68 1.48 neos-827175 0.44 0.43 9.36 0.29 1.48 neos-2987310-joes 1.59 1.49 1.52 1.00 1.49 bnatt500 0.20 0.21 0.29 0.14 1.50 mushroom-best 0.28 0.30 0.26 0.20 1.50 neos-960392 3.29 3.99 8.93 2.66 1.50 air05 0.29 0.27 0.28 0.18 1.50 reblock115 0.14 0.14 0.16 0.09 1.56 neos-4532248-waihi 1.40 1.43 2.61 0.90 1.59 uct-subprob 0.12 0.13 0.11 0.08 1.62 ns1830653 0.25 0.23 0.42 0.14 1.64 fhnw-binpack4-48 0.09 0.10 0.07 0.06 1.67 neos-3083819-nubu 0.05 0.05 0.06 0.03 1.67 ran14x18-disj-8 0.05 0.05 0.04 0.03 1.67 rocI-4-11 0.09 0.10 0.11 0.06 1.67 roll3000 0.10 0.10 0.12 0.06 1.67 opm2-z10-s4 75.72 75.53 89.23 44.75 1.69 istanbul-no-cutoff 1.63 1.62 0.71 0.94 1.72 neos-3656078-kumeu 0.40 0.40 2.37 0.23 1.74 neos-662469 0.86 0.93 1.65 0.53 1.75 k1mushroom 28.33 29.61 31.00 16.80 1.76 rmatr200-p5 8.10 8.20 7.50 4.61 1.78 sing326 9.33 8.37 9.54 4.69 1.78 neos-933966 9.15 5.17 16.79 2.80 1.85 rmatr100-p10 0.33 0.26 0.27 0.14 1.86 bnatt400 0.13 0.15 0.16 0.08 1.88 mcsched 0.30 0.30 0.27 0.16 1.88 atlanta-ip 12.20 9.04 6.85 4.54 1.99 assign1-5-8 0.02 0.02 0.03 0.01 2.00 b1c1s1 0.06 0.04 0.05 0.02 2.00 bppc4-08 0.04 0.04 0.08 0.02 2.00 eil33-2 0.06 0.06 0.05 0.03 2.00 fhnw-binpack4-4 0.03 0.02 0.03 0.01 2.00 ic97_potential 0.02 0.02 0.02 0.01 2.00 mik-250-20-75-4 0.02 0.02 0.02 0.01 2.00 neos-3024952-loue 0.32 0.40 0.41 0.20 2.00 neos-4954672-berkel 0.02 0.02 0.02 0.01 2.00 pg5_34 0.03 0.02 0.03 0.01 2.00 tr12-30 0.02 0.02 0.03 0.01 2.00 graph20-20-1rand 0.30 0.25 0.23 0.12 2.08 neos-3216931-puriri 4.20 6.83 7.33 3.23 2.11 ns1208400 0.76 0.78 3.96 0.36 2.17 chromaticindex512-7 53.03 46.15 16.62 21.24 2.17 splice1k1 19.92 20.18 21.62 9.17 2.20 eilA101-2 3.29 3.08 2.68 1.39 2.22 nw04 1.71 1.36 0.44 0.61 2.23 sing44 12.54 12.80 9.62 5.69 2.25 gfd-schedulen180f7d50m30k18 15.81 15.35 80.00 6.81 2.25 n2seq36q 0.71 0.56 0.46 0.24 2.33 triptim1 132.50 121.77 71.42 51.43 2.37 netdiversion 22.67 22.84 7.64 9.43 2.42 piperout-08 0.41 0.39 0.39 0.16 2.44 chromaticindex1024-7 262.17 232.72 45.86 93.65 2.48 satellites2-60-fs 6.89 8.34 4.17 3.34 2.50 neos-4387871-tavua 0.08 0.10 0.10 0.04 2.50 neos-950242 0.46 0.45 1.05 0.18 2.50 nu25-pr12 0.04 0.05 0.05 0.02 2.50 rococoB10-011000 0.15 0.15 0.13 0.06 2.50 rococoC10-001000 0.06 0.05 0.04 0.02 2.50 glass-sc 0.33 0.31 0.31 0.12 2.58 seymour1 0.97 1.00 0.83 0.38 2.63 seymour 0.95 1.01 0.83 0.38 2.66 unitcal_7 1.05 1.04 0.96 0.39 2.67 sorrell3 1.14 1.10 1.07 0.40 2.75 bab2 75.19 80.63 300.00 28.97 2.78 bab6 34.02 40.19 143.43 14.06 2.86 neos-1445765 0.25 0.26 0.16 0.09 2.89 net12 0.96 1.03 0.55 0.35 2.94 50v-10 0.03 0.03 0.02 0.01 3.00 binkar10_1 0.03 0.03 0.02 0.01 3.00 cost266-UUE 0.06 0.06 0.03 0.02 3.00 gmu-35-40 0.03 0.03 0.03 0.01 3.00 gmu-35-50 0.04 0.03 0.05 0.01 3.00 lotsize 0.03 0.03 0.03 0.01 3.00 n5-3 0.05 0.06 0.03 0.02 3.00 neos-4338804-snowy 0.02 0.03 0.03 0.01 3.00 pg 0.02 0.03 0.03 0.01 3.00 csched007 0.16 0.17 0.19 0.05 3.40 peg-solitaire-a3 2.14 1.78 1.88 0.52 3.42 milo-v12-6-r2-40-1 0.18 0.18 0.25 0.05 3.60 uccase12 5.76 5.36 72.32 1.38 3.88 app1-1 0.10 0.08 0.08 0.02 4.00 csched008 0.13 0.12 0.11 0.03 4.00 neos-3627168-kasai 0.04 0.04 0.04 0.01 4.00 neos17 0.03 0.04 0.03 0.01 4.00 p200x1188c 0.04 0.04 0.03 0.01 4.00 rail01 300.00 300.00 223.84 71.23 4.21 CMS750_4 0.63 0.63 0.38 0.14 4.50 piperout-27 1.23 1.26 0.67 0.26 4.85 neos-631710 215.36 155.73 300.00 31.97 4.87 neos-4722843-widden 7.67 7.75 1.17 1.54 5.03 irish-electricity 109.17 300.00 181.58 59.53 5.04 ex10 300.00 300.00 300.00 59.25 5.06 fastxgemm-n2r6s0t2 0.40 0.47 0.18 0.08 5.87 neos-2075418-temuka 300.00 300.00 128.30 50.57 5.93 beasleyC3 0.06 0.06 0.05 0.01 6.00 snp-02-004-104 24.10 21.83 14.17 2.83 7.71 app1-2 5.77 5.58 5.03 0.70 7.97 mc11 0.08 0.08 0.06 0.01 8.00 brazil3 61.89 63.40 300.00 7.24 8.76 ex9 300.00 132.66 300.00 14.07 9.43 enlight_hard 0.02 0.01 0.02 0.00 10.00 gen-ip054 0.02 0.01 0.03 0.00 10.00 pk1 0.01 0.01 0.02 0.00 10.00 exp-1-500-5-5 0.02 0.02 0.02 0.00 20.00 gen-ip002 0.01 0.02 0.02 0.00 20.00 glass4 0.02 0.02 0.02 0.00 20.00 graphdraw-domain 0.02 0.02 0.02 0.00 20.00 mad 0.02 0.02 0.02 0.00 20.00 markshare2 0.01 0.02 0.02 0.00 20.00 markshare_4_0 0.01 0.02 0.02 0.00 20.00 mas74 0.01 0.02 0.03 0.00 20.00 mas76 0.02 0.02 0.02 0.00 20.00 neos-2657525-crna 0.03 0.02 0.03 0.00 20.00 neos-3046615-murg 0.02 0.02 0.02 0.00 20.00 neos-911970 0.03 0.02 0.03 0.00 20.00 neos5 0.02 0.02 0.02 0.00 20.00 neos859080 0.01 0.02 0.01 0.00 20.00 supportcase26 0.02 0.02 0.03 0.00 20.00 timtab1 0.02 0.02 0.01 0.00 20.00 neos-3754480-nidda 0.02 0.03 0.02 0.00 30.00 sp150x300d 0.02 0.03 0.02 0.00 30.00 Geomean v9/v10: 1.0157 Shifted(+1s): 1.0209 Geomean Baseline/v10: 1.4393 Shifted(+1s): 1.2823 Geomean v10/HiGHS: 1.5021 Shifted(+1s): 0.9925 (240 problems) --- .../mathematical_optimization/constants.h | 6 +- .../pdlp/solver_settings.hpp | 6 + cpp/src/dual_simplex/basis_updates.cpp | 36 +- cpp/src/dual_simplex/basis_updates.hpp | 8 + .../bound_flipping_ratio_test.cpp | 541 ++++--- .../bound_flipping_ratio_test.hpp | 49 +- cpp/src/dual_simplex/crossover.cpp | 68 +- cpp/src/dual_simplex/phase2.cpp | 1287 +++++++++++----- cpp/src/dual_simplex/phase2.hpp | 66 +- cpp/src/dual_simplex/primal.cpp | 1326 ++++++++++++++--- cpp/src/dual_simplex/primal.hpp | 47 +- cpp/src/dual_simplex/right_looking_lu.cpp | 21 +- cpp/src/dual_simplex/scaling.cpp | 33 + .../dual_simplex/simplex_solver_settings.hpp | 16 +- cpp/src/dual_simplex/solve.cpp | 180 ++- cpp/src/dual_simplex/solve.hpp | 56 +- cpp/src/math_optimization/solver_settings.cu | 3 + cpp/src/pdlp/solve.cu | 78 +- .../solver_settings/solver_settings.pyx | 3 +- 19 files changed, 3009 insertions(+), 821 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/constants.h b/cpp/include/cuopt/mathematical_optimization/constants.h index 64da263a4d..c76867d0c0 100644 --- a/cpp/include/cuopt/mathematical_optimization/constants.h +++ b/cpp/include/cuopt/mathematical_optimization/constants.h @@ -57,6 +57,9 @@ #define CUOPT_ELIMINATE_DENSE_COLUMNS "eliminate_dense_columns" #define CUOPT_CUDSS_DETERMINISTIC "cudss_deterministic" #define CUOPT_PRESOLVE "presolve" +#define CUOPT_INITIAL_PERTURBATION "initial_perturbation" +#define CUOPT_REMOVE_PERTURBATION "remove_perturbation" +#define CUOPT_PRIMAL_PRICING "primal_pricing" #define CUOPT_MIP_PROBING "mip_probing" #define CUOPT_DUAL_POSTSOLVE "dual_postsolve" #define CUOPT_MIP_DETERMINISM_MODE "mip_determinism_mode" @@ -211,7 +214,8 @@ #define CUOPT_METHOD_PDLP 1 #define CUOPT_METHOD_DUAL_SIMPLEX 2 #define CUOPT_METHOD_BARRIER 3 -#define CUOPT_METHOD_UNSET 4 +#define CUOPT_METHOD_PRIMAL 4 +#define CUOPT_METHOD_UNSET 5 #define CUOPT_BARRIER_DUAL_INITIAL_POINT_AUTOMATIC -1 #define CUOPT_BARRIER_DUAL_INITIAL_POINT_LUSTIG_MARSTEN_SHANNO 0 diff --git a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp index 55a3359795..1c4d0ce71b 100644 --- a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp @@ -61,6 +61,7 @@ enum pdlp_solver_mode_t : int { * PDLP: Use the PDLP method. * DualSimplex: Use the dual simplex method. * Barrier: Use the barrier method + * Primal: Use the (experimental) primal simplex method. * Unset: The value was not set. * * @note Default method is Concurrent. @@ -70,6 +71,7 @@ enum method_t : int { PDLP = CUOPT_METHOD_PDLP, DualSimplex = CUOPT_METHOD_DUAL_SIMPLEX, Barrier = CUOPT_METHOD_BARRIER, + Primal = CUOPT_METHOD_PRIMAL, Unset = CUOPT_METHOD_UNSET }; @@ -81,6 +83,7 @@ inline std::string method_to_string(method_t method) case method_t::PDLP: return "PDLP"; case method_t::Barrier: return "Barrier"; case method_t::Concurrent: return "Concurrent"; + case method_t::Primal: return "Primal Simplex"; default: return "Unset"; } } @@ -299,6 +302,9 @@ class pdlp_solver_settings_t { i_t augmented{-1}; i_t dualize{-1}; i_t ordering{-1}; + i_t initial_perturbation{-1}; + i_t remove_perturbation{-1}; + i_t primal_pricing{0}; barrier_dual_initial_point_t barrier_dual_initial_point{barrier_dual_initial_point_t::Automatic}; i_t postsolve_info{-1}; i_t barrier_presolve_bound_free_variables{-1}; // -1 automatic, 0 disabled, 1 enabled diff --git a/cpp/src/dual_simplex/basis_updates.cpp b/cpp/src/dual_simplex/basis_updates.cpp index f81962d054..c2a7027548 100644 --- a/cpp/src/dual_simplex/basis_updates.cpp +++ b/cpp/src/dual_simplex/basis_updates.cpp @@ -1507,7 +1507,7 @@ f_t basis_update_mpf_t::dot_product(i_t col, nz_mark++; } } - work_estimate_ += 2 * nz_mark + (col_end - col_start); + work_estimate_ += 2 * (col_end - col_start) + 2 * nz_mark; return dot; } @@ -1524,7 +1524,7 @@ void basis_update_mpf_t::add_sparse_column(const csc_matrix_t @@ -1549,7 +1549,7 @@ void basis_update_mpf_t::add_sparse_column(const csc_matrix_t @@ -2009,6 +2009,34 @@ i_t basis_update_mpf_t::u_solve(sparse_vector_t& rhs) const return 0; } + +// Compute y = U*x. In the MPF factorization, the rank-1 update factors are absorbed into L, so +// U == U0 and U*x reduces to a sparse matvec against U0. +template +void basis_update_mpf_t::u_multiply(const std::vector& x, std::vector& y) const +{ + const i_t m = L0_.m; + y.assign(m, 0.0); + matrix_vector_multiply(U0_, f_t(1.0), x, f_t(0.0), y); + work_estimate_ += 2 * U0_.col_start[U0_.n]; +} + +// Sparse-in/sparse-out overload of u_multiply. Same semantics as the dense version. +template +void basis_update_mpf_t::u_multiply(const sparse_vector_t& x, + sparse_vector_t& y) const +{ + const i_t m = L0_.m; + // Scatter x into a dense workspace, compute U0 * x, gather back to sparse. + std::vector x_dense; + x.to_dense(x_dense); + std::vector y_dense(m, 0.0); + matrix_vector_multiply(U0_, f_t(1.0), x_dense, f_t(0.0), y_dense); + work_estimate_ += 2 * U0_.col_start[U0_.n]; + y.from_dense(y_dense); + work_estimate_ += m; +} + // Solve for x such that L*x = y template i_t basis_update_mpf_t::l_solve(std::vector& rhs) const @@ -2202,7 +2230,7 @@ i_t basis_update_mpf_t::update(const sparse_vector_t& utilde // Ensure the workspace is sorted. Otherwise, the sparse dot will be incorrect. std::sort(xi_workspace_.begin() + m, xi_workspace_.begin() + m + nz, std::less()); - work_estimate_ += (m + nz) * std::log2(m + nz); + work_estimate_ += nz > 1 ? nz * std::log2(nz) : 0; // Gather the workspace into a column of S i_t S_start; diff --git a/cpp/src/dual_simplex/basis_updates.hpp b/cpp/src/dual_simplex/basis_updates.hpp index d1c623db55..bdedcc4a18 100644 --- a/cpp/src/dual_simplex/basis_updates.hpp +++ b/cpp/src/dual_simplex/basis_updates.hpp @@ -353,6 +353,14 @@ class basis_update_mpf_t { // Solve for x such that U'*x = y i_t u_transpose_solve(sparse_vector_t& rhs) const; + // Compute y = U*x. In the MPF factorization the rank-1 update factors are absorbed into L, so + // U is unchanged from the initial factorization (U == U0), and U*x is just a sparse matvec + // against U0. + void u_multiply(const std::vector& x, std::vector& y) const; + + // Sparse-in/sparse-out overload of u_multiply. + void u_multiply(const sparse_vector_t& x, sparse_vector_t& y) const; + // Replace the column B(:, leaving_index) with the vector abar. Pass in utilde such that L*utilde // = abar i_t update(const std::vector& utilde, const std::vector& etilde, i_t leaving_index); diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index cb0964dc05..d18ed95e90 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -11,12 +11,14 @@ #include #include +#include namespace cuopt::mathematical_optimization::simplex { template i_t bound_flipping_ratio_test_t::compute_breakpoints(std::vector& indicies, - std::vector& ratios) + std::vector& ratios, + std::vector& harris_ratios) { i_t n = n_; i_t m = m_; @@ -33,20 +35,21 @@ i_t bound_flipping_ratio_test_t::compute_breakpoints(std::vector& const i_t k = nonbasic_mark_[j]; if (vstatus_[j] == variable_status_t::NONBASIC_FIXED) { continue; } if (vstatus_[j] == variable_status_t::NONBASIC_LOWER && delta_z_[j] < -pivot_tol) { - indicies[idx] = k; - ratios[idx] = std::max((-dual_tol - z_[j]) / delta_z_[j], 0.0); + indicies[idx] = k; + ratios[idx] = std::max((-z_[j]) / delta_z_[j], 0.0); + harris_ratios[idx] = std::max((-dual_tol - z_[j]) / delta_z_[j], 0.0); if constexpr (verbose) { settings_.log.printf("ratios[%d] = %e\n", idx, ratios[idx]); } idx++; } if (vstatus_[j] == variable_status_t::NONBASIC_UPPER && delta_z_[j] > pivot_tol) { - indicies[idx] = k; - ratios[idx] = std::max((dual_tol - z_[j]) / delta_z_[j], 0.0); + indicies[idx] = k; + ratios[idx] = std::max((-z_[j]) / delta_z_[j], 0.0); + harris_ratios[idx] = std::max((dual_tol - z_[j]) / delta_z_[j], 0.0); if constexpr (verbose) { settings_.log.printf("ratios[%d] = %e\n", idx, ratios[idx]); } idx++; } } - work_estimate_ += 4 * nz; - work_estimate_ += 4 * idx; + work_estimate_ += 5 * nz + 5 * idx; pivot_tol /= 10; } return idx; @@ -57,10 +60,10 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, i_t end, const std::vector& indicies, const std::vector& ratios, - f_t& slope, f_t& step_length, i_t& nonbasic_entering, - i_t& entering_index) + i_t& entering_index, + f_t& max_val) { // Find the minimum ratio f_t min_val = inf; @@ -68,27 +71,19 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, i_t candidate = -1; f_t zero_tol = settings_.zero_tol; i_t k_idx = -1; + max_val = 0.0; - i_t min_found = 0; - i_t harris_found = 0; + i_t min_found = 0; for (i_t k = start; k < end; ++k) { if (ratios[k] < min_val) { min_val = ratios[k]; candidate = indicies[k]; k_idx = k; min_found++; - } else if (ratios[k] < min_val + zero_tol) { - // Use Harris to select variables with larger pivots - const i_t j = nonbasic_list_[indicies[k]]; - if (std::abs(delta_z_[j]) > std::abs(delta_z_[candidate])) { - min_val = ratios[k]; - candidate = indicies[k]; - k_idx = k; - } - harris_found++; } + if (ratios[k] > max_val) { max_val = ratios[k]; } } - work_estimate_ += (end - start) + 2 * min_found + 6 * harris_found; + work_estimate_ += (end - start) + 2 * min_found; step_length = min_val; nonbasic_entering = candidate; @@ -96,37 +91,65 @@ i_t bound_flipping_ratio_test_t::single_pass(i_t start, if (nonbasic_entering == -1) { return RATIO_TEST_NUMERICAL_ISSUES; } const i_t j = entering_index = nonbasic_list_[nonbasic_entering]; - constexpr bool verbose = false; - if (bounded_variables_[j]) { - const f_t interval = upper_[j] - lower_[j]; - const f_t delta_slope = std::abs(delta_z_[j]) * interval; - if constexpr (verbose) { - settings_.log.printf("single pass delta slope %e slope %e after slope %e step length %e\n", - delta_slope, - slope, - slope - delta_slope, - step_length); + if (bounded_variables_[j]) { return k_idx; } + return -1; // we are done. do not increase the step-length further +} + +template +void bound_flipping_ratio_test_t::determine_flips(f_t step_length, + i_t entering_index, + std::vector& flip_indices) +{ + // The piecewise-linear model below assumes that a variable flips bounds as soon as + // its reduced cost crosses zero. In practice, small changes between iterations can + // make a reduced cost oscillate around zero, causing excessive bound flips and + // cycling. We therefore flip only after the violation exceeds dual_tol / 10. + // A bounded variable l_j <= x_j <= u_j contributes l_j*z_j to the dual objective + // when z_j >= 0 and u_j*z_j when z_j < 0. If x_j = l_j and + // -dual_tol/10 <= z_j < 0, the model uses u_j*z_j while the unflipped state uses + // l_j*z_j. Their difference is (u_j - l_j)*|z_j|, bounded by + // (u_j - l_j)*dual_tol/10. For multiple unflipped variables, the discrepancy is + // bounded by sum_j (u_j - l_j)*dual_tol/10. + const f_t flip_tol = settings_.dual_tol / 10; + for (const i_t j : delta_z_indices_) { + if (j == entering_index || !bounded_variables_[j]) { continue; } + const f_t new_z = z_[j] + step_length * delta_z_[j]; + if ((vstatus_[j] == variable_status_t::NONBASIC_LOWER && new_z < -flip_tol) || + (vstatus_[j] == variable_status_t::NONBASIC_UPPER && new_z > flip_tol)) { + flip_indices.push_back(j); } - slope -= delta_slope; - return k_idx; // we should see if we can continue to increase the step-length } - return -1; // we are done. do not increase the step-length further + work_estimate_ += 5 * delta_z_indices_.size() + flip_indices.size(); } template i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, - i_t& nonbasic_entering) + i_t& nonbasic_entering, + std::vector& flip_indices) { const i_t m = m_; const i_t n = n_; const i_t nz = delta_z_indices_.size(); constexpr bool verbose = false; + flip_indices.clear(); // Compute the initial set of breakpoints std::vector indicies(nz); std::vector ratios(nz); - work_estimate_ += 2 * nz; - i_t num_breakpoints = compute_breakpoints(indicies, ratios); + std::vector harris_ratios(nz); + work_estimate_ += 3 * nz; + double t0 = tic(); + i_t num_breakpoints = compute_breakpoints(indicies, ratios, harris_ratios); + time_compute_breakpoints_ += toc(t0); + num_breakpoints_ = num_breakpoints; + // Count zero ratios + num_harris_zero_ = 0; + num_exact_zero_ = 0; + for (i_t k = 0; k < num_breakpoints; k++) { + if (harris_ratios[k] == 0.0) num_harris_zero_++; + if (ratios[k] == 0.0) num_exact_zero_++; + } + work_estimate_ += 2 * num_breakpoints; if constexpr (verbose) { settings_.log.printf("Initial breakpoints %d\n", num_breakpoints); } if (num_breakpoints == 0) { nonbasic_entering = -1; @@ -136,11 +159,23 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, f_t slope = slope_; nonbasic_entering = -1; i_t entering_index = RATIO_TEST_NO_ENTERING_VARIABLE; - - i_t k_idx = single_pass( - 0, num_breakpoints, indicies, ratios, slope, step_length, nonbasic_entering, entering_index); + f_t max_step_length; + + t0 = tic(); + i_t k_idx = single_pass(0, + num_breakpoints, + indicies, + harris_ratios, + step_length, + nonbasic_entering, + entering_index, + max_step_length); + time_single_pass_ += toc(t0); if (k_idx == RATIO_TEST_NUMERICAL_ISSUES) { return RATIO_TEST_NUMERICAL_ISSUES; } - bool continue_search = k_idx >= 0 && num_breakpoints > 1 && slope > 0.0; + // The variable selected by single_pass is guaranteed to be in the first bucket: it + // defines the minimum Harris ratio, and its exact ratio is no greater than its Harris + // ratio. Its slope contribution is therefore applied by the bucket pass below. + bool continue_search = k_idx >= 0 && num_breakpoints > 1; if (!continue_search) { if constexpr (verbose) { settings_.log.printf( @@ -150,6 +185,9 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, entering_index, std::abs(delta_z_[entering_index])); } + num_buckets_used_ = 0; + step_length_result_ = step_length; + determine_flips(step_length, entering_index, flip_indices); return entering_index; } @@ -162,185 +200,280 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, slope); } - // Continue the search using a heap to order the breakpoints - ratios[k_idx] = ratios[num_breakpoints - 1]; - indicies[k_idx] = indicies[num_breakpoints - 1]; - - constexpr bool use_bucket_pass = false; - - if (use_bucket_pass) { - f_t max_ratio = 0.0; - for (i_t k = 0; k < num_breakpoints - 1; ++k) { - if (ratios[k] > max_ratio) { max_ratio = ratios[k]; } + // This code is complicated. There are several important concepts that are needed to understand + // it. + // + // We are trying to compute the maximum step length we can take while: + // 1) Staying mostly dual feasible (we allow ourselves to be infeasible by dual_tol amount) + // 2) Increasing the dual objective + // 3) Selecting a variable with a large pivot (| delta_z[j] |) + // + // Let alpha be the step length. For each nonbasic variable j, we have + // z_j(alpha) = z_j + alpha * delta_z_j + // + // To stay dual feasible, we either need to keep + // z_j(alpha) >= 0, if j is on it's lower bound, or + // z_j(alpha) <= 0, if j is on it's upper bound. + // + // Consider the equation z_j(alpha) = z_j + alpha * delta_z_j = 0. Each variable j puts a bound on + // alpha: + // + // alpha_j <= -z_j / delta_z_j, if x_j = l_j (z_j >= 0) and delta_z_j < 0 + // alpha_j <= -z_j / delta_z_j, if x_j = u_j (z_j <= 0) and delta_z_j > 0 + // + // The code refers to these alpha_j as ratios, since they are the ratio of z_j to delta_z_j. + // + // Now we could take alpha = min_j alpha_j, and remain dual feasible. However, we are allowed to + // increase the step-length if j is a variable such that l_j <= x_j <= u_j. To see why imagine + // that our variable was currenlty on it's lower bound, with z_j > 0 and delta_z_j < 0, if we push + // alpha past alpha_j, than z_j(alpha) < 0. This is fine as long as we flip the variable to be on + // it's upper bound. Thus, we can push alpha past alpha_j for *bounded* variables. + // + // Note that this does not work if we try to increase alpha past alpha_j for a variable with a + // single bound. We would just be making ourselves dual infeasible. So we need to check whether a + // variable is bounded. + // + // The dual objective as a function of the step-length alpha, is piecewise linear and concave. The + // breakpoints of this piecewise linear function occur at each of the alpha_j values. We can keep + // increasing the step-length as long as the slope remains nonnegative. After that we must stop, + // because we could decrease the dual objective. So the code tracks the cumulative slope of the + // dual objective. + // + // Now we don't need to exactly feasible: z_j >= 0 if x_j = l_j and z_j <= 0 if x_j = u_j. We can + // violate these bounds by the dual feasibility tolerance eps. We allow ourselves to be infeasible + // if it would help us get a larger pivot (delta_z_j). Small pivots can cause numerical issues, so + // we would like to avoid them. + // + // With this tolerance we get the equations: + // z_j(alpha) = z_j + alpha * delta_z_j >= -eps if x_j = l_j + // z_j(alpha) = z_j + alpha * delta_z_j <= eps if x_j = u_j + // + // This gives bounds on alpha. We call these alpha_harris_j, for Paula Harris, who proposed this + // method. + // + // alpha_harris_j <= (-eps - z_j) / delta_z_j, if x_j = l_j and delta_z_j < 0 + // alpha_harris_j <= (eps - z_j) / delta_z_j, if x_j = u_j and delta_z_j > 0 + // + // Let alpha_harris = min_j alpha_harris_j. We can select the variable with the largest | + // delta_z_j | from those candidates { j | alpha_j <= alpha_harris }. + // + // We combine these two ideas (increasing the step length for bounded variables) and allowing + // ourselves to be slightly dual infeasible to choose a larger pivot. + // + // We partition the variables into buckets. Let B_k be the set of variables in bucket k. B_0 is + // defined as { j | alpha_j <= alpha_harris }. We then compute alpha_harris_1 = min_{j not in B_0} + // alpha_j. And B_1 is defined as { j not in B_0 | alpha_j <= alpha_harris_1 }. And so on. + // + // We want to balance two different things: + // 1) Taking a larger step length to increase the dual objective as much as possible, + // 2) Choosing a large pivot for numerical stability. + // + // Let max_pivot = max_j | delta_z_j |. We start working our way backward from the largest bucket + // to the smallest bucket, we choose a variable j that satisfies | delta_z_j | >= 0.1 * max_pivot. + // Since we can always choose a smaller step length for the sake of numerical stability. + // + // Now the final thing to understand is that the ratios alpha_j are not sorted in any particular + // order. And we don't want to pay the O(num_breakpoints * log(num_breakpoints)) cost of sorting + // them. + // + // So we set a threshold on the step-length and check if all variables j with alpha_j <= threshold + // have already caused the slope to go negative. If so, we just need to consider those candidate + // variables with alpha_j <= threshold. If not, we multiply the threshold by 10. This cost us + // O(log10(max_step_length/min_step_length) * num_breakpoints) time. So we aren't totally linear. + // But the hope is we are better than a sort. + + // Use a coarse filter to find candidates + f_t minimum_harris_ratio = step_length; + f_t coarse_threshold = (minimum_harris_ratio > 0.0) + ? std::min(10.0 * minimum_harris_ratio, max_step_length) + : max_step_length; + f_t total_slope = slope; + bool found_unbounded = false; + std::vector candidates(num_breakpoints); + std::iota(candidates.begin(), candidates.end(), 0); + work_estimate_ += 2 * num_breakpoints; + i_t scan_start = 0; + i_t num_candidates = 0; + + // This is O( log10(max_step_length/min_step_length) * num_breakpoints) + t0 = tic(); + while (total_slope >= 0.0 && coarse_threshold <= max_step_length && + scan_start < num_breakpoints && !found_unbounded) { + for (i_t h = scan_start; h < num_breakpoints; ++h) { + const i_t k = candidates[h]; + if (ratios[k] <= coarse_threshold) { + // Candidate is less than coarse threshold, move it to the front of the candidate list + std::swap(candidates[h], candidates[num_candidates]); + num_candidates++; + const i_t j = nonbasic_list_[indicies[k]]; + if (!bounded_variables_[j]) { + found_unbounded = true; + } else { + total_slope -= std::abs(delta_z_[j]) * (upper_[j] - lower_[j]); + } + } } - work_estimate_ += 2 * num_breakpoints; - settings_.log.printf( - "Starting heap passes. %d breakpoints max ratio %e\n", num_breakpoints - 1, max_ratio); - bucket_pass( - indicies, ratios, num_breakpoints - 1, slope, step_length, nonbasic_entering, entering_index); + work_estimate_ += 2 * (num_breakpoints - scan_start) + 10 * (num_candidates - scan_start); + scan_start = num_candidates; + coarse_threshold *= 10.0; } + time_coarse_filter_ += toc(t0); - heap_passes( - indicies, ratios, num_breakpoints - 1, slope, step_length, nonbasic_entering, entering_index); + candidates.resize(num_candidates); - if constexpr (verbose) { - settings_.log.printf("BFRT step length %e entering index %d non basic entering %d pivot %e\n", - step_length, - entering_index, - nonbasic_entering, - std::abs(delta_z_[entering_index])); - } - return entering_index; -} - -template -void bound_flipping_ratio_test_t::heap_passes(const std::vector& current_indicies, - const std::vector& current_ratios, - i_t num_breakpoints, - f_t& slope, - f_t& step_length, - i_t& nonbasic_entering, - i_t& entering_index) -{ - std::vector bare_idx(num_breakpoints); - constexpr bool verbose = false; - const f_t dual_tol = settings_.dual_tol; - const f_t zero_tol = settings_.zero_tol; - const std::vector& delta_z = delta_z_; - const std::vector& nonbasic_list = nonbasic_list_; - const i_t N = num_breakpoints; - for (i_t k = 0; k < N; ++k) { - bare_idx[k] = k; - if constexpr (verbose) { - settings_.log.printf("Adding index %d ratio %e pivot %e to heap\n", - current_indicies[k], - current_ratios[k], - std::abs(delta_z[nonbasic_list[current_indicies[k]]])); + // Check for variables with one sided bounds. These define the maximum step length. + if (found_unbounded) { + for (i_t h = 0; h < num_candidates; h++) { + const i_t k = candidates[h]; + const i_t j = nonbasic_list_[indicies[k]]; + if (!bounded_variables_[j]) { max_step_length = std::min(max_step_length, harris_ratios[k]); } } - } - work_estimate_ += N; - - auto compare = [zero_tol, ¤t_ratios, ¤t_indicies, &delta_z, &nonbasic_list]( - const i_t& a, const i_t& b) { - return (current_ratios[a] > current_ratios[b]) || - (current_ratios[b] - current_ratios[a] < zero_tol && - std::abs(delta_z[nonbasic_list[current_indicies[a]]]) > - std::abs(delta_z[nonbasic_list[current_indicies[b]]])); - }; - - std::make_heap(bare_idx.begin(), bare_idx.end(), compare); - work_estimate_ += 3 * bare_idx.size(); - - while (bare_idx.size() > 0 && slope > 0) { - // Remove minimum ratio from the heap and rebalance - i_t heap_index = bare_idx.front(); - std::pop_heap(bare_idx.begin(), bare_idx.end(), compare); - work_estimate_ += 2 * std::log2(bare_idx.size()); - bare_idx.pop_back(); - - nonbasic_entering = current_indicies[heap_index]; - const i_t j = entering_index = nonbasic_list_[nonbasic_entering]; - step_length = current_ratios[heap_index]; - - if (bounded_variables_[j]) { - // We have a bounded variable - const f_t interval = upper_[j] - lower_[j]; - const f_t delta_slope = std::abs(delta_z_[j]) * interval; - const f_t pivot = std::abs(delta_z[j]); - if constexpr (verbose) { - settings_.log.printf( - "heap %d step-length %.12e pivot %e nonbasic entering %d slope %e delta_slope %e new " - "slope %e\n", - bare_idx.size(), - current_ratios[heap_index], - pivot, - nonbasic_entering, - slope, - delta_slope, - slope - delta_slope); + work_estimate_ += 5 * num_candidates; + + // Remove candidates that are greater than the maximum step length + const i_t candidates_before_removal = candidates.size(); + for (i_t h = candidates_before_removal - 1; h >= 0; h--) { + const i_t k = candidates[h]; + const f_t ratio = ratios[k]; + if (ratio > max_step_length) { + // Swap with the last candidate and remove + candidates[h] = candidates.back(); + candidates.pop_back(); } - slope -= delta_slope; - } else { - // The variable is not bounded. Stop the search. - break; } + work_estimate_ += + 2 * candidates_before_removal + 2 * (candidates_before_removal - candidates.size()); + num_candidates = candidates.size(); + } - if (toc(start_time_) > settings_.time_limit) { - entering_index = RATIO_TEST_TIME_LIMIT; - return; - } - if (settings_.concurrent_halt != nullptr && *settings_.concurrent_halt == 1) { - entering_index = CONCURRENT_HALT_RETURN; - return; + // Use a bucket sort to partition candidates into buckets by successive Harris breakpoints + // bucket_start[k] = index in candidates[] where bucket k starts + // Bucket k contains candidates[bucket_start[k]] .. candidates[bucket_start[k+1] - 1] + f_t threshold = minimum_harris_ratio; + i_t num_buckets = 0; + std::vector bucket_start(num_candidates + 1, 0); + f_t cumulative_slope = slope; + scan_start = 0; + work_estimate_ += num_candidates + 1; + + // This is O(num_buckets * num_candidates) + i_t slope_breaker_k = -1; // the candidate k that made slope go negative + t0 = tic(); + while (cumulative_slope >= 0.0 && scan_start < num_candidates && threshold <= max_step_length) { + f_t next_threshold = inf; + i_t write = scan_start; + + for (i_t h = scan_start; h < num_candidates; h++) { + const i_t k = candidates[h]; + const f_t ratio = ratios[k]; + + if (ratio <= threshold) { + const i_t j = nonbasic_list_[indicies[k]]; + if (bounded_variables_[j]) { + cumulative_slope -= std::abs(delta_z_[j]) * (upper_[j] - lower_[j]); + if (cumulative_slope < 0.0 && slope_breaker_k < 0) { slope_breaker_k = k; } + } + std::swap(candidates[h], candidates[write]); + write++; + } else { + const i_t j = nonbasic_list_[indicies[k]]; + const f_t harris_ratio = harris_ratios[k]; + next_threshold = std::min(next_threshold, harris_ratio); + } } - } -} + work_estimate_ += 3 * (num_candidates - scan_start) + 9 * (write - scan_start); -template -void bound_flipping_ratio_test_t::bucket_pass(const std::vector& current_indicies, - const std::vector& current_ratios, - i_t num_breakpoints, - f_t& slope, - f_t& step_length, - i_t& nonbasic_entering, - i_t& entering_index) -{ - const f_t dual_tol = settings_.dual_tol; - const f_t zero_tol = settings_.zero_tol; - const std::vector& delta_z = delta_z_; - const std::vector& nonbasic_list = nonbasic_list_; - const i_t N = num_breakpoints; - - const i_t K = 400; // 0, -16, -15, ...., 0, 1, ...., 400 - 18 = 382 - std::vector buckets(K, 0.0); - std::vector bucket_count(K, 0); - for (i_t k = 0; k < N; ++k) { - const i_t idx = current_indicies[k]; - const f_t ratio = current_ratios[k]; - const f_t min_exponent = -16.0; - const f_t max_exponent = 382.0; - const f_t exponent = std::max(min_exponent, std::min(max_exponent, std::log10(ratio))); - const i_t bucket_idx = ratio == 0.0 ? 0 : static_cast(exponent - min_exponent + 1); - // settings_.log.printf("Ratio %e exponent %e bucket_idx %d\n", ratio, exponent, bucket_idx); - const i_t j = nonbasic_list[idx]; - const f_t interval = upper_[j] - lower_[j]; - const f_t delta_slope = std::abs(delta_z_[j]) * interval; - buckets[bucket_idx] += delta_slope; - bucket_count[bucket_idx]++; - } + bucket_start[++num_buckets] = write; + if (write == scan_start) break; // No progress — prevent infinite loop + scan_start = write; + threshold = next_threshold; - std::vector cumulative_sum(K, 0.0); - cumulative_sum[0] = buckets[0]; - if (cumulative_sum[0] > slope) { - settings_.log.printf( - "Bucket 0. Count in bucket %d. Slope %e. Cumulative sum %e. Bucket value %e\n", - bucket_count[0], - slope, - cumulative_sum[0], - buckets[0]); - return; + if (cumulative_slope < 0.0) break; + } + time_bucket_sort_ += toc(t0); + bucket0_size_ = (num_buckets > 0) ? bucket_start[1] : 0; + + // Compute the maximum pivot + // This is O(num_candidates) + f_t max_pivot = 0.0; + for (i_t h = 0; h < bucket_start[num_buckets]; h++) { + const i_t k = candidates[h]; + const i_t j = nonbasic_list_[indicies[k]]; + const f_t pivot = std::abs(delta_z_[j]); + if (pivot > max_pivot) { max_pivot = pivot; } } - i_t k; - bool exceeded = false; - for (k = 1; k < K; ++k) { - cumulative_sum[k] = cumulative_sum[k - 1] + buckets[k]; - if (cumulative_sum[k] > slope) { - exceeded = true; - break; + work_estimate_ += 4 * bucket_start[num_buckets]; + + // Select the entering variable + // Scan from last bucket to first. Within each bucket, pick the variable with + // the largest |delta_z|, provided |delta_z| > pivot_threshold, breaking ties + // by preferring the larger step length. + f_t pivot_threshold = std::max(settings_.pivot_tol, std::min(0.1 * max_pivot, 1.0)); + i_t entering_k = -1; + + // This is O(num_candidates) + for (i_t b = num_buckets - 1; b >= 0; b--) { + const i_t b_start = bucket_start[b]; + const i_t b_end = bucket_start[b + 1]; + f_t best_pivot = -1.0; + f_t best_ratio = -1.0; + for (i_t h = b_start; h < b_end; h++) { + const i_t k = candidates[h]; + const i_t j = nonbasic_list_[indicies[k]]; + const f_t pivot = std::abs(delta_z_[j]); + if (pivot > pivot_threshold && + (pivot > best_pivot || (pivot == best_pivot && ratios[k] > best_ratio))) { + best_pivot = pivot; + best_ratio = ratios[k]; + entering_k = k; + } } + work_estimate_ += 2 + 5 * (b_end - b_start); + if (entering_k >= 0) break; } - if (exceeded) { - settings_.log.printf( - "Value in bucket %d. Count in buckets %d. Slope %e. Cumulative sum %e. Next sum %e Bucket " - "value %e\n", - k, - bucket_count[k], - slope, - cumulative_sum[k - 1], - cumulative_sum[k], - buckets[k - 1]); + // Step = entering variable's breakpoint ratio + num_buckets_used_ = num_buckets; + if (entering_k < 0) { + // Fallback to single_pass result + used_fallback_ = true; + bucket_selected_ = -1; + step_length_result_ = step_length; + selected_is_slope_breaker_ = false; + determine_flips(step_length, entering_index, flip_indices); + return entering_index; } + step_length = ratios[entering_k]; + nonbasic_entering = indicies[entering_k]; + entering_index = nonbasic_list_[nonbasic_entering]; + + // Record whether we selected the slope breaker + selected_is_slope_breaker_ = (entering_k == slope_breaker_k); + + // Record which bucket was selected + used_fallback_ = false; + i_t pos = -1; + for (i_t b = 0; b < num_buckets; b++) { + if (entering_k >= 0) { + // Find which bucket entering_k is in based on its position in candidates + pos = -1; + for (i_t h = 0; h < num_candidates; h++) { + if (candidates[h] == entering_k) { + pos = h; + break; + } + } + if (pos >= bucket_start[b] && pos < bucket_start[b + 1]) { + bucket_selected_ = b; + break; + } + } + } + work_estimate_ += (bucket_selected_ + 1) * (pos + 3); + step_length_result_ = step_length; + determine_flips(step_length, entering_index, flip_indices); + + return entering_index; } #ifdef DUAL_SIMPLEX_INSTANTIATE_DOUBLE diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp index 2e73d05eff..4587037889 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp @@ -53,35 +53,42 @@ class bound_flipping_ratio_test_t { { } - i_t compute_step_length(f_t& step_length, i_t& nonbasic_entering); + i_t compute_step_length(f_t& step_length, i_t& nonbasic_entering, std::vector& flip_indices); f_t work_estimate() const { return work_estimate_; } + // Timing fields (filled by compute_step_length) + f_t time_compute_breakpoints_{0.0}; + f_t time_single_pass_{0.0}; + f_t time_coarse_filter_{0.0}; + f_t time_bucket_sort_{0.0}; + f_t time_pivot_selection_{0.0}; + + // Diagnostic fields + i_t num_buckets_used_{0}; // number of buckets in bucket sort + i_t bucket_selected_{ + -1}; // which bucket the entering variable came from (-1 = single_pass/fallback) + f_t step_length_result_{0.0}; // the step length chosen + bool used_fallback_{false}; // true if we fell back to single_pass result + i_t bucket0_size_{0}; // size of first bucket (candidates with ratio <= min_harris) + i_t num_breakpoints_{0}; // total breakpoints computed + bool selected_is_slope_breaker_{ + false}; // true if we selected the variable that made slope go negative + i_t num_harris_zero_{0}; // number of harris_ratios that are exactly 0 + i_t num_exact_zero_{0}; // number of exact ratios that are exactly 0 + private: - i_t compute_breakpoints(std::vector& indices, std::vector& ratios); + i_t compute_breakpoints(std::vector& indices, + std::vector& ratios, + std::vector& harris_ratios); i_t single_pass(i_t start, i_t end, const std::vector& indices, const std::vector& ratios, - f_t& slope, f_t& step_length, i_t& nonbasic_entering, - i_t& enetering_index); - void heap_passes(const std::vector& current_indicies, - const std::vector& current_ratios, - i_t num_breakpoints, - f_t& slope, - f_t& step_lenght, - i_t& nonbasic_entering, - i_t& entering_index); - - void bucket_pass(const std::vector& current_indicies, - const std::vector& current_ratios, - i_t num_breakpoints, - f_t& slope, - f_t& step_length, - i_t& nonbasic_entering, - i_t& entering_index); - + i_t& entering_index, + f_t& max_val); + void determine_flips(f_t step_length, i_t entering_index, std::vector& flip_indices); const std::vector& lower_; const std::vector& upper_; const std::vector& bounded_variables_; @@ -100,7 +107,7 @@ class bound_flipping_ratio_test_t { i_t n_; i_t m_; - f_t work_estimate_; + f_t work_estimate_{0.0}; }; } // namespace cuopt::mathematical_optimization::simplex diff --git a/cpp/src/dual_simplex/crossover.cpp b/cpp/src/dual_simplex/crossover.cpp index e1ba272adf..977f5e5511 100644 --- a/cpp/src/dual_simplex/crossover.cpp +++ b/cpp/src/dual_simplex/crossover.cpp @@ -168,9 +168,10 @@ f_t primal_infeasibility(const lp_problem_t& lp, f_t primal_inf = 0; constexpr bool verbose = false; constexpr f_t infeas_tol = 1e-3; + const f_t primal_tol = settings.primal_tol; for (i_t j = 0; j < n; ++j) { - if (x[j] < lp.lower[j]) { - // x_j < l_j => -x_j > -l_j => -x_j + l_j > 0 + if (x[j] < lp.lower[j] - primal_tol) { + // x_j < l_j - tol => violation exceeds per-variable threshold const f_t infeas = -x[j] + lp.lower[j]; primal_inf += infeas; if (verbose && infeas > infeas_tol) { @@ -183,8 +184,8 @@ f_t primal_infeasibility(const lp_problem_t& lp, vstatus[j]); } } - if (x[j] > lp.upper[j]) { - // x_j > u_j => x_j - u_j > 0 + if (x[j] > lp.upper[j] + primal_tol) { + // x_j > u_j + tol => violation exceeds per-variable threshold const f_t infeas = x[j] - lp.upper[j]; primal_inf += infeas; if (verbose && infeas > infeas_tol) { @@ -1423,8 +1424,11 @@ crossover_status_t crossover(const lp_problem_t& lp, } else if (dual_feasible && !primal_feasible) { i_t dual_iter = 0; std::vector edge_norms; - dual_status_t status = - dual_phase2(2, 0, start_time, lp, settings, vstatus, solution, dual_iter, edge_norms); + f_t work_estimate = 0.0; + simplex_solver_settings_t dual_settings = settings; + dual_settings.iteration_limit = std::numeric_limits::max(); + dual_status_t status = dual_phase2( + 2, 0, start_time, lp, dual_settings, vstatus, solution, dual_iter, work_estimate, edge_norms); if (toc(start_time) > settings.time_limit) { settings.log.printf("Time limit exceeded\n"); return crossover_status_t::TIME_LIMIT; @@ -1443,7 +1447,33 @@ crossover_status_t crossover(const lp_problem_t& lp, solution.iterations += dual_iter; primal_feasible = primal_infeas <= primal_tol && primal_res <= primal_tol; dual_feasible = dual_infeas <= dual_tol && dual_res <= dual_tol; + } else if (primal_feasible && !dual_feasible) { + i_t primal_iter = 0; + simplex_solver_settings_t primal_settings = settings; + primal_settings.iteration_limit = std::numeric_limits::max(); + primal_status_t primal_status = + primal_phase2(2, start_time, lp, primal_settings, vstatus, solution, primal_iter); + if (toc(start_time) > settings.time_limit) { + settings.log.printf("Time limit exceeded\n"); + return crossover_status_t::TIME_LIMIT; + } + if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { + if (!settings.inside_mip) { settings.log.printf("Concurrent halt\n"); } + return crossover_status_t::CONCURRENT_LIMIT; + } + primal_infeas = primal_infeasibility(lp, settings, vstatus, solution.x); + dual_infeas = dual_infeasibility(lp, settings, vstatus, solution.z); + primal_res = primal_residual(lp, solution); + dual_res = dual_residual(lp, solution); + if (primal_status != primal_status_t::OPTIMAL) { + print_crossover_info(lp, settings, vstatus, solution, "Primal phase 2 complete"); + } + solution.iterations += primal_iter; + primal_feasible = primal_infeas <= primal_tol && primal_res <= primal_tol; + dual_feasible = dual_infeas <= dual_tol && dual_res <= dual_tol; } else { + simplex_solver_settings_t dual_settings = settings; + dual_settings.iteration_limit = std::numeric_limits::max(); lp_problem_t phase1_problem(lp.handle_ptr, 1, 1, 1); create_phase1_problem(lp, phase1_problem); std::vector phase1_vstatus(n); @@ -1469,8 +1499,17 @@ crossover_status_t crossover(const lp_problem_t& lp, i_t iter = 0; lp_solution_t phase1_solution(phase1_problem.num_rows, phase1_problem.num_cols); std::vector junk; - dual_status_t phase1_status = dual_phase2( - 1, 1, start_time, phase1_problem, settings, phase1_vstatus, phase1_solution, iter, junk); + f_t phase1_work_estimate = 0.0; + dual_status_t phase1_status = dual_phase2(1, + 1, + start_time, + phase1_problem, + dual_settings, + phase1_vstatus, + phase1_solution, + iter, + phase1_work_estimate, + junk); if (phase1_status == dual_status_t::NUMERICAL || phase1_status == dual_status_t::DUAL_UNBOUNDED) { settings.log.printf("Failed in Phase 1\n"); @@ -1585,8 +1624,17 @@ crossover_status_t crossover(const lp_problem_t& lp, dual_status_t status = dual_status_t::NUMERICAL; if (dual_infeas <= settings.dual_tol) { std::vector edge_norms; - status = dual_phase2( - 2, iter == 0 ? 1 : 0, start_time, lp, settings, vstatus, solution, iter, edge_norms); + f_t phase2_work_estimate = 0.0; + status = dual_phase2(2, + iter == 0 ? 1 : 0, + start_time, + lp, + dual_settings, + vstatus, + solution, + iter, + phase2_work_estimate, + edge_norms); if (toc(start_time) > settings.time_limit) { settings.log.printf("Time limit exceeded\n"); return crossover_status_t::TIME_LIMIT; diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index a5f10c3229..9eb3224817 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -11,6 +11,7 @@ #include #include #include +#include #include #include #include @@ -160,7 +161,7 @@ void compute_delta_z(const csr_matrix_t& Arow, } } work_estimate += 4 * nz_delta_y; - work_estimate += 4 * nnz_processed; + work_estimate += 5 * nnz_processed; work_estimate += 2 * delta_z_indices.size(); // delta_zB = sigma*ei @@ -454,43 +455,70 @@ template void initial_perturbation(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& vstatus, + bool strongly_degenerate, std::vector& objective) { - const i_t m = lp.num_rows; const i_t n = lp.num_cols; f_t max_abs_obj_coeff = 0.0; for (i_t j = 0; j < n; ++j) { max_abs_obj_coeff = std::max(max_abs_obj_coeff, std::abs(lp.objective[j])); } - const f_t dual_tol = settings.dual_tol; + // Dampen large costs + if (max_abs_obj_coeff > 100.0) { max_abs_obj_coeff = std::sqrt(std::sqrt(max_abs_obj_coeff)); } + // Ensure a minimum perturbation even for tiny-cost problems + if (max_abs_obj_coeff < 1.0) { max_abs_obj_coeff = 1.0; } + + // If few boxed variables, cap max_abs_obj_coeff at 1.0 + i_t num_boxed = 0; + for (i_t j = 0; j < n; ++j) { + if (lp.lower[j] > -inf && lp.upper[j] < inf && lp.lower[j] != lp.upper[j]) { num_boxed++; } + } + if (static_cast(num_boxed) / n < 0.01) { + max_abs_obj_coeff = std::min(max_abs_obj_coeff, f_t(1.0)); + } + + // Sub-tolerance perturbations are less disruptive on ordinary problems, but + // are too small to separate reduced costs when a substantial part of the + // nonbasic set is dual degenerate. Use a stronger, still temporary shift in + // that case. The original costs are restored before declaring optimality. + const f_t perturbation_base = (strongly_degenerate ? 1e-5 : 5e-7) * max_abs_obj_coeff; + + settings.log.printf( + "Perturbation debug: max_abs_obj_coeff=%e (dampened), perturbation_base=%e, n=%d, " + "num_boxed=%d\n", + max_abs_obj_coeff, + perturbation_base, + n, + num_boxed); objective.resize(n); f_t sum_perturb = 0.0; i_t num_perturb = 0; - random_t random(settings.seed); + random_t random(settings.random_seed); for (i_t j = 0; j < n; ++j) { f_t obj = objective[j] = lp.objective[j]; const f_t lower = lp.lower[j]; const f_t upper = lp.upper[j]; - if (vstatus[j] == variable_status_t::NONBASIC_FIXED || - vstatus[j] == variable_status_t::NONBASIC_FREE || lower == upper || - lower == -inf && upper == inf) { - continue; - } + // Skip truly fixed variables and free variables + if (lower == upper || (lower == -inf && upper == inf)) { continue; } - const f_t rand_val = random.random(); - const f_t perturb = - (1e-5 * std::abs(obj) + 1e-7 * max_abs_obj_coeff + 10 * dual_tol) * (1.0 + rand_val); + const f_t rand_val = random.random(); + const f_t cost_factor = std::min(std::abs(obj) + 1.0, max_abs_obj_coeff + 1.0); + const f_t perturb = (1.0 + rand_val) * cost_factor * perturbation_base; - if (vstatus[j] == variable_status_t::NONBASIC_LOWER || lower > -inf && upper < inf && obj > 0) { + if (vstatus[j] == variable_status_t::BASIC) { + // Skip basic variables + continue; + } else if (vstatus[j] == variable_status_t::NONBASIC_LOWER || + vstatus[j] == variable_status_t::NONBASIC_FIXED) { + // NONBASIC_FIXED from phase 1 for boxed variables — treat as at lower bound objective[j] = obj + perturb; sum_perturb += perturb; num_perturb++; - } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER || - lower > -inf && upper < inf && obj < 0) { + } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER) { objective[j] = obj - perturb; sum_perturb += perturb; num_perturb++; @@ -904,7 +932,7 @@ bool update_primal_infeasibilities(const lp_problem_t& lp, primal_inf); if (old_val != 0.0 && squared_infeasibilities[j] == 0.0) { became_feasible = true; } } - work_estimate += 8 * nz; + work_estimate += 9 * nz; return became_feasible; } @@ -1205,13 +1233,8 @@ i_t phase2_ratio_test(const lp_problem_t& lp, template i_t flip_bounds(const lp_problem_t& lp, - const simplex_solver_settings_t& settings, const std::vector& bounded_variables, - const std::vector& objective, - const std::vector& z, - const std::vector& delta_z_indices, - const std::vector& nonbasic_list, - i_t entering_index, + const std::vector& flip_indices, std::vector& vstatus, std::vector& delta_x, std::vector& mark, @@ -1220,15 +1243,9 @@ i_t flip_bounds(const lp_problem_t& lp, f_t& work_estimate) { i_t num_flipped = 0; - for (i_t k = 0; k < delta_z_indices.size(); ++k) { - const i_t j = delta_z_indices[k]; - if (j == entering_index) { continue; } - if (!bounded_variables[j]) { continue; } - // x_j is now a nonbasic bounded variable that will not enter the basis this - // iteration - const f_t dual_tol = - settings.dual_tol; // lower to 1e-7 or less will cause 25fv47 and d2q06c to cycle - if (vstatus[j] == variable_status_t::NONBASIC_LOWER && z[j] < -dual_tol) { + for (const i_t j : flip_indices) { + assert(bounded_variables[j]); + if (vstatus[j] == variable_status_t::NONBASIC_LOWER) { const f_t delta = lp.upper[j] - lp.lower[j]; const size_t atilde_start_size = atilde_index.size(); scatter_dense(lp.A, j, -delta, atilde, mark, atilde_index); @@ -1236,12 +1253,9 @@ i_t flip_bounds(const lp_problem_t& lp, 4 * (lp.A.col_start[j + 1] - lp.A.col_start[j]) + 10; delta_x[j] += delta; vstatus[j] = variable_status_t::NONBASIC_UPPER; -#ifdef BOUND_FLIP_DEBUG - settings.log.printf( - "Flipping nonbasic %d from lo %e to up %e. z %e\n", j, lp.lower[j], lp.upper[j], z[j]); -#endif num_flipped++; - } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER && z[j] > dual_tol) { + } else { + assert(vstatus[j] == variable_status_t::NONBASIC_UPPER); const f_t delta = lp.lower[j] - lp.upper[j]; const size_t atilde_start_size = atilde_index.size(); scatter_dense(lp.A, j, -delta, atilde, mark, atilde_index); @@ -1249,13 +1263,10 @@ i_t flip_bounds(const lp_problem_t& lp, 4 * (lp.A.col_start[j + 1] - lp.A.col_start[j]) + 10; delta_x[j] += delta; vstatus[j] = variable_status_t::NONBASIC_LOWER; -#ifdef BOUND_FLIP_DEBUG - settings.log.printf( - "Flipping nonbasic %d from up %e to lo %e. z %e\n", j, lp.upper[j], lp.lower[j], z[j]); -#endif num_flipped++; } } + work_estimate += 2 * flip_indices.size(); return num_flipped; } @@ -1454,7 +1465,7 @@ i_t update_steepest_edge_norms(const simplex_solver_settings_t& settin work_estimate += 2 * v_sparse.i.size(); } v_sparse.scatter(v); - work_estimate += 2 * v_sparse.i.size(); + work_estimate += 4 * v_sparse.i.size(); const i_t leaving_index = basic_list[basic_leaving_index]; const f_t prev_dy_norm_squared = delta_y_steepest_edge[leaving_index]; @@ -1506,7 +1517,7 @@ i_t update_steepest_edge_norms(const simplex_solver_settings_t& settin delta_y_steepest_edge[j] = new_val; } } - work_estimate += 5 * scaled_delta_xB_nz; + work_estimate += 6 * scaled_delta_xB_nz; const i_t v_nz = v_sparse.i.size(); for (i_t k = 0; k < v_nz; ++k) { @@ -1541,13 +1552,64 @@ i_t check_steepest_edge_norms(const simplex_solver_settings_t& setting return 0; } +// Remove the perturbation from a variable that is leaving the basis. Since it +// is nonbasic, its cost affects only its own reduced cost. If removing the +// perturbation would violate dual feasibility, the perturbation is left in +// place (for boxed variables) or reduced to the minimum needed (for one-sided +// variables). +template +void remove_leaving_perturbation(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + i_t leaving_index, + i_t direction, + std::vector& z, + std::vector& objective) +{ + const f_t perturb = objective[leaving_index] - lp.objective[leaving_index]; + if (perturb == 0.0) return; + + const f_t lower = lp.lower[leaving_index]; + const f_t upper = lp.upper[leaving_index]; + const bool boxed = (lower > -inf && upper < inf); + + if (boxed) { + // Only remove if it won't create dual infeasibility. + // direction=1 means going to lower bound (needs z >= 0 after removal) + // direction=-1 means going to upper bound (needs z <= 0 after removal) + const f_t new_z = z[leaving_index] - perturb; + if (direction == 1 && new_z < -settings.tight_tol) { return; } + if (direction == -1 && new_z > settings.tight_tol) { return; } + z[leaving_index] = new_z; + objective[leaving_index] = lp.objective[leaving_index]; + } else { + z[leaving_index] -= perturb; + objective[leaving_index] = lp.objective[leaving_index]; + + // Restore dual feasibility if needed for one-sided variables + if (upper == inf && lower > -inf && z[leaving_index] < -settings.tight_tol) { + // At lower bound, needs z >= 0 + const f_t correction = -z[leaving_index]; + z[leaving_index] = 0.0; + objective[leaving_index] += correction; + } else if (lower == -inf && upper < inf && z[leaving_index] > settings.tight_tol) { + // At upper bound, needs z <= 0 + const f_t correction = z[leaving_index]; + z[leaving_index] = 0.0; + objective[leaving_index] -= correction; + } + } +} + template i_t compute_perturbation(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& delta_z_indices, + const std::vector& vstatus, std::vector& z, std::vector& objective, f_t& sum_perturb, + i_t entering_index, + f_t step_length, f_t& work_estimate) { const i_t n = lp.num_cols; @@ -1563,32 +1625,27 @@ i_t compute_perturbation(const lp_problem_t& lp, objective[j] += violation; num_perturb++; sum_perturb += violation; -#ifdef PERTURBATION_DEBUG - if (violation > 1e-1) { - settings.log.printf( - "perturbation: violation %e j %d lower %e\n", violation, j, lp.lower[j]); - } -#endif } else if (lp.lower[j] == -inf && lp.upper[j] < inf && z[j] > tight_tol) { const f_t violation = z[j]; z[j] -= violation; // z[j] <- 0 objective[j] -= violation; num_perturb++; sum_perturb += violation; -#ifdef PERTURBATION_DEWBUG - if (violation > 1e-1) { - settings.log.printf( - "perturbation: violation %e j %d upper %e\n", violation, j, lp.upper[j]); - } -#endif } } - work_estimate += 7 * delta_z_indices.size(); -#ifdef PERTURBATION_DEBUG - if (num_perturb > 0) { - settings.log.printf("Perturbed %d dual variables by %e\n", num_perturb, sum_perturb); + // On degenerate steps, shift the entering variable's cost (like HiGHS) + // This accumulates shifts that break degeneracy at the next refactorization + if (entering_index >= 0 && step_length == 0.0) { + assert(vstatus[entering_index] != variable_status_t::BASIC); + const f_t shift = -z[entering_index]; + if (shift != 0.0) { + objective[entering_index] += shift; + z[entering_index] = 0.0; + sum_perturb += std::abs(shift); + num_perturb++; + } } -#endif + work_estimate += 7 * delta_z_indices.size(); return 0; } @@ -2210,19 +2267,26 @@ void bound_info(const lp_problem_t& lp, } template -void set_primal_variables_on_bounds(const lp_problem_t& lp, - const simplex_solver_settings_t& settings, - const std::vector& z, - std::vector& vstatus, - std::vector& x) +i_t set_primal_variables_on_bounds(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& z, + std::vector& vstatus, + std::vector& x, + i_t degen_type = 0) { PHASE2_NVTX_RANGE("DualSimplex::set_primal_variables_on_bounds"); - const i_t n = lp.num_cols; - f_t tol = 1e-10; + const i_t n = lp.num_cols; + f_t tol = 1e-10; + i_t num_fixed_to_lower = 0; + i_t num_fixed_to_upper = 0; + i_t num_lower_to_upper = 0; + i_t num_upper_to_lower = 0; + i_t num_set_fixed = 0; for (i_t j = 0; j < n; ++j) { // We set z_j = 0 for basic variables // But we explicitally skip setting basic variables here if (vstatus[j] == variable_status_t::BASIC) { continue; } + const variable_status_t old_vstatus = vstatus[j]; // We will flip the status of variables between nonbasic lower and nonbasic // upper here to improve dual feasibility const f_t fixed_tolerance = settings.fixed_tol; @@ -2243,29 +2307,69 @@ void set_primal_variables_on_bounds(const lp_problem_t& lp, vstatus[j] == variable_status_t::NONBASIC_UPPER) { x[j] = lp.upper[j]; } else if (z[j] >= 0 && lp.lower[j] > -inf) { - if (vstatus[j] != variable_status_t::NONBASIC_LOWER) { - settings.log.debug( - "Setting nonbasic lower variable (zj %e) %d to %e (current %e). vstatus %d\n", - z[j], - j, - lp.lower[j], - x[j], - static_cast(vstatus[j])); + // For boxed variables with degenerate z, use heuristic based on degen_type + if (degen_type >= 1 && std::abs(z[j]) < settings.dual_tol && lp.upper[j] < inf) { + if (degen_type == 1) { + // Column-sum heuristic + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + f_t col_sum = 0.0; + for (i_t k = col_start; k < col_end; k++) { + col_sum += lp.A.x[k]; + } + if (col_sum < 0.0) { + x[j] = lp.upper[j]; + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } else { + x[j] = lp.lower[j]; + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } + } else { + // degen_type == 3: abs_bound (prefer bound closer to zero, like HiGHS) + if (std::abs(lp.upper[j]) < std::abs(lp.lower[j])) { + x[j] = lp.upper[j]; + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } else { + x[j] = lp.lower[j]; + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } + } + } else { + x[j] = lp.lower[j]; + vstatus[j] = variable_status_t::NONBASIC_LOWER; } - x[j] = lp.lower[j]; - vstatus[j] = variable_status_t::NONBASIC_LOWER; } else if (z[j] <= 0 && lp.upper[j] < inf) { - if (vstatus[j] != variable_status_t::NONBASIC_UPPER) { - settings.log.debug( - "Setting nonbasic upper variable (zj %e) %d to %e (current %e). vstatus %d\n", - z[j], - j, - lp.upper[j], - x[j], - static_cast(vstatus[j])); + // For boxed variables with degenerate z, use heuristic based on degen_type + if (degen_type >= 1 && std::abs(z[j]) < settings.dual_tol && lp.lower[j] > -inf) { + if (degen_type == 1) { + // Column-sum heuristic + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + f_t col_sum = 0.0; + for (i_t k = col_start; k < col_end; k++) { + col_sum += lp.A.x[k]; + } + if (col_sum > 0.0) { + x[j] = lp.lower[j]; + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } else { + x[j] = lp.upper[j]; + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } + } else { + // degen_type == 3: abs_bound (prefer bound closer to zero) + if (std::abs(lp.lower[j]) < std::abs(lp.upper[j])) { + x[j] = lp.lower[j]; + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } else { + x[j] = lp.upper[j]; + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } + } + } else { + x[j] = lp.upper[j]; + vstatus[j] = variable_status_t::NONBASIC_UPPER; } - x[j] = lp.upper[j]; - vstatus[j] = variable_status_t::NONBASIC_UPPER; } else if (lp.upper[j] == inf && lp.lower[j] > -inf && z[j] < 0) { // dual infeasible if (vstatus[j] != variable_status_t::NONBASIC_LOWER) { @@ -2299,7 +2403,38 @@ void set_primal_variables_on_bounds(const lp_problem_t& lp, } else { assert(1 == 0); } + // Track changes + if (old_vstatus != vstatus[j]) { + if (old_vstatus == variable_status_t::NONBASIC_FIXED && + vstatus[j] == variable_status_t::NONBASIC_LOWER) + num_fixed_to_lower++; + else if (old_vstatus == variable_status_t::NONBASIC_FIXED && + vstatus[j] == variable_status_t::NONBASIC_UPPER) + num_fixed_to_upper++; + else if (old_vstatus == variable_status_t::NONBASIC_LOWER && + vstatus[j] == variable_status_t::NONBASIC_UPPER) + num_lower_to_upper++; + else if (old_vstatus == variable_status_t::NONBASIC_UPPER && + vstatus[j] == variable_status_t::NONBASIC_LOWER) + num_upper_to_lower++; + else if (vstatus[j] == variable_status_t::NONBASIC_FIXED) + num_set_fixed++; + } + } + i_t total_changes = num_fixed_to_lower + num_fixed_to_upper + num_lower_to_upper + + num_upper_to_lower + num_set_fixed; + if (total_changes > 0) { + settings.log.printf( + "set_primal_variables_on_bounds: %d changes (fixed->lower=%d, fixed->upper=%d, " + "lower->upper=%d, upper->lower=%d, ->fixed=%d)\n", + total_changes, + num_fixed_to_lower, + num_fixed_to_upper, + num_lower_to_upper, + num_upper_to_lower, + num_set_fixed); } + return total_changes; } template @@ -2324,6 +2459,192 @@ f_t amount_of_perturbation(const lp_problem_t& lp, const std::vector +i_t attempt_to_remove_perturbations(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + basis_update_mpf_t& ft, + const std::vector& basic_list, + const std::vector& nonbasic_list, + std::vector& vstatus, + std::vector& objective, + std::vector& z, + std::vector& y, + std::vector& x, + std::vector& xB_workspace, + std::vector& squared_infeasibilities, + std::vector& infeasibility_indices, + f_t& primal_infeasibility, + f_t& primal_infeasibility_squared, + f_t& work_estimate) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + const i_t n_minus_m = n - m; + + // Check if there's any perturbation + const f_t perturbation = amount_of_perturbation(lp, objective); + if (perturbation <= 1e-6) return 0; // OPTIMAL + + // Count perturbations on basic vs nonbasic variables + i_t num_basic_perturbed = 0; + i_t num_nonbasic_boxed_perturbed = 0; + i_t num_nonbasic_other_perturbed = 0; + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + if (objective[j] != lp.objective[j]) num_basic_perturbed++; + } + for (i_t k = 0; k < n_minus_m; ++k) { + const i_t j = nonbasic_list[k]; + if (objective[j] != lp.objective[j]) { + const f_t lower = lp.lower[j]; + const f_t upper = lp.upper[j]; + if (lower > -inf && upper < inf && lower != upper) { + num_nonbasic_boxed_perturbed++; + } else { + num_nonbasic_other_perturbed++; + } + } + } + + if (num_basic_perturbed == 0 && num_nonbasic_other_perturbed == 0) { + // Safe path: perturbation only on nonbasic boxed variables. + // y is unaffected; z[j] - perturb gives exact unperturbed reduced cost. + i_t num_flipped = 0; + for (i_t k = 0; k < n_minus_m; ++k) { + const i_t j = nonbasic_list[k]; + const f_t perturb = objective[j] - lp.objective[j]; + if (perturb == 0.0) continue; + const f_t new_z = z[j] - perturb; + if (vstatus[j] == variable_status_t::NONBASIC_LOWER && new_z < -settings.dual_tol) { + vstatus[j] = variable_status_t::NONBASIC_UPPER; + z[j] = new_z; + objective[j] = lp.objective[j]; + num_flipped++; + } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER && new_z > settings.dual_tol) { + vstatus[j] = variable_status_t::NONBASIC_LOWER; + z[j] = new_z; + objective[j] = lp.objective[j]; + num_flipped++; + } else { + z[j] = new_z; + objective[j] = lp.objective[j]; + } + } + work_estimate += 5 * n_minus_m; + + // Recompute x_B with flipped statuses + compute_primal_solution_from_basis( + lp, ft, basic_list, nonbasic_list, vstatus, x, xB_workspace, work_estimate); + work_estimate += 2 * n; + primal_infeasibility_squared = compute_initial_primal_infeasibilities(lp, + settings, + basic_list, + x, + squared_infeasibilities, + infeasibility_indices, + primal_infeasibility); + work_estimate += 4 * m + 2 * n; + + if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL + settings.log.printf("Removed perturbation. Continuing dual simplex (primal_inf=%.2e)\n", + primal_infeasibility); + return 1; // CONTINUE_DUAL + } + + // Perturbation on basic (or one-sided nonbasic) variables: need to recompute (y, z). + std::vector unperturbed_y(m); + std::vector unperturbed_z(n); + compute_dual_solution_from_basis( + lp, ft, basic_list, nonbasic_list, unperturbed_y, unperturbed_z, work_estimate); + + // Check if removal is clean (no dual infeasibility) + const f_t dual_infeas = + dual_infeasibility(lp, settings, vstatus, unperturbed_z, settings.tight_tol, settings.dual_tol); + work_estimate += 3 * n; + if (dual_infeas <= settings.dual_tol) { + settings.log.printf("Removed perturbation of %.2e.\n", perturbation); + z = unperturbed_z; + y = unperturbed_y; + objective = lp.objective; + work_estimate += 3 * n + 2 * m; + return 0; // OPTIMAL + } + + // Flip boxed nonbasics that are dual infeasible, and check for one-sided infeasibility + std::vector new_vstatus = vstatus; + i_t num_flipped = 0; + f_t residual_dual_infeas = 0.0; + for (i_t k = 0; k < n_minus_m; ++k) { + const i_t j = nonbasic_list[k]; + const f_t zj = unperturbed_z[j]; + const f_t lower = lp.lower[j]; + const f_t upper = lp.upper[j]; + const bool boxed = (lower > -inf && upper < inf && lower != upper); + + if (new_vstatus[j] == variable_status_t::NONBASIC_LOWER && zj < -settings.dual_tol) { + if (boxed) { + new_vstatus[j] = variable_status_t::NONBASIC_UPPER; + num_flipped++; + } else { + residual_dual_infeas = std::max(residual_dual_infeas, -zj); + } + } else if (new_vstatus[j] == variable_status_t::NONBASIC_UPPER && zj > settings.dual_tol) { + if (boxed) { + new_vstatus[j] = variable_status_t::NONBASIC_LOWER; + num_flipped++; + } else { + residual_dual_infeas = std::max(residual_dual_infeas, zj); + } + } + } + work_estimate += 5 * n_minus_m; + + if (residual_dual_infeas > settings.dual_tol) { + // One-sided infeasibility remains — can't continue with dual simplex. + // new_vstatus is discarded; vstatus unchanged. + settings.log.printf( + "Perturbation removal: %d flips, residual_dual_infeas=%.2e (PRIMAL_CLEANUP)\n", + num_flipped, + residual_dual_infeas); + return 2; // PRIMAL_CLEANUP + } + + // All infeasibility was on boxed variables — accept unperturbed solution + vstatus = new_vstatus; + z = unperturbed_z; + y = unperturbed_y; + objective = lp.objective; + work_estimate += 3 * n + 2 * m; + + // Recompute x_B with flipped statuses + compute_primal_solution_from_basis( + lp, ft, basic_list, nonbasic_list, vstatus, x, xB_workspace, work_estimate); + work_estimate += 2 * n; + primal_infeasibility_squared = compute_initial_primal_infeasibilities(lp, + settings, + basic_list, + x, + squared_infeasibilities, + infeasibility_indices, + primal_infeasibility); + work_estimate += 4 * m + 2 * n; + + settings.log.printf( + "Perturbation removal: %d flips, primal_inf=%.2e\n", num_flipped, primal_infeasibility); + if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL + settings.log.printf("Continuing dual after flip (primal_inf=%.2e)\n", primal_infeasibility); + return 1; // CONTINUE_DUAL +} + template void prepare_optimality(i_t info, f_t orig_primal_infeas, @@ -2331,54 +2652,44 @@ void prepare_optimality(i_t info, const simplex_solver_settings_t& settings, basis_update_mpf_t& ft, const std::vector& objective, - const std::vector& basic_list, - const std::vector& nonbasic_list, - const std::vector& vstatus, + // Primal cleanup below pivots, so the basis, the statuses + // and the iteration count are updated in place. + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, int phase, f_t start_time, f_t max_val, - i_t iter, + f_t& work_estimate, + i_t& iter, const std::vector& x, std::vector& y, std::vector& z, lp_solution_t& sol) { - const i_t m = lp.num_rows; - const i_t n = lp.num_cols; - f_t work_estimate = 0; // Work in this function is not captured - - sol.objective = compute_objective(lp, sol.x); - sol.user_objective = compute_user_objective(lp, sol.objective); - f_t perturbation = amount_of_perturbation(lp, objective); - f_t orig_perturbation = perturbation; - if (perturbation > 1e-6 && phase == 2) { - // Try to remove perturbation - std::vector unperturbed_y(m); - std::vector unperturbed_z(n); - compute_dual_solution_from_basis( - lp, ft, basic_list, nonbasic_list, unperturbed_y, unperturbed_z, work_estimate); - { - const f_t dual_infeas = dual_infeasibility( - lp, settings, vstatus, unperturbed_z, settings.tight_tol, settings.dual_tol); - if (dual_infeas <= settings.dual_tol) { - settings.log.printf("Removed perturbation of %.2e.\n", perturbation); - z = unperturbed_z; - y = unperturbed_y; - perturbation = 0.0; - } else { - settings.log.printf("Failed to remove perturbation of %.2e.\n", perturbation); - } - } - } + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + + sol.objective = compute_objective(lp, sol.x); + sol.user_objective = compute_user_objective(lp, sol.objective); + const f_t perturbation = amount_of_perturbation(lp, objective); - sol.l2_primal_residual = l2_primal_residual(lp, sol); - sol.l2_dual_residual = l2_dual_residual(lp, sol); - const f_t dual_infeas = dual_infeasibility(lp, settings, vstatus, z, 0.0, 0.0); - const f_t primal_infeas = primal_infeasibility(lp, settings, vstatus, x); + sol.l2_primal_residual = l2_primal_residual(lp, sol); + sol.l2_dual_residual = l2_dual_residual(lp, sol); + const f_t dual_infeas = dual_infeasibility(lp, settings, vstatus, z, 0.0, 0.0); + // Compute max primal infeasibility for reporting + f_t primal_infeas = 0.0; + for (i_t j = 0; j < n; ++j) { + if (x[j] < lp.lower[j]) { primal_infeas = std::max(primal_infeas, lp.lower[j] - x[j]); } + if (x[j] > lp.upper[j]) { primal_infeas = std::max(primal_infeas, x[j] - lp.upper[j]); } + } if (phase == 1 && iter > 0) { settings.log.printf("Dual phase I complete. Iterations %d. Time %.2f\n", iter, toc(start_time)); } if (phase == 2) { + if (settings.inside_mip == 0 || settings.inside_mip == 1) { + settings.log.printf("Work estimate: %.2e\n", work_estimate); + } if (!settings.inside_mip) { settings.log.printf("\n"); settings.log.printf( @@ -2399,20 +2710,34 @@ void prepare_optimality(i_t info, primal_infeasibility_breakdown( lp, settings, vstatus, x, basic_infeas, nonbasic_infeas, basic_over); settings.log.printf( - "Primal infeasibility %e/%e (Basic %e, Nonbasic %e, Basic over %e). Perturbation %e/%e. Info " + "Primal infeasibility %e/%e (Basic %e, Nonbasic %e, Basic over %e). Perturbation %e. Info " "%d\n", primal_infeas, orig_primal_infeas, basic_infeas, nonbasic_infeas, basic_over, - orig_perturbation, perturbation, info); } #endif } +template +struct work_timer_t { + work_timer_t(f_t t) : time(t) {} + f_t time{0.0}; + f_t work{0.0}; +}; + +template +work_timer_t& operator+=(work_timer_t& lhs, const work_timer_t& rhs) +{ + lhs.time += rhs.time; + lhs.work += rhs.work; + return lhs; +} + template class phase2_timers_t { public: @@ -2435,60 +2760,131 @@ class phase2_timers_t { { } - void start_timer() + void start_timer(f_t work) { if (!record_time) { return; } start_time = tic(); + start_work = work; + } + + work_timer_t stop_timer(f_t stop_work) + { + if (!record_time) { return work_timer_t(0.0); } + work_timer_t result(toc(start_time)); + result.work = stop_work - start_work; + return result; } - f_t stop_timer() + void print_one(const simplex_solver_settings_t& settings, + const char* name, + const work_timer_t& t, + f_t total_time, + f_t total_work) const { - if (!record_time) { return 0.0; } - return toc(start_time); + const f_t work_per_sec = t.time > 0.0 ? t.work / t.time : f_t(0); + settings.log.printf("%-15s %.2fs %4.1f%% (%.2e work %4.1f%% %.2e/s)\n", + name, + t.time, + total_time > 0.0 ? 100.0 * t.time / total_time : 0.0, + t.work, + total_work > 0.0 ? 100.0 * t.work / total_work : 0.0, + work_per_sec); } void print_timers(const simplex_solver_settings_t& settings) const { if (!record_time) { return; } - const f_t total_time = bfrt_time + pricing_time + btran_time + ftran_time + flip_time + - delta_z_time + lu_update_time + lu_factorization_time + se_norms_time + - se_entering_time + perturb_time + vector_time + objective_time + - update_infeasibility_time; + const f_t total_time = bfrt_time.time + pricing_time.time + btran_time.time + ftran_time.time + + flip_time.time + delta_z_time.time + lu_update_time.time + + lu_factorization_time.time + se_norms_time.time + se_entering_time.time + + perturb_time.time + vector_time.time + objective_time.time + + update_infeasibility_time.time; + const f_t total_work = bfrt_time.work + pricing_time.work + btran_time.work + ftran_time.work + + flip_time.work + delta_z_time.work + lu_update_time.work + + lu_factorization_time.work + se_norms_time.work + se_entering_time.work + + perturb_time.work + vector_time.work + objective_time.work + + update_infeasibility_time.work; // clang-format off - settings.log.printf("BFRT time %.2fs %4.1f%\n", bfrt_time, 100.0 * bfrt_time / total_time); - settings.log.printf("Pricing time %.2fs %4.1f%\n", pricing_time, 100.0 * pricing_time / total_time); - settings.log.printf("BTran time %.2fs %4.1f%\n", btran_time, 100.0 * btran_time / total_time); - settings.log.printf("FTran time %.2fs %4.1f%\n", ftran_time, 100.0 * ftran_time / total_time); - settings.log.printf("Flip time %.2fs %4.1f%\n", flip_time, 100.0 * flip_time / total_time); - settings.log.printf("Delta_z time %.2fs %4.1f%\n", delta_z_time, 100.0 * delta_z_time / total_time); - settings.log.printf("LU update time %.2fs %4.1f%\n", lu_update_time, 100.0 * lu_update_time / total_time); - settings.log.printf("LU factor time %.2fs %4.1f%\n", lu_factorization_time, 100.0 * lu_factorization_time / total_time); - settings.log.printf("SE norms time %.2fs %4.1f%\n", se_norms_time, 100.0 * se_norms_time / total_time); - settings.log.printf("SE enter time %.2fs %4.1f%\n", se_entering_time, 100.0 * se_entering_time / total_time); - settings.log.printf("Perturb time %.2fs %4.1f%\n", perturb_time, 100.0 * perturb_time / total_time); - settings.log.printf("Vector time %.2fs %4.1f%\n", vector_time, 100.0 * vector_time / total_time); - settings.log.printf("Objective time %.2fs %4.1f%\n", objective_time, 100.0 * objective_time / total_time); - settings.log.printf("Inf update time %.2fs %4.1f%\n", update_infeasibility_time, 100.0 * update_infeasibility_time / total_time); - settings.log.printf("Sum %.2fs\n", total_time); + print_one(settings, "BFRT time", bfrt_time, total_time, total_work); + if (bfrt_time.time > 0.1) { + settings.log.printf(" BFRT breakpoints: %.2fs\n", bfrt_breakpoints_time); + settings.log.printf(" BFRT single_pass: %.2fs\n", bfrt_single_pass_time); + settings.log.printf(" BFRT coarse: %.2fs\n", bfrt_coarse_time); + settings.log.printf(" BFRT bucket: %.2fs\n", bfrt_bucket_time); + settings.log.printf(" BFRT select: %.2fs\n", bfrt_select_time); + } + if (bfrt_calls > 0) { + settings.log.printf(" BFRT calls: %d, zero_steps: %d (%.1f%%), single_pass_only: %d, bucket_used: %d, not_last_bucket: %d, fallback: %d\n", + bfrt_calls, bfrt_zero_steps, 100.0 * bfrt_zero_steps / bfrt_calls, + bfrt_single_pass_only, bfrt_bucket_used, bfrt_not_last_bucket, bfrt_fallback); + settings.log.printf(" BFRT slope_breaker: %d, not_slope_breaker: %d (%.1f%%)\n", + bfrt_selected_slope_breaker, bfrt_not_slope_breaker, + bfrt_bucket_used > 0 ? 100.0 * bfrt_not_slope_breaker / bfrt_bucket_used : 0.0); + if (bfrt_zero_steps > 0) { + settings.log.printf(" BFRT zero-step avg: num_buckets=%.1f, bucket0_size=%.1f, num_breakpoints=%.1f, harris_zero=%.1f, exact_zero=%.1f\n", + 1.0 * bfrt_zero_step_num_buckets_sum / bfrt_zero_steps, + 1.0 * bfrt_zero_step_bucket0_sum / bfrt_zero_steps, + 1.0 * bfrt_zero_step_num_breakpoints_sum / bfrt_zero_steps, + 1.0 * bfrt_zero_step_harris_zero_sum / bfrt_zero_steps, + 1.0 * bfrt_zero_step_exact_zero_sum / bfrt_zero_steps); + } + } + print_one(settings, "Pricing time", pricing_time, total_time, total_work); + print_one(settings, "BTran time", btran_time, total_time, total_work); + print_one(settings, "FTran time", ftran_time, total_time, total_work); + print_one(settings, "Flip time", flip_time, total_time, total_work); + print_one(settings, "Delta_z time", delta_z_time, total_time, total_work); + print_one(settings, "LU update time", lu_update_time, total_time, total_work); + print_one(settings, "LU factor time", lu_factorization_time, total_time, total_work); + print_one(settings, "SE norms time", se_norms_time, total_time, total_work); + print_one(settings, "SE enter time", se_entering_time, total_time, total_work); + print_one(settings, "Perturb time", perturb_time, total_time, total_work); + print_one(settings, "Vector time", vector_time, total_time, total_work); + print_one(settings, "Objective time", objective_time, total_time, total_work); + print_one(settings, "Inf update time", update_infeasibility_time, total_time, total_work); + settings.log.printf("Sum %.2fs (%.2e work %.2e/s)\n", + total_time, + total_work, + total_time > 0.0 ? total_work / total_time : f_t(0)); // clang-format on } - f_t bfrt_time; - f_t pricing_time; - f_t btran_time; - f_t ftran_time; - f_t flip_time; - f_t delta_z_time; - f_t se_norms_time; - f_t se_entering_time; - f_t lu_update_time; - f_t lu_factorization_time; - f_t perturb_time; - f_t vector_time; - f_t objective_time; - f_t update_infeasibility_time; + work_timer_t bfrt_time; + f_t bfrt_breakpoints_time{0.0}; + f_t bfrt_single_pass_time{0.0}; + f_t bfrt_coarse_time{0.0}; + f_t bfrt_bucket_time{0.0}; + f_t bfrt_select_time{0.0}; + // BFRT diagnostic counters + i_t bfrt_calls{0}; + i_t bfrt_zero_steps{0}; // step_length == 0 + i_t bfrt_single_pass_only{0}; // no bound flips (single_pass decided) + i_t bfrt_bucket_used{0}; // bucket sort was used + i_t bfrt_not_last_bucket{0}; // selected from a bucket other than the last + i_t bfrt_fallback{0}; // fell back to single_pass result after bucket sort + i_t bfrt_zero_step_num_buckets_sum{0}; // sum of num_buckets on zero-step iters + i_t bfrt_zero_step_bucket0_sum{0}; // sum of bucket0 size on zero-step iters + i_t bfrt_zero_step_num_breakpoints_sum{0}; // sum of num_breakpoints on zero-step iters + i_t bfrt_zero_step_harris_zero_sum{0}; // sum of harris_ratios==0 on zero-step iters + i_t bfrt_zero_step_exact_zero_sum{0}; // sum of exact ratios==0 on zero-step iters + i_t bfrt_selected_slope_breaker{0}; // times we selected the slope breaker + i_t bfrt_not_slope_breaker{0}; // times we selected something else + work_timer_t pricing_time; + work_timer_t btran_time; + work_timer_t ftran_time; + work_timer_t flip_time; + work_timer_t delta_z_time; + work_timer_t se_norms_time; + work_timer_t se_entering_time; + work_timer_t lu_update_time; + work_timer_t lu_factorization_time; + work_timer_t perturb_time; + work_timer_t vector_time; + work_timer_t objective_time; + work_timer_t update_infeasibility_time; private: f_t start_time; + f_t start_work; bool record_time; }; @@ -2503,6 +2899,7 @@ dual_status_t dual_phase2(i_t phase, std::vector& vstatus, lp_solution_t& sol, i_t& iter, + f_t& work_estimate, std::vector& delta_y_steepest_edge, work_limit_context_t* work_unit_context) { @@ -2525,6 +2922,7 @@ dual_status_t dual_phase2(i_t phase, nonbasic_list, sol, iter, + work_estimate, delta_y_steepest_edge, work_unit_context); } @@ -2542,6 +2940,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, std::vector& nonbasic_list, lp_solution_t& sol, i_t& iter, + f_t& phase2_work_estimate, std::vector& delta_y_steepest_edge, work_limit_context_t* work_unit_context) { @@ -2556,7 +2955,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, assert(lp.lower.size() == n); assert(lp.upper.size() == n); assert(lp.rhs.size() == m); - f_t phase2_work_estimate = 0.0; ft.clear_work_estimate(); std::vector& x = sol.x; @@ -2639,8 +3037,145 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, assert(dual_res_norm < 1e-3); #endif - phase2::set_primal_variables_on_bounds(lp, settings, z, vstatus, x); - phase2_work_estimate += 5 * (n - m); + // Count degenerate NONBASIC_FIXED variables before bound assignment + i_t num_degen = 0; + { + i_t num_fixed = 0; + for (i_t j = 0; j < n; ++j) { + if (vstatus[j] == variable_status_t::NONBASIC_FIXED) { + if (std::abs(lp.lower[j] - lp.upper[j]) >= settings.fixed_tol) { + num_fixed++; + if (std::abs(z[j]) < settings.dual_tol) num_degen++; + } + } + } + settings.log.printf( + "NONBASIC_FIXED boxed: %d, degenerate (|z_j| < dual_tol): %d\n", num_fixed, num_degen); + } + + // Try 3 strategies for degenerate bound assignment, pick best + f_t best_sum_infeas = inf; + i_t best_num_infeas = m; + i_t best_degen_type = 0; + std::vector best_vstatus; + std::vector best_x; + const char* degen_names[] = {"default", "column-sum", "abs-bound"}; + const i_t degen_types[] = {0, 1, 3}; + f_t all_sum_infeas[3]; + i_t all_num_infeas[3]; + + for (i_t di = 0; di < 3; di++) { + const i_t dt = degen_types[di]; + std::vector try_vstatus = vstatus; + std::vector try_x = x; + phase2::set_primal_variables_on_bounds(lp, settings, z, try_vstatus, try_x, dt); + phase2::compute_primal_variables(ft, + lp.rhs, + lp.A, + basic_list, + nonbasic_list, + settings.tight_tol, + try_x, + xB_workspace, + phase2_work_estimate); + f_t sum_infeas = 0.0; + i_t num_infeas = 0; + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + f_t infeas = std::max(lp.lower[j] - try_x[j], try_x[j] - lp.upper[j]); + if (infeas > 0.0) { + sum_infeas += infeas; + num_infeas++; + } + } + all_sum_infeas[di] = sum_infeas; + all_num_infeas[di] = num_infeas; + if (di == 0) { + // Default is the baseline + best_sum_infeas = sum_infeas; + best_num_infeas = num_infeas; + best_degen_type = 0; + best_vstatus = try_vstatus; + best_x = try_x; + } else { + // Only pick alternative if BOTH fewer infeasibilities AND lower sum + if (num_infeas <= best_num_infeas && sum_infeas < best_sum_infeas) { + best_sum_infeas = sum_infeas; + best_num_infeas = num_infeas; + best_degen_type = di; + best_vstatus = try_vstatus; + best_x = try_x; + } + } + if (phase == 1 || num_degen == 0) { + for (i_t t = 1; t < 3; t++) { + all_sum_infeas[t] = sum_infeas; + all_num_infeas[t] = num_infeas; + } + break; + } + } + vstatus = best_vstatus; + x = best_x; + settings.log.printf( + "Bound assignment: default(%d/%.2e) colsum(%d/%.2e) abs-bound(%d/%.2e) -> %s\n", + all_num_infeas[0], + all_sum_infeas[0], + all_num_infeas[1], + all_sum_infeas[1], + all_num_infeas[2], + all_sum_infeas[2], + degen_names[best_degen_type]); + phase2_work_estimate += 15 * (n - m); + + // Near-optimality check: decide whether to apply initial perturbation + if (settings.initial_perturbation != 0 && phase == 2) { + i_t num_primal_infeas = 0; + f_t max_primal_infeas = 0.0; + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + f_t infeas = std::max(lp.lower[j] - x[j], x[j] - lp.upper[j]); + if (infeas > settings.primal_tol) { + num_primal_infeas++; + max_primal_infeas = std::max(max_primal_infeas, infeas); + } + } + bool near_optimal = (num_primal_infeas < 1000 && max_primal_infeas < 1e-3); + bool apply_perturbation = (settings.initial_perturbation == 1) || !near_optimal; + settings.log.printf( + "Near-optimal check: num_primal_infeas=%d, max_primal_infeas=%.2e, near_optimal=%d, " + "apply_perturbation=%d\n", + num_primal_infeas, + max_primal_infeas, + near_optimal, + apply_perturbation); + if (apply_perturbation) { + const bool strongly_degenerate = num_degen > n / 20; + phase2::initial_perturbation(lp, settings, vstatus, strongly_degenerate, objective); + // Recompute y, z with perturbed objective + for (i_t k = 0; k < m; ++k) { + c_basic[k] = objective[basic_list[k]]; + } + phase2_work_estimate += 3 * m; + ft.b_transpose_solve(c_basic, y); + phase2::compute_reduced_costs( + objective, lp.A, y, basic_list, nonbasic_list, z, phase2_work_estimate); + // Reassign bounds based on perturbed z (breaks degeneracy) + i_t num_bound_changes2 = phase2::set_primal_variables_on_bounds(lp, settings, z, vstatus, x); + phase2_work_estimate += 5 * (n - m); + if (num_bound_changes2 > 0) { + phase2::compute_primal_variables(ft, + lp.rhs, + lp.A, + basic_list, + nonbasic_list, + settings.tight_tol, + x, + xB_workspace, + phase2_work_estimate); + } + } + } #ifdef PRINT_VSTATUS_CHANGES i_t num_vstatus_changes; @@ -2664,16 +3199,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } phase2_work_estimate += 3 * n; - phase2::compute_primal_variables(ft, - lp.rhs, - lp.A, - basic_list, - nonbasic_list, - settings.tight_tol, - x, - xB_workspace, - phase2_work_estimate); - if (toc(start_time) > settings.time_limit) { return dual_status_t::TIME_LIMIT; } if (print_norms) { settings.log.printf("|| x || %e\n", vector_norm2(x)); } @@ -2792,8 +3317,9 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t dense_delta_z = 0; i_t num_refactors = 0; i_t total_bound_flips = 0; + i_t max_bound_flips = 0; f_t delta_y_nz_percentage = 0.0; - phase2::phase2_timers_t timers(false); + phase2::phase2_timers_t timers(true); // Sparse vectors for main loop (declared outside loop for instrumentation) sparse_vector_t delta_y_sparse(m, 0); @@ -2811,10 +3337,11 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += ft.work_estimate(); ft.clear_work_estimate(); + f_t last_work_reported = 0.0; if (work_unit_context) { work_unit_context->record_work_sync_on_horizon((phase2_work_estimate) / 1e8); + last_work_reported = phase2_work_estimate; } - phase2_work_estimate = 0.0; if (phase == 2) { settings.log.printf("%5d %+.16e %7d %.8e %.2e %.2f\n", @@ -2835,7 +3362,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t basic_leaving_index = -1; i_t leaving_index = -1; f_t max_val; - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); { PHASE2_NVTX_RANGE("DualSimplex::pricing"); if (settings.use_steepest_edge_pricing) { @@ -2856,7 +3383,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, lp, settings, x, basic_list, direction, basic_leaving_index, primal_infeasibility); } } - timers.pricing_time += timers.stop_timer(); + timers.pricing_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); if (leaving_index == -1) { #ifdef CHECK_BASIS_UPDATE for (i_t k = 0; k < basic_list.size(); k++) { @@ -2969,6 +3496,63 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } } + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); + + // Before declaring optimal, attempt to remove perturbation. + if (phase == 2) { + i_t removal_status = phase2::attempt_to_remove_perturbations(lp, + settings, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + z, + y, + x, + xB_workspace, + squared_infeasibilities, + infeasibility_indices, + primal_infeasibility, + primal_infeasibility_squared, + phase2_work_estimate); + if (removal_status == 1) { // CONTINUE_DUAL + obj = phase2::compute_perturbed_objective(objective, x); + phase2_work_estimate += 2 * n; + continue; + } + if (removal_status == 2) { // PRIMAL_CLEANUP + const f_t perturbation = phase2::amount_of_perturbation(lp, objective); + settings.log.printf("Failed to remove perturbation of %.2e.\n", perturbation); + settings.log.printf("Num updates: %d\n", ft.num_updates()); + settings.log.printf("Iterations: %d\n", iter); + i_t dual_iter = iter; + primal_status_t primal_status = primal_phase2_with_advanced_basis(2, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + phase2_work_estimate, + false); + if (primal_status == primal_status_t::OPTIMAL) { + settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); + objective = lp.objective; + } else { + settings.log.printf("Primal cleanup failed.\n"); + const f_t dual_infeas = phase2::dual_infeasibility( + lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); + if (dual_infeas > 10.0 * settings.dual_tol) { return dual_status_t::NUMERICAL; } + } + } + // removal_status == 0 (OPTIMAL) or primal cleanup done: fall through to prepare_optimality + } + phase2::prepare_optimality(0, primal_infeasibility, lp, @@ -2981,6 +3565,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase, start_time, max_val, + phase2_work_estimate, iter, x, y, @@ -2998,7 +3583,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, // BTran // BT*delta_y = -delta_zB = -sigma*ei - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); delta_y_sparse.clear(); UTsol_sparse.clear(); f_t btran_start_work = ft.work_estimate(); @@ -3006,7 +3591,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, PHASE2_NVTX_RANGE("DualSimplex::btran"); phase2::compute_delta_y(ft, basic_leaving_index, direction, delta_y_sparse, UTsol_sparse); } - timers.btran_time += timers.stop_timer(); + timers.btran_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); solve_work += (ft.work_estimate() - btran_start_work); if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { @@ -3030,7 +3615,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, continue; } - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); i_t delta_y_nz0 = 0; const i_t nz_delta_y = delta_y_sparse.i.size(); for (i_t k = 0; k < nz_delta_y; k++) { @@ -3069,7 +3654,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate); } } - timers.delta_z_time += timers.stop_timer(); + timers.delta_z_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { return dual_status_t::CONCURRENT_LIMIT; } @@ -3090,6 +3675,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, f_t step_length; i_t entering_index = -1; i_t nonbasic_entering_index = -1; + std::vector flip_indices; const bool harris_ratio = settings.use_harris_ratio; const bool bound_flip_ratio = settings.use_bound_flip_ratio; { @@ -3105,7 +3691,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, step_length, nonbasic_entering_index); } else if (bound_flip_ratio) { - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); f_t slope = direction == 1 ? (lp.lower[leaving_index] - x[leaving_index]) : (x[leaving_index] - lp.upper[leaving_index]); bound_flipping_ratio_test_t bfrt(settings, @@ -3122,13 +3708,44 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, delta_z, delta_z_indices, nonbasic_mark); - entering_index = bfrt.compute_step_length(step_length, nonbasic_entering_index); + entering_index = + bfrt.compute_step_length(step_length, nonbasic_entering_index, flip_indices); phase2_work_estimate += bfrt.work_estimate(); if (entering_index == RATIO_TEST_NUMERICAL_ISSUES) { settings.log.printf("Numerical issues encountered in ratio test.\n"); return dual_status_t::NUMERICAL; } - timers.bfrt_time += timers.stop_timer(); + timers.bfrt_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); + timers.bfrt_breakpoints_time += bfrt.time_compute_breakpoints_; + timers.bfrt_single_pass_time += bfrt.time_single_pass_; + timers.bfrt_coarse_time += bfrt.time_coarse_filter_; + timers.bfrt_bucket_time += bfrt.time_bucket_sort_; + timers.bfrt_select_time += bfrt.time_pivot_selection_; + // BFRT diagnostics + timers.bfrt_calls++; + if (step_length == 0.0) { + timers.bfrt_zero_steps++; + timers.bfrt_zero_step_num_buckets_sum += bfrt.num_buckets_used_; + timers.bfrt_zero_step_bucket0_sum += bfrt.bucket0_size_; + timers.bfrt_zero_step_num_breakpoints_sum += bfrt.num_breakpoints_; + timers.bfrt_zero_step_harris_zero_sum += bfrt.num_harris_zero_; + timers.bfrt_zero_step_exact_zero_sum += bfrt.num_exact_zero_; + } + if (bfrt.num_buckets_used_ == 0) { + timers.bfrt_single_pass_only++; + } else { + timers.bfrt_bucket_used++; + if (bfrt.used_fallback_) { + timers.bfrt_fallback++; + } else if (bfrt.bucket_selected_ < bfrt.num_buckets_used_ - 1) { + timers.bfrt_not_last_bucket++; + } + if (bfrt.selected_is_slope_breaker_) { + timers.bfrt_selected_slope_breaker++; + } else { + timers.bfrt_not_slope_breaker++; + } + } } else { entering_index = phase2::phase2_ratio_test( lp, settings, vstatus, nonbasic_list, z, delta_z, step_length, nonbasic_entering_index); @@ -3143,131 +3760,63 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += 2 * n; if (perturbation > 0.0 && phase == 2) { - // Try to remove perturbation - std::vector unperturbed_y(m); - std::vector unperturbed_z(n); - phase2_work_estimate += m + n; - phase2::compute_dual_solution_from_basis( - lp, ft, basic_list, nonbasic_list, unperturbed_y, unperturbed_z, phase2_work_estimate); - { - const f_t dual_infeas = phase2::dual_infeasibility( - lp, settings, vstatus, unperturbed_z, settings.tight_tol, settings.dual_tol); - phase2_work_estimate += 3 * n; - settings.log.printf("Dual infeasibility after removing perturbation %e\n", dual_infeas); - if (dual_infeas <= settings.dual_tol) { - settings.log.printf("Removed perturbation of %.2e.\n", perturbation); - z = unperturbed_z; - y = unperturbed_y; - phase2_work_estimate += 2 * n + 2 * m; - perturbation = 0.0; - - std::vector unperturbed_x(n); - phase2_work_estimate += n; - phase2::compute_primal_solution_from_basis(lp, - ft, - basic_list, - nonbasic_list, - vstatus, - unperturbed_x, - xB_workspace, - phase2_work_estimate); - x = unperturbed_x; - primal_infeasibility_squared = - phase2::compute_initial_primal_infeasibilities(lp, - settings, - basic_list, - x, - squared_infeasibilities, - infeasibility_indices, - primal_infeasibility); - phase2_work_estimate += 4 * m + 2 * n; - settings.log.printf("Updated primal infeasibility: %e\n", primal_infeasibility); - - objective = lp.objective; - phase2_work_estimate += 2 * n; - // Need to reset the objective value, since we have recomputed x - obj = phase2::compute_perturbed_objective(objective, x); - phase2_work_estimate += 2 * n; - if (dual_infeas <= settings.dual_tol && primal_infeasibility <= settings.primal_tol) { - phase2::prepare_optimality(1, - primal_infeasibility, - lp, - settings, - ft, - objective, - basic_list, - nonbasic_list, - vstatus, - phase, - start_time, - max_val, - iter, - x, - y, - z, - sol); - status = dual_status_t::OPTIMAL; - break; - } - settings.log.printf( - "Continuing with perturbation removed and steepest edge norms reset\n"); - // Clear delta_z before restarting the iteration - phase2_work_estimate += 3 * delta_z_indices.size(); - phase2::clear_delta_z( - entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); - continue; - } else { - std::vector unperturbed_x(n); - phase2_work_estimate += n; - phase2::compute_primal_solution_from_basis(lp, - ft, - basic_list, - nonbasic_list, - vstatus, - unperturbed_x, - xB_workspace, - phase2_work_estimate); - x = unperturbed_x; - phase2_work_estimate += 2 * n; - primal_infeasibility_squared = - phase2::compute_initial_primal_infeasibilities(lp, - settings, - basic_list, - x, - squared_infeasibilities, - infeasibility_indices, - primal_infeasibility); - phase2_work_estimate += 4 * m + 2 * n; - - const f_t orig_dual_infeas = phase2::dual_infeasibility( - lp, settings, vstatus, z, settings.tight_tol, settings.dual_tol); - phase2_work_estimate += 3 * n; - - if (primal_infeasibility <= settings.primal_tol && - orig_dual_infeas <= settings.dual_tol) { - phase2::prepare_optimality(2, - primal_infeasibility, - lp, - settings, - ft, - objective, - basic_list, - nonbasic_list, - vstatus, - phase, - start_time, - max_val, - iter, - x, - y, - z, - sol); - status = dual_status_t::OPTIMAL; - break; - } - settings.log.printf("Failed to remove perturbation of %.2e.\n", perturbation); + i_t removal_status = phase2::attempt_to_remove_perturbations(lp, + settings, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + z, + y, + x, + xB_workspace, + squared_infeasibilities, + infeasibility_indices, + primal_infeasibility, + primal_infeasibility_squared, + phase2_work_estimate); + if (removal_status == 0) { // OPTIMAL + obj = phase2::compute_perturbed_objective(objective, x); + phase2_work_estimate += 2 * n; + if (primal_infeasibility <= settings.primal_tol) { + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); + phase2::prepare_optimality(1, + primal_infeasibility, + lp, + settings, + ft, + objective, + basic_list, + nonbasic_list, + vstatus, + phase, + start_time, + max_val, + phase2_work_estimate, + iter, + x, + y, + z, + sol); + status = dual_status_t::OPTIMAL; + break; } + settings.log.printf("Continuing with perturbation removed\n"); + phase2_work_estimate += 3 * delta_z_indices.size(); + phase2::clear_delta_z( + entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); + continue; + } else if (removal_status == 1) { // CONTINUE_DUAL + obj = phase2::compute_perturbed_objective(objective, x); + phase2_work_estimate += 2 * n; + phase2_work_estimate += 3 * delta_z_indices.size(); + phase2::clear_delta_z( + entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); + continue; } + // removal_status == 2 (PRIMAL_CLEANUP): fall through to existing logic below } if (perturbation == 0.0 && phase == 2) { @@ -3317,7 +3866,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, return dual_status_t::DUAL_UNBOUNDED; } - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // Update dual variables // y <- y + steplength * delta_y // z <- z + steplength * delta_z @@ -3333,7 +3882,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, settings.log.printf("Numerical issues encountered in update_dual_variables.\n"); return dual_status_t::NUMERICAL; } - timers.vector_time += timers.stop_timer(); + timers.vector_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); #ifdef COMPUTE_DUAL_RESIDUAL std::vector dual_res1; @@ -3344,29 +3893,26 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } #endif - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // Update primal variable - const i_t num_flipped = phase2::flip_bounds(lp, - settings, - bounded_variables, - objective, - z, - delta_z_indices, - nonbasic_list, - entering_index, - vstatus, - delta_x_flip, - atilde_mark, - atilde, - atilde_index, - phase2_work_estimate); - - timers.flip_time += timers.stop_timer(); + const i_t num_flipped = bound_flip_ratio ? phase2::flip_bounds(lp, + bounded_variables, + flip_indices, + vstatus, + delta_x_flip, + atilde_mark, + atilde, + atilde_index, + phase2_work_estimate) + : 0; + + timers.flip_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); total_bound_flips += num_flipped; + if (num_flipped > max_bound_flips) max_bound_flips = num_flipped; delta_xB_0_sparse.clear(); if (num_flipped > 0) { - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); phase2::adjust_for_flips(ft, basic_list, delta_z_indices, @@ -3378,10 +3924,10 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, delta_x_flip, x, phase2_work_estimate); - timers.ftran_time += timers.stop_timer(); + timers.ftran_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); } - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); utilde_sparse.clear(); scaled_delta_xB_sparse.clear(); rhs_sparse.from_csc_column(lp.A, entering_index); @@ -3408,7 +3954,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } } solve_work += (ft.work_estimate() - ftran_start_work); - timers.ftran_time += timers.stop_timer(); + timers.ftran_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { return dual_status_t::CONCURRENT_LIMIT; } @@ -3420,7 +3966,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (primal_step_err > 1e-4) { settings.log.printf("|| A * dx || %e\n", primal_step_err); } #endif - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); f_t se_norms_start_work = ft.work_estimate(); const i_t steepest_edge_status = phase2::update_steepest_edge_norms(settings, basic_list, @@ -3442,18 +3988,18 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } #endif assert(steepest_edge_status == 0); - timers.se_norms_time += timers.stop_timer(); + timers.se_norms_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); solve_work += (ft.work_estimate() - se_norms_start_work); if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { return dual_status_t::CONCURRENT_LIMIT; } - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // x <- x + delta_x phase2::update_primal_variables( scaled_delta_xB_sparse, basic_list, delta_x, entering_index, x, phase2_work_estimate); - timers.vector_time += timers.stop_timer(); + timers.vector_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); #ifdef COMPUTE_PRIMAL_RESIDUAL residual = lp.rhs; @@ -3464,7 +4010,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } #endif - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // TODO(CMM): Do I also need to update the objective due to the bound flips? // TODO(CMM): I'm using the unperturbed objective here, should this be the perturbed objective? phase2::update_objective(basic_list, @@ -3474,9 +4020,9 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, entering_index, obj, phase2_work_estimate); - timers.objective_time += timers.stop_timer(); + timers.objective_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // Update primal infeasibilities due to changes in basic variables // from flipping bounds #ifdef CHECK_BASIC_INFEASIBILITIES @@ -3529,17 +4075,29 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::check_primal_infeasibilities( lp, settings, basic_list, x, squared_infeasibilities, infeasibility_indices); #endif - timers.update_infeasibility_time += timers.stop_timer(); + timers.update_infeasibility_time += + timers.stop_timer(phase2_work_estimate + ft.work_estimate()); // Clear delta_x phase2::clear_delta_x( basic_list, entering_index, scaled_delta_xB_sparse, delta_x, phase2_work_estimate); - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); + if (settings.remove_perturbation != 0) { + phase2::remove_leaving_perturbation(lp, settings, leaving_index, direction, z, objective); + } f_t sum_perturb = 0.0; - phase2::compute_perturbation( - lp, settings, delta_z_indices, z, objective, sum_perturb, phase2_work_estimate); - timers.perturb_time += timers.stop_timer(); + phase2::compute_perturbation(lp, + settings, + delta_z_indices, + vstatus, + z, + objective, + sum_perturb, + entering_index, + step_length, + phase2_work_estimate); + timers.perturb_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); // Update basis information vstatus[entering_index] = variable_status_t::BASIC; @@ -3562,7 +4120,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::check_basic_infeasibilities(basic_list, basic_mark, infeasibility_indices, 5); #endif - timers.start_timer(); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); // Refactor or update the basis factorization { PHASE2_NVTX_RANGE("DualSimplex::basis_update"); @@ -3578,8 +4136,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::check_update(lp, settings, ft, basic_list, basic_leaving_index); #endif should_refactor = recommend_refactor == 1; - timers.lu_update_time += timers.stop_timer(); - timers.start_timer(); + timers.lu_update_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); + timers.start_timer(phase2_work_estimate + ft.work_estimate()); } #ifdef CHECK_BASIC_INFEASIBILITIES @@ -3657,7 +4215,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::check_basic_infeasibilities(basic_list, basic_mark, infeasibility_indices, 7); #endif } - timers.lu_factorization_time += timers.stop_timer(); + timers.lu_factorization_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); #ifdef STEEPEST_EDGE_DEBUG if (iter < 100 || iter % 100 == 0)) @@ -3676,16 +4234,19 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += 3 * delta_z_indices.size(); phase2::clear_delta_z(entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); + // Flush basis update work into the total work estimate every iteration + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); + f_t now = toc(start_time); // Feature logging for regression training (every FEATURE_LOG_INTERVAL iterations) if ((iter % FEATURE_LOG_INTERVAL) == 0 && work_unit_context) { [[maybe_unused]] i_t iters_elapsed = iter - last_feature_log_iter; - phase2_work_estimate += ft.work_estimate(); - ft.clear_work_estimate(); - work_unit_context->record_work_sync_on_horizon(phase2_work_estimate / 1e8); - phase2_work_estimate = 0.0; + work_unit_context->record_work_sync_on_horizon((phase2_work_estimate - last_work_reported) / + 1e8); + last_work_reported = phase2_work_estimate; last_feature_log_iter = iter; } @@ -3717,16 +4278,32 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, return dual_status_t::WORK_LIMIT; } - if (now > settings.time_limit) { return dual_status_t::TIME_LIMIT; } + if (now > settings.time_limit) { + status = dual_status_t::TIME_LIMIT; + break; + } if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { return dual_status_t::CONCURRENT_LIMIT; } } - if (iter >= iter_limit) { status = dual_status_t::ITERATION_LIMIT; } + if (status != dual_status_t::TIME_LIMIT && iter >= iter_limit) { + status = dual_status_t::ITERATION_LIMIT; + } + + // Flush any remaining work from the basis update into the total work estimate + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); if (phase == 2) { timers.print_timers(settings); + i_t num_iters = iter - start_iter; + if (num_iters > 0) { + settings.log.printf("Bound flips: total=%d, avg=%.1f, max=%d\n", + total_bound_flips, + 1.0 * total_bound_flips / num_iters, + max_bound_flips); + } constexpr bool print_stats = false; if constexpr (print_stats) { settings.log.printf("Sparse delta_z %8d %8.2f%\n", @@ -3737,10 +4314,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, 100.0 * dense_delta_z / (sparse_delta_z + dense_delta_z)); ft.print_stats(); } - if (settings.inside_mip == 1 && settings.concurrent_halt != nullptr) { - settings.log.debug("Setting concurrent halt in Dual Simplex Phase 2\n"); - *settings.concurrent_halt = 1; - } } return status; } @@ -3756,6 +4329,7 @@ template dual_status_t dual_phase2( std::vector& vstatus, lp_solution_t& sol, int& iter, + double& work_estimate, std::vector& steepest_edge_norms, work_limit_context_t* work_unit_context); @@ -3772,6 +4346,7 @@ template dual_status_t dual_phase2_with_advanced_basis( std::vector& nonbasic_list, lp_solution_t& sol, int& iter, + double& work_estimate, std::vector& steepest_edge_norms, work_limit_context_t* work_unit_context); diff --git a/cpp/src/dual_simplex/phase2.hpp b/cpp/src/dual_simplex/phase2.hpp index daa946e019..cfa46d8152 100644 --- a/cpp/src/dual_simplex/phase2.hpp +++ b/cpp/src/dual_simplex/phase2.hpp @@ -51,6 +51,19 @@ static std::string dual_status_to_string(dual_status_t status) return "UNKNOWN"; } +template +dual_status_t dual_phase2(i_t phase, + i_t slack_basis, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + lp_solution_t& sol, + i_t& iter, + f_t& work_estimate, + std::vector& steepest_edge_norms, + work_limit_context_t* work_unit_context = nullptr); + template dual_status_t dual_phase2(i_t phase, i_t slack_basis, @@ -61,7 +74,38 @@ dual_status_t dual_phase2(i_t phase, lp_solution_t& sol, i_t& iter, std::vector& steepest_edge_norms, - work_limit_context_t* work_unit_context = nullptr); + work_limit_context_t* work_unit_context = nullptr) +{ + f_t work_estimate = 0.0; + return dual_phase2(phase, + slack_basis, + start_time, + lp, + settings, + vstatus, + sol, + iter, + work_estimate, + steepest_edge_norms, + work_unit_context); +} + +template +dual_status_t dual_phase2_with_advanced_basis(i_t phase, + i_t slack_basis, + bool initialize_basis, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& ft, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + i_t& iter, + f_t& work_estimate, + std::vector& delta_y_steepest_edge, + work_limit_context_t* work_unit_context = nullptr); template dual_status_t dual_phase2_with_advanced_basis(i_t phase, @@ -77,7 +121,25 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, lp_solution_t& sol, i_t& iter, std::vector& delta_y_steepest_edge, - work_limit_context_t* work_unit_context = nullptr); + work_limit_context_t* work_unit_context = nullptr) +{ + f_t work_estimate = 0.0; + return dual_phase2_with_advanced_basis(phase, + slack_basis, + initialize_basis, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + work_estimate, + delta_y_steepest_edge, + work_unit_context); +} template void compute_reduced_cost_update(const lp_problem_t& lp, diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 78c7107ca3..1a67956e47 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -14,18 +14,128 @@ #include #include +#include + namespace cuopt::mathematical_optimization::simplex { +template +struct primal_work_timer_t { + primal_work_timer_t(f_t t) : time(t) {} + f_t time{0.0}; + f_t work{0.0}; +}; + +template +primal_work_timer_t& operator+=(primal_work_timer_t& lhs, + const primal_work_timer_t& rhs) +{ + lhs.time += rhs.time; + lhs.work += rhs.work; + return lhs; +} + +template +class primal_timers_t { + public: + primal_timers_t(bool should_time) + : record_time(should_time), + pricing_time(0), + ftran_time(0), + ratio_test_time(0), + btran_time(0), + delta_z_time(0), + update_duals_time(0), + lu_update_time(0), + lu_factorization_time(0), + update_x_time(0) + { + } + + void start_timer(f_t work) + { + if (!record_time) { return; } + start_time_ = tic(); + start_work_ = work; + } + + primal_work_timer_t stop_timer(f_t stop_work) + { + if (!record_time) { return primal_work_timer_t(0.0); } + primal_work_timer_t result(toc(start_time_)); + result.work = stop_work - start_work_; + return result; + } + + void print_one(const simplex_solver_settings_t& settings, + const char* name, + const primal_work_timer_t& t, + f_t total_time, + f_t total_work) const + { + const f_t work_per_sec = t.time > 0.0 ? t.work / t.time : f_t(0); + settings.log.printf("%-15s %.2fs %4.1f%% (%.2e work %4.1f%% %.2e/s)\n", + name, + t.time, + total_time > 0.0 ? 100.0 * t.time / total_time : 0.0, + t.work, + total_work > 0.0 ? 100.0 * t.work / total_work : 0.0, + work_per_sec); + } + + void print_timers(const simplex_solver_settings_t& settings) const + { + if (!record_time) { return; } + const f_t total_time = pricing_time.time + ftran_time.time + ratio_test_time.time + + btran_time.time + delta_z_time.time + update_duals_time.time + + lu_update_time.time + lu_factorization_time.time + update_x_time.time; + const f_t total_work = pricing_time.work + ftran_time.work + ratio_test_time.work + + btran_time.work + delta_z_time.work + update_duals_time.work + + lu_update_time.work + lu_factorization_time.work + update_x_time.work; + // clang-format off + print_one(settings, "Pricing time", pricing_time, total_time, total_work); + print_one(settings, "FTran time", ftran_time, total_time, total_work); + print_one(settings, "Ratio test", ratio_test_time, total_time, total_work); + print_one(settings, "BTran time", btran_time, total_time, total_work); + print_one(settings, "Delta_z time", delta_z_time, total_time, total_work); + print_one(settings, "Update duals", update_duals_time, total_time, total_work); + print_one(settings, "LU update time", lu_update_time, total_time, total_work); + print_one(settings, "LU factor time", lu_factorization_time, total_time, total_work); + print_one(settings, "Update x time", update_x_time, total_time, total_work); + settings.log.printf("Sum %.2fs (%.2e work %.2e/s)\n", + total_time, + total_work, + total_time > 0.0 ? total_work / total_time : f_t(0)); + // clang-format on + } + + primal_work_timer_t pricing_time; + primal_work_timer_t ftran_time; + primal_work_timer_t ratio_test_time; + primal_work_timer_t btran_time; + primal_work_timer_t delta_z_time; + primal_work_timer_t update_duals_time; + primal_work_timer_t lu_update_time; + primal_work_timer_t lu_factorization_time; + primal_work_timer_t update_x_time; + + private: + f_t start_time_; + f_t start_work_; + bool record_time; +}; + namespace { template void set_primal_variables_on_bounds(const lp_problem_t& lp, const simplex_solver_settings_t& settings, - const std::vector& z, std::vector& vstatus, - std::vector& x) + std::vector& x, + f_t& work_estimate) { - const i_t n = lp.num_cols; + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + constexpr f_t diff_tol = 1e-6; for (i_t j = 0; j < n; ++j) { if (vstatus[j] == variable_status_t::BASIC) { continue; } @@ -53,18 +163,20 @@ void set_primal_variables_on_bounds(const lp_problem_t& lp, assert(1 == 0); } } + work_estimate += n + 3.0 * (n - m); } template f_t dual_infeasibility(const lp_problem_t& lp, const std::vector& vstatus, - const std::vector& z) + const std::vector& z, + f_t tight_tol, + i_t& num_infeasible, + f_t& work_estimate) { const i_t n = lp.num_cols; - const i_t m = lp.num_rows; - i_t num_infeasible = 0; + num_infeasible = 0; f_t sum_infeasible = 0.0; - constexpr f_t tight_tol = 0; i_t lower_bound_inf = 0; i_t upper_bound_inf = 0; i_t free_inf = 0; @@ -102,6 +214,7 @@ f_t dual_infeasibility(const lp_problem_t& lp, non_basic_upper_inf++; } } + work_estimate += 8 * n; return sum_infeasible; } @@ -111,9 +224,11 @@ i_t phase2_pricing(const lp_problem_t& lp, const std::vector& z, const std::vector& nonbasic_list, const std::vector& vstatus, + f_t dual_tol, i_t& direction, i_t& basic_entering, - f_t& dual_inf) + f_t& dual_inf, + f_t& work_estimate) { const i_t m = lp.num_rows; const i_t n = lp.num_cols; @@ -121,8 +236,7 @@ i_t phase2_pricing(const lp_problem_t& lp, f_t max_infeas = 0.0; dual_inf = 0.0; for (i_t k = 0; k < n - m; ++k) { - const i_t j = nonbasic_list[k]; - constexpr f_t dual_tol = 1e-6; + const i_t j = nonbasic_list[k]; if (vstatus[j] == variable_status_t::NONBASIC_FIXED) { continue; } if ((vstatus[j] == variable_status_t::NONBASIC_LOWER || vstatus[j] == variable_status_t::NONBASIC_FREE) && @@ -148,69 +262,78 @@ i_t phase2_pricing(const lp_problem_t& lp, } } } + work_estimate += 5 * (n - m); return entering_index; } template -i_t ratio_test(const lp_problem_t& lp, - const std::vector& vstatus, - const std::vector& basic_list, - std::vector& x, - std::vector& delta_x, - f_t& step_length, - i_t& basic_leaving) +i_t devex_pricing(const lp_problem_t& lp, + const std::vector& z, + const std::vector& devex_weight, + const std::vector& nonbasic_list, + const std::vector& vstatus, + f_t dual_tol, + i_t& direction, + i_t& basic_entering, + f_t& dual_inf, + f_t& work_estimate) { - const i_t m = lp.num_rows; - const i_t n = lp.num_cols; - basic_leaving = -1; - i_t leaving_index = -1; - f_t min_val = inf; - constexpr f_t pivot_tol = 1e-8; - for (i_t k = 0; k < m; ++k) { - const i_t j = basic_list[k]; - if (delta_x[j] == 0.0) { continue; } - if (lp.lower[j] > -inf && x[j] >= lp.lower[j] && delta_x[j] < -pivot_tol) { - // xj + step * delta_x[j] >= lp.lower[j] - // step * delta_x[j] >= lp.lower[j] - x[j] - // step <= (lp.lower[j] - x[j]) / delta_x[j], delta_x[j] < 0 - const f_t neum = lp.lower[j] - x[j]; - f_t ratio = neum / delta_x[j]; - if (ratio < min_val) { - min_val = ratio; - basic_leaving = k; - leaving_index = j; - } + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + i_t entering_index = -1; + f_t max_score = 0.0; + dual_inf = 0.0; + for (i_t k = 0; k < n - m; ++k) { + const i_t j = nonbasic_list[k]; + if (vstatus[j] == variable_status_t::NONBASIC_FIXED) { continue; } + f_t infeas = 0.0; + i_t dir = 0; + if ((vstatus[j] == variable_status_t::NONBASIC_LOWER || + vstatus[j] == variable_status_t::NONBASIC_FREE) && + z[j] < -dual_tol) { + infeas = -z[j]; + dir = 1; + } else if ((vstatus[j] == variable_status_t::NONBASIC_UPPER || + vstatus[j] == variable_status_t::NONBASIC_FREE) && + z[j] > dual_tol) { + infeas = z[j]; + dir = -1; } - if (lp.upper[j] < inf && x[j] <= lp.upper[j] && delta_x[j] > pivot_tol) { - // xj + step * delta_x[j] <= lp.upper[j] - // step * delta_x[j] <= lp.upper[j] - x[j] - // step <= (lp.upper[j] - x[j]) / delta_x[j], delta_x[j] > 0 - const f_t neum = lp.upper[j] - x[j]; - f_t ratio = neum / delta_x[j]; - if (ratio < min_val) { - min_val = ratio; - basic_leaving = k; - leaving_index = j; + if (infeas > 0.0) { + dual_inf += infeas; + const f_t score = (infeas * infeas) / devex_weight[j]; + if (score > max_score) { + max_score = score; + basic_entering = k; + entering_index = j; + direction = dir; } } } - step_length = min_val; - return leaving_index; + work_estimate += 7 * (n - m); + return entering_index; } template f_t primal_infeasibility(const lp_problem_t& lp, const simplex_solver_settings_t& settings, const std::vector& vstatus, - const std::vector& x) + const std::vector& x, + i_t& num_infeasible, + f_t& work_estimate) { + const i_t m = lp.num_rows; const i_t n = lp.num_cols; f_t primal_inf = 0; + num_infeasible = 0; for (i_t j = 0; j < n; ++j) { - if (x[j] < lp.lower[j]) { + // Nonbasics are pinned to a bound; only basics can be (legitimately) infeasible. + if (vstatus[j] != variable_status_t::BASIC) { continue; } + if (x[j] < lp.lower[j] - settings.primal_tol) { // x_j < l_j => -x_j > -l_j => -x_j + l_j > 0 const f_t infeas = -x[j] + lp.lower[j]; primal_inf += infeas; + num_infeasible++; if (infeas > 1e-6) { settings.log.debug("x %d infeas %e lo %e val %e up %e vstatus %hhd\n", j, @@ -221,10 +344,11 @@ f_t primal_infeasibility(const lp_problem_t& lp, vstatus[j]); } } - if (x[j] > lp.upper[j]) { + if (x[j] > lp.upper[j] + settings.primal_tol) { // x_j > u_j => x_j - u_j > 0 const f_t infeas = x[j] - lp.upper[j]; primal_inf += infeas; + num_infeasible++; if (infeas > 1e-6) { settings.log.debug("x %d infeas %e lo %e val %e up %e vstatus %hhd\n", j, @@ -236,15 +360,366 @@ f_t primal_infeasibility(const lp_problem_t& lp, } } } + work_estimate += n + 4 * m; return primal_inf; } +template +f_t primal_infeasibility(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& vstatus, + const std::vector& x, + f_t& work_estimate) +{ + i_t num_infeasible = 0; + return primal_infeasibility(lp, settings, vstatus, x, num_infeasible, work_estimate); +} + +// work estimate: n-m + 4 * m +template +void compute_phase1_objective(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& vstatus, + const std::vector& x, + std::vector& objective, + f_t& work_estimate) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + for (i_t j = 0; j < n; ++j) { + if (vstatus[j] != variable_status_t::BASIC) { + objective[j] = 0.0; + } else if (x[j] < lp.lower[j] - settings.primal_tol) { + objective[j] = -1.0; + } else if (x[j] > lp.upper[j] + settings.primal_tol) { + objective[j] = 1.0; + } else { + objective[j] = 0.0; + } + } + work_estimate += n - m + 4 * m; +} + +template +void compute_delta_y(const basis_update_mpf_t& basis_update, + i_t basic_leaving, + sparse_vector_t& delta_y, + sparse_vector_t& etilde) +{ + const i_t m = delta_y.n; + sparse_vector_t ei(m, 1); + ei.i[0] = basic_leaving; + ei.x[0] = 1.0; + delta_y.clear(); + etilde.clear(); + basis_update.b_transpose_solve(ei, delta_y, etilde); +} + +template +void compute_delta_z(const csr_matrix_t& Arow, + const std::vector& vstatus, + const sparse_vector_t& delta_y, + std::vector& delta_z, + f_t& work_estimate) +{ + // A^T delta_y + delta_z = 0 + // delta_z = -A^T delta_y = - sum_i A(i, :) * delta_y_i + std::fill(delta_z.begin(), delta_z.end(), 0.0); + work_estimate += delta_z.size(); + for (i_t k = 0; k < static_cast(delta_y.i.size()); ++k) { + const i_t i = delta_y.i[k]; + const f_t delta_y_i = delta_y.x[k]; + const i_t row_start = Arow.row_start[i]; + const i_t row_end = Arow.row_start[i + 1]; + for (i_t p = row_start; p < row_end; ++p) { + const i_t j = Arow.j[p]; + if (vstatus[j] != variable_status_t::BASIC) { delta_z[j] -= Arow.x[p] * delta_y_i; } + } + work_estimate += 5 * (row_end - row_start); + } + work_estimate += 4 * delta_y.i.size(); +} + +template +f_t compute_dual_step_length(f_t entering_reduced_cost, f_t pivot) +{ + assert(pivot != 0.0); + return entering_reduced_cost / pivot; +} + +template +void update_y(f_t dual_step_length, + const sparse_vector_t& delta_y, + std::vector& y, + f_t& work_estimate) +{ + for (i_t k = 0; k < static_cast(delta_y.i.size()); ++k) { + const i_t i = delta_y.i[k]; + y[i] += dual_step_length * delta_y.x[k]; + } + work_estimate += 3 * delta_y.i.size(); +} + +template +void update_z(f_t dual_step_length, + const std::vector& nonbasic_list, + i_t entering_index, + const std::vector& delta_z, + std::vector& z, + f_t& work_estimate) +{ + for (i_t k = 0; k < static_cast(nonbasic_list.size()); ++k) { + const i_t j = nonbasic_list[k]; + z[j] += dual_step_length * delta_z[j]; + } + work_estimate += 3 * nonbasic_list.size(); + z[entering_index] = 0.0; +} + +template +void compute_dual_variables(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& objective, + const std::vector& basic_list, + const std::vector& nonbasic_list, + basis_update_mpf_t& ft, + std::vector& c_basic, + std::vector& y, + std::vector& z, + f_t& work_estimate) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + // Solve for y such that B'*y = c_B + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + c_basic[k] = objective[j]; + } + work_estimate += 3 * m; + ft.b_transpose_solve(c_basic, y); + // zN = cN - N'*y + for (i_t k = 0; k < n - m; k++) { + const i_t j = nonbasic_list[k]; + // z_j <- c_j + z[j] = objective[j]; + + // z_j <- z_j - A(:, j)'*y + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + f_t dot = 0.0; + for (i_t p = col_start; p < col_end; ++p) { + dot += lp.A.x[p] * y[lp.A.i[p]]; + } + work_estimate += 3.0 * (col_end - col_start); + z[j] -= dot; + } + work_estimate += 6 * (n - m); + // zB = 0 + for (i_t k = 0; k < m; ++k) { + z[basic_list[k]] = 0.0; + } + work_estimate += 2 * m; +} + +template +void compute_basic_primal_variables(const lp_problem_t& lp, + const basis_update_mpf_t& basis_update, + const std::vector& basic_list, + const std::vector& nonbasic_list, + std::vector& x, + f_t& work_estimate) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + std::vector rhs = lp.rhs; + for (i_t k = 0; k < n - m; ++k) { + const i_t j = nonbasic_list[k]; + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + const f_t xj = x[j]; + for (i_t p = col_start; p < col_end; ++p) { + rhs[lp.A.i[p]] -= xj * lp.A.x[p]; + } + work_estimate += 3.0 * (col_end - col_start); + } + work_estimate += 4 * (n - m); + std::vector xB(m); + work_estimate += m; + basis_update.b_solve(rhs, xB); + for (i_t k = 0; k < m; ++k) { + x[basic_list[k]] = xB[k]; + } + work_estimate += 3 * m; +} + +template +f_t primal_constraint_residual(const lp_problem_t& lp, const std::vector& x) +{ + std::vector residual = lp.rhs; + matrix_vector_multiply(lp.A, 1.0, x, -1.0, residual); + return vector_norm_inf(residual); +} + } // namespace -// Note this implementation of primal simplex is experimental -// It is meant only to serve as a method to remove the perturbation to the objective -// after dual simplex has found a primal feasible solution -// The implementation currently cycles. So is not enabled at this time. +template +i_t primal_ratio_test(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& vstatus, + const std::vector& basic_list, + std::vector& x, + std::vector& delta_x, + f_t& step_length, + i_t& basic_leaving, + i_t entering_index, + i_t direction, + f_t& work_estimate) +{ + const i_t m = lp.num_rows; + basic_leaving = -1; + i_t leaving_index = -1; + constexpr f_t pivot_tol = 1e-8; + constexpr f_t harris_tol = 1e-8; + + // Harris ratio test: two passes. + // Pass 1: find the maximum step length alpha_1 such that no variable + // moves more than harris_tol past its bound. + // Pass 2: among all candidates with ratio <= alpha_1, pick the one + // with the largest pivot (|delta_x[j]|). + + f_t alpha_1 = inf; + + // Entering variable can hit its opposite bound: limit step by that + if (direction > 0 && lp.upper[entering_index] < inf) { + const f_t limit = lp.upper[entering_index] - x[entering_index]; + if (limit >= 0 && limit < alpha_1) { alpha_1 = limit; } + } else if (direction < 0 && lp.lower[entering_index] > -inf) { + const f_t limit = x[entering_index] - lp.lower[entering_index]; + if (limit >= 0 && limit < alpha_1) { alpha_1 = limit; } + } + + // Pass 1: compute alpha_1 (Harris step) + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + if (std::abs(delta_x[j]) <= pivot_tol) { continue; } + + // Already below lower and moving back up: stop exactly at the bound. + // No harris tolerance here — these variables are already infeasible + // and must not overshoot their bound (needed for Phase I correctness). + if (x[j] < lp.lower[j] && delta_x[j] > pivot_tol && lp.lower[j] > -inf) { + const f_t ratio = (lp.lower[j] - x[j]) / delta_x[j]; + if (ratio >= 0 && ratio < alpha_1) { alpha_1 = ratio; } + } + // Already above upper and moving back down: stop exactly at the bound. + if (x[j] > lp.upper[j] && delta_x[j] < -pivot_tol && lp.upper[j] < inf) { + const f_t ratio = (lp.upper[j] - x[j]) / delta_x[j]; + if (ratio >= 0 && ratio < alpha_1) { alpha_1 = ratio; } + } + + if (lp.lower[j] > -inf && delta_x[j] < -pivot_tol) { + // xj + step * delta_x[j] >= lp.lower[j] - harris_tol + f_t neum = lp.lower[j] - x[j] - harris_tol; + if (neum > 0 && neum <= settings.primal_tol) { neum = 0.0; } + f_t ratio = neum / delta_x[j]; + if (ratio >= 0 && ratio < alpha_1) { alpha_1 = ratio; } + } + if (lp.upper[j] < inf && delta_x[j] > pivot_tol) { + // xj + step * delta_x[j] <= lp.upper[j] + harris_tol + f_t neum = lp.upper[j] - x[j] + harris_tol; + if (neum < 0 && -neum <= settings.primal_tol) { neum = 0.0; } + f_t ratio = neum / delta_x[j]; + if (ratio >= 0 && ratio < alpha_1) { alpha_1 = ratio; } + } + } + + // Pass 2: among candidates with exact ratio <= alpha_1, pick largest pivot + f_t best_pivot = 0.0; + step_length = alpha_1; + + // Check entering variable bound (no pivot selection needed — it's fixed at direction) + if (direction > 0 && lp.upper[entering_index] < inf) { + const f_t limit = lp.upper[entering_index] - x[entering_index]; + if (limit >= 0 && limit <= alpha_1) { + // Entering hits its own bound — this is always pivot = 1.0 effectively + step_length = limit; + leaving_index = -1; + basic_leaving = -1; + best_pivot = inf; // Always prefer this if it's within alpha_1 + } + } else if (direction < 0 && lp.lower[entering_index] > -inf) { + const f_t limit = x[entering_index] - lp.lower[entering_index]; + if (limit >= 0 && limit <= alpha_1) { + step_length = limit; + leaving_index = -1; + basic_leaving = -1; + best_pivot = inf; + } + } + + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + if (std::abs(delta_x[j]) <= pivot_tol) { continue; } + + const f_t abs_dx = std::abs(delta_x[j]); + + // Already below lower and moving back up: stop when we reach the lower bound. + // Without this, phase I can take an unbounded step (false unbounded) or skip the + // breakpoint of the piecewise phase-I objective and stall still infeasible. + if (x[j] < lp.lower[j] && delta_x[j] > pivot_tol && lp.lower[j] > -inf) { + const f_t ratio = (lp.lower[j] - x[j]) / delta_x[j]; + if (ratio >= 0 && ratio <= alpha_1 && abs_dx > best_pivot) { + best_pivot = abs_dx; + step_length = ratio; + basic_leaving = k; + leaving_index = j; + } + } + // Already above upper and moving back down + if (x[j] > lp.upper[j] && delta_x[j] < -pivot_tol && lp.upper[j] < inf) { + const f_t ratio = (lp.upper[j] - x[j]) / delta_x[j]; + if (ratio >= 0 && ratio <= alpha_1 && abs_dx > best_pivot) { + best_pivot = abs_dx; + step_length = ratio; + basic_leaving = k; + leaving_index = j; + } + } + + if (lp.lower[j] > -inf && delta_x[j] < -pivot_tol) { + // xj + step * delta_x[j] >= lp.lower[j] + // step <= (lp.lower[j] - x[j]) / delta_x[j], delta_x[j] < 0 + f_t neum = lp.lower[j] - x[j]; + // A basic sitting below its bound (within the primal tolerance) is on + // the bound numerically. Treat it as a zero-length block. + if (neum > 0 && neum <= settings.primal_tol) { neum = 0.0; } + f_t ratio = neum / delta_x[j]; + if (ratio >= 0 && ratio <= alpha_1 && abs_dx > best_pivot) { + best_pivot = abs_dx; + step_length = ratio; + basic_leaving = k; + leaving_index = j; + } + } + if (lp.upper[j] < inf && delta_x[j] > pivot_tol) { + // xj + step * delta_x[j] <= lp.upper[j] + // step <= (lp.upper[j] - x[j]) / delta_x[j], delta_x[j] > 0 + f_t neum = lp.upper[j] - x[j]; + // Mirror of the lower bound case: slightly above the bound is on the bound. + if (neum < 0 && -neum <= settings.primal_tol) { neum = 0.0; } + f_t ratio = neum / delta_x[j]; + if (ratio >= 0 && ratio <= alpha_1 && abs_dx > best_pivot) { + best_pivot = abs_dx; + step_length = ratio; + basic_leaving = k; + leaving_index = j; + } + } + } + + work_estimate += 10 * m; + return leaving_index; +} + template primal_status_t primal_phase2(i_t phase, f_t start_time, @@ -256,34 +731,14 @@ primal_status_t primal_phase2(i_t phase, { const i_t m = lp.num_rows; const i_t n = lp.num_cols; - assert(m <= n); - assert(vstatus.size() == n); - assert(lp.A.m == m); - assert(lp.A.n == n); - assert(lp.objective.size() == n); - assert(lp.lower.size() == n); - assert(lp.upper.size() == n); - assert(lp.rhs.size() == m); + f_t work_estimate = 0; std::vector basic_list(m); std::vector nonbasic_list; std::vector superbasic_list; - std::vector bound_info(n - m); - - std::vector& x = sol.x; - std::vector& y = sol.y; - std::vector& z = sol.z; - - std::vector incoming_x = x; - std::vector incoming_vstatus = vstatus; - - settings.log.printf("Primal Simplex Phase %d\n", phase); - settings.log.printf("Solving a problem with %d constraints %d variables %d nonzeros\n", - lp.num_rows, - lp.num_cols, - lp.A.col_start[lp.num_cols]); get_basis_from_vstatus(m, vstatus, basic_list, nonbasic_list, superbasic_list); + work_estimate += 2 * n; assert(superbasic_list.size() == 0); assert(nonbasic_list.size() == n - m); @@ -308,6 +763,7 @@ primal_status_t primal_phase2(i_t phase, slacks_needed, work_estimate); if (rank == CONCURRENT_HALT_RETURN) { + settings.log.printf("Concurrent halt in primal phase2\n"); return primal_status_t::CONCURRENT_LIMIT; } else if (rank == TIME_LIMIT_RETURN) { return primal_status_t::TIME_LIMIT; @@ -352,47 +808,64 @@ primal_status_t primal_phase2(i_t phase, } } reorder_basic_list(q, basic_list); - reorder_basic_list(q, basic_list); - basis_update_t ft(L, U, p); - - std::vector c_basic(m); - for (i_t k = 0; k < m; ++k) { - const i_t j = basic_list[k]; - c_basic[k] = lp.objective[j]; - } + basis_update_mpf_t ft(L, U, p, settings.refactor_frequency); - // Solve B'*y = cB - ft.b_transpose_solve(c_basic, y); - settings.log.printf( - "|| y || %e || cB || %e\n", vector_norm_inf(y), vector_norm_inf(c_basic)); - - // zN = cN - N'*y - for (i_t k = 0; k < n - m; k++) { - const i_t j = nonbasic_list[k]; - // z_j <- c_j - z[j] = lp.objective[j]; - - // z_j <- z_j - A(:, j)'*y - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - f_t dot = 0.0; - for (i_t p = col_start; p < col_end; ++p) { - dot += lp.A.x[p] * y[lp.A.i[p]]; - } - z[j] -= dot; - } - // zB = 0 - for (i_t k = 0; k < m; ++k) { - z[basic_list[k]] = 0.0; - } - settings.log.printf("|| z || %e\n", vector_norm_inf(z)); + return primal_phase2_with_advanced_basis(phase, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + work_estimate); +} +// Note this implementation of primal simplex is experimental +// It is meant only to serve as a method to remove the perturbation to the objective +// after dual simplex has found a primal feasible solution +template +primal_status_t primal_phase2_with_advanced_basis( + i_t phase, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& basis_update, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + i_t& iter, + f_t& work_estimate, + bool print_summary) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + assert(m <= n); + assert(vstatus.size() == n); + assert(lp.A.m == m); + assert(lp.A.n == n); + assert(lp.objective.size() == n); + assert(lp.lower.size() == n); + assert(lp.upper.size() == n); + assert(lp.rhs.size() == m); - set_primal_variables_on_bounds(lp, settings, z, vstatus, x); + std::vector& x = sol.x; + std::vector& y = sol.y; + std::vector& z = sol.z; - const f_t init_dual_inf = dual_infeasibility(lp, vstatus, z); - settings.log.printf("Initial dual infeasibility %e\n", init_dual_inf); + std::vector incoming_x = x; + std::vector incoming_vstatus = vstatus; + work_estimate += 2.0 * n; + settings.log.printf("Primal Simplex\n"); + settings.log.printf("Pricing: %s\n", settings.primal_pricing == 1 ? "Devex" : "Dantzig"); + // Nonbasics must be on their bounds before forming B x_B = b - A_N x_N. + // Setting them after the solve leaves ||A*x - b|| large whenever x_N != 0. + set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); std::vector rhs = lp.rhs; + work_estimate += m; // rhs = b - sum_{j : x_j = l_j} A(:, j) l(j) - sum_{j : x_j = u_j} A(:, j) * // u(j) for (i_t k = 0; k < n - m; ++k) { @@ -403,150 +876,601 @@ primal_status_t primal_phase2(i_t phase, for (i_t p = col_start; p < col_end; ++p) { rhs[lp.A.i[p]] -= xj * lp.A.x[p]; } + work_estimate += 3.0 * (col_end - col_start); } + work_estimate += 4 * (n - m); std::vector xB(m); - ft.b_solve(rhs, xB); + work_estimate += m; + + basis_update.b_solve(rhs, xB); for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; x[j] = xB[k]; } - settings.log.printf("|| x || %e\n", vector_norm2(x)); + work_estimate += 3 * m; + + constexpr bool print_norms = false; + if constexpr (print_norms) { settings.log.printf("|| x || %e\n", vector_norm2(x)); } std::vector residual = lp.rhs; + work_estimate += m; matrix_vector_multiply(lp.A, 1.0, x, -1.0, residual); + work_estimate += m + 2 * n + 4.0 * lp.A.col_start[lp.A.n]; f_t primal_residual = vector_norm_inf(residual); - if (primal_residual > 1e-6) { settings.log.printf("|| A*x - b || %e\n", primal_residual); } - f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); - settings.log.printf("Initial primal infeasibility %e\n", primal_inf); + work_estimate += m; + if (primal_residual > settings.primal_tol) { + settings.log.printf("|| A*x - b || %e\n", primal_residual); + } + + std::vector objective = lp.objective; + work_estimate += 2 * n; + const f_t primal_tol = settings.primal_tol; + f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x, work_estimate); + if (primal_inf > primal_tol) { + // We are primal infeasible. Switch to phase 1 + compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); + settings.log.printf("Phase 1\n"); + settings.log.printf("Initial primal infeasibility %e\n", primal_inf); + phase = 1; + } else { + settings.log.printf("Phase 2\n"); + phase = 2; + } - const i_t iter_limit = iter + 1000; - std::vector delta_y(m); + std::vector c_basic(m); + work_estimate += m; + compute_dual_variables( + lp, settings, objective, basic_list, nonbasic_list, basis_update, c_basic, y, z, work_estimate); + if constexpr (print_norms) { settings.log.printf("|| z || %e\n", vector_norm_inf(z)); } + + i_t num_dual_inf = 0; + i_t num_primal_inf = 0; + const f_t init_dual_inf = + dual_infeasibility(lp, vstatus, z, settings.dual_tol, num_dual_inf, work_estimate); + if (num_dual_inf > 0) { settings.log.printf("Initial dual infeasibility %e\n", init_dual_inf); } + + csr_matrix_t Arow(m, n, lp.A.nnz()); + work_estimate += n + 2 * lp.A.nnz(); + lp.A.to_compressed_row(Arow); + work_estimate += m + 6 * lp.A.nnz(); + + const i_t iter_limit = settings.iteration_limit; + const i_t start_iter = iter; + sparse_vector_t delta_y(m, 0); + sparse_vector_t etilde(m, 0); std::vector delta_z(n); std::vector delta_x(n); + std::vector devex_weight(n, 1.0); + i_t num_bad_devex_weight = 0; + work_estimate += 2 * m + 3 * n; + + f_t dual_inf = init_dual_inf; + f_t obj = compute_objective(lp, x); + work_estimate += 2 * n; + f_t pricing_dual_tol = settings.dual_tol; + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); + settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); + settings.log.printf("%5d %+.16e %7d %.8e %.2f\n", + iter, + compute_user_objective(lp, obj), + phase == 1 ? num_primal_inf : num_dual_inf, + phase == 1 ? primal_inf : dual_inf, + toc(start_time)); + bool switched_phase = false; + + work_estimate += basis_update.work_estimate(); + basis_update.clear_work_estimate(); + + if (work_estimate > settings.work_limit) { return primal_status_t::WORK_LIMIT; } + + primal_timers_t timers(false); - settings.log.printf("Iter Objective Primal inf Dual Inf. Step Entering Leaving\n"); while (iter < iter_limit) { + timers.start_timer(work_estimate + basis_update.work_estimate()); i_t nonbasic_entering = -1; - f_t dual_inf; i_t direction; - i_t entering_index = - phase2_pricing(lp, z, nonbasic_list, vstatus, direction, nonbasic_entering, dual_inf); + i_t entering_index; + if (settings.primal_pricing == 1) { + entering_index = devex_pricing(lp, + z, + devex_weight, + nonbasic_list, + vstatus, + pricing_dual_tol, + direction, + nonbasic_entering, + dual_inf, + work_estimate); + } else { + entering_index = phase2_pricing(lp, + z, + nonbasic_list, + vstatus, + pricing_dual_tol, + direction, + nonbasic_entering, + dual_inf, + work_estimate); + } + timers.pricing_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); if (entering_index == -1) { - f_t obj = compute_objective(lp, x); - f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); - settings.log.printf( - "Optimal solution found. Objective %e. Dual infeas %e. Primal " - "infeasibility %e. Iterations %d\n", - compute_user_objective(lp, obj), - dual_inf, - primal_inf, - iter); - return primal_status_t::OPTIMAL; + if (phase == 2) { + // Verify optimality with a consistent basic solution: refactor, put + // nonbasics exactly on their status bounds, rebuild x_B so Ax = b, and + // refresh duals. If that point is not primal/dual feasible, continue. + if (basis_update.num_updates() > 0) { + i_t rank = basis_update.refactor_basis( + lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); + if (rank == CONCURRENT_HALT_RETURN) { return primal_status_t::CONCURRENT_LIMIT; } + if (rank == TIME_LIMIT_RETURN) { return primal_status_t::TIME_LIMIT; } + if (rank != 0) { + settings.log.printf("Failed to refactor basis at optimality check. Iteration %d\n", + iter); + return primal_status_t::NUMERICAL; + } + work_estimate += basis_update.work_estimate(); + basis_update.clear_work_estimate(); + } + set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); + compute_basic_primal_variables( + lp, basis_update, basic_list, nonbasic_list, x, work_estimate); + compute_dual_variables(lp, + settings, + objective, + basic_list, + nonbasic_list, + basis_update, + c_basic, + y, + z, + work_estimate); + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); + dual_inf = + dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf, work_estimate); + if (primal_inf > primal_tol) { + compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); + phase = 1; + pricing_dual_tol = settings.dual_tol; + compute_dual_variables(lp, + settings, + objective, + basic_list, + nonbasic_list, + basis_update, + c_basic, + y, + z, + work_estimate); + settings.log.printf( + "Switching to Primal Simplex Phase 1 after near optimality. " + "Primal infeasibility %e\n", + primal_inf); + settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); + switched_phase = true; + continue; + } + if (num_dual_inf > 0) { + // The refreshed reduced costs contain a candidate visible at the active + // pricing tolerance. + continue; + } + + i_t num_tight_dual_inf = 0; + const f_t tight_dual_inf = + dual_infeasibility(lp, vstatus, z, f_t(0.0), num_tight_dual_inf, work_estimate); + if (tight_dual_inf > settings.dual_tol) { + // No candidate is visible at the active pricing tolerance, but the + // zero-tolerance residual is still material. Try tighter pricing before + // accepting optimality. This is needed for problems such as cycle, + // where many small reduced-cost violations lead to improving pivots. + f_t retry_dual_tol = pricing_dual_tol; + f_t retry_dual_inf = 0.0; + i_t retry_entering = -1; + while (retry_entering == -1 && retry_dual_tol > f_t(1e-10)) { + retry_dual_tol *= f_t(0.1); + retry_entering = phase2_pricing(lp, + z, + nonbasic_list, + vstatus, + retry_dual_tol, + direction, + nonbasic_entering, + retry_dual_inf, + work_estimate); + } + if (retry_entering != -1) { + pricing_dual_tol = retry_dual_tol; + continue; + } + } + // Report the unfiltered residual at the accepted solution. + dual_inf = tight_dual_inf; + num_dual_inf = num_tight_dual_inf; + obj = compute_objective(lp, x); + work_estimate += 2 * n; + sol.objective = obj; + sol.user_objective = compute_user_objective(lp, obj); + if (!settings.inside_mip && print_summary) { + settings.log.printf("\n"); + settings.log.printf( + "Optimal solution found in %d iterations and %.2fs\n", iter, toc(start_time)); + settings.log.printf("Objective %+.8e\n", sol.user_objective); + settings.log.printf("\n"); + settings.log.printf("Primal infeasibility (abs): %.2e\n", primal_inf); + settings.log.printf("Dual infeasibility (abs): %.2e\n", dual_inf); + settings.log.printf("Primal residual ||Ax-b||: %.2e\n", + primal_constraint_residual(lp, x)); + } + timers.print_timers(settings); + return primal_status_t::OPTIMAL; + } else { + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); + + if (primal_inf > primal_tol) { + // Incremental duals may be stale relative to the current phase-I + // objective. Refresh objective and duals, then retry pricing with + // successively tighter dual tolerances. + settings.log.printf("Refreshing phase-I objective and duals. Num updates %d. Iter %d\n", + basis_update.num_updates(), + iter); + compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); + compute_dual_variables(lp, + settings, + objective, + basic_list, + nonbasic_list, + basis_update, + c_basic, + y, + z, + work_estimate); + f_t retry_dual_tol = pricing_dual_tol; + while (entering_index == -1 && retry_dual_tol > f_t(1e-10)) { + retry_dual_tol *= f_t(0.1); + settings.log.printf("Retrying phase-I pricing with dual_tol %e\n", retry_dual_tol); + entering_index = phase2_pricing(lp, + z, + nonbasic_list, + vstatus, + retry_dual_tol, + direction, + nonbasic_entering, + dual_inf, + work_estimate); + } + if (entering_index == -1) { + settings.log.printf( + "No entering variable found with large " + "infeasibility %e (%d).\n", + primal_inf, + num_primal_inf); + return primal_status_t::PRIMAL_INFEASIBLE; + } + pricing_dual_tol = retry_dual_tol; + } else { + // Restore the objective to the original objective + objective = lp.objective; + phase = 2; + pricing_dual_tol = settings.dual_tol; + settings.log.printf( + "Primal phase I complete. Iterations %d. Time %.2f\n", iter, toc(start_time)); + settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); + compute_dual_variables(lp, + settings, + objective, + basic_list, + nonbasic_list, + basis_update, + c_basic, + y, + z, + work_estimate); + obj = compute_objective(lp, x); + work_estimate += 2 * n; + dual_inf = + dual_infeasibility(lp, vstatus, z, settings.dual_tol, num_dual_inf, work_estimate); + iter++; + // Print here: continue may hit dual-optimal Phase 2 and return before + // the end-of-loop log checks switched_phase. + settings.log.printf("%5d %+.16e %7d %.8e %.2f\n", + iter, + compute_user_objective(lp, obj), + num_dual_inf, + dual_inf, + toc(start_time)); + continue; + } + } } + sparse_vector_t rhs_sparse(lp.A, entering_index); + work_estimate += 3 * rhs_sparse.i.size(); + sparse_vector_t scaled_delta_xB_sparse(m, 0); + sparse_vector_t utilde_sparse(m, 0); + timers.start_timer(work_estimate + basis_update.work_estimate()); + basis_update.b_solve(rhs_sparse, scaled_delta_xB_sparse, utilde_sparse); std::vector scaled_delta_xB(m); - std::vector rhs(m); - const i_t col_start = lp.A.col_start[entering_index]; - const i_t col_end = lp.A.col_start[entering_index + 1]; - for (i_t p = col_start; p < col_end; ++p) { - rhs[lp.A.i[p]] = lp.A.x[p]; - } - std::vector utilde(m); - ft.b_solve(rhs, scaled_delta_xB, utilde); + scaled_delta_xB_sparse.to_dense(scaled_delta_xB); + work_estimate += m + scaled_delta_xB_sparse.i.size(); for (i_t k = 0; k < m; ++k) { const i_t j = basic_list[k]; delta_x[j] = -direction * scaled_delta_xB[k]; } + work_estimate += 3 * m; for (i_t k = 0; k < n - m; ++k) { const i_t j = nonbasic_list[k]; delta_x[j] = 0.0; } + work_estimate += 2 * (n - m); delta_x[entering_index] = direction; + timers.ftran_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); - std::vector residual(m); - matrix_vector_multiply(lp.A, 1.0, delta_x, 1.0, residual); +#ifdef CHECK_NULLSPACE + std::vector residual(m, 0.0); + matrix_vector_multiply(lp.A, 1.0, delta_x, 0.0, residual); f_t primal_step_err = vector_norm_inf(residual); - if (primal_step_err > 1e-3) { printf("|| A * dx || %e\n", primal_step_err); } + if (primal_step_err > 1e-3) { + settings.log.printf("|| A * dx || %e at iter %d (updates %d)\n", + primal_step_err, + iter, + basis_update.num_updates()); + } +#endif + timers.start_timer(work_estimate + basis_update.work_estimate()); i_t basic_leaving; f_t step_length; - i_t leaving_index = ratio_test(lp, vstatus, basic_list, x, delta_x, step_length, basic_leaving); - if (leaving_index == -1) { + i_t leaving_index = primal_ratio_test(lp, + settings, + vstatus, + basic_list, + x, + delta_x, + step_length, + basic_leaving, + entering_index, + direction, + work_estimate); + timers.ratio_test_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); + if (leaving_index == -1 && step_length >= inf) { settings.log.printf("No leaving variable. Primal unbounded?\n"); return primal_status_t::PRIMAL_UNBOUNDED; } - assert(step_length >= 0.0); - // Update the primal variables + const bool basis_updated = (leaving_index != -1); + bool recompute_duals = false; + timers.start_timer(work_estimate + basis_update.work_estimate()); for (i_t j = 0; j < n; ++j) { x[j] += step_length * delta_x[j]; } + work_estimate += 2 * n; + timers.update_x_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); + +#ifdef COMPUTE_RESIDUAL + f_t debug_primal_residual = primal_constraint_residual(lp, x); + if (debug_primal_residual > 1e-6) { + settings.log.printf("|| A * x - b || %e at iteration %d (updates %d)\n", + debug_primal_residual, + iter, + basis_update.num_updates()); + } +#endif - // Update the factorization - ft.update(utilde, basic_leaving); + if (basis_updated) { + assert(step_length >= 0.0); + + bool should_refactor = basis_update.num_updates() > settings.refactor_frequency; + f_t dual_step_length = 0.0; + if (!should_refactor) { + timers.start_timer(work_estimate + basis_update.work_estimate()); + compute_delta_y(basis_update, basic_leaving, delta_y, etilde); + timers.btran_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); + const f_t pivot = scaled_delta_xB[basic_leaving]; + dual_step_length = compute_dual_step_length(z[entering_index], pivot); + } + + basic_list[basic_leaving] = entering_index; + nonbasic_list[nonbasic_entering] = leaving_index; + vstatus[entering_index] = variable_status_t::BASIC; + // Place the leaver on its leaving bound. If that bound is far from the + // current value (typical after a zero-step leave of an already-infeasible + // basic), rebuild x_B after the factor matches the new basis so Ax = b; + // phase handling below may then (re)enter Phase I if basics are infeasible. + bool rebuild_x_after_bound_snap = false; + f_t leave_bound = 0.0; + if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { + vstatus[leaving_index] = variable_status_t::NONBASIC_FIXED; + leave_bound = lp.lower[leaving_index]; + } else { + // Classify by which bound was hit. Using sign(delta_x) is wrong when the + // variable approached the bound from the infeasible side (phase I). + const f_t x_leave = x[leaving_index]; + const f_t dist_to_lower = std::abs(x_leave - lp.lower[leaving_index]); + const f_t dist_to_upper = std::abs(x_leave - lp.upper[leaving_index]); + if (lp.lower[leaving_index] > -inf && + (lp.upper[leaving_index] >= inf || dist_to_lower <= dist_to_upper)) { + vstatus[leaving_index] = variable_status_t::NONBASIC_LOWER; + leave_bound = lp.lower[leaving_index]; + } else { + vstatus[leaving_index] = variable_status_t::NONBASIC_UPPER; + leave_bound = lp.upper[leaving_index]; + } + } + if (std::abs(x[leaving_index] - leave_bound) > settings.primal_tol) { + rebuild_x_after_bound_snap = true; + } + x[leaving_index] = leave_bound; - // Update the basis - basic_list[basic_leaving] = entering_index; - nonbasic_list[nonbasic_entering] = leaving_index; - vstatus[entering_index] = variable_status_t::BASIC; - if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { - vstatus[leaving_index] = variable_status_t::NONBASIC_FIXED; - } else if (direction == 1) { - vstatus[leaving_index] = variable_status_t::NONBASIC_LOWER; + if (!should_refactor) { + timers.start_timer(work_estimate + basis_update.work_estimate()); + compute_delta_z(Arow, vstatus, delta_y, delta_z, work_estimate); + timers.delta_z_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); + timers.start_timer(work_estimate + basis_update.work_estimate()); + update_y(dual_step_length, delta_y, y, work_estimate); + update_z(dual_step_length, nonbasic_list, entering_index, delta_z, z, work_estimate); + timers.update_duals_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); + + // Devex weight update (only when using Devex pricing) + if (settings.primal_pricing == 1) { + const f_t pivot = scaled_delta_xB[basic_leaving]; + const f_t pivot_sq = pivot * pivot; + const f_t w_enter = devex_weight[entering_index]; + // Exact pivot weight for entering variable is 1/pivot_sq + // Check if stored weight was a bad approximation + const f_t exact_pivot_weight = 1.0 / pivot_sq; + if (w_enter > 3.0 * exact_pivot_weight) { num_bad_devex_weight++; } + // Update weights for all nonbasic columns using the pivot row (delta_z) + // After compute_delta_z and update_z, delta_z[j] still holds the raw + // pivot row entries (update_z multiplies by dual_step_length into z, not delta_z) + for (i_t k = 0; k < n - m; ++k) { + const i_t j = nonbasic_list[k]; + const f_t a_j = delta_z[j]; + const f_t candidate = (a_j * a_j / pivot_sq) * w_enter; + if (candidate > devex_weight[j]) { devex_weight[j] = candidate; } + } + // Weight for leaving variable (now nonbasic) + devex_weight[leaving_index] = std::max(1.0 / pivot_sq, f_t(1e-4)); + // Weight for entering variable (now basic) — reset + devex_weight[entering_index] = 1.0; + work_estimate += 5 * (n - m); + // Reset framework if too many bad weights + if (num_bad_devex_weight > 3) { + std::fill(devex_weight.begin(), devex_weight.end(), f_t(1.0)); + num_bad_devex_weight = 0; + } + } + + timers.start_timer(work_estimate + basis_update.work_estimate()); + should_refactor = basis_update.update(utilde_sparse, etilde, basic_leaving) == 1; + timers.lu_update_time += timers.stop_timer(work_estimate + basis_update.work_estimate()); + } + if (should_refactor) { + timers.start_timer(work_estimate + basis_update.work_estimate()); + i_t rank = basis_update.refactor_basis( + lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); + if (rank == CONCURRENT_HALT_RETURN) { return primal_status_t::CONCURRENT_LIMIT; } + if (rank == TIME_LIMIT_RETURN) { return primal_status_t::TIME_LIMIT; } + if (rank != 0) { + settings.log.printf("Failed to refactor basis. Iteration %d\n", iter); + return primal_status_t::NUMERICAL; + } + work_estimate += basis_update.work_estimate(); + basis_update.clear_work_estimate(); + recompute_duals = true; + // Factor matches basic_list: rebuild x_B so Ax = b exactly. + set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); + compute_basic_primal_variables( + lp, basis_update, basic_list, nonbasic_list, x, work_estimate); + timers.lu_factorization_time += + timers.stop_timer(work_estimate + basis_update.work_estimate()); + } else if (rebuild_x_after_bound_snap) { + // FT update already matches the new basis; recompute x_B with the leaving variable + // snapped onto its bound. + compute_basic_primal_variables( + lp, basis_update, basic_list, nonbasic_list, x, work_estimate); + } } else { - vstatus[leaving_index] = variable_status_t::NONBASIC_UPPER; + if (direction > 0) { + vstatus[entering_index] = variable_status_t::NONBASIC_UPPER; + x[entering_index] = lp.upper[entering_index]; + } else { + vstatus[entering_index] = variable_status_t::NONBASIC_LOWER; + x[entering_index] = lp.lower[entering_index]; + } } - // Solve for y such that B'*y = c_B - for (i_t k = 0; k < m; ++k) { - const i_t j = basic_list[k]; - c_basic[k] = lp.objective[j]; - } - ft.b_transpose_solve(y, c_basic); - // zN = cN - N'*y - for (i_t k = 0; k < n - m; k++) { - const i_t j = nonbasic_list[k]; - // z_j <- c_j - z[j] = lp.objective[j]; - - // z_j <- z_j - A(:, j)'*y - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - f_t dot = 0.0; - for (i_t p = col_start; p < col_end; ++p) { - dot += lp.A.x[p] * y[lp.A.i[p]]; + primal_inf = primal_infeasibility(lp, settings, vstatus, x, num_primal_inf, work_estimate); + if (primal_inf > primal_tol) { + if (phase != 1) { + settings.log.printf( + "Switching to Primal Simplex Phase 1. Iteration %d. Primal infeasibility %e\n", + iter, + primal_inf); + settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); + switched_phase = true; } - z[j] -= dot; + compute_phase1_objective(lp, settings, vstatus, x, objective, work_estimate); + phase = 1; + recompute_duals = true; + } else if (phase == 1) { + objective = lp.objective; + phase = 2; + pricing_dual_tol = settings.dual_tol; + recompute_duals = true; + settings.log.printf( + "Primal phase I complete. Iterations %d. Time %.2f\n", iter, toc(start_time)); + settings.log.printf(" Iter Objective Num Inf. Sum Inf. Time\n"); + switched_phase = true; } - // zB = 0 - for (i_t k = 0; k < m; ++k) { - z[basic_list[k]] = 0.0; + + if (recompute_duals) { + compute_dual_variables(lp, + settings, + objective, + basic_list, + nonbasic_list, + basis_update, + c_basic, + y, + z, + work_estimate); } - const f_t obj = compute_objective(lp, x); - const f_t primal_inf = primal_infeasibility(lp, settings, vstatus, x); - settings.log.printf("%3d %.10e %.2e %.2e %.2e %d %d\n", - iter, - compute_user_objective(lp, obj), - primal_inf, - dual_inf, - step_length, - entering_index, - leaving_index); + obj = compute_objective(lp, x); + work_estimate += 2 * n; + dual_inf = dual_infeasibility(lp, vstatus, z, pricing_dual_tol, num_dual_inf, work_estimate); iter++; + + f_t now = toc(start_time); + if ((iter - start_iter) < settings.first_iteration_log || + (iter % settings.iteration_log_frequency) == 0 || switched_phase) { + const f_t user_obj = compute_user_objective(lp, obj); + settings.log.printf("%5d %+.16e %7d %.8e %.2f\n", + iter, + user_obj, + phase == 1 ? num_primal_inf : num_dual_inf, + phase == 1 ? primal_inf : dual_inf, + now); + switched_phase = false; + } + + work_estimate += basis_update.work_estimate(); + basis_update.clear_work_estimate(); + + if (now > settings.time_limit) { + timers.print_timers(settings); + return primal_status_t::TIME_LIMIT; + } + if (work_estimate > settings.work_limit) { + timers.print_timers(settings); + return primal_status_t::WORK_LIMIT; + } } - if (iter == iter_limit) { return primal_status_t::ITERATION_LIMIT; } + timers.print_timers(settings); + if (iter >= iter_limit) { return primal_status_t::ITERATION_LIMIT; } return primal_status_t::NUMERICAL; } #ifdef DUAL_SIMPLEX_INSTANTIATE_DOUBLE +template int primal_ratio_test(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& vstatus, + const std::vector& basic_list, + std::vector& x, + std::vector& delta_x, + double& step_length, + int& basic_leaving, + int entering_index, + int direction, + double& work_estimate); + template primal_status_t primal_phase2( int phase, double start_time, @@ -556,6 +1480,20 @@ template primal_status_t primal_phase2( lp_solution_t& sol, int& iter); +template primal_status_t primal_phase2_with_advanced_basis( + int phase, + double start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& basis_update, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + int& iter, + double& work_estimate, + bool print_summary); + #endif } // namespace cuopt::mathematical_optimization::simplex diff --git a/cpp/src/dual_simplex/primal.hpp b/cpp/src/dual_simplex/primal.hpp index 930958a802..fc47d90368 100644 --- a/cpp/src/dual_simplex/primal.hpp +++ b/cpp/src/dual_simplex/primal.hpp @@ -7,6 +7,7 @@ #pragma once +#include #include #include #include @@ -18,15 +19,47 @@ namespace cuopt::mathematical_optimization::simplex { enum class primal_status_t { - OPTIMAL = 0, - PRIMAL_UNBOUNDED = 1, - NUMERICAL = 2, - NOT_LOADED = 3, - TIME_LIMIT = 4, - ITERATION_LIMIT = 5, - CONCURRENT_LIMIT = 6 + OPTIMAL = 0, + PRIMAL_UNBOUNDED = 1, + PRIMAL_INFEASIBLE = 2, + NUMERICAL = 3, + TIME_LIMIT = 5, + ITERATION_LIMIT = 6, + CONCURRENT_LIMIT = 7, + WORK_LIMIT = 8, + NOT_LOADED = 9 }; +template +i_t primal_ratio_test(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + const std::vector& vstatus, + const std::vector& basic_list, + std::vector& x, + std::vector& delta_x, + f_t& step_length, + i_t& basic_leaving, + i_t entering_index, + i_t direction, + f_t& work_estimate); + +template +primal_status_t primal_phase2_with_advanced_basis( + i_t phase, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& basis_update, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + i_t& iter, + f_t& work_estimate, + // Callers that print their own summary (dual simplex perturbation cleanup) + // suppress this one, so optimality is not reported twice. + bool print_summary = true); + template primal_status_t primal_phase2(i_t phase, f_t start_time, diff --git a/cpp/src/dual_simplex/right_looking_lu.cpp b/cpp/src/dual_simplex/right_looking_lu.cpp index 6a717cd257..63f5cb7c0f 100644 --- a/cpp/src/dual_simplex/right_looking_lu.cpp +++ b/cpp/src/dual_simplex/right_looking_lu.cpp @@ -209,7 +209,8 @@ class trailing_matrix_t { const f_t max_in_col = max_in_column_[j]; const i_t c_start = col_start_[j]; const i_t c_end = col_end_[j]; - for (i_t p = c_start; p < c_end; p++) { + i_t p; + for (p = c_start; p < c_end; p++) { const i_t i = c_i_[p]; const f_t val = c_x_[p]; const i_t rdeg = row_counts_.get_count(i); @@ -224,7 +225,7 @@ class trailing_matrix_t { if (markowitz <= markowitz_lower_bound) { break; } } } - work_estimate_ += 3 * (c_end - c_start); + work_estimate_ += 3 * (p - c_start); nsearch++; if (markowitz <= markowitz_lower_bound) { break; } } @@ -241,19 +242,21 @@ class trailing_matrix_t { assert(rdeg == nz); const i_t r_start = row_start_[i]; const i_t r_end = row_end_[i]; - for (i_t p = r_start; p < r_end; p++) { + i_t p; + for (p = r_start; p < r_end; p++) { const i_t j = r_j_[p]; // Look up the value from the column copy of j f_t val = 0; const i_t c_start = col_start_[j]; const i_t c_end = col_end_[j]; - for (i_t q = c_start; q < c_end; q++) { + i_t q; + for (q = c_start; q < c_end; q++) { if (c_i_[q] == i) { val = c_x_[q]; break; } } - work_estimate_ += 2 * (c_end - c_start); + work_estimate_ += 2 * (q - c_start); const f_t max_in_col = max_in_column_[j]; const i_t cdeg = col_counts_.get_count(j); assert(cdeg >= 0); @@ -267,7 +270,7 @@ class trailing_matrix_t { if (markowitz <= markowitz_lower_bound) { break; } } } - work_estimate_ += 5 * (r_end - r_start); + work_estimate_ += 5 * (p - r_start); nsearch++; if (markowitz <= markowitz_lower_bound) { break; } } @@ -334,7 +337,7 @@ class trailing_matrix_t { } } } - work_estimate_ += 2 * (c_end - c_start) + 6 * (pivot_col_count - n_fillin); + work_estimate_ += 2 * (c_end - c_start) + 5 * (pivot_col_count - n_fillin); // Step 2b: Remove cancellations (entries that became zero). if (n_cancel > 0) { @@ -1285,12 +1288,14 @@ class symmetric_trailing_matrix_t { const i_t j = r_j_[rp]; // Look up A(pivot_p, j) from column j f_t val = 0; - for (i_t q = col_start_[j]; q < col_end_[j]; q++) { + i_t q; + for (q = col_start_[j]; q < col_end_[j]; q++) { if (c_i_[q] == pivot_p) { val = c_x_[q]; break; } } + work_estimate_ += 2 * (q - col_start_[j]); const f_t lj = val / pivot_val; pivot_col_val_[j] = lj; pivot_col_mark_[j] = 1; diff --git a/cpp/src/dual_simplex/scaling.cpp b/cpp/src/dual_simplex/scaling.cpp index 98c409a630..102036e635 100644 --- a/cpp/src/dual_simplex/scaling.cpp +++ b/cpp/src/dual_simplex/scaling.cpp @@ -253,6 +253,39 @@ i_t scaling(const lp_problem_t& unscaled, return 0; } + // MIP performs integer-aware row scaling before presolve, while QP and SOCP + // use the Ruiz path above. Apply this simpler equilibration only to LPs. + const bool use_lp_row_scaling = + !settings.inside_mip && unscaled.second_order_cone_dims.empty() && unscaled.Q.n == 0; + if (use_lp_row_scaling) { + csr_matrix_t Arow(0, 0, 0); + scaled.A.to_compressed_row(Arow); + std::vector row_norm(m, 1.0); + f_t max_row_norm = 0.0; + f_t min_row_norm = inf; + for (i_t i = 0; i < m; ++i) { + for (i_t p = Arow.row_start[i]; p < Arow.row_start[i + 1]; ++p) { + row_norm[i] = std::max(row_norm[i], std::abs(Arow.x[p])); + } + max_row_norm = std::max(max_row_norm, row_norm[i]); + min_row_norm = std::min(min_row_norm, row_norm[i]); + } + if (min_row_norm > 0.0 && max_row_norm / min_row_norm > 10.0) { + settings.log.printf("Applying row scaling. Maximum row norm %e, minimum row norm %e\n", + max_row_norm, + min_row_norm); + for (i_t j = 0; j < n; ++j) { + for (i_t p = scaled.A.col_start[j]; p < scaled.A.col_start[j + 1]; ++p) { + scaled.A.x[p] /= row_norm[scaled.A.i[p]]; + } + } + for (i_t i = 0; i < m; ++i) { + scaled.rhs[i] /= row_norm[i]; + row_scaling[i] = row_norm[i]; + } + } + } + column_scaling.resize(n); f_t max = 0; f_t min = std::numeric_limits::max(); diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 8b3eba56d3..54af1b7d83 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -81,6 +81,9 @@ struct simplex_solver_settings_t { augmented(0), dualize(-1), ordering(-1), + initial_perturbation(-1), + remove_perturbation(-1), + primal_pricing(0), barrier_dual_initial_point(barrier_dual_initial_point_t::Automatic), postsolve_info(-1), barrier_presolve_bound_free_variables(-1), @@ -88,6 +91,7 @@ struct simplex_solver_settings_t { barrier_initial_point_safeguard(10.0), check_Q(false), crossover(false), + unscaled_max_abs_obj_coeff(-1.0), refactor_frequency(100), iteration_log_frequency(1000), first_iteration_log(2), @@ -188,6 +192,9 @@ struct simplex_solver_settings_t { i_t augmented; // -1 automatic, 0 to solve with ADAT, 1 to solve with augmented system i_t dualize; // -1 automatic, 0 to not dualize, 1 to dualize i_t ordering; // -1 automatic, 0 to use nested dissection, 1 to use AMD + i_t initial_perturbation; // -1 automatic, 0 to not perturb, 1 to perturb + i_t remove_perturbation; // -1 automatic, 0 disabled, 1 enabled + i_t primal_pricing; // 0 Dantzig (default), 1 Devex barrier_dual_initial_point_t barrier_dual_initial_point; // -1 automatic, 0 Lustig-Marsten-Shanno, // 1 dual least squares, 2 SeDuMi mu-based @@ -198,6 +205,7 @@ struct simplex_solver_settings_t { // the interior of the nonnegative orthant / SOC bool check_Q; // true to check if Q is positive semidefinite bool crossover; // true to do crossover, false to not + f_t unscaled_max_abs_obj_coeff; // max |c_j| before scaling (-1 = not set, compute from lp) i_t refactor_frequency; // number of basis updates before refactorization i_t iteration_log_frequency; // number of iterations between log updates i_t first_iteration_log; // number of iterations to log at beginning of solve @@ -215,10 +223,10 @@ struct simplex_solver_settings_t { i_t strong_chvatal_gomory_cuts; // -1 automatic, 0 to disable, >0 to enable strong Chvatal Gomory // cuts i_t symmetry; // -1 automatic, 0 to disable, >0 to enable different symmetry methods - i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost - // strengthening - f_t cut_change_threshold; // threshold for cut change - f_t cut_min_orthogonality; // minimum orthogonality for cuts + i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost + // strengthening + f_t cut_change_threshold; // threshold for cut change + f_t cut_min_orthogonality; // minimum orthogonality for cuts i_t mip_batch_pdlp_strong_branching; // 0 = DS only, 1 = cooperative DS + PDLP, 2 = batch PDLP only i_t mip_batch_pdlp_reliability_branching; // 0 = DS only, 1 = cooperative DS + PDLP, 2 = batch diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index 388bb43b35..867225d79e 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -61,6 +61,53 @@ void write_matlab(const std::string& filename, const simplex::lp_problem_t +void initialize_slack_basis_vstatus(const lp_problem_t& lp, + std::vector& vstatus) +{ + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + vstatus.resize(n); + for (i_t j = 0; j < n; ++j) { + if (lp.lower[j] == -inf && lp.upper[j] == inf) { + vstatus[j] = variable_status_t::NONBASIC_FREE; + } else if (std::abs(lp.upper[j] - lp.lower[j]) < 1e-12) { + vstatus[j] = variable_status_t::NONBASIC_FIXED; + } else if (lp.lower[j] > -inf) { + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } else { + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } + } + i_t num_basic = 0; + for (i_t j = n - 1; j >= 0; --j) { + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + const i_t nz = col_end - col_start; + if (nz == 1 && std::abs(lp.A.x[col_start]) == 1.0) { + vstatus[j] = variable_status_t::BASIC; + num_basic++; + } + if (num_basic == m) { break; } + } + assert(num_basic == m); +} + } // namespace template @@ -117,6 +164,7 @@ lp_status_t solve_linear_program_advanced(const lp_problem_t& original lp_solution_t& original_solution, std::vector& vstatus, std::vector& edge_norms, + f_t& work_estimate, work_limit_context_t* work_unit_context) { raft::common::nvtx::range scope("DualSimplex::solve_lp"); @@ -135,6 +183,7 @@ lp_status_t solve_linear_program_advanced(const lp_problem_t& original nonbasic_list, vstatus, edge_norms, + work_estimate, work_unit_context); return result; } @@ -150,6 +199,7 @@ lp_status_t solve_linear_program_with_advanced_basis( std::vector& nonbasic_list, std::vector& vstatus, std::vector& edge_norms, + f_t& work_estimate, work_limit_context_t* work_unit_context) { lp_status_t lp_status = lp_status_t::UNSET; @@ -177,6 +227,14 @@ lp_status_t solve_linear_program_with_advanced_basis( presolved_lp.A.col_start[presolved_lp.num_cols]); std::vector column_scales; std::vector row_scales_simplex; + // Compute max |c_j| before scaling for perturbation calibration + if (settings.unscaled_max_abs_obj_coeff < 0.0) { + f_t max_obj = 0.0; + for (i_t j = 0; j < presolved_lp.num_cols; ++j) { + max_obj = std::max(max_obj, std::abs(presolved_lp.objective[j])); + } + const_cast&>(settings).unscaled_max_abs_obj_coeff = max_obj; + } { raft::common::nvtx::range scope_scaling("DualSimplex::scaling"); scaling(presolved_lp, settings, lp, column_scales, row_scales_simplex); @@ -217,6 +275,7 @@ lp_status_t solve_linear_program_with_advanced_basis( phase1_vstatus, phase1_solution, iter, + work_estimate, edge_norms, work_unit_context); } @@ -255,6 +314,7 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, solution, iter, + work_estimate, edge_norms, work_unit_context); if (status == dual_status_t::NUMERICAL) { @@ -275,6 +335,7 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, phase1_solution, iter, + work_estimate, edge_norms, work_unit_context); vstatus = phase1_vstatus; @@ -291,12 +352,19 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, solution, iter, + work_estimate, edge_norms, work_unit_context); } constexpr bool primal_cleanup = false; if (status == dual_status_t::OPTIMAL && primal_cleanup) { + settings.log.printf("Running primal cleanup\n"); primal_phase2(2, start_time, lp, settings, vstatus, solution, iter); + // TODO: We need to update ft if the basis changed + } + if (settings.inside_mip == 1 && settings.concurrent_halt != nullptr) { + settings.log.debug("Setting concurrent halt to 1 inside_mip\n"); + *settings.concurrent_halt = 1; } if (status == dual_status_t::OPTIMAL) { std::vector unscaled_x(lp.num_cols); @@ -704,6 +772,104 @@ lp_status_t solve_linear_program_with_barrier(const user_problem_t& us return solve_linear_program_with_barrier(user_problem, settings, start_time, solution); } +template +lp_status_t solve_linear_program_with_primal(const user_problem_t& user_problem, + const simplex_solver_settings_t& settings, + f_t start_time, + lp_solution_t& solution) +{ + raft::common::nvtx::range scope("PrimalSimplex::solve_lp"); + lp_problem_t original_lp(user_problem.handle_ptr, 1, 1, 1); + std::vector new_slacks; + dualize_info_t dualize_info; + convert_user_problem(user_problem, settings, original_lp, new_slacks, dualize_info); + + solution.resize(user_problem.num_rows, user_problem.num_cols); + lp_solution_t original_solution(original_lp.num_rows, original_lp.num_cols); + + // Presolve adds/retains artificial variables so a full slack basis exists. + lp_problem_t presolved_lp(original_lp.handle_ptr, 1, 1, 1); + presolve_info_t presolve_info; + const i_t ok = presolve(original_lp, settings, presolved_lp, presolve_info); + if (ok == CONCURRENT_HALT_RETURN) { return lp_status_t::CONCURRENT_LIMIT; } + if (ok == TIME_LIMIT_RETURN) { return lp_status_t::TIME_LIMIT; } + if (ok == -1) { return lp_status_t::INFEASIBLE; } + + lp_problem_t lp(original_lp.handle_ptr, + presolved_lp.num_rows, + presolved_lp.num_cols, + presolved_lp.A.col_start[presolved_lp.num_cols]); + std::vector column_scales; + std::vector row_scales; + scaling(presolved_lp, settings, lp, column_scales, row_scales); + + std::vector vstatus; + initialize_slack_basis_vstatus(lp, vstatus); + + lp_solution_t lp_solution(lp.num_rows, lp.num_cols); + i_t iter = 0; + const primal_status_t primal_status = + primal_phase2(2, start_time, lp, settings, vstatus, lp_solution, iter); + lp_solution.iterations = iter; + original_solution.iterations = iter; + + if (primal_status == primal_status_t::CONCURRENT_LIMIT) { + solution.iterations = iter; + return lp_status_t::CONCURRENT_LIMIT; + } + + if (primal_status == primal_status_t::OPTIMAL) { + lp_solution.objective = compute_objective(lp, lp_solution.x); + lp_solution.user_objective = compute_user_objective(lp, lp_solution.objective); + + std::vector residual = lp.rhs; + matrix_vector_multiply(lp.A, 1.0, lp_solution.x, -1.0, residual); + lp_solution.l2_primal_residual = vector_norm2(residual); + + std::vector dual_residual = lp_solution.z; + for (i_t j = 0; j < lp.num_cols; ++j) { + dual_residual[j] -= lp.objective[j]; + } + matrix_transpose_vector_multiply(lp.A, 1.0, lp_solution.y, 1.0, dual_residual); + lp_solution.l2_dual_residual = vector_norm2(dual_residual); + + std::vector unscaled_x(lp.num_cols); + std::vector unscaled_y(lp.num_rows); + std::vector unscaled_z(lp.num_cols); + unscale_solution(column_scales, + row_scales, + lp_solution.x, + lp_solution.y, + lp_solution.z, + unscaled_x, + unscaled_y, + unscaled_z); + uncrush_solution(presolve_info, + settings, + original_lp, + unscaled_x, + unscaled_y, + unscaled_z, + original_solution.x, + original_solution.y, + original_solution.z); + original_solution.objective = lp_solution.objective; + original_solution.user_objective = lp_solution.user_objective; + original_solution.l2_primal_residual = lp_solution.l2_primal_residual; + original_solution.l2_dual_residual = lp_solution.l2_dual_residual; + } + + uncrush_primal_solution(user_problem, original_lp, original_solution.x, solution.x); + uncrush_dual_solution( + user_problem, original_lp, original_solution.y, original_solution.z, solution.y, solution.z); + solution.objective = original_solution.objective; + solution.user_objective = original_solution.user_objective; + solution.iterations = original_solution.iterations; + solution.l2_primal_residual = original_solution.l2_primal_residual; + solution.l2_dual_residual = original_solution.l2_dual_residual; + return map_primal_status_to_lp_status(primal_status); +} + template lp_status_t solve_linear_program(const user_problem_t& user_problem, const simplex_solver_settings_t& settings, @@ -718,8 +884,9 @@ lp_status_t solve_linear_program(const user_problem_t& user_problem, lp_solution_t lp_solution(original_lp.num_rows, original_lp.num_cols); std::vector vstatus; std::vector edge_norms; + f_t work_estimate = 0.0; lp_status_t status = solve_linear_program_advanced( - original_lp, start_time, settings, lp_solution, vstatus, edge_norms); + original_lp, start_time, settings, lp_solution, vstatus, edge_norms, work_estimate); if (status == lp_status_t::CONCURRENT_LIMIT) { solution.iterations = lp_solution.iterations; return lp_status_t::CONCURRENT_LIMIT; @@ -771,8 +938,9 @@ i_t solve(const user_problem_t& problem, lp_solution_t solution(original_lp.num_rows, original_lp.num_cols); std::vector vstatus; std::vector edge_norms; + f_t work_estimate = 0.0; lp_status_t lp_status = solve_linear_program_advanced( - original_lp, start_time, settings, solution, vstatus, edge_norms); + original_lp, start_time, settings, solution, vstatus, edge_norms, work_estimate); primal_solution = solution.x; if (lp_status == lp_status_t::OPTIMAL) { status = 0; @@ -828,6 +996,7 @@ template lp_status_t solve_linear_program_advanced( lp_solution_t& original_solution, std::vector& vstatus, std::vector& edge_norms, + double& work_estimate, work_limit_context_t* work_unit_context); template lp_status_t solve_linear_program_with_advanced_basis( @@ -840,6 +1009,7 @@ template lp_status_t solve_linear_program_with_advanced_basis( std::vector& nonbasic_list, std::vector& vstatus, std::vector& edge_norms, + double& work_estimate, work_limit_context_t* work_unit_context); template lp_status_t solve_linear_program_with_barrier( @@ -853,6 +1023,12 @@ template lp_status_t solve_linear_program_with_barrier( double start_time, lp_solution_t& solution); +template lp_status_t solve_linear_program_with_primal( + const user_problem_t& user_problem, + const simplex_solver_settings_t& settings, + double start_time, + lp_solution_t& solution); + template lp_status_t solve_linear_program_with_barrier( const user_problem_t& user_problem, const simplex_solver_settings_t& settings, diff --git a/cpp/src/dual_simplex/solve.hpp b/cpp/src/dual_simplex/solve.hpp index 308c462de5..4bbc908976 100644 --- a/cpp/src/dual_simplex/solve.hpp +++ b/cpp/src/dual_simplex/solve.hpp @@ -73,7 +73,28 @@ lp_status_t solve_linear_program_advanced(const lp_problem_t& original lp_solution_t& original_solution, std::vector& vstatus, std::vector& edge_norms, - work_limit_context_t* work_unit_context = nullptr); + f_t& work_estimate, + work_limit_context_t* work_unit_context = nullptr); + +template +lp_status_t solve_linear_program_advanced(const lp_problem_t& original_lp, + const f_t start_time, + const simplex_solver_settings_t& settings, + lp_solution_t& original_solution, + std::vector& vstatus, + std::vector& edge_norms, + work_limit_context_t* work_unit_context = nullptr) +{ + f_t work_estimate = 0.0; + return solve_linear_program_advanced(original_lp, + start_time, + settings, + original_solution, + vstatus, + edge_norms, + work_estimate, + work_unit_context); +} // Solve the LP using dual simplex and keep the `basis_update_mpf_t` // for future use. @@ -88,8 +109,36 @@ lp_status_t solve_linear_program_with_advanced_basis( std::vector& nonbasic_list, std::vector& vstatus, std::vector& edge_norms, + f_t& work_estimate, work_limit_context_t* work_unit_context = nullptr); +template +lp_status_t solve_linear_program_with_advanced_basis( + const lp_problem_t& original_lp, + const f_t start_time, + const simplex_solver_settings_t& settings, + lp_solution_t& original_solution, + basis_update_mpf_t& ft, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + std::vector& edge_norms, + work_limit_context_t* work_unit_context = nullptr) +{ + f_t work_estimate = 0.0; + return solve_linear_program_with_advanced_basis(original_lp, + start_time, + settings, + original_solution, + ft, + basic_list, + nonbasic_list, + vstatus, + edge_norms, + work_estimate, + work_unit_context); +} + template lp_status_t solve_linear_program_with_barrier(const user_problem_t& user_problem, const simplex_solver_settings_t& settings, @@ -102,6 +151,11 @@ lp_status_t solve_linear_program_with_barrier(const user_problem_t& us lp_solution_t& solution); template +lp_status_t solve_linear_program_with_primal(const user_problem_t& user_problem, + const simplex_solver_settings_t& settings, + f_t start_time, + lp_solution_t& solution); +template lp_status_t solve_linear_program_with_barrier(const user_problem_t& user_problem, const simplex_solver_settings_t& settings, f_t start_time, diff --git a/cpp/src/math_optimization/solver_settings.cu b/cpp/src/math_optimization/solver_settings.cu index 4fb401318c..e6ec6e0927 100644 --- a/cpp/src/math_optimization/solver_settings.cu +++ b/cpp/src/math_optimization/solver_settings.cu @@ -179,6 +179,9 @@ solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings {CUOPT_FOLDING, &pdlp_settings.folding, -1, 1, -1}, {CUOPT_DUALIZE, &pdlp_settings.dualize, -1, 1, -1}, {CUOPT_ORDERING, &pdlp_settings.ordering, -1, 1, -1}, + {CUOPT_INITIAL_PERTURBATION, &pdlp_settings.initial_perturbation, -1, 1, -1}, + {CUOPT_REMOVE_PERTURBATION, &pdlp_settings.remove_perturbation, -1, 1, -1}, + {CUOPT_PRIMAL_PRICING, &pdlp_settings.primal_pricing, 0, 1, 0}, {CUOPT_BARRIER_DUAL_INITIAL_POINT, reinterpret_cast(&pdlp_settings.barrier_dual_initial_point), -1, 2, -1}, {CUOPT_POSTSOLVE_INFO, &pdlp_settings.postsolve_info, -1, 1, -1}, {CUOPT_MIP_CUT_PASSES, &mip_settings.max_cut_passes, -1, std::numeric_limits::max(), 10}, diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index b995dc3f12..63b2f9681e 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -436,7 +436,7 @@ optimization_problem_solution_t convert_dual_simplex_sol( termination_status != pdlp_termination_status_t::TimeLimit && termination_status != pdlp_termination_status_t::ConcurrentLimit) { CUOPT_LOG_INFO("%s Solve status %s", - method == method_t::DualSimplex ? "Dual Simplex" : "Barrier", + method_to_string(method).c_str(), sol.get_termination_status_string().c_str()); } @@ -603,9 +603,12 @@ std::tuple, simplex::lp_status_t, f_t, f_t, f_t f_t norm_rhs = vector_norm2(user_problem.rhs); simplex::simplex_solver_settings_t dual_simplex_settings; - dual_simplex_settings.time_limit = settings.time_limit; - dual_simplex_settings.iteration_limit = settings.iteration_limit; - dual_simplex_settings.concurrent_halt = settings.concurrent_halt; + dual_simplex_settings.time_limit = settings.time_limit; + dual_simplex_settings.iteration_limit = settings.iteration_limit; + dual_simplex_settings.concurrent_halt = settings.concurrent_halt; + dual_simplex_settings.initial_perturbation = settings.initial_perturbation; + dual_simplex_settings.remove_perturbation = settings.remove_perturbation; + dual_simplex_settings.primal_pricing = settings.primal_pricing; if (dual_simplex_settings.concurrent_halt != nullptr) { // Don't show the dual simplex log in concurrent mode. Show the PDLP log instead dual_simplex_settings.log.log = false; @@ -648,6 +651,61 @@ optimization_problem_solution_t run_dual_simplex( method_t::DualSimplex); } +template +std::tuple, simplex::lp_status_t, f_t, f_t, f_t> run_primal( + simplex::user_problem_t& user_problem, + pdlp_solver_settings_t const& settings, + const timer_t& timer) +{ + f_t norm_user_objective = vector_norm2(user_problem.objective); + f_t norm_rhs = vector_norm2(user_problem.rhs); + + simplex::simplex_solver_settings_t primal_settings; + primal_settings.time_limit = settings.time_limit; + primal_settings.iteration_limit = settings.iteration_limit; + primal_settings.concurrent_halt = settings.concurrent_halt; + primal_settings.primal_pricing = settings.primal_pricing; + if (primal_settings.concurrent_halt != nullptr) { + // Don't show the primal simplex log in concurrent mode. Show the PDLP log instead + primal_settings.log.log = false; + } + + simplex::lp_solution_t solution(user_problem.num_rows, user_problem.num_cols); + auto status = simplex::solve_linear_program_with_primal( + user_problem, primal_settings, timer.get_tic_start(), solution); + + CUOPT_LOG_CONDITIONAL_INFO( + !settings.inside_mip, "Primal simplex finished in %.2f seconds", timer.elapsed_time()); + + if (settings.concurrent_halt != nullptr && + (status == simplex::lp_status_t::OPTIMAL || status == simplex::lp_status_t::UNBOUNDED || + status == simplex::lp_status_t::INFEASIBLE || + status == simplex::lp_status_t::UNBOUNDED_OR_INFEASIBLE)) { + // We finished. Tell PDLP to stop if it is still running. + *settings.concurrent_halt = 1; + } + + return {std::move(solution), status, timer.elapsed_time(), norm_user_objective, norm_rhs}; +} + +template +optimization_problem_solution_t run_primal( + mip::problem_t& problem, + pdlp_solver_settings_t const& settings, + const timer_t& timer) +{ + simplex::user_problem_t primal_problem = + cuopt_problem_to_user_problem(problem.handle_ptr, problem); + auto sol_primal = run_primal(primal_problem, settings, timer); + return convert_dual_simplex_sol(problem, + std::get<0>(sol_primal), + std::get<1>(sol_primal), + std::get<2>(sol_primal), + std::get<3>(sol_primal), + std::get<4>(sol_primal), + method_t::Primal); +} + #if PDLP_INSTANTIATE_FLOAT || CUOPT_INSTANTIATE_FLOAT template @@ -1815,19 +1873,27 @@ optimization_problem_solution_t solve_lp_with_method( if constexpr (std::is_same_v) { if (settings.method == method_t::DualSimplex) { return run_dual_simplex(problem, settings, timer); + } else if (settings.method == method_t::Primal) { + return run_primal(problem, settings, timer); } else if (settings.method == method_t::Barrier) { return run_barrier(problem, settings, timer); } else if (settings.method == method_t::Concurrent) { return run_concurrent(problem, settings, timer, is_batch_mode); + } else if (settings.method == method_t::PDLP) { + return run_pdlp(problem, settings, timer, is_batch_mode); } else { + cuopt_expects(false, + error_type_t::ValidationError, + "Invalid LP method. Valid values: Concurrent(0), PDLP(1), DualSimplex(2), " + "Barrier(3), Primal(4)."); return run_pdlp(problem, settings, timer, is_batch_mode); } } else { // Float precision only supports PDLP without presolve/crossover cuopt_expects(settings.method == method_t::PDLP, error_type_t::ValidationError, - "Float precision only supports PDLP method. DualSimplex, Barrier, and Concurrent " - "require double precision."); + "Float precision only supports PDLP method. DualSimplex, Primal, Barrier, and " + "Concurrent require double precision."); return run_pdlp(problem, settings, timer, is_batch_mode); } } diff --git a/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx b/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx index a5dcc78d18..73a2ccccf9 100644 --- a/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx +++ b/python/cuopt/cuopt/linear_programming/solver_settings/solver_settings.pyx @@ -1,4 +1,4 @@ -# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # noqa +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 # cython: profile=False @@ -62,6 +62,7 @@ class SolverMethod(IntEnum): PDLP = auto() DualSimplex = auto() Barrier = auto() + Primal = auto() Unset = auto() def __str__(self): From 553b1fc0d609b5c804a27e4e190f2ed1da481332 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 10 Sep 2026 13:26:10 -0700 Subject: [PATCH 038/113] Remove tiny perturbations before retrying dual simplex --- cpp/src/dual_simplex/phase2.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 9eb3224817..7213861fe4 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2492,7 +2492,7 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, // Check if there's any perturbation const f_t perturbation = amount_of_perturbation(lp, objective); - if (perturbation <= 1e-6) return 0; // OPTIMAL + if (perturbation == 0.0) return 0; // OPTIMAL // Count perturbations on basic vs nonbasic variables i_t num_basic_perturbed = 0; From 62638a77ad36ba4c20ac796c1a386f7a932520ec Mon Sep 17 00:00:00 2001 From: Bulle Mostovoi <135296650+Bubullzz@users.noreply.github.com> Date: Thu, 3 Sep 2026 14:18:22 +0200 Subject: [PATCH 039/113] Replaced cusparse wrappers with simple unique_ptrs and more RAII (#1342) Currently in CuOpt we have many wrappers around cusparse data structures to implement the current behaviour: - limit ownership to a single point - allow for a non-owning access to be passed as arguments to fonctions - custom destructor calling the associated CuSparse destruction API All of these features are handmade using an internal `need_destruction` state. This seems dangerous and easily replacable by a unique_ptr. In this PR I replace these wrappers with unique_ptr when they are owned and I replace them with direct pointers, alliased as *cusparse_object*_view when passed as argument to functions. Authors: - Bulle Mostovoi (https://github.com/Bubullzz) Approvers: - Trevor McKay (https://github.com/tmckayus) - Alice Boucher (https://github.com/aliceb-nv) URL: https://github.com/NVIDIA/cuopt/pull/1342 --- cpp/src/barrier/barrier.cu | 75 +- cpp/src/barrier/cusparse_info.hpp | 36 +- cpp/src/barrier/cusparse_view.cu | 82 +- cpp/src/barrier/cusparse_view.hpp | 15 +- cpp/src/barrier/sparse_matrix_kernels.cuh | 86 +- cpp/src/pdlp/cusparse_view.cu | 920 +++++++----------- cpp/src/pdlp/cusparse_view.hpp | 303 +++--- .../distributed_algorithms.cu | 12 +- .../pdlp/distributed_pdlp/multi_gpu_engine.cu | 18 +- .../distributed_pdlp/multi_gpu_engine.hpp | 8 +- .../optimal_batch_size_handler.cu | 126 +-- cpp/src/pdlp/pdhg.cu | 96 +- cpp/src/pdlp/pdlp.cu | 201 ++-- .../restart_strategy/pdlp_restart_strategy.cu | 18 +- .../adaptive_step_size_strategy.cu | 12 +- .../convergence_information.cu | 24 +- .../infeasibility_information.cu | 24 +- 17 files changed, 917 insertions(+), 1139 deletions(-) diff --git a/cpp/src/barrier/barrier.cu b/cpp/src/barrier/barrier.cu index dca523669c..2d38688f86 100644 --- a/cpp/src/barrier/barrier.cu +++ b/cpp/src/barrier/barrier.cu @@ -1888,13 +1888,13 @@ class iteration_data_t { // v = alpha * A * Dinv * A^T * y + beta * v void gpu_adat_multiply(f_t alpha, const rmm::device_uvector& y, - pdlp::cusparse_dn_vec_descr_wrapper_t const& cusparse_y, + pdlp::cusparse_dn_vec_descr_view cusparse_y, f_t beta, rmm::device_uvector& v, - pdlp::cusparse_dn_vec_descr_wrapper_t const& cusparse_v, + pdlp::cusparse_dn_vec_descr_view cusparse_v, rmm::device_uvector& u, - pdlp::cusparse_dn_vec_descr_wrapper_t const& cusparse_u, + pdlp::cusparse_dn_vec_descr_view cusparse_u, cusparse_view_t& cusparse_view, const rmm::device_uvector& d_inv_diag) const { @@ -2196,20 +2196,20 @@ class iteration_data_t { cusparse_info_t cusparse_info; cusparse_view_t cusparse_view_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_tmp4_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_h_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_dx_residual_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_dy_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_dx_residual_5_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_dx_residual_6_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_dx_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_dx_residual_3_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_dx_residual_4_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_r1_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_dual_residual_; - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_y_residual_; + pdlp::cusparse_dn_vec_uptr cusparse_tmp4_; + pdlp::cusparse_dn_vec_uptr cusparse_h_; + pdlp::cusparse_dn_vec_uptr cusparse_dx_residual_; + pdlp::cusparse_dn_vec_uptr cusparse_dy_; + pdlp::cusparse_dn_vec_uptr cusparse_dx_residual_5_; + pdlp::cusparse_dn_vec_uptr cusparse_dx_residual_6_; + pdlp::cusparse_dn_vec_uptr cusparse_dx_; + pdlp::cusparse_dn_vec_uptr cusparse_dx_residual_3_; + pdlp::cusparse_dn_vec_uptr cusparse_dx_residual_4_; + pdlp::cusparse_dn_vec_uptr cusparse_r1_; + pdlp::cusparse_dn_vec_uptr cusparse_dual_residual_; + pdlp::cusparse_dn_vec_uptr cusparse_y_residual_; // GPU ADAT multiply - pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_u_; + pdlp::cusparse_dn_vec_uptr cusparse_u_; // Device vectors @@ -2736,7 +2736,7 @@ void barrier_solver_t::gpu_compute_residuals(const rmm::device_uvector auto cusparse_d_x = data.cusparse_view_.create_vector(d_x); auto descr_primal_residual = data.cusparse_view_.create_vector(data.d_primal_residual_); - data.cusparse_view_.spmv(-1.0, cusparse_d_x, 1.0, descr_primal_residual); + data.cusparse_view_.spmv(-1.0, cusparse_d_x.get(), 1.0, descr_primal_residual.get()); // Compute bound_residual = E'*u - w - E'*x if (data.n_upper_bounds > 0) { @@ -2760,10 +2760,12 @@ void barrier_solver_t::gpu_compute_residuals(const rmm::device_uvector stream_view_.value()); RAFT_CHECK_CUDA(stream_view_); auto descr_dual_residual = data.cusparse_view_.create_vector(data.d_dual_residual_); - if (data.Q.n > 0) { data.cusparse_Q_view_.spmv(1.0, cusparse_d_x, 1.0, descr_dual_residual); } + if (data.Q.n > 0) { + data.cusparse_Q_view_.spmv(1.0, cusparse_d_x.get(), 1.0, descr_dual_residual.get()); + } // Compute dual_residual = c - A'*y - z + E*v auto cusparse_d_y = data.cusparse_view_.create_vector(d_y); - data.cusparse_view_.transpose_spmv(-1.0, cusparse_d_y, 1.0, descr_dual_residual); + data.cusparse_view_.transpose_spmv(-1.0, cusparse_d_y.get(), 1.0, descr_dual_residual.get()); if (data.n_upper_bounds > 0) { cub::DeviceTransform::Transform( @@ -3144,7 +3146,7 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t(d_dx_residual_6, stream_view_); @@ -3334,8 +3339,9 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::compute_residual_norms_mu_and_objective( if (data.Q.n > 0) { auto cusparse_d_x = data.cusparse_view_.create_vector(data.d_x_); auto cusparse_Qx = data.cusparse_view_.create_vector(data.d_Qx_); - data.cusparse_Q_view_.spmv(1.0, cusparse_d_x, 0.0, cusparse_Qx); + data.cusparse_Q_view_.spmv(1.0, cusparse_d_x.get(), 0.0, cusparse_Qx.get()); rh.xTQx_async(data.d_Qx_, data.d_x_, cublas_handle, stream_view_); } diff --git a/cpp/src/barrier/cusparse_info.hpp b/cpp/src/barrier/cusparse_info.hpp index d88522e79a..7be723c619 100644 --- a/cpp/src/barrier/cusparse_info.hpp +++ b/cpp/src/barrier/cusparse_info.hpp @@ -7,6 +7,7 @@ #pragma once +#include #include #include @@ -17,8 +18,21 @@ #include +#include +#include + namespace cuopt::mathematical_optimization::barrier { +struct cusparse_spgemm_deleter_t { + void operator()(cusparseSpGEMMDescr_t descr) const noexcept + { + if (descr) { CUOPT_CUSPARSE_TRY_NO_THROW(cusparseSpGEMM_destroyDescr(descr)); } + } +}; + +using cusparse_spgemm_uptr = + std::unique_ptr, cusparse_spgemm_deleter_t>; + template struct cusparse_info_t { cusparse_info_t(raft::handle_t const* handle) @@ -35,24 +49,10 @@ struct cusparse_info_t { beta.set_value_async(v, handle->get_stream()); } - ~cusparse_info_t() - { - if (spgemm_descr != nullptr) { - CUOPT_CUSPARSE_TRY_NO_THROW(cusparseSpGEMM_destroyDescr(spgemm_descr)); - } - if (matA_descr != nullptr) { CUOPT_CUSPARSE_TRY_NO_THROW(cusparseDestroySpMat(matA_descr)); } - if (matDAT_descr != nullptr) { - CUOPT_CUSPARSE_TRY_NO_THROW(cusparseDestroySpMat(matDAT_descr)); - } - if (matADAT_descr != nullptr) { - CUOPT_CUSPARSE_TRY_NO_THROW(cusparseDestroySpMat(matADAT_descr)); - } - } - - cusparseSpMatDescr_t matA_descr{nullptr}; - cusparseSpMatDescr_t matDAT_descr{nullptr}; - cusparseSpMatDescr_t matADAT_descr{nullptr}; - cusparseSpGEMMDescr_t spgemm_descr{nullptr}; + pdlp::cusparse_sp_mat_uptr matA_descr; + pdlp::cusparse_sp_mat_uptr matDAT_descr; + pdlp::cusparse_sp_mat_uptr matADAT_descr; + cusparse_spgemm_uptr spgemm_descr; rmm::device_scalar alpha; rmm::device_scalar beta; rmm::device_uvector buffer_size; diff --git a/cpp/src/barrier/cusparse_view.cu b/cpp/src/barrier/cusparse_view.cu index 477200c5e9..03d1b3a13d 100644 --- a/cpp/src/barrier/cusparse_view.cu +++ b/cpp/src/barrier/cusparse_view.cu @@ -200,59 +200,27 @@ cusparse_view_t::cusparse_view_t(raft::handle_t const* handle_ptr, A_T_indices_ = device_copy(A.i, handle_ptr->get_stream()); A_T_data_ = device_copy(A.x, handle_ptr->get_stream()); - cusparseCreateCsr(&A_, - rows_, - cols, - nnz, - A_offsets_.data(), - A_indices_.data(), - A_data_.data(), - CUSPARSE_INDEX_32I, - CUSPARSE_INDEX_32I, - CUSPARSE_INDEX_BASE_ZERO, - CUDA_R_64F); - - cusparseCreateCsr(&A_T_, - cols, - rows_, - nnz, - A_T_offsets_.data(), - A_T_indices_.data(), - A_T_data_.data(), - CUSPARSE_INDEX_32I, - CUSPARSE_INDEX_32I, - CUSPARSE_INDEX_BASE_ZERO, - CUDA_R_64F); - - // Tmp just to init the buffer size and preprocess - cusparseDnVecDescr_t x; - cusparseDnVecDescr_t y; + A_ = pdlp::make_csr( + rows_, cols, nnz, A_offsets_.data(), A_indices_.data(), A_data_.data()); + A_T_ = pdlp::make_csr( + cols, rows_, nnz, A_T_offsets_.data(), A_T_indices_.data(), A_T_data_.data()); + + // Temporary vectors used to initialize the SpMV buffers and preprocessing data. rmm::device_uvector d_x(cols, handle_ptr_->get_stream()); rmm::device_uvector d_y(rows_, handle_ptr_->get_stream()); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsecreatednvec(&x, d_x.size(), d_x.data())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsecreatednvec(&y, d_y.size(), d_y.data())); - - init_spmv_buffer_and_preprocess(A_, x, y, spmv_buffer_, rows_); - init_spmv_buffer_and_preprocess(A_T_, y, x, spmv_buffer_transpose_, A_T_offsets_.size() - 1); - - RAFT_CUSPARSE_TRY(cusparseDestroyDnVec(x)); - RAFT_CUSPARSE_TRY(cusparseDestroyDnVec(y)); -} + auto x = pdlp::make_dnvec(d_x.size(), d_x.data()); + auto y = pdlp::make_dnvec(d_y.size(), d_y.data()); -template -cusparse_view_t::~cusparse_view_t() -{ - CUOPT_CUSPARSE_TRY_NO_THROW(cusparseDestroySpMat(A_)); - if (A_T_ != nullptr) { CUOPT_CUSPARSE_TRY_NO_THROW(cusparseDestroySpMat(A_T_)); } + init_spmv_buffer_and_preprocess(A_.get(), x.get(), y.get(), spmv_buffer_, rows_); + init_spmv_buffer_and_preprocess( + A_T_.get(), y.get(), x.get(), spmv_buffer_transpose_, A_T_offsets_.size() - 1); } template -pdlp::cusparse_dn_vec_descr_wrapper_t cusparse_view_t::create_vector( +pdlp::cusparse_dn_vec_uptr cusparse_view_t::create_vector( rmm::device_uvector const& vec) { - pdlp::cusparse_dn_vec_descr_wrapper_t descr; - descr.create(vec.size(), const_cast(vec.data())); - return descr; + return pdlp::make_dnvec(vec.size(), const_cast(vec.data())); } template @@ -274,16 +242,16 @@ void cusparse_view_t::spmv(f_t alpha, f_t beta, rmm::device_uvector& y) { - pdlp::cusparse_dn_vec_descr_wrapper_t x_cusparse = create_vector(x); - pdlp::cusparse_dn_vec_descr_wrapper_t y_cusparse = create_vector(y); - spmv(alpha, x_cusparse, beta, y_cusparse); + pdlp::cusparse_dn_vec_uptr x_cusparse = create_vector(x); + pdlp::cusparse_dn_vec_uptr y_cusparse = create_vector(y); + spmv(alpha, x_cusparse.get(), beta, y_cusparse.get()); } template void cusparse_view_t::spmv(f_t alpha, - pdlp::cusparse_dn_vec_descr_wrapper_t const& x, + pdlp::cusparse_dn_vec_descr_view x, f_t beta, - pdlp::cusparse_dn_vec_descr_wrapper_t const& y) + pdlp::cusparse_dn_vec_descr_view y) { // Would be simpler if we could pass host data direclty but other cusparse calls with the same // handler depend on device data @@ -298,7 +266,7 @@ void cusparse_view_t::spmv(f_t alpha, raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, (alpha == 1) ? d_one_.data() : d_minus_one_.data(), - A_, + A_.get(), x, d_beta->data(), y, @@ -328,16 +296,16 @@ void cusparse_view_t::transpose_spmv(f_t alpha, rmm::device_uvector& y) { cuopt_assert(A_T_ != nullptr, "transpose_spmv requires an A^T descriptor"); - pdlp::cusparse_dn_vec_descr_wrapper_t x_cusparse = create_vector(x); - pdlp::cusparse_dn_vec_descr_wrapper_t y_cusparse = create_vector(y); - transpose_spmv(alpha, x_cusparse, beta, y_cusparse); + pdlp::cusparse_dn_vec_uptr x_cusparse = create_vector(x); + pdlp::cusparse_dn_vec_uptr y_cusparse = create_vector(y); + transpose_spmv(alpha, x_cusparse.get(), beta, y_cusparse.get()); } template void cusparse_view_t::transpose_spmv(f_t alpha, - pdlp::cusparse_dn_vec_descr_wrapper_t const& x, + pdlp::cusparse_dn_vec_descr_view x, f_t beta, - pdlp::cusparse_dn_vec_descr_wrapper_t const& y) + pdlp::cusparse_dn_vec_descr_view y) { cuopt_assert(A_T_ != nullptr, "transpose_spmv requires an A^T descriptor"); // Would be simpler if we could pass host data direct;y but other cusparse calls with the same @@ -353,7 +321,7 @@ void cusparse_view_t::transpose_spmv(f_t alpha, raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, (alpha == 1) ? d_one_.data() : d_minus_one_.data(), - A_T_, + A_T_.get(), x, d_beta->data(), y, diff --git a/cpp/src/barrier/cusparse_view.hpp b/cpp/src/barrier/cusparse_view.hpp index ea6bf363b9..8a3c38e33c 100644 --- a/cpp/src/barrier/cusparse_view.hpp +++ b/cpp/src/barrier/cusparse_view.hpp @@ -29,9 +29,8 @@ class cusparse_view_t { // Copy CSC -> owned CSR + CSC-transpose, with preprocess. Supports forward and transpose SpMV. // TMP matrix data should already be on the GPU and in CSR not CSC cusparse_view_t(raft::handle_t const* handle_ptr, const csc_matrix_t& A); - ~cusparse_view_t(); - pdlp::cusparse_dn_vec_descr_wrapper_t create_vector(rmm::device_uvector const& vec); + pdlp::cusparse_dn_vec_uptr create_vector(rmm::device_uvector const& vec); template void spmv(f_t alpha, @@ -40,9 +39,9 @@ class cusparse_view_t { std::vector& y); void spmv(f_t alpha, rmm::device_uvector const& x, f_t beta, rmm::device_uvector& y); void spmv(f_t alpha, - pdlp::cusparse_dn_vec_descr_wrapper_t const& x, + pdlp::cusparse_dn_vec_descr_view x, f_t beta, - pdlp::cusparse_dn_vec_descr_wrapper_t const& y); + pdlp::cusparse_dn_vec_descr_view y); template void transpose_spmv(f_t alpha, const std::vector& x, @@ -53,9 +52,9 @@ class cusparse_view_t { f_t beta, rmm::device_uvector& y); void transpose_spmv(f_t alpha, - pdlp::cusparse_dn_vec_descr_wrapper_t const& x, + pdlp::cusparse_dn_vec_descr_view x, f_t beta, - pdlp::cusparse_dn_vec_descr_wrapper_t const& y); + pdlp::cusparse_dn_vec_descr_view y); raft::handle_t const* handle_ptr_{nullptr}; @@ -69,11 +68,11 @@ class cusparse_view_t { rmm::device_uvector A_offsets_; rmm::device_uvector A_indices_; rmm::device_uvector A_data_; - cusparseSpMatDescr_t A_{nullptr}; + pdlp::cusparse_sp_mat_uptr A_; rmm::device_uvector A_T_offsets_; rmm::device_uvector A_T_indices_; rmm::device_uvector A_T_data_; - cusparseSpMatDescr_t A_T_{nullptr}; + pdlp::cusparse_sp_mat_uptr A_T_; rmm::device_buffer spmv_buffer_; rmm::device_buffer spmv_buffer_transpose_; rmm::device_scalar d_one_; diff --git a/cpp/src/barrier/sparse_matrix_kernels.cuh b/cpp/src/barrier/sparse_matrix_kernels.cuh index 0ce8447307..c736e67aa4 100644 --- a/cpp/src/barrier/sparse_matrix_kernels.cuh +++ b/cpp/src/barrier/sparse_matrix_kernels.cuh @@ -29,24 +29,18 @@ void initialize_cusparse_data(raft::handle_t const* handle, f_t chunk_fraction = 0.15; // Create matrix descriptors - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsecreatecsr( - &cusparse_data.matA_descr, A.m, A.n, A_nnz, A.row_start.data(), A.j.data(), A.x.data())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsecreatecsr(&cusparse_data.matDAT_descr, - DAT.n, - DAT.m, - DAT_nnz, - DAT.col_start.data(), - DAT.i.data(), - DAT.x.data())); - - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsecreatecsr(&cusparse_data.matADAT_descr, - ADAT.m, - ADAT.n, - 0, - ADAT.row_start.data(), - ADAT.j.data(), - ADAT.x.data())); - RAFT_CUSPARSE_TRY(cusparseSpGEMM_createDescr(&cusparse_data.spgemm_descr)); + cusparse_data.matA_descr = + pdlp::make_csr(A.m, A.n, A_nnz, A.row_start.data(), A.j.data(), A.x.data()); + cusparse_data.matDAT_descr = pdlp::make_csr( + DAT.n, DAT.m, DAT_nnz, DAT.col_start.data(), DAT.i.data(), DAT.x.data()); + cusparse_data.matADAT_descr = pdlp::make_csr( + ADAT.m, ADAT.n, 0, ADAT.row_start.data(), ADAT.j.data(), ADAT.x.data()); + + { + cusparseSpGEMMDescr_t raw{nullptr}; + RAFT_CUSPARSE_TRY(cusparseSpGEMM_createDescr(&raw)); + cusparse_data.spgemm_descr = cusparse_spgemm_uptr{raw}; + } // Buffer size size_t buffer_size; @@ -54,13 +48,13 @@ void initialize_cusparse_data(raft::handle_t const* handle, CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, cusparse_data.alpha.data(), - cusparse_data.matA_descr, - cusparse_data.matDAT_descr, + cusparse_data.matA_descr.get(), + cusparse_data.matDAT_descr.get(), cusparse_data.beta.data(), - cusparse_data.matADAT_descr, + cusparse_data.matADAT_descr.get(), CUDA_R_64F, CUSPARSE_SPGEMM_ALG3, - cusparse_data.spgemm_descr, + cusparse_data.spgemm_descr.get(), &buffer_size, nullptr)); cusparse_data.buffer_size.resize(buffer_size, handle->get_stream()); @@ -69,31 +63,31 @@ void initialize_cusparse_data(raft::handle_t const* handle, CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, cusparse_data.alpha.data(), - cusparse_data.matA_descr, - cusparse_data.matDAT_descr, + cusparse_data.matA_descr.get(), + cusparse_data.matDAT_descr.get(), cusparse_data.beta.data(), - cusparse_data.matADAT_descr, + cusparse_data.matADAT_descr.get(), CUDA_R_64F, CUSPARSE_SPGEMM_ALG3, - cusparse_data.spgemm_descr, + cusparse_data.spgemm_descr.get(), &buffer_size, cusparse_data.buffer_size.data())); int64_t num_prods; - RAFT_CUSPARSE_TRY(cusparseSpGEMM_getNumProducts(cusparse_data.spgemm_descr, &num_prods)); + RAFT_CUSPARSE_TRY(cusparseSpGEMM_getNumProducts(cusparse_data.spgemm_descr.get(), &num_prods)); size_t buffer_size_3_size; RAFT_CUSPARSE_TRY(cusparseSpGEMM_estimateMemory(handle->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, cusparse_data.alpha.data(), - cusparse_data.matA_descr, - cusparse_data.matDAT_descr, + cusparse_data.matA_descr.get(), + cusparse_data.matDAT_descr.get(), cusparse_data.beta.data(), - cusparse_data.matADAT_descr, + cusparse_data.matADAT_descr.get(), CUDA_R_64F, CUSPARSE_SPGEMM_ALG3, - cusparse_data.spgemm_descr, + cusparse_data.spgemm_descr.get(), chunk_fraction, &buffer_size_3_size, nullptr, @@ -104,13 +98,13 @@ void initialize_cusparse_data(raft::handle_t const* handle, CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, cusparse_data.alpha.data(), - cusparse_data.matA_descr, - cusparse_data.matDAT_descr, + cusparse_data.matA_descr.get(), + cusparse_data.matDAT_descr.get(), cusparse_data.beta.data(), - cusparse_data.matADAT_descr, + cusparse_data.matADAT_descr.get(), CUDA_R_64F, CUSPARSE_SPGEMM_ALG3, - cusparse_data.spgemm_descr, + cusparse_data.spgemm_descr.get(), chunk_fraction, &buffer_size_3_size, cusparse_data.buffer_size_3.data(), @@ -131,20 +125,20 @@ void multiply_kernels(raft::handle_t const* handle, CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, cusparse_data.alpha.data(), - cusparse_data.matA_descr, // non-const descriptor supported - cusparse_data.matDAT_descr, // non-const descriptor supported + cusparse_data.matA_descr.get(), // non-const descriptor supported + cusparse_data.matDAT_descr.get(), // non-const descriptor supported cusparse_data.beta.data(), - cusparse_data.matADAT_descr, + cusparse_data.matADAT_descr.get(), CUDA_R_64F, CUSPARSE_SPGEMM_ALG3, - cusparse_data.spgemm_descr, + cusparse_data.spgemm_descr.get(), &cusparse_data.buffer_size_2_size, cusparse_data.buffer_size_2.data())); // get matrix C non-zero entries C_nnz1 int64_t ADAT_num_rows, ADAT_num_cols, ADAT_nnz1; - RAFT_CUSPARSE_TRY( - cusparseSpMatGetSize(cusparse_data.matADAT_descr, &ADAT_num_rows, &ADAT_num_cols, &ADAT_nnz1)); + RAFT_CUSPARSE_TRY(cusparseSpMatGetSize( + cusparse_data.matADAT_descr.get(), &ADAT_num_rows, &ADAT_num_cols, &ADAT_nnz1)); // cuSPARSE sizes the product in 64 bits while the CSR arrays are indexed by i_t; narrowing would // reach RMM as a negative count and surface as an unrelated device_uvector overflow. if (ADAT_nnz1 > std::numeric_limits::max()) { @@ -160,19 +154,19 @@ void multiply_kernels(raft::handle_t const* handle, // update matC with the new pointers RAFT_CUSPARSE_TRY(cusparseCsrSetPointers( - cusparse_data.matADAT_descr, ADAT.row_start.data(), ADAT.j.data(), ADAT.x.data())); + cusparse_data.matADAT_descr.get(), ADAT.row_start.data(), ADAT.j.data(), ADAT.x.data())); RAFT_CUSPARSE_TRY(cusparseSpGEMM_copy(handle->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, cusparse_data.alpha.data(), - cusparse_data.matA_descr, - cusparse_data.matDAT_descr, + cusparse_data.matA_descr.get(), + cusparse_data.matDAT_descr.get(), cusparse_data.beta.data(), - cusparse_data.matADAT_descr, + cusparse_data.matADAT_descr.get(), CUDA_R_64F, CUSPARSE_SPGEMM_ALG3, - cusparse_data.spgemm_descr)); + cusparse_data.spgemm_descr.get())); handle->sync_stream(); } diff --git a/cpp/src/pdlp/cusparse_view.cu b/cpp/src/pdlp/cusparse_view.cu index 9d3a0cc67c..d0802ae0b0 100644 --- a/cpp/src/pdlp/cusparse_view.cu +++ b/cpp/src/pdlp/cusparse_view.cu @@ -30,129 +30,9 @@ struct double_to_float_functor { namespace cuopt::mathematical_optimization::pdlp { -// cusparse_sp_mat_descr_wrapper_t implementation -template -cusparse_sp_mat_descr_wrapper_t::cusparse_sp_mat_descr_wrapper_t() - : need_destruction_(false) -{ -} - -template -cusparse_sp_mat_descr_wrapper_t::~cusparse_sp_mat_descr_wrapper_t() -{ - if (need_destruction_) { RAFT_CUSPARSE_TRY_NO_THROW(cusparseDestroySpMat(descr_)); } -} - -template -cusparse_sp_mat_descr_wrapper_t::cusparse_sp_mat_descr_wrapper_t( - const cusparse_sp_mat_descr_wrapper_t& other) - : descr_(other.descr_), need_destruction_(false) -{ -} - -template -void cusparse_sp_mat_descr_wrapper_t::create( - int64_t m, int64_t n, int64_t nnz, i_t* offsets, i_t* indices, f_t* values) -{ - RAFT_CUSPARSE_TRY( - raft::sparse::detail::cusparsecreatecsr(&descr_, m, n, nnz, offsets, indices, values)); - need_destruction_ = true; -} - -template -cusparse_sp_mat_descr_wrapper_t::operator cusparseSpMatDescr_t() const -{ - return descr_; -} - -// cusparse_dn_vec_descr_wrapper_t implementation -template -cusparse_dn_vec_descr_wrapper_t::cusparse_dn_vec_descr_wrapper_t() : need_destruction_(false) -{ -} - -template -cusparse_dn_vec_descr_wrapper_t::~cusparse_dn_vec_descr_wrapper_t() -{ - if (need_destruction_) { RAFT_CUSPARSE_TRY_NO_THROW(cusparseDestroyDnVec(descr_)); } -} - -template -cusparse_dn_vec_descr_wrapper_t::cusparse_dn_vec_descr_wrapper_t( - const cusparse_dn_vec_descr_wrapper_t& other) - : descr_(other.descr_), need_destruction_(false) -{ -} - -template -cusparse_dn_vec_descr_wrapper_t& cusparse_dn_vec_descr_wrapper_t::operator=( - cusparse_dn_vec_descr_wrapper_t&& other) -{ - if (need_destruction_) { RAFT_CUSPARSE_TRY(cusparseDestroyDnVec(descr_)); } - descr_ = other.descr_; - need_destruction_ = other.need_destruction_; - other.need_destruction_ = false; - return *this; -} - -template -void cusparse_dn_vec_descr_wrapper_t::create(int64_t size, f_t* values) -{ - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsecreatednvec(&descr_, size, values)); - need_destruction_ = true; -} - -template -cusparse_dn_vec_descr_wrapper_t::operator cusparseDnVecDescr_t() const -{ - return descr_; -} - -// cusparse_dn_mat_descr_wrapper_t implementation -template -cusparse_dn_mat_descr_wrapper_t::cusparse_dn_mat_descr_wrapper_t() : need_destruction_(false) -{ -} - -template -cusparse_dn_mat_descr_wrapper_t::~cusparse_dn_mat_descr_wrapper_t() -{ - if (need_destruction_) { RAFT_CUSPARSE_TRY_NO_THROW(cusparseDestroyDnMat(descr_)); } -} - -template -cusparse_dn_mat_descr_wrapper_t::cusparse_dn_mat_descr_wrapper_t( - const cusparse_dn_mat_descr_wrapper_t& other) - : descr_(other.descr_), need_destruction_(false) -{ -} - -template -cusparse_dn_mat_descr_wrapper_t& cusparse_dn_mat_descr_wrapper_t::operator=( - cusparse_dn_mat_descr_wrapper_t&& other) -{ - if (need_destruction_) { RAFT_CUSPARSE_TRY(cusparseDestroyDnMat(descr_)); } - descr_ = other.descr_; - need_destruction_ = other.need_destruction_; - other.need_destruction_ = false; - return *this; -} - -template -void cusparse_dn_mat_descr_wrapper_t::create( - int64_t row, int64_t col, int64_t ld, f_t* values, cusparseOrder_t order) -{ - if (need_destruction_) { RAFT_CUSPARSE_TRY(cusparseDestroyDnMat(descr_)); } - RAFT_CUSPARSE_TRY( - raft::sparse::detail::cusparsecreatednmat(&descr_, row, col, ld, values, order)); - need_destruction_ = true; -} - -template -cusparse_dn_mat_descr_wrapper_t::operator cusparseDnMatDescr_t() const -{ - return descr_; -} +// All factories and aliases for SpMat/DnVec/DnMat live in the header. +// Deleter operator() bodies for SpMVOpDescr/SpMVOpPlan are defined further down +// because they need dlsym-resolved cuSPARSE symbols. #if CUDA_VER_12_4_UP struct dynamic_load_runtime { @@ -250,10 +130,10 @@ void my_cusparsespmm_preprocess(cusparseHandle_t handle, cusparseOperation_t opA, cusparseOperation_t opB, const T* alpha, - const cusparseSpMatDescr_t matA, - const cusparseDnMatDescr_t matB, + cusparse_sp_mat_descr_view matA, + cusparse_dn_mat_descr_view matB, const T* beta, - const cusparseDnMatDescr_t matC, + cusparse_dn_mat_descr_view matC, cusparseSpMMAlg_t alg, void* externalBuffer, cudaStream_t stream) @@ -265,7 +145,7 @@ void my_cusparsespmm_preprocess(cusparseHandle_t handle, return CUDA_R_64F; } }(); - CUSPARSE_CHECK(cusparseSetStream(handle, stream)); + RAFT_CUSPARSE_TRY(cusparseSetStream(handle, stream)); RAFT_CUSPARSE_TRY(cusparseSpMM_preprocess( handle, opA, opB, alpha, matA, matB, beta, matC, float_type, alg, externalBuffer)); } @@ -305,135 +185,86 @@ using cusparseSpMVOp_sig = cusparse_sig; -cusparseStatus_t cusparse_spmvop_descr_wrapper_t::dlsym_create(cusparseHandle_t handle, - cusparseSpMVOpDescr_t* descr, - cusparseOperation_t opA, - cusparseSpMatDescr_t matA, - cusparseDnVecDescr_t vecX, - cusparseDnVecDescr_t vecY, - cusparseDnVecDescr_t vecZ, - cudaDataType computeType, - void* buffer) +cusparseStatus_t cusparse_spmvop_buffer_size(cusparseHandle_t handle, + cusparseOperation_t opA, + cusparseSpMatDescr_t matA, + cusparseDnVecDescr_t vecX, + cusparseDnVecDescr_t vecY, + cusparseDnVecDescr_t vecZ, + cudaDataType computeType, + size_t* bufferSize) { static const auto fn = - dynamic_load_runtime::function("cusparseSpMVOp_createDescr"); + dynamic_load_runtime::function("cusparseSpMVOp_bufferSize"); return (*fn)( - handle, descr, opA, matA, vecX, vecY, vecZ, computeType, CUSPARSE_SPMVOP_ALG_DEFAULT, buffer); + handle, opA, matA, vecX, vecY, vecZ, computeType, CUSPARSE_SPMVOP_ALG_DEFAULT, bufferSize); } -cusparseStatus_t cusparse_spmvop_descr_wrapper_t::dlsym_destroy(cusparseSpMVOpDescr_t descr) +cusparseStatus_t cusparse_spmvop_create_descr(cusparseHandle_t handle, + cusparseSpMVOpDescr_t* descr, + cusparseOperation_t opA, + cusparseSpMatDescr_t matA, + cusparseDnVecDescr_t vecX, + cusparseDnVecDescr_t vecY, + cusparseDnVecDescr_t vecZ, + cudaDataType computeType, + void* buffer) { static const auto fn = - dynamic_load_runtime::function("cusparseSpMVOp_destroyDescr"); - return (*fn)(descr); + dynamic_load_runtime::function("cusparseSpMVOp_createDescr"); + return (*fn)( + handle, descr, opA, matA, vecX, vecY, vecZ, computeType, CUSPARSE_SPMVOP_ALG_DEFAULT, buffer); } -cusparseStatus_t cusparse_spmvop_plan_wrapper_t::dlsym_create(cusparseHandle_t handle, - cusparseSpMVOpDescr_t descr, - cusparseSpMVOpPlan_t* plan, - char* ltoIRBuf, - size_t ltoIRSize) +void cusparse_spmvop_descr_deleter_t::operator()(cusparseSpMVOpDescr_t descr) const noexcept { + if (!descr) { return; } static const auto fn = - dynamic_load_runtime::function("cusparseSpMVOp_createPlan"); - return (*fn)(handle, descr, plan, ltoIRBuf, ltoIRSize); + dynamic_load_runtime::function("cusparseSpMVOp_destroyDescr"); + if (fn.has_value()) { RAFT_CUSPARSE_TRY_NO_THROW((*fn)(descr)); } } -cusparseStatus_t cusparse_spmvop_plan_wrapper_t::dlsym_destroy(cusparseSpMVOpPlan_t plan) +void cusparse_spmvop_plan_deleter_t::operator()(cusparseSpMVOpPlan_t plan) const noexcept { + if (!plan) { return; } static const auto fn = dynamic_load_runtime::function("cusparseSpMVOp_destroyPlan"); - return (*fn)(plan); -} - -cusparse_spmvop_descr_wrapper_t::cusparse_spmvop_descr_wrapper_t() - : descr_(nullptr), need_destruction_(false) -{ -} - -cusparse_spmvop_descr_wrapper_t::~cusparse_spmvop_descr_wrapper_t() -{ - if (!need_destruction_) { return; } - RAFT_CUSPARSE_TRY_NO_THROW(dlsym_destroy(descr_)); -} - -cusparse_spmvop_descr_wrapper_t::cusparse_spmvop_descr_wrapper_t( - const cusparse_spmvop_descr_wrapper_t& other) - : descr_(other.descr_), need_destruction_(false) -{ -} - -cusparse_spmvop_descr_wrapper_t& cusparse_spmvop_descr_wrapper_t::operator=( - cusparse_spmvop_descr_wrapper_t&& other) -{ - if (need_destruction_) { RAFT_CUSPARSE_TRY(dlsym_destroy(descr_)); } - descr_ = other.descr_; - need_destruction_ = other.need_destruction_; - other.need_destruction_ = false; - return *this; + if (fn.has_value()) { RAFT_CUSPARSE_TRY_NO_THROW((*fn)(plan)); } } -void cusparse_spmvop_descr_wrapper_t::create(cusparseHandle_t handle, +cusparse_spmvop_descr_uptr make_spmvop_descr(cusparseHandle_t handle, cusparseOperation_t opA, - cusparseSpMatDescr_t matA, - cusparseDnVecDescr_t vecX, - cusparseDnVecDescr_t vecY, - cusparseDnVecDescr_t vecZ, + cusparse_sp_mat_descr_view matA, + cusparse_dn_vec_descr_view vecX, + cusparse_dn_vec_descr_view vecY, + cusparse_dn_vec_descr_view vecZ, cudaDataType computeType, rmm::device_uvector& buffer) { - if (need_destruction_) { RAFT_CUSPARSE_TRY(dlsym_destroy(descr_)); } - RAFT_CUSPARSE_TRY( - dlsym_create(handle, &descr_, opA, matA, vecX, vecY, vecZ, computeType, buffer.data())); - need_destruction_ = true; -} - -cusparse_spmvop_descr_wrapper_t::operator cusparseSpMVOpDescr_t() const { return descr_; } - -cusparse_spmvop_plan_wrapper_t::cusparse_spmvop_plan_wrapper_t() - : plan_(nullptr), need_destruction_(false) -{ + cusparseSpMVOpDescr_t descr{nullptr}; + RAFT_CUSPARSE_TRY(cusparse_spmvop_create_descr( + handle, &descr, opA, matA, vecX, vecY, vecZ, computeType, buffer.data())); + return cusparse_spmvop_descr_uptr{descr}; } -cusparse_spmvop_plan_wrapper_t::~cusparse_spmvop_plan_wrapper_t() +cusparse_spmvop_plan_uptr make_spmvop_plan(cusparseHandle_t handle, cusparseSpMVOpDescr_t descr) { - if (!need_destruction_) { return; } - RAFT_CUSPARSE_TRY_NO_THROW(dlsym_destroy(plan_)); -} - -cusparse_spmvop_plan_wrapper_t::cusparse_spmvop_plan_wrapper_t( - const cusparse_spmvop_plan_wrapper_t& other) - : plan_(other.plan_), need_destruction_(false) -{ -} - -cusparse_spmvop_plan_wrapper_t& cusparse_spmvop_plan_wrapper_t::operator=( - cusparse_spmvop_plan_wrapper_t&& other) -{ - if (need_destruction_) { RAFT_CUSPARSE_TRY(dlsym_destroy(plan_)); } - plan_ = other.plan_; - need_destruction_ = other.need_destruction_; - other.need_destruction_ = false; - return *this; -} - -void cusparse_spmvop_plan_wrapper_t::create(cusparseHandle_t handle, cusparseSpMVOpDescr_t descr) -{ - if (need_destruction_) { RAFT_CUSPARSE_TRY(dlsym_destroy(plan_)); } + static const auto fn = + dynamic_load_runtime::function("cusparseSpMVOp_createPlan"); + if (!fn.has_value()) { EXE_CUOPT_FAIL("Unable to resolve cusparseSpMVOp_createPlan at runtime"); } + cusparseSpMVOpPlan_t plan{nullptr}; // cuOpt does not supply user-provided LTO IR; pass nullptr/0 so cuSPARSE JITs internally. - RAFT_CUSPARSE_TRY(dlsym_create(handle, descr, &plan_, /*ltoIRBuf=*/nullptr, /*ltoIRSize=*/0)); - need_destruction_ = true; + RAFT_CUSPARSE_TRY((*fn)(handle, descr, &plan, /*ltoIRBuf=*/nullptr, /*ltoIRSize=*/0)); + return cusparse_spmvop_plan_uptr{plan}; } -cusparse_spmvop_plan_wrapper_t::operator cusparseSpMVOpPlan_t() const { return plan_; } - void cusparse_spmvop_run(cusparseHandle_t handle, cusparseSpMVOpPlan_t plan, const void* alpha, const void* beta, - cusparseDnVecDescr_t vecX, - cusparseDnVecDescr_t vecY, - cusparseDnVecDescr_t vecZ, + cusparse_dn_vec_descr_view vecX, + cusparse_dn_vec_descr_view vecY, + cusparse_dn_vec_descr_view vecZ, cudaStream_t stream) { static const auto func = dynamic_load_runtime::function("cusparseSpMVOp"); @@ -459,18 +290,6 @@ cusparse_view_t::cusparse_view_t( bool enable_mixed_precision_spmv) : batch_mode_(climber_strategies.size() > 1), handle_ptr_(handle_ptr), - A{}, - A_T{}, - c{}, - primal_solution{}, - dual_solution{}, - primal_gradient{}, - dual_gradient{}, - current_AtY{}, - next_AtY{}, - potential_next_dual_solution{}, - tmp_primal{}, - tmp_dual{}, A_T_{op_problem_scaled.reverse_coefficients}, A_T_offsets_{op_problem_scaled.reverse_offsets}, A_T_indices_{op_problem_scaled.reverse_constraints}, @@ -500,113 +319,111 @@ cusparse_view_t::cusparse_view_t( #endif // setup cusparse view - A.create(op_problem_scaled.n_constraints, - op_problem_scaled.n_variables, - static_cast(A_.size()), - const_cast(op_problem_scaled.offsets.data()), - const_cast(op_problem_scaled.variables.data()), - const_cast(op_problem_scaled.coefficients.data())); - - // A_T can have a different nnz than A in multi-GPU shards - // A is just what is needed to compute A_x for owned constraints - // A_T is just what is needed to compute A_T_y for owned variables - A_T.create(op_problem_scaled.n_variables, - op_problem_scaled.n_constraints, - static_cast(A_T_.size()), - const_cast(A_T_offsets_.data()), - const_cast(A_T_indices_.data()), - const_cast(A_T_.data())); - - c.create(op_problem_scaled.n_variables, - const_cast(op_problem_scaled.objective_coefficients.data())); - - primal_solution.create(op_problem_scaled.n_variables, - current_saddle_point_state.get_primal_solution().data()); - dual_solution.create(op_problem_scaled.n_constraints, - current_saddle_point_state.get_dual_solution().data()); - - // TODO batch mdoe: convert those to RAII views + A = make_csr(op_problem_scaled.n_constraints, + op_problem_scaled.n_variables, + static_cast(A_.size()), + const_cast(op_problem_scaled.offsets.data()), + const_cast(op_problem_scaled.variables.data()), + const_cast(op_problem_scaled.coefficients.data())); + + A_T = make_csr(op_problem_scaled.n_variables, + op_problem_scaled.n_constraints, + static_cast(A_T_.size()), + const_cast(A_T_offsets_.data()), + const_cast(A_T_indices_.data()), + const_cast(A_T_.data())); + + c = make_dnvec(op_problem_scaled.n_variables, + const_cast(op_problem_scaled.objective_coefficients.data())); + + primal_solution = make_dnvec(op_problem_scaled.n_variables, + current_saddle_point_state.get_primal_solution().data()); + dual_solution = make_dnvec(op_problem_scaled.n_constraints, + current_saddle_point_state.get_dual_solution().data()); + if (batch_mode_) { [[maybe_unused]] const bool is_cupdlpx = is_cupdlpx_restart(hyper_params); cuopt_assert(is_cupdlpx, "Batch mode only supported with cuPDLPx restart"); - batch_dual_solutions.create(op_problem_scaled.n_constraints, - climber_strategies.size(), - climber_strategies.size(), - current_saddle_point_state.get_dual_solution().data(), - CUSPARSE_ORDER_ROW); - batch_current_AtYs.create(op_problem_scaled.n_variables, - climber_strategies.size(), - climber_strategies.size(), - current_saddle_point_state.get_current_AtY().data(), - CUSPARSE_ORDER_ROW); - batch_potential_next_dual_solution.create(op_problem_scaled.n_constraints, - climber_strategies.size(), - op_problem_scaled.n_constraints, - _potential_next_dual_solution.data(), - CUSPARSE_ORDER_COL); - batch_next_AtYs.create(op_problem_scaled.n_variables, - climber_strategies.size(), - op_problem_scaled.n_variables, - current_saddle_point_state.get_next_AtY().data(), - CUSPARSE_ORDER_COL); + batch_dual_solutions = make_dnmat(op_problem_scaled.n_constraints, + climber_strategies.size(), + climber_strategies.size(), + current_saddle_point_state.get_dual_solution().data(), + CUSPARSE_ORDER_ROW); + batch_current_AtYs = make_dnmat(op_problem_scaled.n_variables, + climber_strategies.size(), + climber_strategies.size(), + current_saddle_point_state.get_current_AtY().data(), + CUSPARSE_ORDER_ROW); + batch_potential_next_dual_solution = make_dnmat(op_problem_scaled.n_constraints, + climber_strategies.size(), + op_problem_scaled.n_constraints, + _potential_next_dual_solution.data(), + CUSPARSE_ORDER_COL); + batch_next_AtYs = make_dnmat(op_problem_scaled.n_variables, + climber_strategies.size(), + op_problem_scaled.n_variables, + current_saddle_point_state.get_next_AtY().data(), + CUSPARSE_ORDER_COL); cuopt_assert(_reflected_primal_solution.size() >= static_cast(op_problem_scaled.n_variables) * climber_strategies.size(), "Reflected primal solution undersized"); - batch_reflected_primal_solutions.create(op_problem_scaled.n_variables, - climber_strategies.size(), - climber_strategies.size(), - _reflected_primal_solution.data(), - CUSPARSE_ORDER_ROW); - batch_dual_gradients.create(op_problem_scaled.n_constraints, - climber_strategies.size(), - climber_strategies.size(), - current_saddle_point_state.get_dual_gradient().data(), - CUSPARSE_ORDER_ROW); + batch_reflected_primal_solutions = make_dnmat(op_problem_scaled.n_variables, + climber_strategies.size(), + climber_strategies.size(), + _reflected_primal_solution.data(), + CUSPARSE_ORDER_ROW); + batch_dual_gradients = make_dnmat(op_problem_scaled.n_constraints, + climber_strategies.size(), + climber_strategies.size(), + current_saddle_point_state.get_dual_gradient().data(), + CUSPARSE_ORDER_ROW); } // Necessary even in non batch mode (because of infeasiblity detection) - batch_delta_primal_solutions.create(op_problem_scaled.n_variables, - climber_strategies.size(), - op_problem_scaled.n_variables, - current_saddle_point_state.get_delta_primal().data(), - CUSPARSE_ORDER_COL); - batch_delta_dual_solutions.create(op_problem_scaled.n_constraints, + batch_delta_primal_solutions = + make_dnmat(op_problem_scaled.n_variables, + climber_strategies.size(), + op_problem_scaled.n_variables, + current_saddle_point_state.get_delta_primal().data(), + CUSPARSE_ORDER_COL); + batch_delta_dual_solutions = make_dnmat(op_problem_scaled.n_constraints, + climber_strategies.size(), + op_problem_scaled.n_constraints, + current_saddle_point_state.get_delta_dual().data(), + CUSPARSE_ORDER_COL); + batch_tmp_duals = make_dnmat(op_problem_scaled.n_constraints, climber_strategies.size(), op_problem_scaled.n_constraints, - current_saddle_point_state.get_delta_dual().data(), + _tmp_dual.data(), CUSPARSE_ORDER_COL); - batch_tmp_duals.create(op_problem_scaled.n_constraints, - climber_strategies.size(), - op_problem_scaled.n_constraints, - _tmp_dual.data(), - CUSPARSE_ORDER_COL); - batch_tmp_primals.create(op_problem_scaled.n_variables, - climber_strategies.size(), - op_problem_scaled.n_variables, - _tmp_primal.data(), - CUSPARSE_ORDER_COL); - - primal_gradient.create( - current_saddle_point_state.get_primal_gradient().size(), // It is 0 in cupdlpx - current_saddle_point_state.get_primal_gradient().data()); - dual_gradient.create(op_problem_scaled.n_constraints, - current_saddle_point_state.get_dual_gradient().data()); - - current_AtY.create(op_problem_scaled.n_variables, - current_saddle_point_state.get_current_AtY().data()); - next_AtY.create(op_problem_scaled.n_variables, current_saddle_point_state.get_next_AtY().data()); - - potential_next_dual_solution.create(op_problem_scaled.n_constraints, - _potential_next_dual_solution.data()); - - tmp_primal.create(op_problem_scaled.n_variables, _tmp_primal.data()); - tmp_dual.create(op_problem_scaled.n_constraints, _tmp_dual.data()); + batch_tmp_primals = make_dnmat(op_problem_scaled.n_variables, + climber_strategies.size(), + op_problem_scaled.n_variables, + _tmp_primal.data(), + CUSPARSE_ORDER_COL); + + primal_gradient = + make_dnvec(current_saddle_point_state.get_primal_gradient().size(), // It is 0 in cupdlpx + current_saddle_point_state.get_primal_gradient().data()); + dual_gradient = make_dnvec(op_problem_scaled.n_constraints, + current_saddle_point_state.get_dual_gradient().data()); + + current_AtY = make_dnvec(op_problem_scaled.n_variables, + current_saddle_point_state.get_current_AtY().data()); + next_AtY = make_dnvec(op_problem_scaled.n_variables, + current_saddle_point_state.get_next_AtY().data()); + + potential_next_dual_solution = + make_dnvec(op_problem_scaled.n_constraints, _potential_next_dual_solution.data()); + + tmp_primal = make_dnvec(op_problem_scaled.n_variables, _tmp_primal.data()); + tmp_dual = make_dnvec(op_problem_scaled.n_constraints, _tmp_dual.data()); if (hyper_params.use_reflected_primal_dual) { cuopt_assert( _reflected_primal_solution.size() >= static_cast(op_problem_scaled.n_variables), "Reflected primal solution undersized"); - reflected_primal_solution.create(op_problem_scaled.n_variables, - _reflected_primal_solution.data()); + reflected_primal_solution = + make_dnvec(op_problem_scaled.n_variables, _reflected_primal_solution.data()); } const rmm::device_scalar alpha{one_v, handle_ptr->get_stream()}; @@ -616,10 +433,10 @@ cusparse_view_t::cusparse_view_t( raft::sparse::detail::cusparsespmv_buffersize(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - c, + A.get(), + c.get(), beta.data(), - dual_solution, + dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_non_transpose, handle_ptr->get_stream())); @@ -630,10 +447,10 @@ cusparse_view_t::cusparse_view_t( raft::sparse::detail::cusparsespmv_buffersize(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - dual_solution, + A_T.get(), + dual_solution.get(), beta.data(), - c, + c.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_transpose, handle_ptr->get_stream())); @@ -647,10 +464,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - batch_delta_dual_solutions, + A_T.get(), + batch_delta_dual_solutions.get(), beta.data(), - batch_tmp_primals, + batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, &buffer_size_transpose_batch, handle_ptr->get_stream())); @@ -662,10 +479,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - batch_delta_primal_solutions, + A.get(), + batch_delta_primal_solutions.get(), beta.data(), - batch_tmp_duals, + batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, &buffer_size_non_transpose_batch, handle_ptr->get_stream())); @@ -679,10 +496,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - batch_dual_solutions, + A_T.get(), + batch_dual_solutions.get(), beta.data(), - batch_current_AtYs, + batch_current_AtYs.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &buffer_size_transpose_batch_row_row, handle_ptr->get_stream())); @@ -694,10 +511,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - batch_reflected_primal_solutions, + A.get(), + batch_reflected_primal_solutions.get(), beta.data(), - batch_dual_gradients, + batch_dual_gradients.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &buffer_size_non_transpose_batch_row_row, handle_ptr->get_stream())); @@ -709,10 +526,10 @@ cusparse_view_t::cusparse_view_t( my_cusparsespmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - c, + A.get(), + c.get(), beta.data(), - dual_solution, + dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_non_transpose.data(), handle_ptr->get_stream()); @@ -720,10 +537,10 @@ cusparse_view_t::cusparse_view_t( my_cusparsespmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - dual_solution, + A_T.get(), + dual_solution.get(), beta.data(), - c, + c.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_transpose.data(), handle_ptr->get_stream()); @@ -731,10 +548,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - batch_delta_dual_solutions, + A_T.get(), + batch_delta_dual_solutions.get(), beta.data(), - batch_tmp_primals, + batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, buffer_transpose_batch.data(), handle_ptr->get_stream()); @@ -743,10 +560,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - batch_delta_primal_solutions, + A.get(), + batch_delta_primal_solutions.get(), beta.data(), - batch_tmp_duals, + batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, buffer_non_transpose_batch.data(), handle_ptr->get_stream()); @@ -756,10 +573,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - batch_dual_solutions, + A_T.get(), + batch_dual_solutions.get(), beta.data(), - batch_current_AtYs, + batch_current_AtYs.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, buffer_transpose_batch_row_row_.data(), handle_ptr->get_stream()); @@ -768,10 +585,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - batch_reflected_primal_solutions, + A.get(), + batch_reflected_primal_solutions.get(), beta.data(), - batch_dual_gradients, + batch_dual_gradients.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, buffer_non_transpose_batch_row_row_.data(), handle_ptr->get_stream()); @@ -797,19 +614,18 @@ cusparse_view_t::cusparse_view_t( double_to_float_functor{}, handle_ptr->get_stream().value())); - A_mixed_.create(op_problem_scaled.n_constraints, - op_problem_scaled.n_variables, - op_problem_scaled.nnz, - const_cast(op_problem_scaled.offsets.data()), - const_cast(op_problem_scaled.variables.data()), - A_float_.data()); - - A_T_mixed_.create(op_problem_scaled.n_variables, - op_problem_scaled.n_constraints, - op_problem_scaled.nnz, - const_cast(A_T_offsets_.data()), - const_cast(A_T_indices_.data()), - A_T_float_.data()); + A_mixed_ = make_csr(op_problem_scaled.n_constraints, + op_problem_scaled.n_variables, + op_problem_scaled.nnz, + const_cast(op_problem_scaled.offsets.data()), + const_cast(op_problem_scaled.variables.data()), + A_float_.data()); + A_T_mixed_ = make_csr(op_problem_scaled.n_variables, + op_problem_scaled.n_constraints, + op_problem_scaled.nnz, + const_cast(A_T_offsets_.data()), + const_cast(A_T_indices_.data()), + A_T_float_.data()); const rmm::device_scalar alpha_d{one_v, handle_ptr->get_stream()}; const rmm::device_scalar beta_d{zero_v, handle_ptr->get_stream()}; @@ -818,10 +634,10 @@ cusparse_view_t::cusparse_view_t( mixed_precision_spmv_buffersize(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha_d.data(), - A_mixed_, - c, + A_mixed_.get(), + c.get(), beta_d.data(), - dual_solution, + dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, handle_ptr->get_stream()); buffer_non_transpose_mixed_.resize(buffer_size_non_transpose_mixed, handle_ptr->get_stream()); @@ -830,10 +646,10 @@ cusparse_view_t::cusparse_view_t( mixed_precision_spmv_buffersize(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha_d.data(), - A_T_mixed_, - dual_solution, + A_T_mixed_.get(), + dual_solution.get(), beta_d.data(), - c, + c.get(), CUSPARSE_SPMV_CSR_ALG2, handle_ptr->get_stream()); buffer_transpose_mixed_.resize(buffer_size_transpose_mixed, handle_ptr->get_stream()); @@ -842,10 +658,10 @@ cusparse_view_t::cusparse_view_t( mixed_precision_spmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha_d.data(), - A_mixed_, - c, + A_mixed_.get(), + c.get(), beta_d.data(), - dual_solution, + dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_non_transpose_mixed_.data(), handle_ptr->get_stream()); @@ -853,10 +669,10 @@ cusparse_view_t::cusparse_view_t( mixed_precision_spmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha_d.data(), - A_T_mixed_, - dual_solution, + A_T_mixed_.get(), + dual_solution.get(), beta_d.data(), - c, + c.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_transpose_mixed_.data(), handle_ptr->get_stream()); @@ -884,15 +700,6 @@ cusparse_view_t::cusparse_view_t( const pdlp::pdlp_hyper_params_t& hyper_params) : batch_mode_(climber_strategies.size() > 1), handle_ptr_(handle_ptr), - A{}, - A_T{}, - c{}, - primal_solution{}, - dual_solution{}, - primal_gradient{}, - dual_gradient{}, - tmp_primal{}, - tmp_dual{}, A_T_{_A_T}, A_T_offsets_{_A_T_offsets}, A_T_indices_{_A_T_indices}, @@ -923,56 +730,57 @@ cusparse_view_t::cusparse_view_t( handle_ptr_->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); // setup cusparse view - A.create(op_problem.n_constraints, - op_problem.n_variables, - static_cast(A_.size()), - const_cast(op_problem.offsets.data()), - const_cast(op_problem.variables.data()), - const_cast(op_problem.coefficients.data())); - - A_T.create(op_problem.n_variables, - op_problem.n_constraints, - static_cast(A_T_.size()), - const_cast(A_T_offsets_.data()), - const_cast(A_T_indices_.data()), - const_cast(A_T_.data())); - - c.create(op_problem.n_variables, const_cast(op_problem.objective_coefficients.data())); + A = make_csr(op_problem.n_constraints, + op_problem.n_variables, + static_cast(A_.size()), + const_cast(op_problem.offsets.data()), + const_cast(op_problem.variables.data()), + const_cast(op_problem.coefficients.data())); + + A_T = make_csr(op_problem.n_variables, + op_problem.n_constraints, + static_cast(A_T_.size()), + const_cast(A_T_offsets_.data()), + const_cast(A_T_indices_.data()), + const_cast(A_T_.data())); + + c = make_dnvec(op_problem.n_variables, + const_cast(op_problem.objective_coefficients.data())); if (!hyper_params.use_adaptive_step_size_strategy) { - primal_solution.create(op_problem.n_variables, _potential_next_primal.data()); - dual_solution.create(op_problem.n_constraints, _potential_next_dual.data()); + primal_solution = make_dnvec(op_problem.n_variables, _potential_next_primal.data()); + dual_solution = make_dnvec(op_problem.n_constraints, _potential_next_dual.data()); } else { - primal_solution.create(op_problem.n_variables, _primal_solution.data()); - dual_solution.create(op_problem.n_constraints, _dual_solution.data()); + primal_solution = make_dnvec(op_problem.n_variables, _primal_solution.data()); + dual_solution = make_dnvec(op_problem.n_constraints, _dual_solution.data()); } - tmp_primal.create(op_problem.n_variables, _tmp_primal.data()); - tmp_dual.create(op_problem.n_constraints, _tmp_dual.data()); + tmp_primal = make_dnvec(op_problem.n_variables, _tmp_primal.data()); + tmp_dual = make_dnvec(op_problem.n_constraints, _tmp_dual.data()); if (batch_mode_) { [[maybe_unused]] const bool is_cupdlpx = is_cupdlpx_restart(hyper_params); cuopt_assert(is_cupdlpx, "Batch mode only supported with cuPDLPx restart"); - batch_primal_solutions.create(op_problem.n_variables, - climber_strategies.size(), - op_problem.n_variables, - _potential_next_primal.data(), - CUSPARSE_ORDER_COL); - batch_dual_solutions.create(op_problem.n_constraints, - climber_strategies.size(), - op_problem.n_constraints, - _potential_next_dual.data(), - CUSPARSE_ORDER_COL); - batch_tmp_duals.create(op_problem.n_constraints, - climber_strategies.size(), - op_problem.n_constraints, - _tmp_dual.data(), - CUSPARSE_ORDER_COL); - batch_tmp_primals.create(op_problem.n_variables, - climber_strategies.size(), - op_problem.n_variables, - _tmp_primal.data(), - CUSPARSE_ORDER_COL); + batch_primal_solutions = make_dnmat(op_problem.n_variables, + climber_strategies.size(), + op_problem.n_variables, + _potential_next_primal.data(), + CUSPARSE_ORDER_COL); + batch_dual_solutions = make_dnmat(op_problem.n_constraints, + climber_strategies.size(), + op_problem.n_constraints, + _potential_next_dual.data(), + CUSPARSE_ORDER_COL); + batch_tmp_duals = make_dnmat(op_problem.n_constraints, + climber_strategies.size(), + op_problem.n_constraints, + _tmp_dual.data(), + CUSPARSE_ORDER_COL); + batch_tmp_primals = make_dnmat(op_problem.n_variables, + climber_strategies.size(), + op_problem.n_variables, + _tmp_primal.data(), + CUSPARSE_ORDER_COL); } const rmm::device_scalar alpha{one_v, handle_ptr->get_stream()}; @@ -982,10 +790,10 @@ cusparse_view_t::cusparse_view_t( raft::sparse::detail::cusparsespmv_buffersize(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - c, + A.get(), + c.get(), beta.data(), - dual_solution, + dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_non_transpose, handle_ptr->get_stream())); @@ -996,10 +804,10 @@ cusparse_view_t::cusparse_view_t( raft::sparse::detail::cusparsespmv_buffersize(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - dual_solution, + A_T.get(), + dual_solution.get(), beta.data(), - c, + c.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_transpose, handle_ptr->get_stream())); @@ -1013,10 +821,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - batch_dual_solutions, + A_T.get(), + batch_dual_solutions.get(), beta.data(), - batch_tmp_primals, + batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, &buffer_size_transpose_batch, handle_ptr->get_stream())); @@ -1027,10 +835,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - batch_primal_solutions, + A.get(), + batch_primal_solutions.get(), beta.data(), - batch_tmp_duals, + batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, &buffer_size_non_transpose_batch, handle_ptr->get_stream())); @@ -1041,10 +849,10 @@ cusparse_view_t::cusparse_view_t( my_cusparsespmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - c, + A.get(), + c.get(), beta.data(), - dual_solution, + dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_non_transpose.data(), handle_ptr->get_stream()); @@ -1052,10 +860,10 @@ cusparse_view_t::cusparse_view_t( my_cusparsespmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - dual_solution, + A_T.get(), + dual_solution.get(), beta.data(), - c, + c.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_transpose.data(), handle_ptr->get_stream()); @@ -1065,10 +873,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - batch_primal_solutions, + A.get(), + batch_primal_solutions.get(), beta.data(), - batch_tmp_duals, + batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, buffer_non_transpose_batch.data(), handle_ptr->get_stream()); @@ -1077,10 +885,10 @@ cusparse_view_t::cusparse_view_t( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - batch_dual_solutions, + A_T.get(), + batch_dual_solutions.get(), beta.data(), - batch_tmp_primals, + batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, buffer_transpose_batch.data(), handle_ptr->get_stream()); @@ -1100,13 +908,6 @@ cusparse_view_t::cusparse_view_t( f_t* _primal_gradient, f_t* _dual_gradient) : handle_ptr_(handle_ptr), - c(existing_cusparse_view.c), - primal_solution{}, - dual_solution{}, - primal_gradient{}, - dual_gradient{}, - tmp_primal(existing_cusparse_view.tmp_primal), - tmp_dual(existing_cusparse_view.tmp_dual), buffer_non_transpose{0, handle_ptr->get_stream()}, buffer_transpose{0, handle_ptr->get_stream()}, buffer_non_transpose_spmvop{0, handle_ptr->get_stream()}, @@ -1142,25 +943,36 @@ cusparse_view_t::cusparse_view_t( // correct pointer // See comment in the PDHG cusparse_view_t ctor: bind the descriptor nnz to // the actual value-buffer length so A and A_T stay symmetric and shard-safe. - A.create(op_problem.n_constraints, - op_problem.n_variables, - static_cast(A_.size()), - const_cast(A_offsets_.data()), - const_cast(A_indices_.data()), - const_cast(A_.data())); - - A_T.create(op_problem.n_variables, - op_problem.n_constraints, - static_cast(existing_cusparse_view.A_T_.size()), - const_cast(existing_cusparse_view.A_T_offsets_.data()), - const_cast(existing_cusparse_view.A_T_indices_.data()), - const_cast(existing_cusparse_view.A_T_.data())); - - primal_solution.create(op_problem.n_variables, _primal_solution); - dual_solution.create(op_problem.n_constraints, _dual_solution); - - primal_gradient.create(op_problem.n_variables, _primal_gradient); - dual_gradient.create(op_problem.n_constraints, _dual_gradient); + A = make_csr(op_problem.n_constraints, + op_problem.n_variables, + static_cast(A_.size()), + const_cast(A_offsets_.data()), + const_cast(A_indices_.data()), + const_cast(A_.data())); + + A_T = make_csr(op_problem.n_variables, + op_problem.n_constraints, + static_cast(existing_cusparse_view.A_T_.size()), + const_cast(existing_cusparse_view.A_T_offsets_.data()), + const_cast(existing_cusparse_view.A_T_indices_.data()), + const_cast(existing_cusparse_view.A_T_.data())); + + c = make_dnvec(op_problem.n_variables, + const_cast(op_problem.objective_coefficients.data())); + + primal_solution = make_dnvec(op_problem.n_variables, _primal_solution); + dual_solution = make_dnvec(op_problem.n_constraints, _dual_solution); + + primal_gradient = make_dnvec(op_problem.n_variables, _primal_gradient); + dual_gradient = make_dnvec(op_problem.n_constraints, _dual_gradient); + + // tmp_primal/tmp_dual alias pdhg-owned scratch, so re-wrap the existing view's buffers instead + // of allocating + void* scratch{nullptr}; + RAFT_CUSPARSE_TRY(cusparseDnVecGetValues(existing_cusparse_view.tmp_primal.get(), &scratch)); + tmp_primal = make_dnvec(op_problem.n_variables, static_cast(scratch)); + RAFT_CUSPARSE_TRY(cusparseDnVecGetValues(existing_cusparse_view.tmp_dual.get(), &scratch)); + tmp_dual = make_dnvec(op_problem.n_constraints, static_cast(scratch)); const rmm::device_scalar alpha{one_v, handle_ptr->get_stream()}; const rmm::device_scalar beta{one_v, handle_ptr->get_stream()}; @@ -1169,10 +981,10 @@ cusparse_view_t::cusparse_view_t( raft::sparse::detail::cusparsespmv_buffersize(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - c, + A.get(), + c.get(), beta.data(), - dual_solution, + dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_non_transpose, handle_ptr->get_stream())); @@ -1183,10 +995,10 @@ cusparse_view_t::cusparse_view_t( raft::sparse::detail::cusparsespmv_buffersize(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - dual_solution, + A_T.get(), + dual_solution.get(), beta.data(), - c, + c.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_transpose, handle_ptr->get_stream())); @@ -1197,10 +1009,10 @@ cusparse_view_t::cusparse_view_t( my_cusparsespmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A, - c, + A.get(), + c.get(), beta.data(), - dual_solution, + dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_non_transpose.data(), handle_ptr->get_stream()); @@ -1208,10 +1020,10 @@ cusparse_view_t::cusparse_view_t( my_cusparsespmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), - A_T, - dual_solution, + A_T.get(), + dual_solution.get(), beta.data(), - c, + c.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_transpose.data(), handle_ptr->get_stream()); @@ -1278,26 +1090,26 @@ template void cusparse_view_t::redirect_cusparse_csr_structure_pointers( const mip::problem_t& original_problem) { - RAFT_CUSPARSE_TRY(cusparseCsrSetPointers(A, + RAFT_CUSPARSE_TRY(cusparseCsrSetPointers(A.get(), const_cast(original_problem.offsets.data()), const_cast(original_problem.variables.data()), const_cast(A_.data()))); RAFT_CUSPARSE_TRY( - cusparseCsrSetPointers(A_T, + cusparseCsrSetPointers(A_T.get(), const_cast(original_problem.reverse_offsets.data()), const_cast(original_problem.reverse_constraints.data()), const_cast(A_T_.data()))); if constexpr (std::is_same_v) { if (mixed_precision_enabled_) { - RAFT_CUSPARSE_TRY(cusparseCsrSetPointers(A_mixed_, + RAFT_CUSPARSE_TRY(cusparseCsrSetPointers(A_mixed_.get(), const_cast(original_problem.offsets.data()), const_cast(original_problem.variables.data()), A_float_.data())); RAFT_CUSPARSE_TRY( - cusparseCsrSetPointers(A_T_mixed_, + cusparseCsrSetPointers(A_T_mixed_.get(), const_cast(original_problem.reverse_offsets.data()), const_cast(original_problem.reverse_constraints.data()), A_T_float_.data())); @@ -1309,10 +1121,10 @@ void cusparse_view_t::redirect_cusparse_csr_structure_pointers( size_t mixed_precision_spmv_buffersize(cusparseHandle_t handle, cusparseOperation_t opA, const double* alpha, - cusparseSpMatDescr_t matA, // FP32 matrix - cusparseDnVecDescr_t vecX, // FP64 vector + cusparse_sp_mat_descr_view matA, // FP32 matrix + cusparse_dn_vec_descr_view vecX, // FP64 vector const double* beta, - cusparseDnVecDescr_t vecY, // FP64 vector + cusparse_dn_vec_descr_view vecY, // FP64 vector cusparseSpMVAlg_t alg, cudaStream_t stream) { @@ -1326,10 +1138,10 @@ size_t mixed_precision_spmv_buffersize(cusparseHandle_t handle, void mixed_precision_spmv(cusparseHandle_t handle, cusparseOperation_t opA, const double* alpha, - cusparseSpMatDescr_t matA, // FP32 matrix - cusparseDnVecDescr_t vecX, // FP64 vector + cusparse_sp_mat_descr_view matA, // FP32 matrix + cusparse_dn_vec_descr_view vecX, // FP64 vector const double* beta, - cusparseDnVecDescr_t vecY, // FP64 vector + cusparse_dn_vec_descr_view vecY, // FP64 vector cusparseSpMVAlg_t alg, void* externalBuffer, cudaStream_t stream) @@ -1343,10 +1155,10 @@ void mixed_precision_spmv(cusparseHandle_t handle, void mixed_precision_spmv_preprocess(cusparseHandle_t handle, cusparseOperation_t opA, const double* alpha, - cusparseSpMatDescr_t matA, // FP32 matrix - cusparseDnVecDescr_t vecX, // FP64 vector + cusparse_sp_mat_descr_view matA, // FP32 matrix + cusparse_dn_vec_descr_view vecX, // FP64 vector const double* beta, - cusparseDnVecDescr_t vecY, // FP64 vector + cusparse_dn_vec_descr_view vecY, // FP64 vector cusparseSpMVAlg_t alg, void* externalBuffer, cudaStream_t stream) @@ -1389,71 +1201,63 @@ void cusparse_view_t::create_spmv_op_plans(bool is_reflected) { #if CUOPT_CUSPARSE_VER_12_8_UP if (!is_cusparse_runtime_spmvop_supported() || !(std::is_same_v)) { return; } - static const auto buffer_size = - dynamic_load_runtime::function("cusparseSpMVOp_bufferSize"); - CUSPARSE_CHECK(cusparseSetStream(handle_ptr_->get_cusparse_handle(), handle_ptr_->get_stream())); + RAFT_CUSPARSE_TRY( + cusparseSetStream(handle_ptr_->get_cusparse_handle(), handle_ptr_->get_stream())); // Prepare buffers for At_y SpMVOp size_t buffer_size_transpose = 0; - RAFT_CUSPARSE_TRY((*buffer_size)(handle_ptr_->get_cusparse_handle(), - CUSPARSE_OPERATION_NON_TRANSPOSE, - A_T, - dual_solution, - current_AtY, - current_AtY, - CUDA_R_64F, - CUSPARSE_SPMVOP_ALG_DEFAULT, - &buffer_size_transpose)); + RAFT_CUSPARSE_TRY(cusparse_spmvop_buffer_size(handle_ptr_->get_cusparse_handle(), + CUSPARSE_OPERATION_NON_TRANSPOSE, + A_T.get(), + dual_solution.get(), + current_AtY.get(), + current_AtY.get(), + CUDA_R_64F, + &buffer_size_transpose)); buffer_transpose_spmvop.resize(buffer_size_transpose, handle_ptr_->get_stream()); - spmv_op_descr_A_t_.create(handle_ptr_->get_cusparse_handle(), - CUSPARSE_OPERATION_NON_TRANSPOSE, - A_T, - dual_solution, - current_AtY, - current_AtY, - CUDA_R_64F, - buffer_transpose_spmvop); + spmv_op_descr_A_t_ = make_spmvop_descr(handle_ptr_->get_cusparse_handle(), + CUSPARSE_OPERATION_NON_TRANSPOSE, + A_T.get(), + dual_solution.get(), + current_AtY.get(), + current_AtY.get(), + CUDA_R_64F, + buffer_transpose_spmvop); - spmv_op_plan_A_t_.create(handle_ptr_->get_cusparse_handle(), spmv_op_descr_A_t_); + spmv_op_plan_A_t_ = + make_spmvop_plan(handle_ptr_->get_cusparse_handle(), spmv_op_descr_A_t_.get()); // Only prepare buffers for A_x if we are using reflected_halpern if (is_reflected) { size_t buffer_size_non_transpose = 0; - RAFT_CUSPARSE_TRY((*buffer_size)(handle_ptr_->get_cusparse_handle(), - CUSPARSE_OPERATION_NON_TRANSPOSE, - A, - reflected_primal_solution, - dual_gradient, - dual_gradient, - CUDA_R_64F, - CUSPARSE_SPMVOP_ALG_DEFAULT, - &buffer_size_non_transpose)); + RAFT_CUSPARSE_TRY(cusparse_spmvop_buffer_size(handle_ptr_->get_cusparse_handle(), + CUSPARSE_OPERATION_NON_TRANSPOSE, + A.get(), + reflected_primal_solution.get(), + dual_gradient.get(), + dual_gradient.get(), + CUDA_R_64F, + &buffer_size_non_transpose)); buffer_non_transpose_spmvop.resize(buffer_size_non_transpose, handle_ptr_->get_stream()); - spmv_op_descr_A_.create(handle_ptr_->get_cusparse_handle(), - CUSPARSE_OPERATION_NON_TRANSPOSE, - A, - reflected_primal_solution, - dual_gradient, - dual_gradient, - CUDA_R_64F, - buffer_non_transpose_spmvop); + spmv_op_descr_A_ = make_spmvop_descr(handle_ptr_->get_cusparse_handle(), + CUSPARSE_OPERATION_NON_TRANSPOSE, + A.get(), + reflected_primal_solution.get(), + dual_gradient.get(), + dual_gradient.get(), + CUDA_R_64F, + buffer_non_transpose_spmvop); - spmv_op_plan_A_.create(handle_ptr_->get_cusparse_handle(), spmv_op_descr_A_); + spmv_op_plan_A_ = make_spmvop_plan(handle_ptr_->get_cusparse_handle(), spmv_op_descr_A_.get()); } #endif // CUOPT_CUSPARSE_VER_12_8_UP } #if MIP_INSTANTIATE_FLOAT || PDLP_INSTANTIATE_FLOAT -template class cusparse_sp_mat_descr_wrapper_t; -template class cusparse_dn_vec_descr_wrapper_t; -template class cusparse_dn_mat_descr_wrapper_t; template class cusparse_view_t; #endif #if MIP_INSTANTIATE_DOUBLE -template class cusparse_sp_mat_descr_wrapper_t; -template class cusparse_dn_vec_descr_wrapper_t; -template class cusparse_dn_mat_descr_wrapper_t; template class cusparse_view_t; #endif diff --git a/cpp/src/pdlp/cusparse_view.hpp b/cpp/src/pdlp/cusparse_view.hpp index 71ac851ef3..0f109c1c0e 100644 --- a/cpp/src/pdlp/cusparse_view.hpp +++ b/cpp/src/pdlp/cusparse_view.hpp @@ -20,136 +20,124 @@ #include +#include +#include + // cuSPARSE 12.8 ships with CUDA Toolkit 13.3 #define CUOPT_CUSPARSE_VER_12_8_UP (CUSPARSE_VERSION >= 12800) namespace cuopt::mathematical_optimization::pdlp { -template -class cusparse_sp_mat_descr_wrapper_t { - public: - cusparse_sp_mat_descr_wrapper_t(); - ~cusparse_sp_mat_descr_wrapper_t(); - - cusparse_sp_mat_descr_wrapper_t(const cusparse_sp_mat_descr_wrapper_t& other); - - cusparse_sp_mat_descr_wrapper_t& operator=(const cusparse_sp_mat_descr_wrapper_t& other) = delete; - - void create(int64_t m, int64_t n, int64_t nnz, i_t* offsets, i_t* indices, f_t* values); - - operator cusparseSpMatDescr_t() const; +// --------------------------------------------------------------------------- +// Deleters and unique_ptr aliases for cuSPARSE opaque handles. +// +// Each cuSPARSE handle (cusparseSpMatDescr_t etc.) is a typedef for a pointer +// to an opaque struct. We use std::remove_pointer_t to feed unique_ptr the +// pointee type so that unique_ptr<...>::pointer matches the cuSPARSE handle. +// --------------------------------------------------------------------------- + +struct cusparse_sp_mat_deleter_t { + void operator()(cusparseSpMatDescr_t descr) const noexcept + { + if (descr) { RAFT_CUSPARSE_TRY_NO_THROW(cusparseDestroySpMat(descr)); } + } +}; - private: - cusparseSpMatDescr_t descr_; - bool need_destruction_; +struct cusparse_dn_vec_deleter_t { + void operator()(cusparseDnVecDescr_t descr) const noexcept + { + if (descr) { RAFT_CUSPARSE_TRY_NO_THROW(cusparseDestroyDnVec(descr)); } + } }; -template -class cusparse_dn_vec_descr_wrapper_t { - public: - cusparse_dn_vec_descr_wrapper_t(); - ~cusparse_dn_vec_descr_wrapper_t(); +struct cusparse_dn_mat_deleter_t { + void operator()(cusparseDnMatDescr_t descr) const noexcept + { + if (descr) { RAFT_CUSPARSE_TRY_NO_THROW(cusparseDestroyDnMat(descr)); } + } +}; - cusparse_dn_vec_descr_wrapper_t(const cusparse_dn_vec_descr_wrapper_t& other); - cusparse_dn_vec_descr_wrapper_t& operator=(cusparse_dn_vec_descr_wrapper_t&& other); - cusparse_dn_vec_descr_wrapper_t& operator=(const cusparse_dn_vec_descr_wrapper_t& other) = delete; +using cusparse_sp_mat_uptr = + std::unique_ptr, cusparse_sp_mat_deleter_t>; +using cusparse_dn_vec_uptr = + std::unique_ptr, cusparse_dn_vec_deleter_t>; +using cusparse_dn_mat_uptr = + std::unique_ptr, cusparse_dn_mat_deleter_t>; - void create(int64_t size, f_t* values); +// Borrowed views: identical to the raw cuSPARSE handle types but the alias makes the non-owning +// intent explicit at API boundaries. Pair with the *_uptr aliases above: +// _uptr -> owns the descriptor; the destructor calls cusparseDestroy* +// _view -> non-owning, just the raw handle, lifetime managed elsewhere +using cusparse_sp_mat_descr_view = cusparseSpMatDescr_t; +using cusparse_dn_vec_descr_view = cusparseDnVecDescr_t; +using cusparse_dn_mat_descr_view = cusparseDnMatDescr_t; - operator cusparseDnVecDescr_t() const; +// Factory functions replacing the old `wrapper.create(...)` two-phase init. - private: - cusparseDnVecDescr_t descr_; - bool need_destruction_; -}; +template +cusparse_sp_mat_uptr make_csr( + int64_t m, int64_t n, int64_t nnz, i_t* offsets, i_t* indices, f_t* values) +{ + cusparseSpMatDescr_t descr{nullptr}; + RAFT_CUSPARSE_TRY( + raft::sparse::detail::cusparsecreatecsr(&descr, m, n, nnz, offsets, indices, values)); + return cusparse_sp_mat_uptr{descr}; +} template -class cusparse_dn_mat_descr_wrapper_t { - public: - cusparse_dn_mat_descr_wrapper_t(); - ~cusparse_dn_mat_descr_wrapper_t(); - - cusparse_dn_mat_descr_wrapper_t(const cusparse_dn_mat_descr_wrapper_t& other); - cusparse_dn_mat_descr_wrapper_t& operator=(cusparse_dn_mat_descr_wrapper_t&& other); - cusparse_dn_mat_descr_wrapper_t& operator=(const cusparse_dn_mat_descr_wrapper_t& other) = delete; +cusparse_dn_vec_uptr make_dnvec(int64_t size, f_t* values) +{ + cusparseDnVecDescr_t descr{nullptr}; + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsecreatednvec(&descr, size, values)); + return cusparse_dn_vec_uptr{descr}; +} - void create(int64_t row, int64_t col, int64_t ld, f_t* values, cusparseOrder_t order); - - operator cusparseDnMatDescr_t() const; - - private: - cusparseDnMatDescr_t descr_; - bool need_destruction_; -}; +template +cusparse_dn_mat_uptr make_dnmat( + int64_t row, int64_t col, int64_t ld, f_t* values, cusparseOrder_t order) +{ + cusparseDnMatDescr_t descr{nullptr}; + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsecreatednmat(&descr, row, col, ld, values, order)); + return cusparse_dn_mat_uptr{descr}; +} #if CUOPT_CUSPARSE_VER_12_8_UP -// RAII wrapper around cusparse SpMVOp objects. All the buffers are owned by the cusparse_view_t. -class cusparse_spmvop_descr_wrapper_t { - public: - cusparse_spmvop_descr_wrapper_t(); - ~cusparse_spmvop_descr_wrapper_t(); - - cusparse_spmvop_descr_wrapper_t(const cusparse_spmvop_descr_wrapper_t& other); - cusparse_spmvop_descr_wrapper_t& operator=(cusparse_spmvop_descr_wrapper_t&& other); - cusparse_spmvop_descr_wrapper_t& operator=(const cusparse_spmvop_descr_wrapper_t& other) = delete; - - void create(cusparseHandle_t handle, - cusparseOperation_t opA, - cusparseSpMatDescr_t matA, - cusparseDnVecDescr_t vecX, - cusparseDnVecDescr_t vecY, - cusparseDnVecDescr_t vecZ, - cudaDataType computeType, - rmm::device_uvector& buffer); - - operator cusparseSpMVOpDescr_t() const; - - private: - // Forwards to cusparseSpMVOp_{create,destroy}Descr resolved via dlsym (cached on first call). - // This is needed because the cusparseSpMVOp_{create,destroy}Descr symbols might not be defined in - // current runtime. - static cusparseStatus_t dlsym_create(cusparseHandle_t handle, - cusparseSpMVOpDescr_t* descr, - cusparseOperation_t opA, - cusparseSpMatDescr_t matA, - cusparseDnVecDescr_t vecX, - cusparseDnVecDescr_t vecY, - cusparseDnVecDescr_t vecZ, - cudaDataType computeType, - void* buffer); - static cusparseStatus_t dlsym_destroy(cusparseSpMVOpDescr_t descr); - - cusparseSpMVOpDescr_t descr_; - bool need_destruction_; +// --------------------------------------------------------------------------- +// SpMVOp descriptor and plan deleters. +// +// The cusparseSpMVOp_{create,destroy}{Descr,Plan} symbols may not be present +// in the runtime cuSPARSE (the compiled CUDA version may differ from the one +// at runtime), so destruction is dispatched through dlsym. The deleters below +// resolve the destroy symbol at first use and cache it via a function-local +// static. +// --------------------------------------------------------------------------- + +struct cusparse_spmvop_descr_deleter_t { + void operator()(cusparseSpMVOpDescr_t descr) const noexcept; }; -class cusparse_spmvop_plan_wrapper_t { - public: - cusparse_spmvop_plan_wrapper_t(); - ~cusparse_spmvop_plan_wrapper_t(); - - cusparse_spmvop_plan_wrapper_t(const cusparse_spmvop_plan_wrapper_t& other); - cusparse_spmvop_plan_wrapper_t& operator=(cusparse_spmvop_plan_wrapper_t&& other); - cusparse_spmvop_plan_wrapper_t& operator=(const cusparse_spmvop_plan_wrapper_t& other) = delete; - - void create(cusparseHandle_t handle, cusparseSpMVOpDescr_t descr); - - operator cusparseSpMVOpPlan_t() const; - - private: - // Forwards to cusparseSpMVOp_{create,destroy}Plan resolved via dlsym (cached on first call). - // This is needed because the cusparseSpMVOp_{create,destroy}Plan symbols might not be defined in - // current runtime. - static cusparseStatus_t dlsym_create(cusparseHandle_t handle, - cusparseSpMVOpDescr_t descr, - cusparseSpMVOpPlan_t* plan, - char* ltoIRBuf, - size_t ltoIRSize); - static cusparseStatus_t dlsym_destroy(cusparseSpMVOpPlan_t plan); - - cusparseSpMVOpPlan_t plan_; - bool need_destruction_; +struct cusparse_spmvop_plan_deleter_t { + void operator()(cusparseSpMVOpPlan_t plan) const noexcept; }; + +using cusparse_spmvop_descr_uptr = + std::unique_ptr, cusparse_spmvop_descr_deleter_t>; +using cusparse_spmvop_plan_uptr = + std::unique_ptr, cusparse_spmvop_plan_deleter_t>; + +// Factories. `make_spmvop_descr` resolves cusparseSpMVOp_createDescr via dlsym. +cusparse_spmvop_descr_uptr make_spmvop_descr(cusparseHandle_t handle, + cusparseOperation_t opA, + cusparse_sp_mat_descr_view matA, + cusparse_dn_vec_descr_view vecX, + cusparse_dn_vec_descr_view vecY, + cusparse_dn_vec_descr_view vecZ, + cudaDataType computeType, + rmm::device_uvector& buffer); + +// `make_spmvop_plan` passes nullptr/0 for ltoIRBuf/ltoIRSize so cuSPARSE JITs +// internally; cuOpt does not supply user-provided LTO IR. +cusparse_spmvop_plan_uptr make_spmvop_plan(cusparseHandle_t handle, cusparseSpMVOpDescr_t descr); #endif // CUOPT_CUSPARSE_VER_12_8_UP template @@ -198,48 +186,47 @@ class cusparse_view_t { raft::handle_t const* handle_ptr_{nullptr}; // cusparse view of linear program - cusparse_sp_mat_descr_wrapper_t A; - cusparse_sp_mat_descr_wrapper_t A_T; - cusparse_dn_vec_descr_wrapper_t c; + cusparse_sp_mat_uptr A; + cusparse_sp_mat_uptr A_T; + cusparse_dn_vec_uptr c; // cusparse view of solutions - cusparse_dn_vec_descr_wrapper_t primal_solution; - cusparse_dn_vec_descr_wrapper_t dual_solution; + cusparse_dn_vec_uptr primal_solution; + cusparse_dn_vec_uptr dual_solution; // cusparse view of gradients - cusparse_dn_vec_descr_wrapper_t primal_gradient; - cusparse_dn_vec_descr_wrapper_t dual_gradient; + cusparse_dn_vec_uptr primal_gradient; + cusparse_dn_vec_uptr dual_gradient; // cusparse view of batch gradients - cusparse_dn_mat_descr_wrapper_t batch_dual_gradients; + cusparse_dn_mat_uptr batch_dual_gradients; // cusparse view of batch solutions - cusparse_dn_mat_descr_wrapper_t batch_primal_solutions; - cusparse_dn_mat_descr_wrapper_t batch_dual_solutions; - cusparse_dn_mat_descr_wrapper_t batch_potential_next_dual_solution; - cusparse_dn_mat_descr_wrapper_t batch_next_AtYs; - cusparse_dn_mat_descr_wrapper_t batch_tmp_duals; - cusparse_dn_mat_descr_wrapper_t batch_reflected_primal_solutions; - cusparse_dn_mat_descr_wrapper_t batch_delta_primal_solutions; - cusparse_dn_mat_descr_wrapper_t batch_delta_dual_solutions; + cusparse_dn_mat_uptr batch_primal_solutions; + cusparse_dn_mat_uptr batch_dual_solutions; + cusparse_dn_mat_uptr batch_potential_next_dual_solution; + cusparse_dn_mat_uptr batch_next_AtYs; + cusparse_dn_mat_uptr batch_tmp_duals; + cusparse_dn_mat_uptr batch_reflected_primal_solutions; + cusparse_dn_mat_uptr batch_delta_primal_solutions; + cusparse_dn_mat_uptr batch_delta_dual_solutions; // cusparse view of At * Y batch computation - cusparse_dn_mat_descr_wrapper_t batch_current_AtYs; + cusparse_dn_mat_uptr batch_current_AtYs; // cusparse view of auxillirary space needed for some spmm computations - cusparse_dn_mat_descr_wrapper_t batch_tmp_primals; + cusparse_dn_mat_uptr batch_tmp_primals; // cusparse view of At * Y computation - cusparse_dn_vec_descr_wrapper_t - current_AtY; // Only used at very first iteration and after each restart to average - cusparse_dn_vec_descr_wrapper_t - next_AtY; // Next value is swapped out with current after each valid PDHG - // step to save the first AtY SpMV in compute next primal - cusparse_dn_vec_descr_wrapper_t potential_next_dual_solution; + cusparse_dn_vec_uptr current_AtY; // Only used at very first iteration and after each restart to + // average + cusparse_dn_vec_uptr next_AtY; // Next value is swapped out with current after each valid PDHG + // step to save the first AtY SpMV in compute next primal + cusparse_dn_vec_uptr potential_next_dual_solution; // cusparse view of auxiliary space needed for some spmv computations - cusparse_dn_vec_descr_wrapper_t tmp_primal; - cusparse_dn_vec_descr_wrapper_t tmp_dual; + cusparse_dn_vec_uptr tmp_primal; + cusparse_dn_vec_uptr tmp_dual; // reuse buffers for cusparse spmv rmm::device_uvector buffer_non_transpose; @@ -251,10 +238,10 @@ class cusparse_view_t { #if CUOPT_CUSPARSE_VER_12_8_UP // SpMVOp descriptors and plans for A and A_T (descr before plan so dtor destroys plan first) - cusparse_spmvop_descr_wrapper_t spmv_op_descr_A_; - cusparse_spmvop_plan_wrapper_t spmv_op_plan_A_; - cusparse_spmvop_descr_wrapper_t spmv_op_descr_A_t_; - cusparse_spmvop_plan_wrapper_t spmv_op_plan_A_t_; + cusparse_spmvop_descr_uptr spmv_op_descr_A_; + cusparse_spmvop_plan_uptr spmv_op_plan_A_; + cusparse_spmvop_descr_uptr spmv_op_descr_A_t_; + cusparse_spmvop_plan_uptr spmv_op_plan_A_t_; #endif // CUOPT_CUSPARSE_VER_12_8_UP // reuse buffers for cusparse spmm rmm::device_uvector buffer_transpose_batch; @@ -262,7 +249,7 @@ class cusparse_view_t { rmm::device_uvector buffer_transpose_batch_row_row_; rmm::device_uvector buffer_non_transpose_batch_row_row_; // Only when using reflection - cusparse_dn_vec_descr_wrapper_t reflected_primal_solution; + cusparse_dn_vec_uptr reflected_primal_solution; // Ref to the A_T found in either // Initial problem, we use it to have an unscaled A_T @@ -284,8 +271,8 @@ class cusparse_view_t { // Only used when mixed_precision_enabled_ is true and f_t = double rmm::device_uvector A_float_; // FP32 copy of A values rmm::device_uvector A_T_float_; // FP32 copy of A_T values - cusparse_sp_mat_descr_wrapper_t A_mixed_; // FP32 matrix descriptor for A - cusparse_sp_mat_descr_wrapper_t A_T_mixed_; // FP32 matrix descriptor for A_T + cusparse_sp_mat_uptr A_mixed_; // FP32 matrix descriptor for A + cusparse_sp_mat_uptr A_T_mixed_; // FP32 matrix descriptor for A_T rmm::device_uvector buffer_non_transpose_mixed_; // SpMV buffer for mixed precision A rmm::device_uvector buffer_transpose_mixed_; // SpMV buffer for mixed precision A_T bool mixed_precision_enabled_{false}; @@ -304,10 +291,10 @@ class cusparse_view_t { void mixed_precision_spmv(cusparseHandle_t handle, cusparseOperation_t opA, const double* alpha, - cusparseSpMatDescr_t matA, // FP32 matrix - cusparseDnVecDescr_t vecX, // FP64 vector + cusparse_sp_mat_descr_view matA, // FP32 matrix + cusparse_dn_vec_descr_view vecX, // FP64 vector const double* beta, - cusparseDnVecDescr_t vecY, // FP64 vector + cusparse_dn_vec_descr_view vecY, // FP64 vector cusparseSpMVAlg_t alg, void* externalBuffer, cudaStream_t stream); @@ -315,10 +302,10 @@ void mixed_precision_spmv(cusparseHandle_t handle, size_t mixed_precision_spmv_buffersize(cusparseHandle_t handle, cusparseOperation_t opA, const double* alpha, - cusparseSpMatDescr_t matA, // FP32 matrix - cusparseDnVecDescr_t vecX, // FP64 vector + cusparse_sp_mat_descr_view matA, // FP32 matrix + cusparse_dn_vec_descr_view vecX, // FP64 vector const double* beta, - cusparseDnVecDescr_t vecY, // FP64 vector + cusparse_dn_vec_descr_view vecY, // FP64 vector cusparseSpMVAlg_t alg, cudaStream_t stream); @@ -326,10 +313,10 @@ size_t mixed_precision_spmv_buffersize(cusparseHandle_t handle, void mixed_precision_spmv_preprocess(cusparseHandle_t handle, cusparseOperation_t opA, const double* alpha, - cusparseSpMatDescr_t matA, // FP32 matrix - cusparseDnVecDescr_t vecX, // FP64 vector + cusparse_sp_mat_descr_view matA, // FP32 matrix + cusparse_dn_vec_descr_view vecX, // FP64 vector const double* beta, - cusparseDnVecDescr_t vecY, // FP64 vector + cusparse_dn_vec_descr_view vecY, // FP64 vector cusparseSpMVAlg_t alg, void* externalBuffer, cudaStream_t stream); @@ -343,10 +330,10 @@ void my_cusparsespmm_preprocess(cusparseHandle_t handle, cusparseOperation_t opA, cusparseOperation_t opB, const T* alpha, - const cusparseSpMatDescr_t matA, - const cusparseDnMatDescr_t matB, + cusparse_sp_mat_descr_view matA, + cusparse_dn_mat_descr_view matB, const T* beta, - const cusparseDnMatDescr_t matC, + cusparse_dn_mat_descr_view matC, cusparseSpMMAlg_t alg, void* externalBuffer, cudaStream_t stream); @@ -366,9 +353,9 @@ void cusparse_spmvop_run(cusparseHandle_t handle, cusparseSpMVOpPlan_t plan, const void* alpha, const void* beta, - cusparseDnVecDescr_t vecX, - cusparseDnVecDescr_t vecY, - cusparseDnVecDescr_t vecZ, + cusparse_dn_vec_descr_view vecX, + cusparse_dn_vec_descr_view vecY, + cusparse_dn_vec_descr_view vecZ, cudaStream_t stream); #endif // CUOPT_CUSPARSE_VER_12_8_UP diff --git a/cpp/src/pdlp/distributed_pdlp/distributed_algorithms.cu b/cpp/src/pdlp/distributed_pdlp/distributed_algorithms.cu index 09cd3cb836..75646e7ce2 100644 --- a/cpp/src/pdlp/distributed_pdlp/distributed_algorithms.cu +++ b/cpp/src/pdlp/distributed_pdlp/distributed_algorithms.cu @@ -224,9 +224,9 @@ f_t multi_gpu_engine_t::distributed_max_singular_value_squared(i_t n_g std::vector> norm_q; std::vector> residual_norm; - std::vector> q_dn(nb); - std::vector> z_dn(nb); - std::vector> atq_dn(nb); + std::vector q_dn(nb); + std::vector z_dn(nb); + std::vector atq_dn(nb); // Per-shard owned-slice spans consumed by the engine's *_bufs helpers. std::vector> q_owned, z_owned; @@ -250,9 +250,9 @@ f_t multi_gpu_engine_t::distributed_max_singular_value_squared(i_t n_g sigma_sq.emplace_back(s.stream.view()); norm_q.emplace_back(s.stream.view()); residual_norm.emplace_back(s.stream.view()); - q_dn[r].create(static_cast(cstr_total), q.back().data()); - z_dn[r].create(static_cast(cstr_total), z.back().data()); - atq_dn[r].create(static_cast(var_total), atq.back().data()); + q_dn[r] = make_dnvec(static_cast(cstr_total), q.back().data()); + z_dn[r] = make_dnvec(static_cast(cstr_total), z.back().data()); + atq_dn[r] = make_dnvec(static_cast(var_total), atq.back().data()); q_owned.emplace_back(q.back().data(), static_cast(n_owned)); z_owned.emplace_back(z.back().data(), static_cast(n_owned)); diff --git a/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.cu b/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.cu index 5cf5b2af46..050c360e0c 100644 --- a/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.cu +++ b/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.cu @@ -430,23 +430,25 @@ void multi_gpu_engine_t::distributed_l2_norm_to_master_buf( template void multi_gpu_engine_t::distributed_spmv_At( std::vector>& in_bufs, - std::vector>& in_descs, - std::vector>& out_descs) + std::vector& in_descs, + std::vector& out_descs) { halo_exchange_cstr_bufs(in_bufs); - for_each_shard( - [&](auto& s, int r) { s.sub_pdlp->pdhg_solver_.spmv_At_into(in_descs[r], out_descs[r]); }); + for_each_shard([&](auto& s, int r) { + s.sub_pdlp->pdhg_solver_.spmv_At_into(in_descs[r].get(), out_descs[r].get()); + }); } template void multi_gpu_engine_t::distributed_spmv_A( std::vector>& in_bufs, - std::vector>& in_descs, - std::vector>& out_descs) + std::vector& in_descs, + std::vector& out_descs) { halo_exchange_var_bufs(in_bufs); - for_each_shard( - [&](auto& s, int r) { s.sub_pdlp->pdhg_solver_.spmv_A_into(in_descs[r], out_descs[r]); }); + for_each_shard([&](auto& s, int r) { + s.sub_pdlp->pdhg_solver_.spmv_A_into(in_descs[r].get(), out_descs[r].get()); + }); } template struct multi_gpu_engine_t; diff --git a/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.hpp b/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.hpp index f5bd95b832..fa77c3aed2 100644 --- a/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.hpp +++ b/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.hpp @@ -444,16 +444,16 @@ struct multi_gpu_engine_t { // (cstr axis, since the input is cstr-shaped), then dispatches each shard's // local spmv_At_into that reads from in_descs[r] and writes into out_descs[r]. void distributed_spmv_At(std::vector>& in_bufs, - std::vector>& in_descs, - std::vector>& out_descs); + std::vector& in_descs, + std::vector& out_descs); // Distributed A @ in on caller-owned scratch. Refreshes the halo of `in_bufs` // (var axis, since the input is var-shaped), then dispatches each shard's // local spmv_A_into. Caller owns / sizes the descriptor vectors as above // (in_descs to var_total, out_descs to cstr_total). void distributed_spmv_A(std::vector>& in_bufs, - std::vector>& in_descs, - std::vector>& out_descs); + std::vector& in_descs, + std::vector& out_descs); // -------- High-level algorithms (defined in distributed_algorithms.cu) --- // Refreshes the halo copies of the cumulative variable + constraint scalings on diff --git a/cpp/src/pdlp/optimal_batch_size_handler/optimal_batch_size_handler.cu b/cpp/src/pdlp/optimal_batch_size_handler/optimal_batch_size_handler.cu index 8c8cdce0b4..c969d0347a 100644 --- a/cpp/src/pdlp/optimal_batch_size_handler/optimal_batch_size_handler.cu +++ b/cpp/src/pdlp/optimal_batch_size_handler/optimal_batch_size_handler.cu @@ -20,8 +20,8 @@ namespace cuopt::mathematical_optimization::pdlp { template struct SpMM_benchmarks_context_t { - SpMM_benchmarks_context_t(cusparse_sp_mat_descr_wrapper_t& A, - cusparse_sp_mat_descr_wrapper_t& A_T, + SpMM_benchmarks_context_t(cusparse_sp_mat_descr_view A, + cusparse_sp_mat_descr_view A_T, int primal_size, int dual_size, size_t current_batch_size, @@ -46,8 +46,8 @@ struct SpMM_benchmarks_context_t { int col_dual = current_batch_size; int ld_dual = current_batch_size; - x_descr.create(rows_primal, col_primal, ld_primal, x.data(), CUSPARSE_ORDER_ROW); - y_descr.create(rows_dual, col_dual, ld_dual, y.data(), CUSPARSE_ORDER_ROW); + x_descr = make_dnmat(rows_primal, col_primal, ld_primal, x.data(), CUSPARSE_ORDER_ROW); + y_descr = make_dnmat(rows_dual, col_dual, ld_dual, y.data(), CUSPARSE_ORDER_ROW); // Init buffers for SpMMs size_t buffer_size_non_transpose_batch = 0; @@ -57,9 +57,9 @@ struct SpMM_benchmarks_context_t { CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), A, - x_descr, + x_descr.get(), beta.data(), - y_descr, + y_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &buffer_size_non_transpose_batch, stream_view)); @@ -71,9 +71,9 @@ struct SpMM_benchmarks_context_t { CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), A_T, - y_descr, + y_descr.get(), beta.data(), - x_descr, + x_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &buffer_size_transpose_batch, stream_view)); @@ -89,9 +89,9 @@ struct SpMM_benchmarks_context_t { CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), A_T, - y_descr, + y_descr.get(), beta.data(), - x_descr, + x_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, buffer_transpose_batch.data(), stream_view); @@ -102,9 +102,9 @@ struct SpMM_benchmarks_context_t { CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), A, - x_descr, + x_descr.get(), beta.data(), - y_descr, + y_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, buffer_non_transpose_batch.data(), stream_view); @@ -124,9 +124,9 @@ struct SpMM_benchmarks_context_t { CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), A, - x_descr, + x_descr.get(), beta.data(), - y_descr, + y_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, (f_t*)buffer_non_transpose_batch.data(), stream_view)); @@ -137,30 +137,30 @@ struct SpMM_benchmarks_context_t { CUSPARSE_OPERATION_NON_TRANSPOSE, alpha.data(), A_T, - y_descr, + y_descr.get(), beta.data(), - x_descr, + x_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, (f_t*)buffer_transpose_batch.data(), stream_view)); } - cusparse_dn_mat_descr_wrapper_t x_descr; - cusparse_dn_mat_descr_wrapper_t y_descr; + cusparse_dn_mat_uptr x_descr; + cusparse_dn_mat_uptr y_descr; rmm::device_uvector x; rmm::device_uvector y; rmm::device_buffer buffer_non_transpose_batch; rmm::device_buffer buffer_transpose_batch; rmm::device_scalar alpha; rmm::device_scalar beta; - cusparse_sp_mat_descr_wrapper_t& A; - cusparse_sp_mat_descr_wrapper_t& A_T; + cusparse_sp_mat_descr_view A; + cusparse_sp_mat_descr_view A_T; raft::handle_t const* handle_ptr; }; template -static double evaluate_node(cusparse_sp_mat_descr_wrapper_t& A, - cusparse_sp_mat_descr_wrapper_t& A_T, +static double evaluate_node(cusparse_sp_mat_descr_view A, + cusparse_sp_mat_descr_view A_T, i_t primal_size, i_t dual_size, int current_batch_size, @@ -224,24 +224,20 @@ int optimal_batch_size_handler(const optimization_problem_t& op_proble mip::problem_t problem(op_problem); // Init cuSparse views - cusparse_sp_mat_descr_wrapper_t A; - cusparse_sp_mat_descr_wrapper_t A_T; - i_t primal_size = problem.n_variables; - i_t dual_size = problem.n_constraints; - - A.create(problem.n_constraints, - problem.n_variables, - problem.nnz, - problem.offsets.data(), - problem.variables.data(), - problem.coefficients.data()); - - A_T.create(problem.n_variables, - problem.n_constraints, - problem.nnz, - problem.reverse_offsets.data(), - problem.reverse_constraints.data(), - problem.reverse_coefficients.data()); + cusparse_sp_mat_uptr A = make_csr(problem.n_constraints, + problem.n_variables, + problem.nnz, + problem.offsets.data(), + problem.variables.data(), + problem.coefficients.data()); + cusparse_sp_mat_uptr A_T = make_csr(problem.n_variables, + problem.n_constraints, + problem.nnz, + problem.reverse_offsets.data(), + problem.reverse_constraints.data(), + problem.reverse_coefficients.data()); + i_t primal_size = problem.n_variables; + i_t dual_size = problem.n_constraints; // Sync before starting anything to make sure everything is done RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view)); @@ -250,18 +246,28 @@ int optimal_batch_size_handler(const optimization_problem_t& op_proble const int left_node = std::max(1, current_batch_size / 2); const int right_node = std::min(current_batch_size * 2, max_batch_size); - double current_ratio = evaluate_node(A, - A_T, + double current_ratio = evaluate_node(A.get(), + A_T.get(), primal_size, dual_size, current_batch_size, benchmark_runs, op_problem.get_handle_ptr()); - double left_ratio = evaluate_node( - A, A_T, primal_size, dual_size, left_node, benchmark_runs, op_problem.get_handle_ptr()); - double right_ratio = evaluate_node( - A, A_T, primal_size, dual_size, right_node, benchmark_runs, op_problem.get_handle_ptr()); - int current_step = 1; + double left_ratio = evaluate_node(A.get(), + A_T.get(), + primal_size, + dual_size, + left_node, + benchmark_runs, + op_problem.get_handle_ptr()); + double right_ratio = evaluate_node(A.get(), + A_T.get(), + primal_size, + dual_size, + right_node, + benchmark_runs, + op_problem.get_handle_ptr()); + int current_step = 1; #ifdef BATCH_VERBOSE_MODE std::cout << "Starting batch size: " << current_batch_size << " and ratio: " << current_ratio @@ -290,8 +296,8 @@ int optimal_batch_size_handler(const optimization_problem_t& op_proble #ifdef BATCH_VERBOSE_MODE std::cout << "Evaluating left node: " << current_batch_size << std::endl; #endif - left_ratio = evaluate_node(A, - A_T, + left_ratio = evaluate_node(A.get(), + A_T.get(), primal_size, dual_size, current_batch_size, @@ -325,8 +331,13 @@ int optimal_batch_size_handler(const optimization_problem_t& op_proble #ifdef BATCH_VERBOSE_MODE std::cout << "Testing one last time between the two at node: " << middle_node << std::endl; #endif - double middle_ratio = evaluate_node( - A, A_T, primal_size, dual_size, middle_node, benchmark_runs, op_problem.get_handle_ptr()); + double middle_ratio = evaluate_node(A.get(), + A_T.get(), + primal_size, + dual_size, + middle_node, + benchmark_runs, + op_problem.get_handle_ptr()); #ifdef BATCH_VERBOSE_MODE std::cout << "Middle node ratio: " << middle_ratio << std::endl; #endif @@ -368,8 +379,8 @@ int optimal_batch_size_handler(const optimization_problem_t& op_proble #ifdef BATCH_VERBOSE_MODE std::cout << "Evaluating right node: " << current_batch_size << std::endl; #endif - right_ratio = evaluate_node(A, - A_T, + right_ratio = evaluate_node(A.get(), + A_T.get(), primal_size, dual_size, current_batch_size, @@ -404,8 +415,13 @@ int optimal_batch_size_handler(const optimization_problem_t& op_proble #ifdef BATCH_VERBOSE_MODE std::cout << "Testing one last time between the two at node: " << middle_node << std::endl; #endif - double middle_ratio = evaluate_node( - A, A_T, primal_size, dual_size, middle_node, benchmark_runs, op_problem.get_handle_ptr()); + double middle_ratio = evaluate_node(A.get(), + A_T.get(), + primal_size, + dual_size, + middle_node, + benchmark_runs, + op_problem.get_handle_ptr()); #ifdef BATCH_VERBOSE_MODE std::cout << "Middle node ratio: " << middle_ratio << std::endl; #endif diff --git a/cpp/src/pdlp/pdhg.cu b/cpp/src/pdlp/pdhg.cu index 81a5a2f0e5..67117e14d0 100644 --- a/cpp/src/pdlp/pdhg.cu +++ b/cpp/src/pdlp/pdhg.cu @@ -405,10 +405,10 @@ void pdhg_solver_t::compute_next_dual_solution(rmm::device_uvectorget_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A_mixed_, - cusparse_view_.tmp_primal, + cusparse_view_.A_mixed_.get(), + cusparse_view_.tmp_primal.get(), reusable_device_scalar_value_0_.data(), - cusparse_view_.dual_gradient, + cusparse_view_.dual_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, cusparse_view_.buffer_non_transpose_mixed_.data(), stream_view_); @@ -419,10 +419,10 @@ void pdhg_solver_t::compute_next_dual_solution(rmm::device_uvectorget_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A, - cusparse_view_.tmp_primal, + cusparse_view_.A.get(), + cusparse_view_.tmp_primal.get(), reusable_device_scalar_value_0_.data(), - cusparse_view_.dual_gradient, + cusparse_view_.dual_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_non_transpose.data(), stream_view_)); @@ -454,12 +454,12 @@ void pdhg_solver_t::spmvop_At_y() #if CUOPT_CUSPARSE_VER_12_8_UP if (is_cusparse_runtime_spmvop_supported()) { cusparse_spmvop_run(handle_ptr_->get_cusparse_handle(), - cusparse_view_.spmv_op_plan_A_t_, + cusparse_view_.spmv_op_plan_A_t_.get(), reusable_device_scalar_value_1_.data(), reusable_device_scalar_value_0_.data(), - cusparse_view_.dual_solution, - cusparse_view_.current_AtY, - cusparse_view_.current_AtY, + cusparse_view_.dual_solution.get(), + cusparse_view_.current_AtY.get(), + cusparse_view_.current_AtY.get(), stream_view_.value()); return; } @@ -467,10 +467,10 @@ void pdhg_solver_t::spmvop_At_y() RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A_T, - cusparse_view_.dual_solution, + cusparse_view_.A_T.get(), + cusparse_view_.dual_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view_.current_AtY, + cusparse_view_.current_AtY.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_transpose.data(), stream_view_)); @@ -482,12 +482,12 @@ void pdhg_solver_t::spmvop_A_x() #if CUOPT_CUSPARSE_VER_12_8_UP if (is_cusparse_runtime_spmvop_supported()) { cusparse_spmvop_run(handle_ptr_->get_cusparse_handle(), - cusparse_view_.spmv_op_plan_A_, + cusparse_view_.spmv_op_plan_A_.get(), reusable_device_scalar_value_1_.data(), reusable_device_scalar_value_0_.data(), - cusparse_view_.reflected_primal_solution, - cusparse_view_.dual_gradient, - cusparse_view_.dual_gradient, + cusparse_view_.reflected_primal_solution.get(), + cusparse_view_.dual_gradient.get(), + cusparse_view_.dual_gradient.get(), stream_view_.value()); return; } @@ -496,10 +496,10 @@ void pdhg_solver_t::spmvop_A_x() raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A, - cusparse_view_.reflected_primal_solution, + cusparse_view_.A.get(), + cusparse_view_.reflected_primal_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view_.dual_gradient, + cusparse_view_.dual_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_non_transpose.data(), stream_view_)); @@ -523,10 +523,10 @@ void pdhg_solver_t::compute_At_y() mixed_precision_spmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A_T_mixed_, - cusparse_view_.dual_solution, + cusparse_view_.A_T_mixed_.get(), + cusparse_view_.dual_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view_.current_AtY, + cusparse_view_.current_AtY.get(), CUSPARSE_SPMV_CSR_ALG2, cusparse_view_.buffer_transpose_mixed_.data(), stream_view_); @@ -538,10 +538,10 @@ void pdhg_solver_t::compute_At_y() raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A_T, - cusparse_view_.dual_solution, + cusparse_view_.A_T.get(), + cusparse_view_.dual_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view_.current_AtY, + cusparse_view_.current_AtY.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_transpose.data(), stream_view_)); @@ -552,10 +552,10 @@ void pdhg_solver_t::compute_At_y() CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A_T, - cusparse_view_.batch_dual_solutions, + cusparse_view_.A_T.get(), + cusparse_view_.batch_dual_solutions.get(), reusable_device_scalar_value_0_.data(), - cusparse_view_.batch_current_AtYs, + cusparse_view_.batch_current_AtYs.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, (f_t*)cusparse_view_.buffer_transpose_batch_row_row_.data(), stream_view_)); @@ -581,10 +581,10 @@ void pdhg_solver_t::compute_A_x() mixed_precision_spmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A_mixed_, - cusparse_view_.reflected_primal_solution, + cusparse_view_.A_mixed_.get(), + cusparse_view_.reflected_primal_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view_.dual_gradient, + cusparse_view_.dual_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, cusparse_view_.buffer_non_transpose_mixed_.data(), stream_view_); @@ -596,10 +596,10 @@ void pdhg_solver_t::compute_A_x() raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A, - cusparse_view_.reflected_primal_solution, + cusparse_view_.A.get(), + cusparse_view_.reflected_primal_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view_.dual_gradient, + cusparse_view_.dual_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_non_transpose.data(), stream_view_)); @@ -610,10 +610,10 @@ void pdhg_solver_t::compute_A_x() CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A, - cusparse_view_.batch_reflected_primal_solutions, + cusparse_view_.A.get(), + cusparse_view_.batch_reflected_primal_solutions.get(), reusable_device_scalar_value_0_.data(), - cusparse_view_.batch_dual_gradients, + cusparse_view_.batch_dual_gradients.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, (f_t*)cusparse_view_.buffer_non_transpose_batch_row_row_.data(), stream_view_)); @@ -630,7 +630,7 @@ void pdhg_solver_t::spmv_At_into(cusparseDnVecDescr_t in_desc, RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A_T, + cusparse_view_.A_T.get(), in_desc, reusable_device_scalar_value_0_.data(), out_desc, @@ -648,7 +648,7 @@ void pdhg_solver_t::spmv_A_into(cusparseDnVecDescr_t in_desc, raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A, + cusparse_view_.A.get(), in_desc, reusable_device_scalar_value_0_.data(), out_desc, @@ -1508,21 +1508,21 @@ void pdhg_solver_t::update_solution( std::swap(current_saddle_point_state_.current_AtY_, current_saddle_point_state_.next_AtY_); // Update cusparse views to point to the new values, cost is marginal - RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view_.current_AtY, + RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view_.current_AtY.get(), current_saddle_point_state_.current_AtY_.data())); - RAFT_CUSPARSE_TRY( - cusparseDnVecSetValues(cusparse_view_.next_AtY, current_saddle_point_state_.next_AtY_.data())); - RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view_.potential_next_dual_solution, + RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view_.next_AtY.get(), + current_saddle_point_state_.next_AtY_.data())); + RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view_.potential_next_dual_solution.get(), potential_next_dual_solution_.data())); - RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view_.primal_solution, + RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view_.primal_solution.get(), current_saddle_point_state_.primal_solution_.data())); - RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view_.dual_solution, + RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view_.dual_solution.get(), current_saddle_point_state_.dual_solution_.data())); RAFT_CUSPARSE_TRY( - cusparseDnVecSetValues(current_op_problem_evaluation_cusparse_view_.primal_solution, + cusparseDnVecSetValues(current_op_problem_evaluation_cusparse_view_.primal_solution.get(), current_saddle_point_state_.primal_solution_.data())); RAFT_CUSPARSE_TRY( - cusparseDnVecSetValues(current_op_problem_evaluation_cusparse_view_.dual_solution, + cusparseDnVecSetValues(current_op_problem_evaluation_cusparse_view_.dual_solution.get(), current_saddle_point_state_.dual_solution_.data())); } diff --git a/cpp/src/pdlp/pdlp.cu b/cpp/src/pdlp/pdlp.cu index 217ea4260a..abc119b1e6 100644 --- a/cpp/src/pdlp/pdlp.cu +++ b/cpp/src/pdlp/pdlp.cu @@ -1940,72 +1940,72 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( // Reset all cusparse view - // Reset cuSparse views for PDHG + // Reset cuSparse views for PDHG. unique_ptr move-assign destroys the old descriptor first. auto& pdhg_cusparse_view = pdhg_solver_.get_cusparse_view(); - pdhg_cusparse_view.batch_dual_solutions.create( - op_problem_scaled_.n_constraints, - climber_strategies_.size(), - climber_strategies_.size(), - pdhg_solver_.get_saddle_point_state().get_dual_solution().data(), - CUSPARSE_ORDER_ROW); - pdhg_cusparse_view.batch_current_AtYs.create( - op_problem_scaled_.n_variables, - climber_strategies_.size(), - climber_strategies_.size(), - pdhg_solver_.get_saddle_point_state().get_current_AtY().data(), - CUSPARSE_ORDER_ROW); - pdhg_cusparse_view.batch_reflected_primal_solutions.create( - op_problem_scaled_.n_variables, - climber_strategies_.size(), - climber_strategies_.size(), - pdhg_solver_.get_reflected_primal().data(), - CUSPARSE_ORDER_ROW); - pdhg_cusparse_view.batch_dual_gradients.create( - op_problem_scaled_.n_constraints, - climber_strategies_.size(), - climber_strategies_.size(), - pdhg_solver_.get_saddle_point_state().get_dual_gradient().data(), - CUSPARSE_ORDER_ROW); + pdhg_cusparse_view.batch_dual_solutions = + make_dnmat(op_problem_scaled_.n_constraints, + climber_strategies_.size(), + climber_strategies_.size(), + pdhg_solver_.get_saddle_point_state().get_dual_solution().data(), + CUSPARSE_ORDER_ROW); + pdhg_cusparse_view.batch_current_AtYs = + make_dnmat(op_problem_scaled_.n_variables, + climber_strategies_.size(), + climber_strategies_.size(), + pdhg_solver_.get_saddle_point_state().get_current_AtY().data(), + CUSPARSE_ORDER_ROW); + pdhg_cusparse_view.batch_reflected_primal_solutions = + make_dnmat(op_problem_scaled_.n_variables, + climber_strategies_.size(), + climber_strategies_.size(), + pdhg_solver_.get_reflected_primal().data(), + CUSPARSE_ORDER_ROW); + pdhg_cusparse_view.batch_dual_gradients = + make_dnmat(op_problem_scaled_.n_constraints, + climber_strategies_.size(), + climber_strategies_.size(), + pdhg_solver_.get_saddle_point_state().get_dual_gradient().data(), + CUSPARSE_ORDER_ROW); // Reset cusparse view used by adaptive step size strategy but owned by PDHG - pdhg_cusparse_view.batch_potential_next_dual_solution.create( - op_problem_scaled_.n_constraints, - climber_strategies_.size(), - op_problem_scaled_.n_constraints, - pdhg_solver_.get_potential_next_dual_solution().data(), - CUSPARSE_ORDER_COL); - pdhg_cusparse_view.batch_next_AtYs.create( - op_problem_scaled_.n_variables, - climber_strategies_.size(), - op_problem_scaled_.n_variables, - pdhg_solver_.get_saddle_point_state().get_next_AtY().data(), - CUSPARSE_ORDER_COL); + pdhg_cusparse_view.batch_potential_next_dual_solution = + make_dnmat(op_problem_scaled_.n_constraints, + climber_strategies_.size(), + op_problem_scaled_.n_constraints, + pdhg_solver_.get_potential_next_dual_solution().data(), + CUSPARSE_ORDER_COL); + pdhg_cusparse_view.batch_next_AtYs = + make_dnmat(op_problem_scaled_.n_variables, + climber_strategies_.size(), + op_problem_scaled_.n_variables, + pdhg_solver_.get_saddle_point_state().get_next_AtY().data(), + CUSPARSE_ORDER_COL); // Reset cusparse view used by convergence information but owned by PDLP - current_op_problem_evaluation_cusparse_view_.batch_primal_solutions.create( - op_problem_scaled_.n_variables, - climber_strategies_.size(), - op_problem_scaled_.n_variables, - pdhg_solver_.get_potential_next_primal_solution().data(), - CUSPARSE_ORDER_COL); - current_op_problem_evaluation_cusparse_view_.batch_dual_solutions.create( - op_problem_scaled_.n_constraints, - climber_strategies_.size(), - op_problem_scaled_.n_constraints, - pdhg_solver_.get_potential_next_dual_solution().data(), - CUSPARSE_ORDER_COL); - current_op_problem_evaluation_cusparse_view_.batch_tmp_duals.create( - op_problem_scaled_.n_constraints, - climber_strategies_.size(), - op_problem_scaled_.n_constraints, - pdhg_solver_.get_dual_tmp_resource().data(), - CUSPARSE_ORDER_COL); - current_op_problem_evaluation_cusparse_view_.batch_tmp_primals.create( - op_problem_scaled_.n_variables, - climber_strategies_.size(), - op_problem_scaled_.n_variables, - pdhg_solver_.get_primal_tmp_resource().data(), - CUSPARSE_ORDER_COL); + current_op_problem_evaluation_cusparse_view_.batch_primal_solutions = + make_dnmat(op_problem_scaled_.n_variables, + climber_strategies_.size(), + op_problem_scaled_.n_variables, + pdhg_solver_.get_potential_next_primal_solution().data(), + CUSPARSE_ORDER_COL); + current_op_problem_evaluation_cusparse_view_.batch_dual_solutions = + make_dnmat(op_problem_scaled_.n_constraints, + climber_strategies_.size(), + op_problem_scaled_.n_constraints, + pdhg_solver_.get_potential_next_dual_solution().data(), + CUSPARSE_ORDER_COL); + current_op_problem_evaluation_cusparse_view_.batch_tmp_duals = + make_dnmat(op_problem_scaled_.n_constraints, + climber_strategies_.size(), + op_problem_scaled_.n_constraints, + pdhg_solver_.get_dual_tmp_resource().data(), + CUSPARSE_ORDER_COL); + current_op_problem_evaluation_cusparse_view_.batch_tmp_primals = + make_dnmat(op_problem_scaled_.n_variables, + climber_strategies_.size(), + op_problem_scaled_.n_variables, + pdhg_solver_.get_primal_tmp_resource().data(), + CUSPARSE_ORDER_COL); // Recalculate SpMM buffer sizes for the new batch dimensions. // cuSparse may require different buffer sizes when the number of columns changes @@ -2019,10 +2019,10 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - pdhg_cusparse_view.A_T, - pdhg_cusparse_view.batch_dual_solutions, + pdhg_cusparse_view.A_T.get(), + pdhg_cusparse_view.batch_dual_solutions.get(), reusable_device_scalar_value_0_.data(), - pdhg_cusparse_view.batch_current_AtYs, + pdhg_cusparse_view.batch_current_AtYs.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &new_buf_size, stream_view_)); @@ -2034,10 +2034,10 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - pdhg_cusparse_view.A, - pdhg_cusparse_view.batch_reflected_primal_solutions, + pdhg_cusparse_view.A.get(), + pdhg_cusparse_view.batch_reflected_primal_solutions.get(), reusable_device_scalar_value_0_.data(), - pdhg_cusparse_view.batch_dual_gradients, + pdhg_cusparse_view.batch_dual_gradients.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &new_buf_size, stream_view_)); @@ -2049,10 +2049,10 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - pdhg_cusparse_view.A_T, - pdhg_cusparse_view.batch_potential_next_dual_solution, + pdhg_cusparse_view.A_T.get(), + pdhg_cusparse_view.batch_potential_next_dual_solution.get(), reusable_device_scalar_value_0_.data(), - pdhg_cusparse_view.batch_next_AtYs, + pdhg_cusparse_view.batch_next_AtYs.get(), CUSPARSE_SPMM_CSR_ALG3, &new_buf_size, stream_view_)); @@ -2064,10 +2064,10 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - current_op_problem_evaluation_cusparse_view_.A_T, - current_op_problem_evaluation_cusparse_view_.batch_dual_solutions, + current_op_problem_evaluation_cusparse_view_.A_T.get(), + current_op_problem_evaluation_cusparse_view_.batch_dual_solutions.get(), reusable_device_scalar_value_0_.data(), - current_op_problem_evaluation_cusparse_view_.batch_tmp_primals, + current_op_problem_evaluation_cusparse_view_.batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, &new_buf_size, stream_view_)); @@ -2080,10 +2080,10 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - current_op_problem_evaluation_cusparse_view_.A, - current_op_problem_evaluation_cusparse_view_.batch_primal_solutions, + current_op_problem_evaluation_cusparse_view_.A.get(), + current_op_problem_evaluation_cusparse_view_.batch_primal_solutions.get(), reusable_device_scalar_value_0_.data(), - current_op_problem_evaluation_cusparse_view_.batch_tmp_duals, + current_op_problem_evaluation_cusparse_view_.batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, &new_buf_size, stream_view_)); @@ -2100,10 +2100,10 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - pdhg_cusparse_view.A_T, - pdhg_cusparse_view.batch_dual_solutions, + pdhg_cusparse_view.A_T.get(), + pdhg_cusparse_view.batch_dual_solutions.get(), reusable_device_scalar_value_0_.data(), - pdhg_cusparse_view.batch_current_AtYs, + pdhg_cusparse_view.batch_current_AtYs.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, pdhg_cusparse_view.buffer_transpose_batch_row_row_.data(), stream_view_); @@ -2112,10 +2112,10 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - pdhg_cusparse_view.A, - pdhg_cusparse_view.batch_reflected_primal_solutions, + pdhg_cusparse_view.A.get(), + pdhg_cusparse_view.batch_reflected_primal_solutions.get(), reusable_device_scalar_value_0_.data(), - pdhg_cusparse_view.batch_dual_gradients, + pdhg_cusparse_view.batch_dual_gradients.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, pdhg_cusparse_view.buffer_non_transpose_batch_row_row_.data(), stream_view_); @@ -2125,10 +2125,10 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - pdhg_cusparse_view.A_T, - pdhg_cusparse_view.batch_potential_next_dual_solution, + pdhg_cusparse_view.A_T.get(), + pdhg_cusparse_view.batch_potential_next_dual_solution.get(), reusable_device_scalar_value_0_.data(), - pdhg_cusparse_view.batch_next_AtYs, + pdhg_cusparse_view.batch_next_AtYs.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)pdhg_cusparse_view.buffer_transpose_batch.data(), stream_view_); @@ -2139,10 +2139,10 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - current_op_problem_evaluation_cusparse_view_.A_T, - current_op_problem_evaluation_cusparse_view_.batch_dual_solutions, + current_op_problem_evaluation_cusparse_view_.A_T.get(), + current_op_problem_evaluation_cusparse_view_.batch_dual_solutions.get(), reusable_device_scalar_value_0_.data(), - current_op_problem_evaluation_cusparse_view_.batch_tmp_primals, + current_op_problem_evaluation_cusparse_view_.batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)current_op_problem_evaluation_cusparse_view_.buffer_transpose_batch.data(), stream_view_); @@ -2152,10 +2152,10 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - current_op_problem_evaluation_cusparse_view_.A, - current_op_problem_evaluation_cusparse_view_.batch_primal_solutions, + current_op_problem_evaluation_cusparse_view_.A.get(), + current_op_problem_evaluation_cusparse_view_.batch_primal_solutions.get(), reusable_device_scalar_value_0_.data(), - current_op_problem_evaluation_cusparse_view_.batch_tmp_duals, + current_op_problem_evaluation_cusparse_view_.batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)current_op_problem_evaluation_cusparse_view_.buffer_non_transpose_batch.data(), stream_view_); @@ -2260,7 +2260,7 @@ void pdlp_solver_t::compute_fixed_error(std::vector& has_restarte auto& sub_cv = sub_pdlp.pdhg_solver_.get_cusparse_view(); RAFT_CUSPARSE_TRY( - cusparseDnVecSetValues(sub_cv.potential_next_dual_solution, + cusparseDnVecSetValues(sub_cv.potential_next_dual_solution.get(), (void*)sub_pdlp.pdhg_solver_.get_reflected_dual().data())); sub_pdlp.step_size_strategy_.compute_interaction_and_movement( @@ -2271,7 +2271,7 @@ void pdlp_solver_t::compute_fixed_error(std::vector& has_restarte shard.rank_data.owned_cstr_size); RAFT_CUSPARSE_TRY(cusparseDnVecSetValues( - sub_cv.potential_next_dual_solution, + sub_cv.potential_next_dual_solution.get(), (void*)sub_pdlp.pdhg_solver_.get_potential_next_dual_solution().data())); }); @@ -2288,12 +2288,13 @@ void pdlp_solver_t::compute_fixed_error(std::vector& has_restarte RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); // Make potential_next_dual_solution point towards reflected dual solution to reuse the code - RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view.potential_next_dual_solution, + RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view.potential_next_dual_solution.get(), (void*)pdhg_solver_.get_reflected_dual().data())); if (batch_mode_) - RAFT_CUSPARSE_TRY(cusparseDnMatSetValues(cusparse_view.batch_potential_next_dual_solution, - (void*)pdhg_solver_.get_reflected_dual().data())); + RAFT_CUSPARSE_TRY( + cusparseDnMatSetValues(cusparse_view.batch_potential_next_dual_solution.get(), + (void*)pdhg_solver_.get_reflected_dual().data())); step_size_strategy_.compute_interaction_and_movement( pdhg_solver_.get_primal_tmp_resource(), cusparse_view, pdhg_solver_.get_saddle_point_state()); @@ -2331,12 +2332,12 @@ void pdlp_solver_t::compute_fixed_error(std::vector& has_restarte // Put back, already done in multi-gpu side if (!is_distributed_master()) { RAFT_CUSPARSE_TRY( - cusparseDnVecSetValues(cusparse_view.potential_next_dual_solution, + cusparseDnVecSetValues(cusparse_view.potential_next_dual_solution.get(), (void*)pdhg_solver_.get_potential_next_dual_solution().data())); } if (batch_mode_) { RAFT_CUSPARSE_TRY( - cusparseDnMatSetValues(cusparse_view.batch_potential_next_dual_solution, + cusparseDnMatSetValues(cusparse_view.batch_potential_next_dual_solution.get(), (void*)pdhg_solver_.get_potential_next_dual_solution().data())); } @@ -3390,7 +3391,7 @@ void pdlp_solver_t::compute_initial_step_size() raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view_.A_T, + cusparse_view_.A_T.get(), vecQ, reusable_device_scalar_value_0_.data(), vecATQ, @@ -3403,7 +3404,7 @@ void pdlp_solver_t::compute_initial_step_size() raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), // 1 - cusparse_view_.A, + cusparse_view_.A.get(), vecATQ, reusable_device_scalar_value_0_.data(), // 1 vecZ, diff --git a/cpp/src/pdlp/restart_strategy/pdlp_restart_strategy.cu b/cpp/src/pdlp/restart_strategy/pdlp_restart_strategy.cu index 954935b06e..01605dfb93 100644 --- a/cpp/src/pdlp/restart_strategy/pdlp_restart_strategy.cu +++ b/cpp/src/pdlp/restart_strategy/pdlp_restart_strategy.cu @@ -2270,10 +2270,10 @@ void pdlp_restart_strategy_t::compute_primal_gradient( RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_neg_1_.data(), - cusparse_view.A_T, - cusparse_view.dual_solution, + cusparse_view.A_T.get(), + cusparse_view.dual_solution.get(), reusable_device_scalar_value_1_.data(), - cusparse_view.primal_gradient, + cusparse_view.primal_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_transpose.data(), stream_view_)); @@ -2338,10 +2338,10 @@ void pdlp_restart_strategy_t::compute_dual_gradient( raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view.A, - cusparse_view.primal_solution, + cusparse_view.A.get(), + cusparse_view.primal_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view.dual_gradient, + cusparse_view.dual_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_non_transpose.data(), stream_view_)); @@ -2395,10 +2395,10 @@ void pdlp_restart_strategy_t::compute_lagrangian_value( RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view.A_T, - cusparse_view.dual_solution, + cusparse_view.A_T.get(), + cusparse_view.dual_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view.tmp_primal, + cusparse_view.tmp_primal.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_transpose.data(), stream_view_)); diff --git a/cpp/src/pdlp/step_size_strategy/adaptive_step_size_strategy.cu b/cpp/src/pdlp/step_size_strategy/adaptive_step_size_strategy.cu index e52c329166..5e1340a590 100644 --- a/cpp/src/pdlp/step_size_strategy/adaptive_step_size_strategy.cu +++ b/cpp/src/pdlp/step_size_strategy/adaptive_step_size_strategy.cu @@ -415,10 +415,10 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), // alpha - cusparse_view.A_T, - cusparse_view.potential_next_dual_solution, + cusparse_view.A_T.get(), + cusparse_view.potential_next_dual_solution.get(), reusable_device_scalar_value_0_.data(), // beta - cusparse_view.next_AtY, + cusparse_view.next_AtY.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_transpose.data(), stream_view_.value())); @@ -429,10 +429,10 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view.A_T, - cusparse_view.batch_potential_next_dual_solution, + cusparse_view.A_T.get(), + cusparse_view.batch_potential_next_dual_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view.batch_next_AtYs, + cusparse_view.batch_next_AtYs.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)cusparse_view.buffer_transpose_batch.data(), stream_view_.value())); diff --git a/cpp/src/pdlp/termination_strategy/convergence_information.cu b/cpp/src/pdlp/termination_strategy/convergence_information.cu index 685870b699..dd7cb925f6 100644 --- a/cpp/src/pdlp/termination_strategy/convergence_information.cu +++ b/cpp/src/pdlp/termination_strategy/convergence_information.cu @@ -720,10 +720,10 @@ void convergence_information_t::compute_primal_residual( raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view.A, - cusparse_view.primal_solution, + cusparse_view.A.get(), + cusparse_view.primal_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view.tmp_dual, + cusparse_view.tmp_dual.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_non_transpose.data(), stream_view_)); @@ -733,10 +733,10 @@ void convergence_information_t::compute_primal_residual( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view.A, - cusparse_view.batch_primal_solutions, + cusparse_view.A.get(), + cusparse_view.batch_primal_solutions.get(), reusable_device_scalar_value_0_.data(), - cusparse_view.batch_tmp_duals, + cusparse_view.batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)cusparse_view.buffer_non_transpose_batch.data(), stream_view_)); @@ -882,10 +882,10 @@ void convergence_information_t::compute_dual_residual( raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view.A_T, - cusparse_view.dual_solution, + cusparse_view.A_T.get(), + cusparse_view.dual_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view.tmp_primal, + cusparse_view.tmp_primal.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_transpose.data(), stream_view_)); @@ -895,10 +895,10 @@ void convergence_information_t::compute_dual_residual( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view.A_T, - cusparse_view.batch_dual_solutions, + cusparse_view.A_T.get(), + cusparse_view.batch_dual_solutions.get(), reusable_device_scalar_value_0_.data(), - cusparse_view.batch_tmp_primals, + cusparse_view.batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)cusparse_view.buffer_transpose_batch.data(), stream_view_)); diff --git a/cpp/src/pdlp/termination_strategy/infeasibility_information.cu b/cpp/src/pdlp/termination_strategy/infeasibility_information.cu index 7e38ffa845..f4f3a45577 100644 --- a/cpp/src/pdlp/termination_strategy/infeasibility_information.cu +++ b/cpp/src/pdlp/termination_strategy/infeasibility_information.cu @@ -321,10 +321,10 @@ void infeasibility_information_t::compute_infeasibility_information( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - scaled_cusparse_view_.A, - scaled_cusparse_view_.batch_delta_primal_solutions, + scaled_cusparse_view_.A.get(), + scaled_cusparse_view_.batch_delta_primal_solutions.get(), reusable_device_scalar_value_0_.data(), - scaled_cusparse_view_.batch_tmp_duals, + scaled_cusparse_view_.batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)scaled_cusparse_view_.buffer_non_transpose_batch.data(), stream_view_)); @@ -333,10 +333,10 @@ void infeasibility_information_t::compute_infeasibility_information( CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - scaled_cusparse_view_.A_T, - scaled_cusparse_view_.batch_delta_dual_solutions, + scaled_cusparse_view_.A_T.get(), + scaled_cusparse_view_.batch_delta_dual_solutions.get(), reusable_device_scalar_value_0_.data(), - scaled_cusparse_view_.batch_tmp_primals, + scaled_cusparse_view_.batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)scaled_cusparse_view_.buffer_transpose_batch.data(), stream_view_)); @@ -552,10 +552,10 @@ void infeasibility_information_t::compute_homogenous_primal_residual( raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_1_.data(), - cusparse_view.A, - cusparse_view.primal_solution, + cusparse_view.A.get(), + cusparse_view.primal_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view.tmp_dual, + cusparse_view.tmp_dual.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_non_transpose.data(), stream_view_)); @@ -622,10 +622,10 @@ void infeasibility_information_t::compute_homogenous_dual_residual( RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, reusable_device_scalar_value_neg_1_.data(), - cusparse_view.A_T, - cusparse_view.dual_solution, + cusparse_view.A_T.get(), + cusparse_view.dual_solution.get(), reusable_device_scalar_value_0_.data(), - cusparse_view.tmp_primal, + cusparse_view.tmp_primal.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_transpose.data(), stream_view_)); From 74865d3c8a84714e8d265f92c43e9ced74504496 Mon Sep 17 00:00:00 2001 From: Trevor McKay Date: Thu, 3 Sep 2026 10:12:12 -0400 Subject: [PATCH 040/113] Forward routing gRPC settings and map node name strings to ints (#1837) This change fixes some gaps in the gRPC routing client: 1) string values for node name types in initial solutions were not turned into ints before sending across the wire 2) verbose_mode and error_logging settings flags were not being forwarded Authors: - Trevor McKay (https://github.com/tmckayus) Approvers: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) URL: https://github.com/NVIDIA/cuopt/pull/1837 --- .../cuopt/cuopt/grpc/client/grpc_client.pyx | 99 ++++++++++++++++--- .../cuopt/cuopt/routing/vehicle_routing.pxd | 4 +- python/cuopt/cuopt/routing/vehicle_routing.py | 10 ++ .../cuopt/routing/vehicle_routing_wrapper.pyx | 8 +- .../test_routing_grpc_serialization.py | 63 ++++++++++++ 5 files changed, 169 insertions(+), 15 deletions(-) diff --git a/python/cuopt/cuopt/grpc/client/grpc_client.pyx b/python/cuopt/cuopt/grpc/client/grpc_client.pyx index 6e1909fc2f..3f9cdcb0e5 100644 --- a/python/cuopt/cuopt/grpc/client/grpc_client.pyx +++ b/python/cuopt/cuopt/grpc/client/grpc_client.pyx @@ -731,6 +731,52 @@ def _to_host(x): return np.asarray(x) +_NODE_TYPE_NAMES = { + "Depot": 0, + "Pickup": 1, + "Delivery": 2, + "Break": 3, +} + + +def _is_node_type_name_array(values): + """True when values are node-type names rather than wire integers. + + Integer and unsigned arrays (including uint8, the local DataModel storage + type) pass through. Name arrays are Unicode (U), byte strings (S), NumPy 2 + variable-width strings (T), or object arrays of str/bytes. Object arrays of + ints must not take the name path. + """ + kind = values.dtype.kind + if kind in "UST": + return True + if kind != "O" or values.size == 0: + return False + # Types are a 1-D sequence. Flatten only for a stray column (n, 1). + sample = values[0] if values.ndim == 1 else values.reshape(-1)[0] + return isinstance(sample, (str, bytes, np.str_, np.bytes_)) + + +def _node_type_name(value): + if isinstance(value, bytes): + return value.decode("utf-8") + return value + + +def _routing_node_types(values): + """Normalize routing node names or enum values to wire integers.""" + values = np.asarray(_to_host(values)) + if not _is_node_type_name_array(values): + return values + try: + return np.asarray( + [_NODE_TYPE_NAMES[_node_type_name(value)] for value in values.ravel()], + dtype=np.int32, + ) + except KeyError as error: + raise ValueError(f"unknown routing node type {error.args[0]!r}") from error + + # --- numpy -> std::vector fillers ------------------------------------------ cdef void _fill_i32(vector[int32_t]& v, arr) except *: @@ -868,7 +914,9 @@ cdef void _populate(cpu_routing_problem_t& p, data_model) except *: elif name == "add_initial_solutions": _fill_i32(p.initial_solutions.vehicle_ids, args[0]) _fill_i32(p.initial_solutions.routes, args[1]) - _fill_i32(p.initial_solutions.types, args[2]) + _fill_i32( + p.initial_solutions.types, _routing_node_types(args[2]) + ) _fill_i32(p.initial_solutions.sol_offsets, args[3]) elif name == "set_min_vehicles": p.min_vehicles = int(args[0]) @@ -965,6 +1013,42 @@ cdef _solution_to_py(cpu_routing_solution_t s): } +# dump_best_results is intentionally not forwarded. The proto and C++ mapper +# already carry dump_best_results_path, but that writes a debug file on the +# gRPC server host rather than returning data to the client. Wire it up later +# if a remote-debug use case appears. +cdef void _apply_routing_settings( + routing_solver_settings_t[int, float]& s, settings +) except *: + if settings is None: + return + if isinstance(settings, dict): + tl = settings.get("time_limit") + if tl is not None: + s.set_time_limit(float(tl)) + verbose = settings.get("verbose_mode", settings.get("verbose")) + if verbose is not None: + s.set_verbose_mode(bool(verbose)) + error_logging = settings.get("error_logging") + if error_logging is not None: + s.set_error_logging_mode(bool(error_logging)) + return + + get_time_limit = getattr(settings, "get_time_limit", None) + if get_time_limit is not None: + tl = get_time_limit() + if tl is not None: + s.set_time_limit(float(tl)) + get_verbose_mode = getattr(settings, "get_verbose_mode", None) + if get_verbose_mode is not None: + s.set_verbose_mode(bool(get_verbose_mode())) + get_error_logging_mode = getattr( + settings, "get_error_logging_mode", None + ) + if get_error_logging_mode is not None: + s.set_error_logging_mode(bool(get_error_logging_mode())) + + cdef class RoutingClient: """Client for solving VRP problems on a remote cuOpt gRPC server.""" @@ -996,18 +1080,7 @@ cdef class RoutingClient: ) cdef _apply_settings(self, routing_solver_settings_t[int, float]& s, settings): - if settings is None: - return - if isinstance(settings, dict): - tl = settings.get("time_limit") - if tl is not None: - s.set_time_limit(float(tl)) - return - get_time_limit = getattr(settings, "get_time_limit", None) - if get_time_limit is not None: - tl = get_time_limit() - if tl is not None: - s.set_time_limit(float(tl)) + _apply_routing_settings(s, settings) def submit(self, data_model, settings=None): """Serialize and submit a VRP problem; return its ``job_id``.""" diff --git a/python/cuopt/cuopt/routing/vehicle_routing.pxd b/python/cuopt/cuopt/routing/vehicle_routing.pxd index 7f89d33ff8..d5e714a41e 100644 --- a/python/cuopt/cuopt/routing/vehicle_routing.pxd +++ b/python/cuopt/cuopt/routing/vehicle_routing.pxd @@ -1,4 +1,4 @@ -# SPDX-FileCopyrightText: Copyright (c) 2021-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # noqa +# SPDX-FileCopyrightText: Copyright (c) 2021-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 @@ -128,6 +128,8 @@ cdef extern from "cuopt/routing/solve.hpp" namespace "cuopt::routing": void dump_best_results(const string &file_path, i_t interval) except+ f_t get_time_limit() except+ + bool get_verbose_mode() except+ + bool get_error_logging_mode() except+ cdef extern from "cuopt/routing/cython/cython.hpp" namespace "cuopt::cython": # noqa cdef unique_ptr[vehicle_routing_ret_t] call_solve( diff --git a/python/cuopt/cuopt/routing/vehicle_routing.py b/python/cuopt/cuopt/routing/vehicle_routing.py index bff3aefc22..0f0433f6c9 100644 --- a/python/cuopt/cuopt/routing/vehicle_routing.py +++ b/python/cuopt/cuopt/routing/vehicle_routing.py @@ -1510,6 +1510,16 @@ def get_time_limit(self): """ return super().get_time_limit() + @catch_cuopt_exception + def get_verbose_mode(self): + """Return whether verbose solver output is enabled.""" + return super().get_verbose_mode() + + @catch_cuopt_exception + def get_error_logging_mode(self): + """Return whether constraint error logging is enabled.""" + return super().get_error_logging_mode() + @catch_cuopt_exception def get_best_results_file_path(self): """ diff --git a/python/cuopt/cuopt/routing/vehicle_routing_wrapper.pyx b/python/cuopt/cuopt/routing/vehicle_routing_wrapper.pyx index a290132d50..972ffd86b0 100644 --- a/python/cuopt/cuopt/routing/vehicle_routing_wrapper.pyx +++ b/python/cuopt/cuopt/routing/vehicle_routing_wrapper.pyx @@ -1,4 +1,4 @@ -# SPDX-FileCopyrightText: Copyright (c) 2021-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # noqa +# SPDX-FileCopyrightText: Copyright (c) 2021-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 @@ -788,6 +788,12 @@ cdef class SolverSettings: def get_time_limit(self): return self.c_solver_settings.get().get_time_limit() + def get_verbose_mode(self): + return self.c_solver_settings.get().get_verbose_mode() + + def get_error_logging_mode(self): + return self.c_solver_settings.get().get_error_logging_mode() + def get_best_results_file_path(self): return self.file_path diff --git a/python/cuopt/cuopt/tests/routing/test_routing_grpc_serialization.py b/python/cuopt/cuopt/tests/routing/test_routing_grpc_serialization.py index fa43287ee2..e7e4abfd16 100644 --- a/python/cuopt/cuopt/tests/routing/test_routing_grpc_serialization.py +++ b/python/cuopt/cuopt/tests/routing/test_routing_grpc_serialization.py @@ -115,6 +115,69 @@ def test_populate_breaks(): assert s["uniform_breaks"] == 1 +def test_populate_initial_solution_node_type_names(): + dm = routing.DataModel(3, 1, 2) + dm.add_cost_matrix(np.eye(3, dtype=np.float32)) + dm.add_initial_solutions( + np.array([0, 0, 0, 0], np.int32), + np.array([0, 0, 1, 0], np.int32), + np.array(["Depot", "Pickup", "Delivery", "Depot"]), + np.array([0, 4], np.int32), + ) + assert problem_summary(dm)["initial_solutions_routes"] == 4 + + +def test_routing_node_types_accepts_string_and_integer_arrays(): + from cuopt.grpc.client import grpc_client as grpc_native + + names = ["Depot", "Pickup", "Delivery", "Break"] + expected = np.array([0, 1, 2, 3], dtype=np.int32) + assert np.array_equal(grpc_native._routing_node_types(names), expected) + assert np.array_equal( + grpc_native._routing_node_types(np.array(names, dtype=object)), + expected, + ) + assert np.array_equal( + grpc_native._routing_node_types(np.array(names)), expected + ) + assert np.array_equal( + grpc_native._routing_node_types( + np.array([b"Depot", b"Pickup", b"Delivery", b"Break"]) + ), + expected, + ) + assert np.array_equal(grpc_native._routing_node_types(expected), expected) + assert np.array_equal( + grpc_native._routing_node_types(expected.astype(np.uint8)), + expected.astype(np.uint8), + ) + assert np.array_equal( + grpc_native._routing_node_types(np.array([0, 1, 2, 3], dtype=object)), + np.array([0, 1, 2, 3], dtype=object), + ) + string_dtype = getattr(np.dtypes, "StringDType", None) + if string_dtype is not None: + assert np.array_equal( + grpc_native._routing_node_types( + np.array(names, dtype=string_dtype()) + ), + expected, + ) + + +def test_routing_settings_object_exposes_values_the_client_forwards(): + # RoutingClient copies these through get_time_limit / get_verbose_mode / + # get_error_logging_mode. The dict branch of _apply_routing_settings reads + # the same fields as time_limit, verbose_mode (or verbose), and error_logging. + settings = routing.SolverSettings() + settings.set_time_limit(3.5) + settings.set_verbose_mode(True) + settings.set_error_logging_mode(False) + assert settings.get_time_limit() == 3.5 + assert settings.get_verbose_mode() is True + assert settings.get_error_logging_mode() is False + + def test_populate_handles_pandas_host_inputs(): """Pandas (host) inputs map identically to numpy. From b43fae04e34c8da1894850d6ffb512633c5b3d2e Mon Sep 17 00:00:00 2001 From: "Nicolas L. Guidotti" Date: Mon, 7 Sep 2026 16:22:38 +0200 Subject: [PATCH 041/113] Set Papilo back to the main branch (#1694) Set Papilo back to the main branch as the issue related with `DualInfer` reduction has been fixed. See scipopt/papilo#89. Closes #1585. Authors: - Nicolas L. Guidotti (https://github.com/nguidotti) Approvers: - Rajesh Gandham (https://github.com/rg20) - James Lamb (https://github.com/jameslamb) - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) URL: https://github.com/NVIDIA/cuopt/pull/1694 --- cpp/CMakeLists.txt | 10 +--- .../unit_tests/presolve_test.cu | 46 ++++++++++++++----- 2 files changed, 37 insertions(+), 19 deletions(-) diff --git a/cpp/CMakeLists.txt b/cpp/CMakeLists.txt index 3d41ca01e0..72e52902e9 100644 --- a/cpp/CMakeLists.txt +++ b/cpp/CMakeLists.txt @@ -270,14 +270,8 @@ endif () FetchContent_Declare( papilo - GIT_REPOSITORY "https://github.com/akifcorduk/papilo.git" - # We would want to get the main branch. However, the main branch - # does not have some of the presolvers and settings that we need - # Mainly, probing and clique merging. - # This is the reason we are using the development branch - # from Oct 12, 2025. Once these changes are merged into the main branch, - #we can switch to the main branch. - GIT_TAG "55d5edece584885061639ecf3a6eb8a4629be9f2" + GIT_REPOSITORY "https://github.com/scipopt/papilo.git" + GIT_TAG "3a0d1d30a0a580530b37495b0ebb5926cf6e8a2c" GIT_PROGRESS TRUE EXCLUDE_FROM_ALL SYSTEM diff --git a/cpp/tests/linear_programming/unit_tests/presolve_test.cu b/cpp/tests/linear_programming/unit_tests/presolve_test.cu index 92e962059d..d8a723b5a9 100644 --- a/cpp/tests/linear_programming/unit_tests/presolve_test.cu +++ b/cpp/tests/linear_programming/unit_tests/presolve_test.cu @@ -1020,10 +1020,8 @@ TEST_P(papilo_problem, round_trip) } // Exercises the MIP presolve path: presolver_t::apply -> third_party_presolve_t::apply -// reduces a user_problem_t in place via PaPILO. ex9 is fully solved by presolve (it collapses -// to a 0x0 problem), so this also checks the OPTIMAL status and that postsolve maps the empty -// reduced solution back to a full-dimension, objective-81 assignment. -TEST(submip_presolve, ex9_fully_reduced) +// reduces a user_problem_t in place via PaPILO. +TEST(submip_presolve, ex9_reduced_to_residual_row) { const raft::handle_t handle_{}; @@ -1044,17 +1042,43 @@ TEST(submip_presolve, ex9_fully_reduced) mip::third_party_presolve_t presolver; auto status = presolver.apply_to_subproblem(user_problem, settings, 120, 8); - // PaPILO solves ex9 entirely during presolve -> empty reduced problem. - EXPECT_EQ(status, mip::third_party_presolve_status_t::OPTIMAL); - EXPECT_EQ(user_problem.num_rows, 0); - EXPECT_EQ(user_problem.num_cols, 0); - EXPECT_EQ(user_problem.A.nnz(), 0); + ASSERT_EQ(status, mip::third_party_presolve_status_t::REDUCED); + ASSERT_EQ(user_problem.num_rows, 1); + ASSERT_EQ(user_problem.num_cols, 4); + ASSERT_EQ(user_problem.A.nnz(), 4); - // Postsolve reconstructs the full original assignment from the (empty) reduced solution. - std::vector reduced_solution; // no reduced columns remain + // The residual is `x0 + x1 + x2 + x3 == 1` over four binaries of unit cost, with the remaining + // 80 of the optimum already banked in obj_constant. + EXPECT_DOUBLE_EQ(user_problem.obj_constant, 80.0); + EXPECT_EQ(user_problem.row_sense[0], 'E'); + EXPECT_DOUBLE_EQ(user_problem.rhs[0], 1.0); + for (int j = 0; j < user_problem.num_cols; ++j) { + EXPECT_DOUBLE_EQ(user_problem.objective[j], 1.0) << "column " << j; + EXPECT_DOUBLE_EQ(user_problem.lower[j], 0.0) << "column " << j; + EXPECT_DOUBLE_EQ(user_problem.upper[j], 1.0) << "column " << j; + ASSERT_EQ(user_problem.A.col_start[j + 1] - user_problem.A.col_start[j], 1) << "column " << j; + EXPECT_EQ(user_problem.A.i[user_problem.A.col_start[j]], 0) << "column " << j; + EXPECT_DOUBLE_EQ(user_problem.A.x[user_problem.A.col_start[j]], 1.0) << "column " << j; + } + + // One column per surviving reduced column, each pointing at a distinct original column. + const auto& reduced_to_original = presolver.get_reduced_to_original_map(); + ASSERT_EQ(reduced_to_original.size(), user_problem.num_cols); + for (int j = 0; j < user_problem.num_cols; ++j) { + EXPECT_GE(reduced_to_original[j], 0) << "column " << j; + EXPECT_LT(reduced_to_original[j], orig_cols) << "column " << j; + } + + // Satisfying the residual row completes the optimum: postsolve reconstructs the full original + // assignment, keeping the reduced values in place and filling in everything presolve removed. + std::vector reduced_solution(user_problem.num_cols, 0.0); + reduced_solution[0] = 1.0; std::vector full_solution; presolver.uncrush_primal_solution(reduced_solution, full_solution); ASSERT_EQ(static_cast(full_solution.size()), orig_cols); + for (int j = 0; j < user_problem.num_cols; ++j) { + EXPECT_DOUBLE_EQ(full_solution[reduced_to_original[j]], reduced_solution[j]) << "column " << j; + } double objective = 0.0; for (int j = 0; j < orig_cols; ++j) { From a0435e54b8fde60ae5c4560f90914c2025714103 Mon Sep 17 00:00:00 2001 From: Bradley Dice Date: Tue, 8 Sep 2026 10:57:18 -0500 Subject: [PATCH 042/113] Adopt CUDA stream compatibility accessors (#1858) Use the `get()` and `sync()` compatibility aliases added in [RMM #2537](https://github.com/rapidsai/rmm/pull/2537). These spellings are shared by `rmm::cuda_stream_view` and `cuda::stream_ref`. This preserves existing stream types and public APIs while extracting mechanical accessor updates from the broader stream migration. It is independently buildable without [RMM #2372](https://github.com/rapidsai/rmm/pull/2372) and leaves the migration PR focused on actual type and signature changes. This updates raw CUDA, library, kernel-launch, and legacy API boundaries throughout routing and mathematical optimization code while preserving current stream types. The remaining signature migration stays in [cuOpt #1828](https://github.com/NVIDIA/cuopt/pull/1828). ## Issue [rapidsai/build-planning#318](https://github.com/rapidsai/build-planning/issues/318) Authors: - Bradley Dice (https://github.com/bdice) Approvers: - Rajesh Gandham (https://github.com/rg20) URL: https://github.com/NVIDIA/cuopt/pull/1858 --- .../mip/solver_settings.hpp | 5 +- .../optimization_problem_solution.hpp | 26 +- .../pdlp/solver_settings.hpp | 19 +- .../solver_settings.hpp | 11 +- .../utilities/segmented_sum_handler.cuh | 10 +- cpp/src/barrier/barrier.cu | 257 +++++++++--------- cpp/src/barrier/csr_kkt_build.cuh | 23 +- cpp/src/barrier/cusparse_view.cu | 15 +- cpp/src/barrier/device_sparse_matrix.cuh | 58 ++-- cpp/src/barrier/iterative_refinement.hpp | 20 +- cpp/src/barrier/second_order_cone_kernels.cuh | 183 ++++++------- .../barrier/second_order_cone_reduction.cuh | 10 +- cpp/src/barrier/sparse_cholesky.cuh | 28 +- cpp/src/linear_algebra/sort_csr.cuh | 8 +- cpp/src/linear_algebra/vector_math.cuh | 8 +- .../diversity/assignment_hash_map.cu | 14 +- .../diversity/recombiners/recombiner.cuh | 10 +- .../feasibility_jump/feasibility_jump.cu | 26 +- .../feasibility_jump/feasibility_jump.cuh | 2 +- .../feasibility_jump_kernels.cu | 68 +++-- .../mip_heuristics/feasibility_jump/utils.cuh | 4 +- .../feasibility_pump/feasibility_pump.cu | 4 +- .../local_search/lagrangian.cuh | 2 +- .../local_search/rounding/bounds_repair.cu | 4 +- .../local_search/rounding/constraint_prop.cu | 6 +- .../local_search/rounding/lb_bounds_repair.cu | 16 +- .../rounding/lb_constraint_prop.cu | 12 +- .../local_search/rounding/simple_rounding.cu | 39 +-- .../mip_heuristics/mip_scaling_strategy.cu | 20 +- cpp/src/mip_heuristics/presolve/block_bve.cu | 24 +- .../presolve/bounds_presolve.cu | 10 +- .../conditional_bound_strengthening.cu | 6 +- .../presolve/lb_probing_cache.cu | 2 +- .../presolve/load_balanced_bounds_presolve.cu | 42 +-- .../load_balanced_bounds_presolve.cuh | 6 +- .../load_balanced_bounds_presolve_helpers.cuh | 20 +- .../load_balanced_partition_helpers.cuh | 2 +- .../mip_heuristics/presolve/multi_probe.cu | 24 +- .../mip_heuristics/presolve/probing_cache.cu | 10 +- .../presolve/third_party_presolve.cpp | 4 +- .../presolve/trivial_presolve.cuh | 16 +- .../problem/load_balanced_problem.cu | 46 ++-- cpp/src/mip_heuristics/problem/problem.cu | 49 ++-- .../problem/problem_helpers.cuh | 10 +- .../solution/feasibility_test.cuh | 10 +- cpp/src/mip_heuristics/solution/solution.cu | 13 +- cpp/src/mip_heuristics/solve.cu | 9 +- cpp/src/mip_heuristics/solver.cu | 7 +- cpp/src/mip_heuristics/solver_solution.cu | 4 +- cpp/src/mip_heuristics/utils.cuh | 6 +- cpp/src/pdlp/cpu_pdlp_warm_start_data.cu | 4 +- cpp/src/pdlp/cuopt_c_internal.hpp | 4 +- cpp/src/pdlp/cusparse_view.cu | 78 +++--- cpp/src/pdlp/cusparse_view.hpp | 1 - .../distributed_algorithms.cu | 4 +- .../pdlp/distributed_pdlp/multi_gpu_engine.cu | 10 +- .../distributed_pdlp/multi_gpu_engine.hpp | 2 +- .../initial_scaling.cu | 99 ++++--- .../optimal_batch_size_handler.cu | 14 +- cpp/src/pdlp/optimization_problem.cu | 16 +- cpp/src/pdlp/pdhg.cu | 56 ++-- cpp/src/pdlp/pdlp.cu | 146 +++++----- cpp/src/pdlp/pdlp_warm_start_data.cu | 27 +- .../localized_duality_gap_container.cu | 6 +- .../localized_duality_gap_container.hpp | 1 - .../restart_strategy/pdlp_restart_strategy.cu | 125 +++++---- .../weighted_average_solution.cu | 40 +-- cpp/src/pdlp/saddle_point.cu | 12 +- cpp/src/pdlp/solve.cu | 19 +- cpp/src/pdlp/solver_solution.cu | 14 +- .../adaptive_step_size_strategy.cu | 67 +++-- cpp/src/pdlp/swap_and_resize_helper.cuh | 2 +- .../convergence_information.cu | 124 ++++----- .../infeasibility_information.cu | 76 +++--- .../termination_strategy.cu | 22 +- cpp/src/pdlp/translate.hpp | 8 +- cpp/src/pdlp/utilities/cython_solve.cu | 48 ++-- cpp/src/pdlp/utils.cuh | 30 +- cpp/src/routing/adapters/adapted_sol.cuh | 2 +- .../routing/adapters/assignment_adapter.cuh | 6 +- cpp/src/routing/adapters/solution_adapter.cuh | 2 +- cpp/src/routing/assignment.cu | 7 +- cpp/src/routing/cpu_routing_problem.cu | 4 +- .../routing/crossovers/optimal_eax_cycles.cu | 26 +- cpp/src/routing/crossovers/ox_recombiner.cuh | 22 +- cpp/src/routing/cuda_graph.cuh | 6 +- .../distance_engine/waypoint_matrix.cpp | 8 +- cpp/src/routing/fleet_info.cu | 10 +- cpp/src/routing/generator/generator.cu | 38 +-- .../routing/ges/compute_fragment_ejections.cu | 2 +- cpp/src/routing/ges/eject_until_feasible.cu | 12 +- cpp/src/routing/ges/ejection_pool.cuh | 2 +- cpp/src/routing/ges/execute_insertion.cu | 28 +- cpp/src/routing/ges/guided_ejection_search.cu | 14 +- .../routing/ges/guided_ejection_search.cuh | 3 +- .../brute_force_lexico.cu | 20 +- .../lexicographic_search.cu | 35 +-- cpp/src/routing/ges/squeeze.cu | 103 +++---- .../routing/local_search/breaks_insertion.cu | 12 +- .../local_search/compute_compatible.cu | 62 +++-- .../local_search/compute_insertions.cu | 12 +- .../local_search/cycle_finder/cycle_finder.cu | 68 ++--- .../cycle_finder/cycle_finder.hpp | 3 +- .../routing/local_search/fill_gpu_graph.cu | 6 +- .../local_search/hvrp/vehicle_assignment.cu | 35 +-- cpp/src/routing/local_search/local_search.cu | 4 +- cpp/src/routing/local_search/perform_moves.cu | 17 +- .../routing/local_search/prize_collection.cu | 15 +- cpp/src/routing/local_search/random_cross.cu | 24 +- cpp/src/routing/local_search/sliding_tsp.cu | 26 +- .../routing/local_search/sliding_window.cu | 26 +- cpp/src/routing/local_search/two_opt.cu | 10 +- .../local_search/vrp/nodes_to_search.cu | 4 +- .../routing/local_search/vrp/vrp_execute.cu | 10 +- .../routing/local_search/vrp/vrp_search.cu | 4 +- cpp/src/routing/order_info.cu | 10 +- cpp/src/routing/problem/problem.cu | 2 +- cpp/src/routing/route/capacity_route.cuh | 2 +- cpp/src/routing/solution/pool_allocator.cuh | 2 +- cpp/src/routing/solution/solution.cu | 42 +-- cpp/src/routing/solution/solution_handle.cuh | 4 +- .../util_kernels/compute_backward_forward.cu | 6 +- .../routing/util_kernels/runtime_checks.cu | 10 +- .../routing/util_kernels/set_initial_nodes.cu | 14 +- cpp/src/routing/utilities/check_input.cu | 30 +- cpp/src/routing/utilities/cython.cu | 4 +- cpp/src/utilities/copy_helpers.hpp | 6 +- cpp/src/utilities/event_handler.cuh | 8 +- cpp/src/utilities/manual_cuda_graph.cuh | 12 +- cpp/src/utilities/vector_helpers.cuh | 10 +- .../distance_engine/waypoint_matrix_test.cpp | 8 +- .../dual_simplex/unit_tests/solve_barrier.cu | 7 +- cpp/tests/linear_programming/pdlp_test.cu | 24 +- .../unit_tests/solution_interface_test.cu | 6 +- cpp/tests/mip/bounds_standardization_test.cu | 7 +- cpp/tests/mip/elim_var_remap_test.cu | 7 +- cpp/tests/mip/feasibility_jump_tests.cu | 7 +- cpp/tests/mip/load_balancing_test.cu | 7 +- cpp/tests/mip/multi_probe_test.cu | 7 +- cpp/tests/routing/level0/l0_routing_test.cu | 14 +- .../routing/level0/l0_vehicle_order_match.cu | 2 +- .../unit_tests/local_search_cand_test.cu | 4 +- cpp/tests/routing/unit_tests/top_k.cu | 31 ++- .../routing/utilities/check_constraints.cu | 2 - cpp/tests/socp/general_quadratic_test.cu | 7 +- cpp/tests/socp/second_order_cone_kernels.cu | 24 +- cpp/tests/socp/solve_barrier_socp.cu | 7 +- cpp/tests/socp/sparse_augmented_kkt_test.cu | 14 +- 148 files changed, 1735 insertions(+), 1583 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp index f12c33e818..89e8e89886 100644 --- a/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/mip/solver_settings.hpp @@ -19,6 +19,8 @@ #include #include +#include + #include #include @@ -90,7 +92,8 @@ class mip_solver_settings_t { */ void add_initial_solution(const f_t* initial_solution, i_t size, - rmm::cuda_stream_view stream = rmm::cuda_stream_default); + rmm::cuda_stream_view stream = cuda::stream_ref{ + cudaStream_t{cudaStreamDefault}}); /** * @brief Get the callback for the user solution diff --git a/cpp/include/cuopt/mathematical_optimization/optimization_problem_solution.hpp b/cpp/include/cuopt/mathematical_optimization/optimization_problem_solution.hpp index 577d5727ec..b3706473b3 100644 --- a/cpp/include/cuopt/mathematical_optimization/optimization_problem_solution.hpp +++ b/cpp/include/cuopt/mathematical_optimization/optimization_problem_solution.hpp @@ -65,7 +65,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { solution_.get_primal_solution().data(), solution_.get_primal_solution().size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -77,7 +77,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { solution_.get_dual_solution().data(), solution_.get_dual_solution().size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -88,7 +88,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { auto stream = reduced_cost.stream(); std::vector result(reduced_cost.size()); raft::copy(result.data(), reduced_cost.data(), reduced_cost.size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -154,7 +154,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { ws.current_primal_solution_.data(), ws.current_primal_solution_.size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -167,7 +167,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { std::vector result(ws.current_dual_solution_.size()); raft::copy( result.data(), ws.current_dual_solution_.data(), ws.current_dual_solution_.size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -180,7 +180,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { std::vector result(ws.initial_primal_average_.size()); raft::copy( result.data(), ws.initial_primal_average_.data(), ws.initial_primal_average_.size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -193,7 +193,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { std::vector result(ws.initial_dual_average_.size()); raft::copy( result.data(), ws.initial_dual_average_.data(), ws.initial_dual_average_.size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -205,7 +205,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { auto stream = ws.current_ATY_.stream(); std::vector result(ws.current_ATY_.size()); raft::copy(result.data(), ws.current_ATY_.data(), ws.current_ATY_.size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -218,7 +218,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { std::vector result(ws.sum_primal_solutions_.size()); raft::copy( result.data(), ws.sum_primal_solutions_.data(), ws.sum_primal_solutions_.size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -230,7 +230,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { auto stream = ws.sum_dual_solutions_.stream(); std::vector result(ws.sum_dual_solutions_.size()); raft::copy(result.data(), ws.sum_dual_solutions_.data(), ws.sum_dual_solutions_.size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -245,7 +245,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { ws.last_restart_duality_gap_primal_solution_.data(), ws.last_restart_duality_gap_primal_solution_.size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -260,7 +260,7 @@ class gpu_lp_solution_t : public lp_solution_interface_t { ws.last_restart_duality_gap_dual_solution_.data(), ws.last_restart_duality_gap_dual_solution_.size(), stream); - stream.synchronize(); + stream.sync(); return result; } @@ -406,7 +406,7 @@ class gpu_mip_solution_t : public mip_solution_interface_t { std::vector result(solution_.get_solution().size()); raft::copy( result.data(), solution_.get_solution().data(), solution_.get_solution().size(), stream); - stream.synchronize(); + stream.sync(); return result; } diff --git a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp index cf6424fb6f..1c4d0ce71b 100644 --- a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp @@ -8,6 +8,8 @@ #pragma once #include + +#include #include #include #include @@ -154,7 +156,8 @@ class pdlp_solver_settings_t { */ void set_initial_primal_solution(const f_t* initial_primal_solution, i_t size, - rmm::cuda_stream_view stream = rmm::cuda_stream_default); + rmm::cuda_stream_view stream = cuda::stream_ref{ + cudaStream_t{cudaStreamDefault}}); /** * @brief Set an initial dual solution. @@ -168,7 +171,8 @@ class pdlp_solver_settings_t { */ void set_initial_dual_solution(const f_t* initial_dual_solution, i_t size, - rmm::cuda_stream_view stream = rmm::cuda_stream_default); + rmm::cuda_stream_view stream = cuda::stream_ref{ + cudaStream_t{cudaStreamDefault}}); /** TODO batch mode: tmp * @brief Set an initial step size. @@ -203,11 +207,12 @@ class pdlp_solver_settings_t { * @param constraint_mapping Constraints indices to scatter to in case the new * problem has less constraints */ - void set_pdlp_warm_start_data(pdlp_warm_start_data_t& pdlp_warm_start_data_view, - const rmm::device_uvector& var_mapping = - rmm::device_uvector{0, rmm::cuda_stream_default}, - const rmm::device_uvector& constraint_mapping = - rmm::device_uvector{0, rmm::cuda_stream_default}); + void set_pdlp_warm_start_data( + pdlp_warm_start_data_t& pdlp_warm_start_data_view, + const rmm::device_uvector& var_mapping = + rmm::device_uvector{0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}}}, + const rmm::device_uvector& constraint_mapping = rmm::device_uvector{ + 0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}}}); // Same but for the Cython interface void set_pdlp_warm_start_data(const f_t* current_primal_solution, diff --git a/cpp/include/cuopt/mathematical_optimization/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/solver_settings.hpp index 6b47805702..cfe265edcf 100644 --- a/cpp/include/cuopt/mathematical_optimization/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/solver_settings.hpp @@ -10,6 +10,8 @@ #include #include +#include + #include #include @@ -52,10 +54,12 @@ class solver_settings_t { void set_initial_pdlp_primal_solution(const f_t* initial_primal_solution, i_t size, - rmm::cuda_stream_view stream = rmm::cuda_stream_default); + rmm::cuda_stream_view stream = cuda::stream_ref{ + cudaStream_t{cudaStreamDefault}}); void set_initial_pdlp_dual_solution(const f_t* initial_dual_solution, i_t size, - rmm::cuda_stream_view stream = rmm::cuda_stream_default); + rmm::cuda_stream_view stream = cuda::stream_ref{ + cudaStream_t{cudaStreamDefault}}); void set_pdlp_warm_start_data(const f_t* current_primal_solution, const f_t* current_dual_solution, const f_t* initial_primal_average, @@ -82,7 +86,8 @@ class solver_settings_t { // MIP Settings void add_initial_mip_solution(const f_t* initial_solution, i_t size, - rmm::cuda_stream_view stream = rmm::cuda_stream_default); + rmm::cuda_stream_view stream = cuda::stream_ref{ + cudaStream_t{cudaStreamDefault}}); void set_mip_callback(internals::base_solution_callback_t* callback = nullptr, void* user_data = nullptr); diff --git a/cpp/include/cuopt/mathematical_optimization/utilities/segmented_sum_handler.cuh b/cpp/include/cuopt/mathematical_optimization/utilities/segmented_sum_handler.cuh index dd4284dab3..aad0329a7f 100644 --- a/cpp/include/cuopt/mathematical_optimization/utilities/segmented_sum_handler.cuh +++ b/cpp/include/cuopt/mathematical_optimization/utilities/segmented_sum_handler.cuh @@ -22,7 +22,7 @@ struct segmented_sum_handler_t { i_t problem_size) { cub::DeviceSegmentedReduce::Sum( - nullptr, byte_needed_, input, output, batch_size, problem_size, stream_view_); + nullptr, byte_needed_, input, output, batch_size, problem_size, stream_view_.get()); segmented_sum_storage_.resize(byte_needed_, stream_view_); @@ -32,7 +32,7 @@ struct segmented_sum_handler_t { output, batch_size, problem_size, - stream_view_); + stream_view_.get()); } template @@ -51,9 +51,9 @@ struct segmented_sum_handler_t { problem_size, reduction_op, initial_value, - stream_view_.value()); + stream_view_.get()); - segmented_sum_storage_.resize(byte_needed_, stream_view_.value()); + segmented_sum_storage_.resize(byte_needed_, stream_view_.get()); cub::DeviceSegmentedReduce::Reduce(segmented_sum_storage_.data(), byte_needed_, @@ -63,7 +63,7 @@ struct segmented_sum_handler_t { problem_size, reduction_op, initial_value, - stream_view_.value()); + stream_view_.get()); } size_t byte_needed_; diff --git a/cpp/src/barrier/barrier.cu b/cpp/src/barrier/barrier.cu index 2d38688f86..bd55ecfa33 100644 --- a/cpp/src/barrier/barrier.cu +++ b/cpp/src/barrier/barrier.cu @@ -139,7 +139,7 @@ template f_t* a, f_t* b, f_t* out, int size, rmm::cuda_stream_view stream) { cub::DeviceTransform::Transform( - cuda::std::make_tuple(a, b), out, size, cuda::std::multiplies<>{}, stream.value()); + cuda::std::make_tuple(a, b), out, size, cuda::std::multiplies<>{}, stream.get()); } // out[i] = is_direct_free_linear[i] ? 0 : a[i] * b[i] @@ -152,7 +152,7 @@ template out, size, [] __host__ __device__(f_t x_j, f_t d_j, int free_j) { return free_j ? f_t{0} : x_j * d_j; }, - stream.value()); + stream.get()); } template @@ -164,7 +164,7 @@ template out, size, [alpha, beta] __host__ __device__(f_t a, f_t b) { return alpha * a + beta * b; }, - stream.value()); + stream.get()); } // Step size computation for nonnegative and free variables. Fuses two independent @@ -224,8 +224,8 @@ static void recover_linear_orthant_dz(raft::device_span target, if (is_direct_free) return f_t(0); return target_val - (z_val * dx_val) / x_val; }, - stream.value()); - RAFT_CHECK_CUDA(stream); + stream.get()); + RAFT_CHECK_CUDA(stream.get()); } template @@ -235,7 +235,7 @@ static void negate_complementarity_rhs(raft::device_span out, { if (out.empty()) return; cub::DeviceTransform::Transform( - residual.data(), out.data(), out.size(), [] HD(f_t rhs) { return -rhs; }, stream.value()); + residual.data(), out.data(), out.size(), [] HD(f_t rhs) { return -rhs; }, stream.get()); } template @@ -254,8 +254,8 @@ static void fill_linear_cc_rhs(raft::device_span out, [new_mu] HD(f_t dx_aff_val, f_t dz_aff_val, i_t is_direct_free_linear) { return is_direct_free_linear ? f_t(0) : (-(dx_aff_val * dz_aff_val) + new_mu); }, - stream.value()); - RAFT_CHECK_CUDA(stream); + stream.get()); + RAFT_CHECK_CUDA(stream.get()); } // Batches the independent GPU reductions/dot-products needed by @@ -343,7 +343,7 @@ class barrier_reduce_helper_t { void sync(rmm::cuda_stream_view stream_view) { raft::copy(h_results_.data(), d_results_.data(), static_cast(kCount), stream_view); - stream_view.synchronize(); + stream_view.sync(); } // Raw reduced values; the caller combines these into residual norms, mu, and objectives. @@ -383,14 +383,15 @@ class barrier_reduce_helper_t { { f_t* out = d_results_.data() + slot; if (size == 0) { - RAFT_CUDA_TRY(cudaMemsetAsync(out, 0, sizeof(f_t), stream_view.value())); + RAFT_CUDA_TRY(cudaMemsetAsync(out, 0, sizeof(f_t), stream_view.get())); return; } size_t temp_storage_bytes = 0; - cub::DeviceReduce::Reduce(nullptr, temp_storage_bytes, in, out, size, op, init, stream_view); + cub::DeviceReduce::Reduce( + nullptr, temp_storage_bytes, in, out, size, op, init, stream_view.get()); d_temp_storage_.resize(temp_storage_bytes, stream_view); cub::DeviceReduce::Reduce( - d_temp_storage_.data(), temp_storage_bytes, in, out, size, op, init, stream_view); + d_temp_storage_.data(), temp_storage_bytes, in, out, size, op, init, stream_view.get()); } void norm_inf_async(Slot slot, const f_t* in, i_t size, rmm::cuda_stream_view stream_view) @@ -407,9 +408,10 @@ class barrier_reduce_helper_t { { f_t* out = d_results_.data() + slot; size_t temp_storage_bytes = 0; - cub::DeviceReduce::Sum(nullptr, temp_storage_bytes, in, out, size, stream_view); + cub::DeviceReduce::Sum(nullptr, temp_storage_bytes, in, out, size, stream_view.get()); d_temp_storage_.resize(temp_storage_bytes, stream_view); - cub::DeviceReduce::Sum(d_temp_storage_.data(), temp_storage_bytes, in, out, size, stream_view); + cub::DeviceReduce::Sum( + d_temp_storage_.data(), temp_storage_bytes, in, out, size, stream_view.get()); } void dot_async(Slot slot, @@ -418,8 +420,14 @@ class barrier_reduce_helper_t { cublasHandle_t cublas_handle, rmm::cuda_stream_view stream_view) { - RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot( - cublas_handle, a.size(), a.data(), 1, b.data(), 1, d_results_.data() + slot, stream_view)); + RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(cublas_handle, + a.size(), + a.data(), + 1, + b.data(), + 1, + d_results_.data() + slot, + stream_view.get())); } rmm::device_uvector d_results_; @@ -659,7 +667,7 @@ class iteration_data_t { d_inv_diag_prime.data(), d_num_flag.data(), inv_diag.size(), - stream_view_)); + stream_view_.get())); d_flag_buffer.resize(flag_buffer_size, stream_view_); } @@ -847,7 +855,7 @@ class iteration_data_t { raft::copy( device_A_x_values.data(), device_AD.x.data(), device_AD.x.size(), handle_ptr->get_stream()); device_AD.to_compressed_row(device_A, handle_ptr->get_stream()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { return; } @@ -1031,7 +1039,7 @@ class iteration_data_t { const f_t d_j = span_diag[j]; span_x[span_diag_indices[j]] = -q_diag - d_j - dual_perturb_value; }); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); thrust::for_each_n(rmm::exec_policy(handle_ptr->get_stream()), thrust::make_counting_iterator(n), @@ -1041,7 +1049,7 @@ class iteration_data_t { primal_perturb_value = primal_perturb] __device__(i_t j) { span_x[span_diag_indices[j]] = primal_perturb_value; }); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); if (has_soc) { if (cones().has_sparse_cones()) { @@ -1057,7 +1065,7 @@ class iteration_data_t { cone_kkt_data_.sparse_expansion_D, handle_ptr->get_stream(), dual_perturb); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } if (cones().n_dense_cones() > 0) { scatter_dense_hessian_into_augmented(cones(), @@ -1068,7 +1076,7 @@ class iteration_data_t { cone_kkt_data_.dense_cone_ids, handle_ptr->get_stream(), dual_perturb); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } } handle_ptr->sync_stream(); @@ -1103,8 +1111,8 @@ class iteration_data_t { d_inv_diag_prime.data(), d_num_flag.data(), d_inv_diag.size(), - stream_view_); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); } else { d_inv_diag_prime.resize(inv_diag.size(), stream_view_); raft::copy(d_inv_diag_prime.data(), d_inv_diag.data(), inv_diag.size(), stream_view_); @@ -1124,7 +1132,7 @@ class iteration_data_t { span_col_ind = cuopt::make_span(device_AD.col_index)] __device__(i_t i) { span_x[i] *= span_scale[span_col_ind[i]]; }); - RAFT_CHECK_CUDA(stream_view_); + RAFT_CHECK_CUDA(stream_view_.get()); } if (settings_.concurrent_halt != nullptr && *settings_.concurrent_halt == 1) { return; } if (first_call) { @@ -1539,7 +1547,7 @@ class iteration_data_t { return chol->solve(d_b, d_x); } else { raft::copy(inv_diag.data(), d_inv_diag.data(), d_inv_diag.size(), stream_view_); - stream_view_.synchronize(); + stream_view_.sync(); dense_vector_t b = host_copy(d_b, stream_view_); dense_vector_t x = host_copy(d_x, stream_view_); @@ -1549,7 +1557,7 @@ class iteration_data_t { raft::copy(d_b.data(), b.data(), b.size(), stream_view_); d_x.resize(x.size(), stream_view_); raft::copy(d_x.data(), x.data(), x.size(), stream_view_); - stream_view_.synchronize(); // host x can go out of scope before copy finishes + stream_view_.sync(); // host x can go out of scope before copy finishes return out; } @@ -1917,8 +1925,8 @@ class iteration_data_t { u.data(), u.size(), cuda::std::multiplies<>{}, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); // y = alpha * A * w + beta * v = alpha * A * Dinv * A^T * y + beta * v cusparse_view.spmv(alpha, cusparse_u, beta, cusparse_v); @@ -1939,8 +1947,8 @@ class iteration_data_t { u.data(), u.size(), cuda::std::multiplies<>{}, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); cusparse_view_.spmv(alpha, u, beta, v); } @@ -2023,7 +2031,7 @@ class iteration_data_t { d_r1_.data(), linear_n, stream_view_); - RAFT_CHECK_CUDA(stream_view_); + RAFT_CHECK_CUDA(stream_view_.get()); } // r1 <- D * x_1 + H x_1 on cone rows @@ -2044,7 +2052,7 @@ class iteration_data_t { n, m, handle_ptr->get_stream()); - RAFT_CHECK_CUDA(stream_view_); + RAFT_CHECK_CUDA(stream_view_.get()); } if (cones().n_dense_cones() > 0) { launch_dense_hessian_matvec( @@ -2052,7 +2060,7 @@ class iteration_data_t { cones(), raft::device_span(d_r1_.data() + cone_start(), m_c), stream_view_); - RAFT_CHECK_CUDA(stream_view_); + RAFT_CHECK_CUDA(stream_view_.get()); } } @@ -2542,7 +2550,7 @@ int barrier_solver_t::initial_point(iteration_data_t& data) // x = Dinv*(F*u - A'*q) // Fu <- -1.0 * A' * q + 1.0 * Fu data.cusparse_view_.transpose_spmv(-1.0, q, 1.0, Fu); - data.handle_ptr->get_stream().synchronize(); + data.handle_ptr->get_stream().sync(); // x <- Dinv * (F*u - A'*q) data.inv_diag.pairwise_product(Fu, data.x); @@ -2560,7 +2568,7 @@ int barrier_solver_t::initial_point(iteration_data_t& data) dense_vector_t init_primal_residual(lp.num_rows); init_primal_residual = lp.rhs; data.cusparse_view_.spmv(1.0, data.x, -1.0, init_primal_residual); - data.handle_ptr->get_stream().synchronize(); + data.handle_ptr->get_stream().sync(); #ifdef PRINT_INFO settings.log.printf("||b - A * x||: %.16e\n", vector_norm2(init_primal_residual)); #endif @@ -2748,8 +2756,8 @@ void barrier_solver_t::gpu_compute_residuals(const rmm::device_uvector data.d_bound_residual_.data(), data.d_upper_bounds_.size(), [] HD(f_t upper_j, f_t w_k, f_t x_j) { return upper_j - w_k - x_j; }, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); } // Compute dual_residual = c - A'*y - z + E*v + Q*x @@ -2757,8 +2765,8 @@ void barrier_solver_t::gpu_compute_residuals(const rmm::device_uvector data.d_dual_residual_.data(), data.d_dual_residual_.size(), cuda::std::minus<>{}, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); auto descr_dual_residual = data.cusparse_view_.create_vector(data.d_dual_residual_); if (data.Q.n > 0) { data.cusparse_Q_view_.spmv(1.0, cusparse_d_x.get(), 1.0, descr_dual_residual.get()); @@ -2775,8 +2783,8 @@ void barrier_solver_t::gpu_compute_residuals(const rmm::device_uvector thrust::make_permutation_iterator(data.d_dual_residual_.data(), data.d_upper_bounds_.data()), data.d_upper_bounds_.size(), [] HD(f_t dual_residual_j, f_t v_k) { return dual_residual_j + v_k; }, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); } // Compute complementarity_xz_residual = x.*z @@ -2784,15 +2792,15 @@ void barrier_solver_t::gpu_compute_residuals(const rmm::device_uvector data.d_complementarity_xz_residual_.data(), data.d_complementarity_xz_residual_.size(), cuda::std::multiplies<>{}, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); // Compute complementarity_wv_residual = w.*v cub::DeviceTransform::Transform(cuda::std::make_tuple(d_w.data(), d_v.data()), data.d_complementarity_wv_residual_.data(), data.d_complementarity_wv_residual_.size(), cuda::std::multiplies<>{}, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); } template @@ -2893,15 +2901,15 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t{}, - stream_view_.value()); + stream_view_.get()); } - RAFT_CHECK_CUDA(stream_view_); + RAFT_CHECK_CUDA(stream_view_.get()); // Upper-bound slacks: D_j += v_k/w_k. if (data.n_upper_bounds > 0) { @@ -2913,8 +2921,8 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t f_t(0)) return diag_j; return diag_j + free_var_reg; }, - stream_view_.value()); + stream_view_.get()); } else { cub::DeviceTransform::Transform( cuda::std::make_tuple(data.d_diag_.data(), data.d_is_direct_free_linear_.data()), @@ -2947,9 +2955,9 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t 0) { cub::DeviceTransform::Transform( cuda::std::make_tuple(data.d_bound_rhs_.data(), @@ -3036,8 +3044,8 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t(data.d_xz_residual_.data(), linear_size), @@ -3447,7 +3455,7 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t(data.d_dz_.data(), linear_size), raft::device_span(data.d_dx_.data(), linear_size), raft::device_span(data.d_x_.data(), linear_size)); - RAFT_CHECK_CUDA(stream_view_); + RAFT_CHECK_CUDA(stream_view_.get()); const f_t xz_residual_norm = device_vector_norm_inf(data.d_xz_residual_, stream_view_); max_residual = std::max(max_residual, xz_residual_norm); @@ -3470,8 +3478,8 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t(d_dv_residual, stream_view_); max_residual = std::max(max_residual, dv_residual_norm); if (dv_residual_norm > 1e-2) { @@ -3531,8 +3539,8 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t(data.d_dual_residual_, stream_view_); max_residual = std::max(max_residual, dual_residual_norm); @@ -3552,8 +3560,8 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t::gpu_compute_search_direction(iteration_data_t(data.d_dw_residual_, stream_view_); max_residual = std::max(max_residual, dw_residual_norm); @@ -3592,8 +3600,8 @@ i_t barrier_solver_t::gpu_compute_search_direction(iteration_data_t(data.d_wv_residual_, stream_view_); max_residual = std::max(max_residual, wv_residual_norm); @@ -3621,8 +3629,8 @@ void fill_linear_complementarity_target(iteration_data_t& data, if (is_direct_free_linear) return f_t(0); return complementarity_xz_rhs / x_val; }, - stream.value()); - RAFT_CHECK_CUDA(stream); + stream.get()); + RAFT_CHECK_CUDA(stream.get()); } template @@ -3638,9 +3646,9 @@ void fill_affine_cone_complementarity_target(iteration_data_t& data, auto cone_target = raft::device_span(data.d_complementarity_target_.data() + cone_var_start, m_c); cub::DeviceTransform::Transform( - cones.z.data(), cone_target.data(), m_c, [] HD(f_t z_val) { return -z_val; }, stream.value()); + cones.z.data(), cone_target.data(), m_c, [] HD(f_t z_val) { return -z_val; }, stream.get()); RAFT_CUDA_TRY(cudaPeekAtLastError()); - RAFT_CHECK_CUDA(stream); + RAFT_CHECK_CUDA(stream.get()); } template @@ -3704,7 +3712,7 @@ void barrier_solver_t::compute_affine_rhs(iteration_data_t& raft::device_span(data.d_complementarity_wv_residual_.data(), data.d_complementarity_wv_residual_.size()), stream_view_); - RAFT_CHECK_CUDA(stream_view_); + RAFT_CHECK_CUDA(stream_view_.get()); fill_linear_complementarity_target( data, @@ -3858,17 +3866,18 @@ void barrier_solver_t::compute_cc_rhs(iteration_data_t& data data.d_complementarity_wv_rhs_.data(), data.d_complementarity_wv_rhs_.size(), [new_mu] HD(f_t dw_aff, f_t dv_aff) { return -(dw_aff * dv_aff) + new_mu; }, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); // Zero the corrector RHS on device - RAFT_CUDA_TRY(cudaMemsetAsync(data.d_h_.data(), 0, sizeof(f_t) * data.d_h_.size(), stream_view_)); + RAFT_CUDA_TRY( + cudaMemsetAsync(data.d_h_.data(), 0, sizeof(f_t) * data.d_h_.size(), stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync( - data.d_dual_rhs_.data(), 0, sizeof(f_t) * data.d_dual_rhs_.size(), stream_view_)); + data.d_dual_rhs_.data(), 0, sizeof(f_t) * data.d_dual_rhs_.size(), stream_view_.get())); if (data.n_upper_bounds > 0) { RAFT_CUDA_TRY(cudaMemsetAsync( - data.d_bound_rhs_.data(), 0, sizeof(f_t) * data.d_bound_rhs_.size(), stream_view_)); + data.d_bound_rhs_.data(), 0, sizeof(f_t) * data.d_bound_rhs_.size(), stream_view_.get())); RAFT_CUDA_TRY( - cudaMemsetAsync(data.d_dw_.data(), 0, sizeof(f_t) * data.d_dw_.size(), stream_view_)); + cudaMemsetAsync(data.d_dw_.data(), 0, sizeof(f_t) * data.d_dw_.size(), stream_view_.get())); } data.cone_combined_step_ = has_soc; data.cone_sigma_mu_ = has_soc ? new_mu : f_t(0); @@ -3901,8 +3910,8 @@ void barrier_solver_t::compute_final_direction(iteration_data_t thrust::tuple { return {dw + dw_aff, dv + dv_aff}; }, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); cub::DeviceTransform::Transform( cuda::std::make_tuple( data.d_dx_aff_.data(), data.d_dz_aff_.data(), data.d_dx_.data(), data.d_dz_.data()), @@ -3911,15 +3920,15 @@ void barrier_solver_t::compute_final_direction(iteration_data_t thrust::tuple { return {dx + dx_aff, dz + dz_aff}; }, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); cub::DeviceTransform::Transform( cuda::std::make_tuple(data.d_dy_aff_.data(), data.d_dy_.data()), data.d_dy_.data(), data.d_dy_.size(), [] HD(f_t dy_aff, f_t dy) { return dy + dy_aff; }, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); } template @@ -3975,8 +3984,8 @@ void barrier_solver_t::compute_next_iterate(iteration_data_t [step_primal, step_dual] HD(f_t w, f_t v, f_t dw, f_t dv) -> thrust::tuple { return {w + step_primal * dw, v + step_dual * dv}; }, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); cub::DeviceTransform::Transform( cuda::std::make_tuple(data.d_x_.data(), data.d_z_.data(), data.d_dx_.data(), data.d_dz_.data()), thrust::make_zip_iterator(data.d_x_.data(), data.d_z_.data()), @@ -3984,15 +3993,15 @@ void barrier_solver_t::compute_next_iterate(iteration_data_t [step_primal, step_dual] HD(f_t x, f_t z, f_t dx, f_t dz) -> thrust::tuple { return {x + step_primal * dx, z + step_dual * dz}; }, - stream_view_.value()); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); cub::DeviceTransform::Transform( cuda::std::make_tuple(data.d_y_.data(), data.d_dy_.data()), data.d_y_.data(), data.d_y_.size(), [step_dual] HD(f_t y, f_t dy) { return y + step_dual * dy; }, - stream_view_); - RAFT_CHECK_CUDA(stream_view_); + stream_view_.get()); + RAFT_CHECK_CUDA(stream_view_.get()); // Do not handle free variables for quadratic problems i_t num_free_variables = presolve_info.free_variable_pairs.size() / 2; if (num_free_variables > 0 && data.Q.n == 0) { @@ -4098,7 +4107,7 @@ void barrier_solver_t::compute_residual_norms_mu_and_objective( data.d_z_.data(), 1, d_xz.data(), - stream_view_)); + stream_view_.get())); RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(lp.handle_ptr->get_cublas_handle(), data.d_w_.size(), data.d_w_.data(), @@ -4106,7 +4115,7 @@ void barrier_solver_t::compute_residual_norms_mu_and_objective( data.d_v_.data(), 1, d_wv.data(), - stream_view_)); + stream_view_.get())); RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(lp.handle_ptr->get_cublas_handle(), data.d_x_.size(), data.d_x_.data(), @@ -4114,7 +4123,7 @@ void barrier_solver_t::compute_residual_norms_mu_and_objective( data.d_dual_residual_.data(), 1, d_rdx.data(), - stream_view_)); + stream_view_.get())); RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(lp.handle_ptr->get_cublas_handle(), data.d_y_.size(), data.d_y_.data(), @@ -4122,7 +4131,7 @@ void barrier_solver_t::compute_residual_norms_mu_and_objective( data.d_primal_residual_.data(), 1, d_rpy.data(), - stream_view_)); + stream_view_.get())); RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(lp.handle_ptr->get_cublas_handle(), data.d_bound_residual_.size(), data.d_bound_residual_.data(), @@ -4130,7 +4139,7 @@ void barrier_solver_t::compute_residual_norms_mu_and_objective( data.d_v_.data(), 1, d_rwv.data(), - stream_view_)); + stream_view_.get())); RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(lp.handle_ptr->get_cublas_handle(), data.d_primal_residual_.size(), data.d_primal_residual_.data(), @@ -4138,7 +4147,7 @@ void barrier_solver_t::compute_residual_norms_mu_and_objective( data.d_primal_residual_.data(), 1, d_p.data(), - stream_view_)); + stream_view_.get())); RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(lp.handle_ptr->get_cublas_handle(), data.d_y_.size(), data.d_y_.data(), @@ -4146,7 +4155,7 @@ void barrier_solver_t::compute_residual_norms_mu_and_objective( data.d_y_.data(), 1, d_y.data(), - stream_view_)); + stream_view_.get())); f_t xz = d_xz.value(stream_view_); f_t wv = d_wv.value(stream_view_); f_t rdx = d_rdx.value(stream_view_); @@ -4155,7 +4164,7 @@ void barrier_solver_t::compute_residual_norms_mu_and_objective( f_t p = d_p.value(stream_view_); f_t y = d_y.value(stream_view_); - stream_view_.synchronize(); + stream_view_.sync(); f_t objective_gap_1 = primal_objective - dual_objective; f_t objective_gap_2 = xz + wv + rdx - rpy + rwv; @@ -4198,7 +4207,7 @@ lp_status_t barrier_solver_t::check_for_suboptimal_solution( raft::copy(data.y.data(), data.d_y_.data(), data.d_y_.size(), stream_view_); raft::copy(data.z.data(), data.d_z_.data(), data.d_z_.size(), stream_view_); raft::copy(data.v.data(), data.d_v_.data(), data.d_v_.size(), stream_view_); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); data.to_solution(lp, iter, primal_objective, @@ -4589,7 +4598,7 @@ lp_status_t barrier_solver_t::solve(f_t start_time, lp_solution_t::solve(f_t start_time, lp_solution_t::solve(f_t start_time, lp_solution_t& cones, i_t(-1)); if (n_sparse > 0) { const size_t grid = raft::ceildiv(n_sparse, augmented_csr_block_size); - scatter_sparse_ids_by_cone_kernel<<>>( + scatter_sparse_ids_by_cone_kernel<<>>( cuopt::make_span(metadata.sparse_ids_by_cone), cuopt::make_span(cones.sparse_cone_ids), n_sparse); @@ -548,14 +548,14 @@ void build_augmented_csr_metadata(const cone_data_t& cones, rmm::exec_policy(stream), is_dense_cone.begin(), is_dense_cone.end(), dense_prefix.begin()); const size_t grid = raft::ceildiv(n_cones, augmented_csr_block_size); - build_dense_ids_by_cone_kernel<<>>( + build_dense_ids_by_cone_kernel<<>>( cuopt::make_span(metadata.dense_ids_by_cone), cuopt::make_span(cones.cone_is_sparse), cuopt::make_span(dense_prefix), n_cones); RAFT_CUDA_TRY(cudaPeekAtLastError()); - compact_dense_cone_ids_kernel<<>>( + compact_dense_cone_ids_kernel<<>>( cuopt::make_span(metadata.dense_cone_ids), cuopt::make_span(dense_prefix), cuopt::make_span(cones.cone_is_sparse), @@ -564,12 +564,11 @@ void build_augmented_csr_metadata(const cone_data_t& cones, rmm::device_uvector dense_block_sizes(n_dense, stream); const size_t dense_grid = raft::ceildiv(n_dense, augmented_csr_block_size); - build_dense_block_sizes_kernel - <<>>( - cuopt::make_span(dense_block_sizes), - cuopt::make_span(metadata.dense_cone_ids), - cuopt::make_span(cones.cone_offsets), - n_dense); + build_dense_block_sizes_kernel<<>>( + cuopt::make_span(dense_block_sizes), + cuopt::make_span(metadata.dense_cone_ids), + cuopt::make_span(cones.cone_offsets), + n_dense); RAFT_CUDA_TRY(cudaPeekAtLastError()); thrust::exclusive_scan(rmm::exec_policy(stream), @@ -612,7 +611,7 @@ void build_augmented_csr_metadata(const cone_data_t& cones, const size_t entry_grid = raft::ceildiv(m_c, augmented_csr_block_size); build_dense_cone_entry_rank_kernel - <<>>( + <<>>( cuopt::make_span(metadata.dense_cone_entry_rank), cuopt::make_span(cones.element_cone_ids), cuopt::make_span(cones.cone_is_sparse), @@ -650,7 +649,7 @@ i_t build_augmented_csr_on_device(i_t n, { raft::common::nvtx::range scope("Barrier: augmented: device CSR count"); const size_t grid = raft::ceildiv(factorization_size, augmented_csr_block_size); - count_augmented_row_nnz_kernel<<>>( + count_augmented_row_nnz_kernel<<>>( factorization_size, n, m, @@ -715,7 +714,7 @@ i_t build_augmented_csr_on_device(i_t n, raft::common::nvtx::range scope("Barrier: augmented: device CSR fill"); auto views = make_cone_kkt_views(cone_data, augmented_diagonal_indices); const size_t grid = raft::ceildiv(factorization_size, augmented_csr_block_size); - fill_augmented_csr_row_kernel<<>>( + fill_augmented_csr_row_kernel<<>>( factorization_size, n, m, diff --git a/cpp/src/barrier/cusparse_view.cu b/cpp/src/barrier/cusparse_view.cu index 03d1b3a13d..c7a9cac067 100644 --- a/cpp/src/barrier/cusparse_view.cu +++ b/cpp/src/barrier/cusparse_view.cu @@ -145,7 +145,7 @@ void cusparse_view_t::init_spmv_buffer_and_preprocess(cusparseSpMatDes y, spmv_alg, &buffer_size_spmv, - handle_ptr_->get_stream())); + handle_ptr_->get_stream().get())); buffer.resize(buffer_size_spmv, handle_ptr_->get_stream()); my_cusparsespmv_preprocess(handle_ptr_->get_cusparse_handle(), @@ -157,7 +157,7 @@ void cusparse_view_t::init_spmv_buffer_and_preprocess(cusparseSpMatDes y, spmv_alg, buffer.data(), - handle_ptr_->get_stream()); + handle_ptr_->get_stream().get()); } template @@ -177,9 +177,10 @@ cusparse_view_t::cusparse_view_t(raft::handle_t const* handle_ptr, d_zero_(zero_v, handle_ptr->get_stream()) { RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); // TMP matrix data should already be on the GPU constexpr bool debug = false; if (debug) { printf("A hash: %zu\n", A.hash()); } @@ -272,7 +273,7 @@ void cusparse_view_t::spmv(f_t alpha, y, get_spmv_alg(rows_), (f_t*)spmv_buffer_.data(), - handle_ptr_->get_stream()); + handle_ptr_->get_stream().get()); } template @@ -327,7 +328,7 @@ void cusparse_view_t::transpose_spmv(f_t alpha, y, get_spmv_alg(A_T_offsets_.size() - 1), (f_t*)spmv_buffer_transpose_.data(), - handle_ptr_->get_stream()); + handle_ptr_->get_stream().get()); } template class cusparse_view_t; diff --git a/cpp/src/barrier/device_sparse_matrix.cuh b/cpp/src/barrier/device_sparse_matrix.cuh index 974e2b0f4a..4012da7a6f 100644 --- a/cpp/src/barrier/device_sparse_matrix.cuh +++ b/cpp/src/barrier/device_sparse_matrix.cuh @@ -43,9 +43,10 @@ struct sum_reduce_helper_t { f_t sum(InputIteratorT input, i_t size, rmm::cuda_stream_view stream_view) { buffer_size = 0; - cub::DeviceReduce::Sum(nullptr, buffer_size, input, out.data(), size, stream_view); + cub::DeviceReduce::Sum(nullptr, buffer_size, input, out.data(), size, stream_view.get()); buffer_data.resize(buffer_size, stream_view); - cub::DeviceReduce::Sum(buffer_data.data(), buffer_size, input, out.data(), size, stream_view); + cub::DeviceReduce::Sum( + buffer_data.data(), buffer_size, input, out.data(), size, stream_view.get()); return out.value(stream_view); } }; @@ -69,8 +70,15 @@ struct transform_reduce_helper_t { i_t size, rmm::cuda_stream_view stream_view) { - cub::DeviceReduce::TransformReduce( - nullptr, buffer_size, input, out.data(), size, reduce_op, transform_op, init, stream_view); + cub::DeviceReduce::TransformReduce(nullptr, + buffer_size, + input, + out.data(), + size, + reduce_op, + transform_op, + init, + stream_view.get()); buffer_data.resize(buffer_size, stream_view); @@ -82,7 +90,7 @@ struct transform_reduce_helper_t { reduce_op, transform_op, init, - stream_view); + stream_view.get()); return out.value(stream_view); } @@ -123,8 +131,15 @@ struct transform_reduce_pair_helper_t { rmm::cuda_stream_view stream_view) { f2_min_t reduce_op{}; - cub::DeviceReduce::TransformReduce( - nullptr, buffer_size, input, out.data(), size, reduce_op, transform_op, init, stream_view); + cub::DeviceReduce::TransformReduce(nullptr, + buffer_size, + input, + out.data(), + size, + reduce_op, + transform_op, + init, + stream_view.get()); buffer_data.resize(buffer_size, stream_view); @@ -136,7 +151,7 @@ struct transform_reduce_pair_helper_t { reduce_op, transform_op, init, - stream_view); + stream_view.get()); return out.value(stream_view); } @@ -240,7 +255,8 @@ class device_csc_matrix_t { void form_col_index(rmm::cuda_stream_view stream) { col_index.resize(x.size(), stream); - RAFT_CUDA_TRY(cudaMemsetAsync(col_index.data(), 0, sizeof(i_t) * col_index.size(), stream)); + RAFT_CUDA_TRY( + cudaMemsetAsync(col_index.data(), 0, sizeof(i_t) * col_index.size(), stream.get())); // Scatter 1 when there is a col start in col_index if (col_start.size() > 2) { @@ -259,17 +275,21 @@ class device_csc_matrix_t { // Inclusive cumulative sum to have the corresponding column for each entry rmm::device_buffer d_temp_storage; size_t temp_storage_bytes{0}; - cub::DeviceScan::InclusiveSum( - nullptr, temp_storage_bytes, col_index.data(), col_index.data(), col_index.size(), stream); + cub::DeviceScan::InclusiveSum(nullptr, + temp_storage_bytes, + col_index.data(), + col_index.data(), + col_index.size(), + stream.get()); d_temp_storage.resize(temp_storage_bytes, stream); cub::DeviceScan::InclusiveSum(d_temp_storage.data(), temp_storage_bytes, col_index.data(), col_index.data(), col_index.size(), - stream); + stream.get()); // Have to sync since InclusiveSum is being run on local data (d_temp_storage) - stream.synchronize(); + stream.sync(); } csc_view_t view() @@ -394,13 +414,13 @@ void device_csc_matrix_t::to_compressed_row(device_csr_matrix_t row_counts(m, stream); - RAFT_CUDA_TRY(cudaMemsetAsync(row_counts.data(), 0, sizeof(i_t) * m, stream)); + RAFT_CUDA_TRY(cudaMemsetAsync(row_counts.data(), 0, sizeof(i_t) * m, stream.get())); thrust::for_each(exec, thrust::make_counting_iterator(0), @@ -413,13 +433,13 @@ void device_csc_matrix_t::to_compressed_row(device_csr_matrix_tget_stream()); + RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream().get()); op.solve(r, delta_x); thrust::transform(op.data_.handle_ptr->get_thrust_policy(), @@ -89,7 +89,7 @@ f_t iterative_refinement_simple(T& op, delta_x.data(), x.data(), thrust::plus()); - RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream().get()); // r = b - Ax raft::copy(r.data(), b.data(), b.size(), x.stream()); op.a_multiply(-1.0, x, 1.0, r); @@ -183,7 +183,7 @@ f_t iterative_refinement_gmres(T& op, V[0].data() + V[0].size(), V[0].data(), scale_op{inv_rnorm}); - RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream().get()); e1.assign(m + 1, 0.0); e1[0] = rnorm; @@ -211,7 +211,7 @@ f_t iterative_refinement_gmres(T& op, V[k + 1].data() + x.size(), V[j].data(), f_t(0)); - RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream().get()); H[j][k] = hij; // w -= H[j][k] * V[j] thrust::transform(op.data_.handle_ptr->get_thrust_policy(), @@ -220,7 +220,7 @@ f_t iterative_refinement_gmres(T& op, V[j].data(), V[k + 1].data(), subtract_scaled_op{hij}); - RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream().get()); } // H[k+1][k] = ||w|| @@ -247,7 +247,7 @@ f_t iterative_refinement_gmres(T& op, V[k + 1].data() + x.size(), V[k + 1].data(), scale_op{inv_h}); - RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream().get()); // Apply Given's rotations to new column for (int i = 0; i < k; ++i) { @@ -298,7 +298,7 @@ f_t iterative_refinement_gmres(T& op, delta_x.data(), delta_x.data() + delta_x.size(), 0.0); - RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream().get()); for (int j = 0; j < k; ++j) { thrust::transform(op.data_.handle_ptr->get_thrust_policy(), delta_x.data(), @@ -306,7 +306,7 @@ f_t iterative_refinement_gmres(T& op, Z[j].data(), delta_x.data(), axpy_op{y[j]}); - RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream().get()); } // Update x = x + delta_x @@ -316,7 +316,7 @@ f_t iterative_refinement_gmres(T& op, delta_x.data(), x.data(), thrust::plus()); - RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(op.data_.handle_ptr->get_stream().get()); // r = b - A*x raft::copy(r.data(), b.data(), b.size(), x.stream()); op.a_multiply(-1.0, x, 1.0, r); @@ -375,7 +375,7 @@ f_t iterative_refinement(T& op, raft::copy(x.data(), d_x.data(), x.size(), op.data_.handle_ptr->get_stream()); - RAFT_CUDA_TRY(cudaStreamSynchronize(op.data_.handle_ptr->get_stream())); + op.data_.handle_ptr->get_stream().sync(); return err; } diff --git a/cpp/src/barrier/second_order_cone_kernels.cuh b/cpp/src/barrier/second_order_cone_kernels.cuh index b605b04192..16ab9b65e8 100644 --- a/cpp/src/barrier/second_order_cone_kernels.cuh +++ b/cpp/src/barrier/second_order_cone_kernels.cuh @@ -475,15 +475,14 @@ void launch_nt_scaling(cone_data_t& cones, rmm::cuda_stream_view strea const size_t cone_grid_dim = raft::ceildiv(static_cast(cones.n_cones), soc_block_size); - nt_finalize_scaling_scalars_kernel - <<>>( - cones.x, cones.z, x_scale, z_scale, cuopt::make_span(cones.eta), cone_offsets, cones.n_cones); + nt_finalize_scaling_scalars_kernel<<>>( + cones.x, cones.z, x_scale, z_scale, cuopt::make_span(cones.eta), cone_offsets, cones.n_cones); RAFT_CUDA_TRY(cudaPeekAtLastError()); const size_t element_grid_dim = raft::ceildiv(cones.n_cone_entries, soc_block_size); auto w = cuopt::make_span(cones.w); - nt_write_w_kernel<<>>( + nt_write_w_kernel<<>>( cones.x, cones.z, x_scale, z_scale, w, cone_offsets, element_cone_ids); RAFT_CUDA_TRY(cudaPeekAtLastError()); @@ -495,24 +494,24 @@ void launch_nt_scaling(cone_data_t& cones, rmm::cuda_stream_view strea }); cones.segmented_sum(unnormalized_tail_sq_terms, w_scale, stream); - nt_finalize_w_scale_kernel<<>>( + nt_finalize_w_scale_kernel<<>>( w, w_scale, w_scale, cone_offsets, cones.n_cones); RAFT_CUDA_TRY(cudaPeekAtLastError()); nt_normalize_w_kernel - <<>>(w, w_scale, element_cone_ids); + <<>>(w, w_scale, element_cone_ids); RAFT_CUDA_TRY(cudaPeekAtLastError()); // Persist lambda while w_scale still stores sqrt(det_J(w_tmp)). nt_write_lambda_kernel - <<>>(cones.x, - cones.z, - x_scale, - z_scale, - w_scale, - cuopt::make_span(cones.lambda), - cone_offsets, - element_cone_ids); + <<>>(cones.x, + cones.z, + x_scale, + z_scale, + w_scale, + cuopt::make_span(cones.lambda), + cone_offsets, + element_cone_ids); RAFT_CUDA_TRY(cudaPeekAtLastError()); // w_scale is overwritten from here @@ -524,7 +523,7 @@ void launch_nt_scaling(cone_data_t& cones, rmm::cuda_stream_view strea }); cones.segmented_sum(normalized_tail_terms, w_scale, stream); - nt_finalize_head_kernel<<>>( + nt_finalize_head_kernel<<>>( cuopt::make_span(cones.w), w_scale, cone_offsets, cones.n_cones); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -604,16 +603,16 @@ void launch_update_scaling_sparse(cone_data_t& cones, rmm::cuda_stream const i_t n_sparse = cones.n_sparse_cones; update_scaling_sparse_kernel - <<>>(cuopt::make_span(cones.w), - cuopt::make_span(cones.eta), - cuopt::make_span(cones.d), - cuopt::make_span(cones.sparse_v), - cuopt::make_span(cones.sparse_u), - cuopt::make_span(cones.cone_offsets), - cuopt::make_span(cones.sparse_cone_dims), - cuopt::make_span(cones.sparse_cone_ids), - cuopt::make_span(cones.sparse_entry_offsets), - n_sparse); + <<>>(cuopt::make_span(cones.w), + cuopt::make_span(cones.eta), + cuopt::make_span(cones.d), + cuopt::make_span(cones.sparse_v), + cuopt::make_span(cones.sparse_u), + cuopt::make_span(cones.cone_offsets), + cuopt::make_span(cones.sparse_cone_dims), + cuopt::make_span(cones.sparse_cone_ids), + cuopt::make_span(cones.sparse_entry_offsets), + n_sparse); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -850,7 +849,7 @@ void apply_w_inv(raft::device_span v, cones.segmented_sum(tail_terms, tail_dot, stream); const size_t grid_dim = raft::ceildiv(out.size(), soc_block_size); - apply_w_inv_write_kernel<<>>( + apply_w_inv_write_kernel<<>>( v, out, w, eta, tail_dot, cone_offsets, element_cone_ids); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -886,7 +885,7 @@ void apply_w(raft::device_span v, cones.segmented_sum(tail_terms, tail_dot, stream); const size_t grid_dim = raft::ceildiv(out.size(), soc_block_size); - apply_w_write_kernel<<>>( + apply_w_write_kernel<<>>( v, out, w, eta, tail_dot, cone_offsets, element_cone_ids); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -921,18 +920,18 @@ void apply_hessian(raft::device_span v, const size_t grid_dim = raft::ceildiv(out.size(), soc_block_size); apply_hessian_kernel - <<>>(v, - out, - w, - eta, - wv_dot, - cone_offsets, - element_cone_ids, - cuopt::make_span(cones.cone_is_sparse), - dense_cones_only, - bias, - output_scale, - bias_scale); + <<>>(v, + out, + w, + eta, + wv_dot, + cone_offsets, + element_cone_ids, + cuopt::make_span(cones.cone_is_sparse), + dense_cones_only, + bias, + output_scale, + bias_scale); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -1062,24 +1061,23 @@ void scatter_sparse_hessian_into_augmented(cone_data_t& cones, const size_t E = cones.n_sparse_cone_entries; const size_t entry_grid = raft::ceildiv(E, soc_block_size); scatter_sparse_hessian_into_augmented_kernel - <<>>( - cuopt::make_span(augmented_x), - cuopt::make_span(Hs_diag), - cuopt::make_span(cones.eta), - cuopt::make_span(cones.d), - cuopt::make_span(cones.sparse_cone_ids), - cuopt::make_span(cones.sparse_entry_offsets), - n_sparse, - cuopt::make_span(hessian_diag_csr_indices), - cuopt::make_span(q_values), - cuopt::make_span(cones.sparse_v), - cuopt::make_span(cones.sparse_u), - cuopt::make_span(exp_v_col), - cuopt::make_span(exp_u_col), - cuopt::make_span(exp_v_row), - cuopt::make_span(exp_u_row), - cuopt::make_span(sparse_expansion_D), - dual_perturb); + <<>>(cuopt::make_span(augmented_x), + cuopt::make_span(Hs_diag), + cuopt::make_span(cones.eta), + cuopt::make_span(cones.d), + cuopt::make_span(cones.sparse_cone_ids), + cuopt::make_span(cones.sparse_entry_offsets), + n_sparse, + cuopt::make_span(hessian_diag_csr_indices), + cuopt::make_span(q_values), + cuopt::make_span(cones.sparse_v), + cuopt::make_span(cones.sparse_u), + cuopt::make_span(exp_v_col), + cuopt::make_span(exp_u_col), + cuopt::make_span(exp_v_row), + cuopt::make_span(exp_u_row), + cuopt::make_span(sparse_expansion_D), + dual_perturb); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -1181,21 +1179,21 @@ void launch_sparse_augmented_matvec(raft::device_span x, "expansion output size mismatch"); sparse_augmented_matvec_kernel - <<>>(x, - r1, - y_exp, - Hs_diag, - cuopt::make_span(cones.sparse_v), - cuopt::make_span(cones.sparse_u), - cuopt::make_span(cones.eta), - cuopt::make_span(cones.sparse_cone_ids), - cuopt::make_span(cones.sparse_cone_dims), - cuopt::make_span(cones.sparse_entry_offsets), - cuopt::make_span(cones.cone_offsets), - cone_var_start, - n_primal, - m_constraints, - n_sparse); + <<>>(x, + r1, + y_exp, + Hs_diag, + cuopt::make_span(cones.sparse_v), + cuopt::make_span(cones.sparse_u), + cuopt::make_span(cones.eta), + cuopt::make_span(cones.sparse_cone_ids), + cuopt::make_span(cones.sparse_cone_dims), + cuopt::make_span(cones.sparse_entry_offsets), + cuopt::make_span(cones.cone_offsets), + cone_var_start, + n_primal, + m_constraints, + n_sparse); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -1261,16 +1259,16 @@ void scatter_dense_hessian_into_augmented(const cone_data_t& cones, const i_t n_dense = cones.n_dense_cones(); const size_t grid = raft::ceildiv(count, soc_block_size); scatter_dense_hessian_into_augmented_kernel - <<>>(cuopt::make_span(augmented_x), - cuopt::make_span(csr_indices), - cuopt::make_span(q_values), - cuopt::make_span(cones.w), - cuopt::make_span(cones.eta), - cuopt::make_span(cones.cone_offsets), - cuopt::make_span(dense_block_offsets), - cuopt::make_span(dense_cone_ids), - n_dense, - dual_perturb_value); + <<>>(cuopt::make_span(augmented_x), + cuopt::make_span(csr_indices), + cuopt::make_span(q_values), + cuopt::make_span(cones.w), + cuopt::make_span(cones.eta), + cuopt::make_span(cones.cone_offsets), + cuopt::make_span(dense_block_offsets), + cuopt::make_span(dense_cone_ids), + n_dense, + dual_perturb_value); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -1467,7 +1465,7 @@ void launch_cone_step_length(segmented_sum_t& partitions, const auto n_small = partitions.small_cone_ids.size(); const auto grid = (n_small + warps_per_cta - 1) / warps_per_cta; step_length_small_kernel - <<>>( + <<>>( u, du, alpha, @@ -1480,7 +1478,7 @@ void launch_cone_step_length(segmented_sum_t& partitions, if (!partitions.medium_cone_ids.is_empty()) { constexpr int medium_block_dim = 256; step_length_medium_kernel - <<>>( + <<>>( u, du, alpha, @@ -1512,14 +1510,14 @@ void launch_cone_step_length(segmented_sum_t& partitions, input, large_sums.data() + i, dim, - stream.value())); + stream.get())); } raft::device_span> large_sums_c(large_sums.data(), large_sums.size()); constexpr int large_solve_block_dim = 256; const auto grid = raft::ceildiv(n_large, static_cast(large_solve_block_dim)); - step_length_large_solve_kernel<<>>( + step_length_large_solve_kernel<<>>( u, du, alpha, @@ -1608,16 +1606,15 @@ void compute_combined_cone_rhs_term(raft::device_span dx_aff, // Stage both head vectors first because every tail entry needs them. const size_t cone_grid_dim = raft::ceildiv(static_cast(cones.n_cones), soc_block_size); - gather_cone_heads_kernel<<>>( + gather_cone_heads_kernel<<>>( scaled_dx, slot_1, cone_offsets, cones.n_cones); - gather_cone_heads_kernel<<>>( + gather_cone_heads_kernel<<>>( scaled_dz, slot_2, cone_offsets, cones.n_cones); RAFT_CUDA_TRY(cudaPeekAtLastError()); const size_t element_grid_dim = raft::ceildiv(cones.n_cone_entries, soc_block_size); - combined_cone_shift_write_kernel - <<>>( - out, scaled_dx, scaled_dz, slot_0, slot_1, slot_2, cone_offsets, element_cone_ids, sigma_mu); + combined_cone_shift_write_kernel<<>>( + out, scaled_dx, scaled_dz, slot_0, slot_1, slot_2, cone_offsets, element_cone_ids, sigma_mu); RAFT_CUDA_TRY(cudaPeekAtLastError()); auto shift = raft::device_span(out.data(), out.size()); @@ -1641,13 +1638,13 @@ void compute_combined_cone_rhs_term(raft::device_span dx_aff, cones.segmented_sum(lambda_tail_sq_terms, slot_1, stream); jordan_divide_by_lambda_scalar_kernel - <<>>( + <<>>( shift, nt_point, slot_0, slot_1, slot_0, slot_1, cone_offsets, cones.n_cones); RAFT_CUDA_TRY(cudaPeekAtLastError()); // Note that we implicitly multiply by -1 here since we are writing -p. jordan_divide_by_lambda_write_kernel - <<>>( + <<>>( shift, nt_point, slot_0, slot_1, cone_offsets, element_cone_ids, scratch_cone); RAFT_CUDA_TRY(cudaPeekAtLastError()); diff --git a/cpp/src/barrier/second_order_cone_reduction.cuh b/cpp/src/barrier/second_order_cone_reduction.cuh index bed06572a9..3e31e5b8be 100644 --- a/cpp/src/barrier/second_order_cone_reduction.cuh +++ b/cpp/src/barrier/second_order_cone_reduction.cuh @@ -95,7 +95,7 @@ struct segmented_sum_t { input + large_cone_offsets[i], output + large_cone_ids[i], large_cone_dimensions[i], - stream.value())); + stream.get())); cub_workspace_bytes = std::max(cub_workspace_bytes, temp_storage_bytes); } @@ -122,7 +122,7 @@ struct segmented_sum_t { const auto n_small = small_cone_ids.size(); const auto grid = (n_small + warps_per_cta - 1) / warps_per_cta; warp_per_cone_reduce_kernel - <<>>( + <<>>( input, cuopt::make_span(small_cone_ids), cone_offsets, output, init); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -131,7 +131,7 @@ struct segmented_sum_t { constexpr int medium_block_dim = 256; const auto n_medium = medium_cone_ids.size(); block_per_cone_reduce_kernel - <<>>( + <<>>( input, cuopt::make_span(medium_cone_ids), cone_offsets, output, init); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -147,7 +147,7 @@ struct segmented_sum_t { input + large_cone_offsets[i], output + large_cone_ids[i], large_cone_dimensions[i], - stream.value())); + stream.get())); } } } @@ -199,7 +199,7 @@ struct segmented_sum_t { cuopt::device_copy(large_cone_ids_device, large_cone_ids, stream); need_sync = true; } - if (need_sync) { stream.synchronize(); } + if (need_sync) { stream.sync(); } } }; diff --git a/cpp/src/barrier/sparse_cholesky.cuh b/cpp/src/barrier/sparse_cholesky.cuh index 01045847d1..dc51cc282d 100644 --- a/cpp/src/barrier/sparse_cholesky.cuh +++ b/cpp/src/barrier/sparse_cholesky.cuh @@ -144,7 +144,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { positive_definite(true), A_created(false), settings_(settings), - stream(handle_ptr->get_stream()) + stream(handle_ptr->get_stream().get()) { int major, minor, patch; cudssGetProperty(MAJOR_VERSION, &major); @@ -221,7 +221,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { // 4. Create the green context and stream for that green // context CUstream barrier_green_ctx_stream; i_t stream_priority; - cudaStream_t cuda_stream = handle_ptr_->get_stream(); + cudaStream_t cuda_stream = handle_ptr_->get_stream().get(); cudaError_t priority_result = cudaStreamGetPriority(cuda_stream, &stream_priority); RAFT_CUDA_TRY(priority_result); auto cuGreenCtxCreate_func = cuopt::get_driver_entry_point("cuGreenCtxCreate"); @@ -347,7 +347,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { status, "cudssMatrixCreateDn for x"); #endif - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); } ~sparse_cholesky_cudss_t() override @@ -381,7 +381,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { CU_CHECK( reinterpret_cast(cuGreenCtxDestroy_func)(barrier_green_ctx), reinterpret_cast(cuGetErrorString_func)); - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); } #endif } @@ -522,7 +522,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { // TODO: Is there any way to get nonzeros in the factors? // TODO: Is there any way to get flops for the factorization? RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); return 0; } @@ -582,7 +582,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { status, "cudssDataGet for info"); - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); if (info != 0) { settings_.log.printf("Factorization failed info %d\n", info); @@ -717,7 +717,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { settings_.log.printf("Symbolic factorization time : %.2fs\n", symbolic_time); if (settings_.concurrent_halt != nullptr && *settings_.concurrent_halt == 1) { RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); return CONCURRENT_HALT_RETURN; } int64_t lu_nz = 0; @@ -728,7 +728,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { "cudssDataGet for LU_NNZ"); settings_.log.printf("Symbolic nonzeros in factor : %.2e\n", static_cast(lu_nz) / 2.0); RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); // TODO: Is there any way to get nonzeros in the factors? // TODO: Is there any way to get flops for the factorization? @@ -753,7 +753,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { "cudaMemcpy for csr_values"); CUDA_CALL_AND_CHECK(cudaStreamSynchronize(stream), "cudaStreamSynchronize"); - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); CUDSS_CALL_AND_CHECK( cudssMatrixSetValues(A, csr_values_d), status, "cudssMatrixSetValues for A"); @@ -777,7 +777,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { status, "cudssDataGet for info"); RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); if (info != 0) { settings_.log.printf("Factorization failed info %d\n", info); return -1; @@ -798,13 +798,13 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { { auto d_b = cuopt::device_copy(b, handle_ptr_->get_stream()); auto d_x = cuopt::device_copy(x, handle_ptr_->get_stream()); - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); i_t out = solve(d_b, d_x); raft::copy(x.data(), d_x.data(), d_x.size(), handle_ptr_->get_stream()); // Sync so that data is on the host - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); for (i_t i = 0; i < n; i++) { if (x[i] != x[i]) { return -1; } @@ -815,7 +815,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { i_t solve(rmm::device_uvector& b, rmm::device_uvector& x) override { - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); if (static_cast(b.size()) != n) { settings_.log.printf("Error: b.size() %d != n %d\n", b.size(), n); return -1; @@ -843,7 +843,7 @@ class sparse_cholesky_cudss_t : public sparse_cholesky_base_t { } CUDA_CALL_AND_CHECK(cudaStreamSynchronize(stream), "cudaStreamSynchronize"); - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); #ifdef PRINT_RHS_AND_SOLUTION_HASH dense_vector_t b_host(n); diff --git a/cpp/src/linear_algebra/sort_csr.cuh b/cpp/src/linear_algebra/sort_csr.cuh index 23b9fd2d57..1b57bb3163 100644 --- a/cpp/src/linear_algebra/sort_csr.cuh +++ b/cpp/src/linear_algebra/sort_csr.cuh @@ -37,7 +37,7 @@ void sort_csr(optimization_problem_t& op_problem) num_segments, op_problem.get_constraint_matrix_offsets().data(), op_problem.get_constraint_matrix_offsets().data() + 1, - stream_view); + stream_view.get()); d_tmp_storage_bytes.resize(tmp_storage_bytes, stream_view); cub::DeviceSegmentedSort::SortPairs(d_tmp_storage_bytes.data(), tmp_storage_bytes, @@ -49,9 +49,9 @@ void sort_csr(optimization_problem_t& op_problem) num_segments, op_problem.get_constraint_matrix_offsets().data(), op_problem.get_constraint_matrix_offsets().data() + 1, - stream_view); - RAFT_CHECK_CUDA(stream_view); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view)); + stream_view.get()); + RAFT_CHECK_CUDA(stream_view.get()); + stream_view.sync(); } } // namespace mathematical_optimization diff --git a/cpp/src/linear_algebra/vector_math.cuh b/cpp/src/linear_algebra/vector_math.cuh index ac9d24001b..85c90c5172 100644 --- a/cpp/src/linear_algebra/vector_math.cuh +++ b/cpp/src/linear_algebra/vector_math.cuh @@ -53,7 +53,7 @@ f_t device_custom_vector_norm_inf(InputIteratorT in, i_t size, rmm::cuda_stream_ size, custom_op, init, - stream_view); + stream_view.get()); d_temp_storage.resize(temp_storage_bytes, stream_view); @@ -64,7 +64,7 @@ f_t device_custom_vector_norm_inf(InputIteratorT in, i_t size, rmm::cuda_stream_ size, custom_op, init, - stream_view); + stream_view.get()); return d_out.value(stream_view); } @@ -109,7 +109,7 @@ f_t vector_norm_inf(const rmm::device_uvector& x) [] __host__ __device__(f_t val) { return abs(val); }, static_cast(0), thrust::maximum{}); - RAFT_CHECK_CUDA(x.stream()); + RAFT_CHECK_CUDA(x.stream().get()); return max_abs; } @@ -125,7 +125,7 @@ f_t vector_norm2(const rmm::device_uvector& x) [] __host__ __device__(f_t val) { return val * val; }, f_t(0), thrust::plus{}); - RAFT_CHECK_CUDA(x.stream()); + RAFT_CHECK_CUDA(x.stream().get()); return std::sqrt(sum_of_squares); } diff --git a/cpp/src/mip_heuristics/diversity/assignment_hash_map.cu b/cpp/src/mip_heuristics/diversity/assignment_hash_map.cu index c9d20c97fe..db8e8825d7 100644 --- a/cpp/src/mip_heuristics/diversity/assignment_hash_map.cu +++ b/cpp/src/mip_heuristics/diversity/assignment_hash_map.cu @@ -84,10 +84,12 @@ size_t assignment_hash_map_t::hash_solution(solution_t& solu fill_integer_assignment(solution); thrust::fill( solution.handle_ptr->get_thrust_policy(), reduction_buffer.begin(), reduction_buffer.end(), 0); - hash_solution_kernel - <<<(integer_assignment.size() + TPB - 1) / TPB, TPB, 0, solution.handle_ptr->get_stream()>>>( - cuopt::make_span(integer_assignment), cuopt::make_span(reduction_buffer)); - RAFT_CHECK_CUDA(solution.handle_ptr->get_stream()); + hash_solution_kernel<<<(integer_assignment.size() + TPB - 1) / TPB, + TPB, + 0, + solution.handle_ptr->get_stream().get()>>>( + cuopt::make_span(integer_assignment), cuopt::make_span(reduction_buffer)); + RAFT_CHECK_CUDA(solution.handle_ptr->get_stream().get()); // Get the number of blocks used in the hash_solution_kernel int num_blocks = (integer_assignment.size() + TPB - 1) / TPB; @@ -103,7 +105,7 @@ size_t assignment_hash_map_t::hash_solution(solution_t& solu num_blocks, combine_hash(), 0, - solution.handle_ptr->get_stream()); + solution.handle_ptr->get_stream().get()); // Allocate temporary storage temp_storage.resize(temp_storage_bytes, solution.handle_ptr->get_stream()); @@ -117,7 +119,7 @@ size_t assignment_hash_map_t::hash_solution(solution_t& solu num_blocks, combine_hash(), 0, - solution.handle_ptr->get_stream()); + solution.handle_ptr->get_stream().get()); // Return early since we've already computed the hash sum return hash_sum.value(solution.handle_ptr->get_stream()); diff --git a/cpp/src/mip_heuristics/diversity/recombiners/recombiner.cuh b/cpp/src/mip_heuristics/diversity/recombiners/recombiner.cuh index f3faca1f28..fe9fb8ebec 100644 --- a/cpp/src/mip_heuristics/diversity/recombiners/recombiner.cuh +++ b/cpp/src/mip_heuristics/diversity/recombiners/recombiner.cuh @@ -90,11 +90,11 @@ class recombiner_t { const i_t TPB = 128; i_t n_blocks = (a.problem_ptr->n_integer_vars + TPB - 1) / TPB; assign_same_variables_kernel - <<get_stream()>>>(a.view(), - b.view(), - offspring.view(), - cuopt::make_span(remaining_indices), - n_remaining.data()); + <<get_stream().get()>>>(a.view(), + b.view(), + offspring.view(), + cuopt::make_span(remaining_indices), + n_remaining.data()); i_t remaining_variables = this->n_remaining.value(a.handle_ptr->get_stream()); auto vec_remaining_indices = diff --git a/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump.cu b/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump.cu index d164d2bfdb..0fa6b3c3d3 100644 --- a/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump.cu +++ b/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump.cu @@ -455,7 +455,7 @@ void fj_t::climber_init(i_t climber_idx, const rmm::cuda_stream_view& f_t excess = climber->violation_score.value(climber_stream); climber->best_excess.set_value_async(excess, climber_stream); } - climber_stream.synchronize(); + climber_stream.sync(); climber->break_condition.set_value_to_zero_async(climber_stream); climber->temp_break_condition.set_value_to_zero_async(climber_stream); @@ -471,9 +471,9 @@ void fj_t::climber_init(i_t climber_idx, const rmm::cuda_stream_view& climber->iterations_until_feasible_counter.set_value_to_zero_async(climber_stream); climber->small_move_tabu.set_value_to_zero_async(climber_stream); - climber_stream.synchronize(); + climber_stream.sync(); - climber_stream.synchronize(); + climber_stream.sync(); view = climber->view(); @@ -499,7 +499,7 @@ void fj_t::climber_init(i_t climber_idx, const rmm::cuda_stream_view& row_size_it_bin, row_size_bin_prefix_sum.data(), pb_ptr->binary_indices.size(), - climber_stream); + climber_stream.get()); if (i == 0 && temp_storage_bytes > climber->cub_storage_bytes.size()) climber->cub_storage_bytes.resize(temp_storage_bytes, climber_stream); } @@ -510,7 +510,7 @@ void fj_t::climber_init(i_t climber_idx, const rmm::cuda_stream_view& row_size_it_nonbin, row_size_nonbin_prefix_sum.data(), pb_ptr->nonbinary_indices.size(), - climber_stream); + climber_stream.get()); if (i == 0 && temp_storage_bytes > climber->cub_storage_bytes.size()) climber->cub_storage_bytes.resize(temp_storage_bytes, climber_stream); } @@ -533,7 +533,7 @@ void fj_t::climber_init(i_t climber_idx, const rmm::cuda_stream_view& pb_ptr->n_variables, pb_ptr->related_variables_offsets.begin(), pb_ptr->related_variables_offsets.begin() + 1, - climber_stream); + climber_stream.get()); if (i == 0 && temp_storage_bytes > climber->cub_storage_bytes.size()) climber->cub_storage_bytes.resize(temp_storage_bytes, climber_stream); } @@ -723,7 +723,7 @@ void fj_t::run_step_device(const rmm::cuda_stream_view& climber_stream data.candidate_variables.contents.data(), data.candidate_variables.set_size.data(), pb_ptr->n_variables, - climber_stream); + climber_stream.get()); if (compaction_temp_storage_bytes > data.cub_storage_bytes.size()) { data.cub_storage_bytes.resize(compaction_temp_storage_bytes, climber_stream); } @@ -771,7 +771,7 @@ void fj_t::run_step_device(const rmm::cuda_stream_view& climber_stream data.candidate_variables.contents.data(), data.candidate_variables.set_size.data(), pb_ptr->n_variables, - climber_stream); + climber_stream.get()); launch_select_variable_kernel(dim3(1), dim3(256), kernel_args, climber_stream); @@ -804,7 +804,7 @@ void fj_t::round_remaining_fractionals(solution_t& solution, data.handle_fractionals_only.set_value_async(handle_fractionals_only, climber_stream); data.break_condition.set_value_to_zero_async(climber_stream); data.temp_break_condition.set_value_to_zero_async(climber_stream); - climber_stream.synchronize(); + climber_stream.sync(); // Run the fractional move selection and assignment update kernels until all have been rounded host_loop(solution, climber_idx); @@ -909,7 +909,7 @@ i_t fj_t::host_loop(solution_t& solution, i_t climber_idx) data.best_assignment.data(), data.best_assignment.size(), climber_stream); - climber_stream.synchronize(); + climber_stream.sync(); // this solution cost computation with the changing(or not changing) weights is needed to // decide whether we reset the best objective on the FIRST_FEASIBLE mode. once we get rid of // FIRST_FEASIBLE mode, we can remove the following too. @@ -927,7 +927,7 @@ i_t fj_t::host_loop(solution_t& solution, i_t climber_idx) solution.assignment.data(), solution.assignment.size(), climber_stream); - climber_stream.synchronize(); + climber_stream.sync(); improvement_callback(user_obj, h_assignment); } } @@ -1110,11 +1110,11 @@ i_t fj_t::solve(solution_t& solution) } climber_init(0); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); handle_ptr->sync_stream(); i_t iterations = host_loop(solution); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); handle_ptr->sync_stream(); f_t effort_rate = (f_t)iterations / timer.elapsed_time(); diff --git a/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump.cuh b/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump.cuh index 0797d51750..a0f3103233 100644 --- a/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump.cuh +++ b/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump.cuh @@ -447,7 +447,7 @@ class fj_t { dot_product_buffer.data(), incumbent_objective.data(), fj.pb_ptr->n_variables, - fj.handle_ptr->get_stream()); + fj.handle_ptr->get_stream().get()); // Allocate temporary storage cub_storage_bytes.resize(temp_storage_bytes, fj.handle_ptr->get_stream()); diff --git a/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump_kernels.cu b/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump_kernels.cu index 441cfcc01f..0469574197 100644 --- a/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump_kernels.cu +++ b/cpp/src/mip_heuristics/feasibility_jump/feasibility_jump_kernels.cu @@ -1442,7 +1442,7 @@ void launch_load_balancing_prepare_iteration(dim3 grid, rmm::cuda_stream_view stream) { RAFT_CUDA_TRY(cudaLaunchCooperativeKernel( - (void*)load_balancing_prepare_iteration, grid, blocks, kernel_args, 0, stream)); + (void*)load_balancing_prepare_iteration, grid, blocks, kernel_args, 0, stream.get())); } template @@ -1460,7 +1460,7 @@ void launch_update_assignment_kernel(dim3 grid, rmm::cuda_stream_view stream) { RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)update_assignment_kernel, grid, blocks, kernel_args, 0, stream)); + (void*)update_assignment_kernel, grid, blocks, kernel_args, 0, stream.get())); } template @@ -1539,7 +1539,7 @@ void launch_compute_mtm_moves_kernel(dim3 grid, blocks, kernel_args, 0, - stream)); + stream.get())); } template @@ -1549,7 +1549,7 @@ void launch_load_balancing_sanity_checks(dim3 grid, rmm::cuda_stream_view stream) { RAFT_CUDA_TRY(cudaLaunchCooperativeKernel( - (void*)load_balancing_sanity_checks, grid, blocks, kernel_args, 0, stream)); + (void*)load_balancing_sanity_checks, grid, blocks, kernel_args, 0, stream.get())); } template @@ -1559,7 +1559,7 @@ void launch_handle_local_minimum_kernel(dim3 grid, rmm::cuda_stream_view stream) { RAFT_CUDA_TRY(cudaLaunchCooperativeKernel( - (void*)handle_local_minimum_kernel, grid, blocks, kernel_args, 0, stream)); + (void*)handle_local_minimum_kernel, grid, blocks, kernel_args, 0, stream.get())); } template @@ -1576,8 +1576,12 @@ void launch_update_changed_constraints_kernel(dim3 grid, void** kernel_args, rmm::cuda_stream_view stream) { - RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)update_changed_constraints_kernel, grid, blocks, kernel_args, 0, stream)); + RAFT_CUDA_TRY(cudaLaunchKernel((void*)update_changed_constraints_kernel, + grid, + blocks, + kernel_args, + 0, + stream.get())); } template @@ -1587,7 +1591,7 @@ void launch_update_lift_moves_kernel(dim3 grid, rmm::cuda_stream_view stream) { RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)update_lift_moves_kernel, grid, blocks, kernel_args, 0, stream)); + (void*)update_lift_moves_kernel, grid, blocks, kernel_args, 0, stream.get())); } template @@ -1597,7 +1601,7 @@ void launch_update_breakthrough_moves_kernel(dim3 grid, rmm::cuda_stream_view stream) { RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)update_breakthrough_moves_kernel, grid, blocks, kernel_args, 0, stream)); + (void*)update_breakthrough_moves_kernel, grid, blocks, kernel_args, 0, stream.get())); } template @@ -1607,7 +1611,7 @@ void launch_select_variable_kernel(dim3 grid, rmm::cuda_stream_view stream) { RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)select_variable_kernel, grid, blocks, kernel_args, 0, stream)); + (void*)select_variable_kernel, grid, blocks, kernel_args, 0, stream.get())); } template @@ -1617,7 +1621,7 @@ void launch_init_lhs_and_violation(dim3 grid, rmm::cuda_stream_view stream) { RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)init_lhs_and_violation, grid, blocks, kernel_args, 0, stream)); + (void*)init_lhs_and_violation, grid, blocks, kernel_args, 0, stream.get())); } template @@ -1627,7 +1631,7 @@ void launch_update_best_solution_kernel(dim3 grid, rmm::cuda_stream_view stream) { RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)update_best_solution_kernel, grid, blocks, kernel_args, 0, stream)); + (void*)update_best_solution_kernel, grid, blocks, kernel_args, 0, stream.get())); } template @@ -1636,8 +1640,12 @@ void launch_load_balancing_compute_workid_mappings(dim3 grid, void** kernel_args, rmm::cuda_stream_view stream) { - RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)load_balancing_compute_workid_mappings, grid, blocks, kernel_args, 0, stream)); + RAFT_CUDA_TRY(cudaLaunchKernel((void*)load_balancing_compute_workid_mappings, + grid, + blocks, + kernel_args, + 0, + stream.get())); } template @@ -1646,8 +1654,12 @@ void launch_load_balancing_init_cstr_bounds_csr(dim3 grid, void** kernel_args, rmm::cuda_stream_view stream) { - RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)load_balancing_init_cstr_bounds_csr, grid, blocks, kernel_args, 0, stream)); + RAFT_CUDA_TRY(cudaLaunchKernel((void*)load_balancing_init_cstr_bounds_csr, + grid, + blocks, + kernel_args, + 0, + stream.get())); } template @@ -1656,8 +1668,12 @@ void launch_load_balancing_compute_scores_binary(dim3 grid, void** kernel_args, rmm::cuda_stream_view stream) { - RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)load_balancing_compute_scores_binary, grid, blocks, kernel_args, 0, stream)); + RAFT_CUDA_TRY(cudaLaunchKernel((void*)load_balancing_compute_scores_binary, + grid, + blocks, + kernel_args, + 0, + stream.get())); } template @@ -1666,8 +1682,12 @@ void launch_load_balancing_mtm_compute_candidates(dim3 grid, void** kernel_args, rmm::cuda_stream_view stream) { - RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)load_balancing_mtm_compute_candidates, grid, blocks, kernel_args, 0, stream)); + RAFT_CUDA_TRY(cudaLaunchKernel((void*)load_balancing_mtm_compute_candidates, + grid, + blocks, + kernel_args, + 0, + stream.get())); } template @@ -1676,8 +1696,12 @@ void launch_load_balancing_mtm_compute_scores(dim3 grid, void** kernel_args, rmm::cuda_stream_view stream) { - RAFT_CUDA_TRY(cudaLaunchKernel( - (void*)load_balancing_mtm_compute_scores, grid, blocks, kernel_args, 0, stream)); + RAFT_CUDA_TRY(cudaLaunchKernel((void*)load_balancing_mtm_compute_scores, + grid, + blocks, + kernel_args, + 0, + stream.get())); } // to save from compilation time, separate those and instantiate separately rather being part of a diff --git a/cpp/src/mip_heuristics/feasibility_jump/utils.cuh b/cpp/src/mip_heuristics/feasibility_jump/utils.cuh index 1b2862d558..7eee62e8a3 100644 --- a/cpp/src/mip_heuristics/feasibility_jump/utils.cuh +++ b/cpp/src/mip_heuristics/feasibility_jump/utils.cuh @@ -45,7 +45,7 @@ struct bitmap_t { void clear(const rmm::cuda_stream_view& stream) { cudaMemsetAsync( - validity_bitmap.data(), 0, sizeof(word_t) * validity_bitmap.size(), stream.value()); + validity_bitmap.data(), 0, sizeof(word_t) * validity_bitmap.size(), stream.get()); } void clear(const raft::handle_t* handle_ptr) { @@ -115,7 +115,7 @@ struct contiguous_set_t { set_size.set_value_to_zero_async(stream); // can't use thrust::fill, needs a memset node in order to be recorded in CUDA graphs // works bcs (uint8_t)-1 == 0xFF => (repeated 4 times) 0xFFFFFFFF == (uint32_t)-1 - cudaMemsetAsync(index_map.data(), -1, sizeof(i_t) * index_map.size(), stream.value()); + cudaMemsetAsync(index_map.data(), -1, sizeof(i_t) * index_map.size(), stream.get()); validity_bitmap.clear(stream); } diff --git a/cpp/src/mip_heuristics/local_search/feasibility_pump/feasibility_pump.cu b/cpp/src/mip_heuristics/local_search/feasibility_pump/feasibility_pump.cu index 9e5a00b175..c2ef6299c7 100644 --- a/cpp/src/mip_heuristics/local_search/feasibility_pump/feasibility_pump.cu +++ b/cpp/src/mip_heuristics/local_search/feasibility_pump/feasibility_pump.cu @@ -200,7 +200,7 @@ bool feasibility_pump_t::linear_project_onto_polytope(solution_tget_stream()); - RAFT_CHECK_CUDA(solution.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(solution.handle_ptr->get_stream().get()); temp_p.presolve_data.objective_offset = obj_offset; // change the precision between 1. and 10-4 depending on the integer ratio // the lp tolerance can be pretty high @@ -447,7 +447,7 @@ void feasibility_pump_t::relax_general_integers(solution_t& var_types[v_idx] = copy_type; }); solution.handle_ptr->sync_stream(); - RAFT_CHECK_CUDA(solution.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(solution.handle_ptr->get_stream().get()); solution.problem_ptr->compute_n_integer_vars(); solution.problem_ptr->compute_binary_var_table(); CUOPT_LOG_DEBUG("Integers are relaxed n_int vars %d n_binary vars %d n_vars %d", diff --git a/cpp/src/mip_heuristics/local_search/lagrangian.cuh b/cpp/src/mip_heuristics/local_search/lagrangian.cuh index 9c814d91d0..5cf6d2a31d 100644 --- a/cpp/src/mip_heuristics/local_search/lagrangian.cuh +++ b/cpp/src/mip_heuristics/local_search/lagrangian.cuh @@ -52,7 +52,7 @@ inline rmm::device_uvector get_weighted_lagrangian_weights( const i_t TPB = 128; const i_t n_blocks = problem.n_variables; compute_lagrangian_weights_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( problem.view(), raft::device_span{cstr_left_weights.data(), cstr_left_weights.size()}, raft::device_span{cstr_right_weights.data(), cstr_right_weights.size()}, diff --git a/cpp/src/mip_heuristics/local_search/rounding/bounds_repair.cu b/cpp/src/mip_heuristics/local_search/rounding/bounds_repair.cu index 2779a757e6..c29e448f3b 100644 --- a/cpp/src/mip_heuristics/local_search/rounding/bounds_repair.cu +++ b/cpp/src/mip_heuristics/local_search/rounding/bounds_repair.cu @@ -254,14 +254,14 @@ void bounds_repair_t::compute_damages(problem_t& problem, i_ CUOPT_LOG_TRACE("Bounds repair: Computing damanges!"); // TODO check performance, we can apply load balancing here const i_t TPB = 256; - compute_damages_kernel<<get_stream()>>>( + compute_damages_kernel<<get_stream().get()>>>( problem.view(), candidates.view(), make_span(cstr_violations_up), make_span(cstr_violations_down), make_span(bound_presolve.upd.min_activity), make_span(bound_presolve.upd.max_activity)); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); auto sort_iterator = thrust::make_zip_iterator( thrust::make_tuple(candidates.cstr_delta.data(), candidates.damage.data())); // sort the best moves so that we can filter diff --git a/cpp/src/mip_heuristics/local_search/rounding/constraint_prop.cu b/cpp/src/mip_heuristics/local_search/rounding/constraint_prop.cu index 286a8224a5..f64545e8c8 100644 --- a/cpp/src/mip_heuristics/local_search/rounding/constraint_prop.cu +++ b/cpp/src/mip_heuristics/local_search/rounding/constraint_prop.cu @@ -91,7 +91,7 @@ void sort_subsections(raft::device_span vars, n_subsections, offsets.data(), offsets.data() + 1, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); // Allocate temporary storage d_temp_storage.resize(temp_storage_bytes, handle_ptr->get_stream()); @@ -107,7 +107,7 @@ void sort_subsections(raft::device_span vars, n_subsections, offsets.data(), offsets.data() + 1, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); handle_ptr->sync_stream(); } @@ -179,7 +179,7 @@ void constraint_prop_t::sort_by_implied_slack_consumption(solution_t - <<get_stream()>>>( + <<get_stream().get()>>>( sol.problem_ptr->view(), vars, min_activity, diff --git a/cpp/src/mip_heuristics/local_search/rounding/lb_bounds_repair.cu b/cpp/src/mip_heuristics/local_search/rounding/lb_bounds_repair.cu index 68e0a5a757..a73da89fda 100644 --- a/cpp/src/mip_heuristics/local_search/rounding/lb_bounds_repair.cu +++ b/cpp/src/mip_heuristics/local_search/rounding/lb_bounds_repair.cu @@ -268,14 +268,14 @@ void lb_bounds_repair_t::compute_damages( // TODO check performance, we can apply load balancing here const i_t TPB = 256; using f_t2 = typename type_2::type; - compute_damages_kernel - <<get_stream()>>>(original_problem.view(), - candidates.view(), - make_span_2(problem.variable_bounds), - make_span(cstr_violations_up), - make_span(cstr_violations_down), - make_span_2(lb_bound_presolve.cnst_slack)); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + compute_damages_kernel<<get_stream().get()>>>( + original_problem.view(), + candidates.view(), + make_span_2(problem.variable_bounds), + make_span(cstr_violations_up), + make_span(cstr_violations_down), + make_span_2(lb_bound_presolve.cnst_slack)); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); auto sort_iterator = thrust::make_zip_iterator( thrust::make_tuple(candidates.cstr_delta.data(), candidates.damage.data())); // sort the best moves so that we can filter diff --git a/cpp/src/mip_heuristics/local_search/rounding/lb_constraint_prop.cu b/cpp/src/mip_heuristics/local_search/rounding/lb_constraint_prop.cu index f1de4d12ca..f4ca208843 100644 --- a/cpp/src/mip_heuristics/local_search/rounding/lb_constraint_prop.cu +++ b/cpp/src/mip_heuristics/local_search/rounding/lb_constraint_prop.cu @@ -372,7 +372,7 @@ void lb_constraint_prop_t::sort_by_implied_slack_consumption( const i_t block_dim = 128; lb_bounds_update.calculate_constraint_slack(original_problem.handle_ptr); compute_implied_slack_consumption_per_var - <<get_stream()>>>( + <<get_stream().get()>>>( original_problem.view(), vars, make_span_2(lb_bounds_update.cnst_slack), @@ -585,7 +585,7 @@ void sort_subsections(raft::device_span vars, n_subsections, offsets.data(), offsets.data() + 1, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); // Allocate temporary storage d_temp_storage.resize(temp_storage_bytes, handle_ptr->get_stream()); @@ -601,7 +601,7 @@ void sort_subsections(raft::device_span vars, n_subsections, offsets.data(), offsets.data() + 1, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); handle_ptr->sync_stream(); } @@ -760,7 +760,7 @@ bool lb_constraint_prop_t::find_integer( timer_t& timer, std::optional>> probing_candidates) { - RAFT_CHECK_CUDA(problem.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(problem.handle_ptr->get_stream().get()); if (orig_sol.problem_ptr->n_integer_vars == 0) { cuopt_func_call(orig_sol.test_variable_bounds()); return orig_sol.compute_feasibility(); @@ -779,7 +779,7 @@ bool lb_constraint_prop_t::find_integer( lb_bounds_update.settings.time_limit = max_timer.remaining_time(); lb_bounds_update.settings.iteration_limit = 20; - RAFT_CHECK_CUDA(problem.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(problem.handle_ptr->get_stream().get()); if (max_timer.check_time_limit()) { CUOPT_LOG_DEBUG("Time limit is reached before bounds prop rounding!"); @@ -793,7 +793,7 @@ bool lb_constraint_prop_t::find_integer( orig_sol.problem_ptr->n_integer_vars, orig_sol.handle_ptr->get_stream()); CUOPT_LOG_DEBUG("LB Bounds propagation rounding: unset vars %lu", unset_integer_vars.size()); - RAFT_CHECK_CUDA(problem.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(problem.handle_ptr->get_stream().get()); // this is needed for the sort inside of the loop // infeasible cnst_slack invalid diff --git a/cpp/src/mip_heuristics/local_search/rounding/simple_rounding.cu b/cpp/src/mip_heuristics/local_search/rounding/simple_rounding.cu index a44872aba9..48b4559772 100644 --- a/cpp/src/mip_heuristics/local_search/rounding/simple_rounding.cu +++ b/cpp/src/mip_heuristics/local_search/rounding/simple_rounding.cu @@ -53,16 +53,18 @@ bool check_brute_force_rounding(solution_t& solution) // // try all configs in parallel and compute feasibility brute_force_check_kernel - <<get_stream()>>>(solution.view(), - n_integers_to_round, - cuopt::make_span(var_map), - cuopt::make_span(constraint_buf), - best_config.data()); + <<get_stream().get()>>>( + solution.view(), + n_integers_to_round, + cuopt::make_span(var_map), + cuopt::make_span(constraint_buf), + best_config.data()); if (best_config.value(solution.handle_ptr->get_stream()) != -1) { CUOPT_LOG_DEBUG("Feasible found during brute force rounding!"); // apply the feasible rounding - apply_feasible_rounding_kernel<<<1, TPB, 0, solution.handle_ptr->get_stream()>>>( - solution.view(), n_integers_to_round, cuopt::make_span(var_map), best_config.data()); + apply_feasible_rounding_kernel + <<<1, TPB, 0, solution.handle_ptr->get_stream().get()>>>( + solution.view(), n_integers_to_round, cuopt::make_span(var_map), best_config.data()); solution.handle_ptr->sync_stream(); bool feas = solution.compute_feasibility(); cuopt_assert(feas, "Solution must be feasible!"); @@ -83,7 +85,7 @@ bool invoke_simple_rounding(solution_t& solution) rmm::device_scalar successful(true_v, solution.handle_ptr->get_stream()); i_t TPB = 128; simple_rounding_kernel - <<<2048, TPB, 0, solution.handle_ptr->get_stream()>>>(solution.view(), successful.data()); + <<<2048, TPB, 0, solution.handle_ptr->get_stream().get()>>>(solution.view(), successful.data()); if (!successful.value(solution.handle_ptr->get_stream())) { CUOPT_LOG_DEBUG("Simple rounding failed"); solution.copy_from(sol_copy); @@ -112,8 +114,8 @@ void invoke_round_nearest(solution_t& solution, uint64_t seed) i_t n_blocks = (solution.problem_ptr->n_integer_vars + TPB - 1) / TPB; nearest_rounding_kernel - <<get_stream()>>>(solution.view(), seed); - RAFT_CHECK_CUDA(solution.handle_ptr->get_stream()); + <<get_stream().get()>>>(solution.view(), seed); + RAFT_CHECK_CUDA(solution.handle_ptr->get_stream().get()); } template @@ -129,8 +131,9 @@ void invoke_random_round_nearest(solution_t& solution, n_integers, solution.problem_ptr->n_integer_vars); rmm::device_scalar n_randomly_rounded(zero_v, solution.handle_ptr->get_stream()); - random_nearest_rounding_kernel<<get_stream()>>>( - solution.view(), seed_rng.next_u64(), n_randomly_rounded.data()); + random_nearest_rounding_kernel + <<get_stream().get()>>>( + solution.view(), seed_rng.next_u64(), n_randomly_rounded.data()); i_t h_n_random_rounds = n_randomly_rounded.value(solution.handle_ptr->get_stream()); CUOPT_LOG_TRACE("Randomly rounded integers %d", h_n_random_rounds); i_t additional_roundings_needed = n_target_random_rounds - h_n_random_rounds; @@ -145,17 +148,17 @@ void invoke_random_round_nearest(solution_t& solution, shuffled_indices.end(), rng); random_rounding_kernel - <<<1, 1, 0, solution.handle_ptr->get_stream()>>>(solution.view(), - seed_rng.next_u64(), - shuffled_indices.data(), - n_randomly_rounded.data(), - additional_roundings_needed); + <<<1, 1, 0, solution.handle_ptr->get_stream().get()>>>(solution.view(), + seed_rng.next_u64(), + shuffled_indices.data(), + n_randomly_rounded.data(), + additional_roundings_needed); h_n_random_rounds = n_randomly_rounded.value(solution.handle_ptr->get_stream()); CUOPT_LOG_TRACE("Randomly rounded integers, after adding close integers too %d", h_n_random_rounds); } solution.round_nearest(seed_rng.next_u64()); - RAFT_CHECK_CUDA(solution.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(solution.handle_ptr->get_stream().get()); } template diff --git a/cpp/src/mip_heuristics/mip_scaling_strategy.cu b/cpp/src/mip_heuristics/mip_scaling_strategy.cu index 8ff8310f61..2ab03ac535 100644 --- a/cpp/src/mip_heuristics/mip_scaling_strategy.cu +++ b/cpp/src/mip_heuristics/mip_scaling_strategy.cu @@ -165,7 +165,7 @@ void compute_row_inf_norm( matrix_offsets.data() + 1, max_op_t{}, f_t(0), - stream_view)); + stream_view.get())); } template @@ -203,7 +203,7 @@ void compute_row_integer_gcd( matrix_offsets.data() + 1, gcd_op_t{}, std::int64_t{0}, - stream_view)); + stream_view.get())); } template @@ -236,7 +236,7 @@ void compute_big_m_skip_rows( matrix_offsets.data() + 1, max_op_t{}, f_t(0), - stream_view)); + stream_view.get())); size_t min_bytes = temp_storage_bytes; RAFT_CUDA_TRY(cub::DeviceSegmentedReduce::Reduce(temp_storage.data(), min_bytes, @@ -247,7 +247,7 @@ void compute_big_m_skip_rows( matrix_offsets.data() + 1, min_op_t{}, std::numeric_limits::infinity(), - stream_view)); + stream_view.get())); size_t count_bytes = temp_storage_bytes; RAFT_CUDA_TRY(cub::DeviceSegmentedReduce::Reduce(temp_storage.data(), count_bytes, @@ -258,7 +258,7 @@ void compute_big_m_skip_rows( matrix_offsets.data() + 1, thrust::plus{}, i_t(0), - stream_view)); + stream_view.get())); auto row_begin = thrust::make_zip_iterator( thrust::make_tuple(row_inf_norm.begin(), row_min_nonzero.begin(), row_nonzero_count.begin())); @@ -435,7 +435,7 @@ size_t dry_run_cub( matrix_offsets.data() + 1, max_op_t{}, f_t(0), - stream_view)); + stream_view.get())); temp_storage_bytes = std::max(temp_storage_bytes, current_required_bytes); auto coeff_nonzero_min_iter = @@ -449,7 +449,7 @@ size_t dry_run_cub( matrix_offsets.data() + 1, min_op_t{}, std::numeric_limits::infinity(), - stream_view)); + stream_view.get())); temp_storage_bytes = std::max(temp_storage_bytes, current_required_bytes); auto coeff_nonzero_count_iter = @@ -463,7 +463,7 @@ size_t dry_run_cub( matrix_offsets.data() + 1, thrust::plus{}, i_t(0), - stream_view)); + stream_view.get())); temp_storage_bytes = std::max(temp_storage_bytes, current_required_bytes); if (variable_types.size() == static_cast(op_problem.get_n_variables())) { @@ -482,7 +482,7 @@ size_t dry_run_cub( matrix_offsets.data() + 1, gcd_op_t{}, std::int64_t{0}, - stream_view)); + stream_view.get())); temp_storage_bytes = std::max(temp_storage_bytes, current_required_bytes); } @@ -678,7 +678,7 @@ void mip_scaling_strategy_t::scale_problem(bool do_objective_scaling) ref_log2_values.data() + median_idx, sizeof(double), cudaMemcpyDeviceToHost, - stream_view_)); + stream_view_.get())); handle_ptr_->sync_stream(); f_t target_norm = static_cast(exp2(h_median_log2)); cuopt_assert(std::isfinite(static_cast(target_norm)), "target_norm must be finite"); diff --git a/cpp/src/mip_heuristics/presolve/block_bve.cu b/cpp/src/mip_heuristics/presolve/block_bve.cu index 6874833e15..c8d6ea3314 100644 --- a/cpp/src/mip_heuristics/presolve/block_bve.cu +++ b/cpp/src/mip_heuristics/presolve/block_bve.cu @@ -762,22 +762,22 @@ double bve_project_batch_gpu(const raft::handle_t& handle, // sentinel 0xFFFFFFFF (every byte 0xFF) marks a boundary pattern with no feasible interior // yet RAFT_CUDA_TRY( - cudaMemsetAsync(d_witness.data(), 0xFF, d_witness.size() * sizeof(uint32_t), stream)); + cudaMemsetAsync(d_witness.data(), 0xFF, d_witness.size() * sizeof(uint32_t), stream.get())); // one warp per row, one CTA per (block, m, am) assignment, grid-strided const int64_t total = (int64_t)num * (int64_t)patterns * ((int64_t)1 << na); const int grid = std::min(total, int64_t{65535}); - bve_enumerate_kernel<<>>(num, - nb, - na, - nrows, - tol, - d_coeffs.data(), - d_local_var.data(), - d_row_start.data(), - d_lower.data(), - d_upper.data(), - d_witness.data()); + bve_enumerate_kernel<<>>(num, + nb, + na, + nrows, + tol, + d_coeffs.data(), + d_local_var.data(), + d_row_start.data(), + d_lower.data(), + d_upper.data(), + d_witness.data()); RAFT_CUDA_TRY(cudaGetLastError()); // Unscaled op counts: host pack/unpack touches + one coeff read per assignment. diff --git a/cpp/src/mip_heuristics/presolve/bounds_presolve.cu b/cpp/src/mip_heuristics/presolve/bounds_presolve.cu index e5a7f249f1..0c84d26fa0 100644 --- a/cpp/src/mip_heuristics/presolve/bounds_presolve.cu +++ b/cpp/src/mip_heuristics/presolve/bounds_presolve.cu @@ -100,7 +100,7 @@ void bound_presolve_t::calculate_activity(problem_t& pb) constexpr auto n_threads = 256; calc_activity_kernel - <<get_stream()>>>(pb.view(), upd.view()); + <<get_stream().get()>>>(pb.view(), upd.view()); } template @@ -122,8 +122,8 @@ bool bound_presolve_t::calculate_bounds_update(problem_t& pb pb.tolerances.absolute_tolerance / context.settings.semi_continuous_big_m; upd.bounds_changed.set_value_async(zero, pb.handle_ptr->get_stream()); update_bounds_kernel - <<get_stream()>>>(pb.view(), upd.view()); - RAFT_CHECK_CUDA(pb.handle_ptr->get_stream()); + <<get_stream().get()>>>(pb.view(), upd.view()); + RAFT_CHECK_CUDA(pb.handle_ptr->get_stream().get()); i_t h_bounds_changed = upd.bounds_changed.value(pb.handle_ptr->get_stream()); return h_bounds_changed != zero; } @@ -165,7 +165,7 @@ void bound_presolve_t::set_bounds( var_ub[pair.first] = pair.second; }); handle_ptr->sync_stream(); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } template @@ -283,7 +283,7 @@ bool bound_presolve_t::calculate_infeasible_redundant_constraints(prob thrust::make_tuple(0, 0), tuple_plus_t{}); - RAFT_CHECK_CUDA(pb.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(pb.handle_ptr->get_stream().get()); if (redund_constraints_count > 0) { CUOPT_LOG_TRACE("Redundant constraint count %d", redund_constraints_count); diff --git a/cpp/src/mip_heuristics/presolve/conditional_bound_strengthening.cu b/cpp/src/mip_heuristics/presolve/conditional_bound_strengthening.cu index b1f27de8a6..0223c77422 100644 --- a/cpp/src/mip_heuristics/presolve/conditional_bound_strengthening.cu +++ b/cpp/src/mip_heuristics/presolve/conditional_bound_strengthening.cu @@ -85,7 +85,7 @@ void spgemm_cusparse([[maybe_unused]] rmm::device_uvector& offsetsA, auto stream = offsetsA.stream(); cusparseHandle_t handle; cusparseCreate(&handle); - cusparseSetStream(handle, stream); + cusparseSetStream(handle, stream.get()); int m = offsetsA.size() - 1; int n = offsetsB.size() - 1; @@ -215,7 +215,7 @@ void spgemm_cusparse([[maybe_unused]] rmm::device_uvector& offsetsA, check_cusparse_status(cusparseSpGEMM_copy( handle, opA, opB, &alpha, matA, matB, &beta, matC, computeType, alg, spgemmDesc)); - stream.synchronize(); + stream.sync(); cusparseSpGEMM_destroyDescr(spgemmDesc); cusparseDestroySpMat(matA); @@ -677,7 +677,7 @@ void conditional_bound_strengthening_t::solve(problem_t& pro update_constraint_bounds_kernel<<>>( problem.view(), cuopt::make_span(constraint_pairs), cuopt::make_span(locks_per_constraint)); - RAFT_CHECK_CUDA(problem.handle_ptr->get_stream()); + RAFT_CHECK_CUDA(problem.handle_ptr->get_stream().get()); problem.handle_ptr->sync_stream(); #ifdef DEBUG_COND_BOUNDS_PROP diff --git a/cpp/src/mip_heuristics/presolve/lb_probing_cache.cu b/cpp/src/mip_heuristics/presolve/lb_probing_cache.cu index 3ac7650615..59202a48dc 100644 --- a/cpp/src/mip_heuristics/presolve/lb_probing_cache.cu +++ b/cpp/src/mip_heuristics/presolve/lb_probing_cache.cu @@ -279,7 +279,7 @@ inline std::vector compute_prioritized_integer_indices( CUOPT_LOG_INFO("prioritized integer_indices n_integer_vars %d", problem.pb->n_integer_vars); // compute the min var slack compute_min_slack_per_var - <<n_integer_vars, 128, 0, problem.handle_ptr->get_stream()>>>( + <<n_integer_vars, 128, 0, problem.handle_ptr->get_stream().get()>>>( problem.pb->view(), make_span_2(bound_presolve.cnst_slack), make_span(min_slack_per_var), diff --git a/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve.cu b/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve.cu index 017bf32e91..290f6bef92 100644 --- a/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve.cu +++ b/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve.cu @@ -172,8 +172,8 @@ bool build_graph(managed_stream_pool& streams, cudaEvent_t fork_stream_event; cudaEventCreate(&fork_stream_event); - cudaStreamBeginCapture(handle_ptr->get_stream(), cudaStreamCaptureModeThreadLocal); - cudaEventRecord(fork_stream_event, handle_ptr->get_stream()); + cudaStreamBeginCapture(handle_ptr->get_stream().get(), cudaStreamCaptureModeThreadLocal); + cudaEventRecord(fork_stream_event, handle_ptr->get_stream().get()); // dry-run - managed pool tracks how many streams were issued d_func(); @@ -184,26 +184,26 @@ bool build_graph(managed_stream_pool& streams, auto activity_done = streams.create_events_on_issued(); streams.reset_issued(); for (auto& e : activity_done) { - cudaStreamWaitEvent(handle_ptr->get_stream(), e); + cudaStreamWaitEvent(handle_ptr->get_stream().get(), e); } - cudaStreamEndCapture(handle_ptr->get_stream(), &graph); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + cudaStreamEndCapture(handle_ptr->get_stream().get(), &graph); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); if (graph_exec != nullptr) { cudaGraphExecDestroy(graph_exec); cudaGraphInstantiate(&graph_exec, graph); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } else { cudaGraphInstantiate(&graph_exec, graph); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } cudaGraphDestroy(graph); graph_created = true; - handle_ptr->get_stream().synchronize(); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + handle_ptr->get_stream().sync(); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); return graph_created; } @@ -215,7 +215,7 @@ void load_balanced_bounds_presolve_t::setup( pb = &problem; auto handle_ptr = pb->handle_ptr; auto stream = handle_ptr->get_stream(); - stream.synchronize(); + stream.sync(); host_bounds.resize(2 * pb->n_variables); cnst_slack.resize(2 * pb->n_constraints, stream); vars_bnd.resize(2 * pb->n_variables, stream); @@ -234,7 +234,7 @@ void load_balanced_bounds_presolve_t::setup( heavy_degree_cutoff, problem.cnst_bin_offsets, problem.offsets); - RAFT_CHECK_CUDA(stream_heavy_cnst); + RAFT_CHECK_CUDA(stream_heavy_cnst.get()); num_blocks_heavy_vars = create_heavy_item_block_segments(stream_heavy_vars, heavy_vars_vertex_ids, @@ -243,7 +243,7 @@ void load_balanced_bounds_presolve_t::setup( heavy_degree_cutoff, problem.vars_bin_offsets, problem.reverse_offsets); - RAFT_CHECK_CUDA(stream_heavy_vars); + RAFT_CHECK_CUDA(stream_heavy_vars.get()); tmp_act.resize(2 * num_blocks_heavy_cnst, stream_heavy_cnst); tmp_bnd.resize(2 * num_blocks_heavy_vars, stream_heavy_vars); @@ -254,7 +254,7 @@ void load_balanced_bounds_presolve_t::setup( std::tie(is_vars_sub_warp_single_bin, vars_sub_warp_count) = sub_warp_meta(stream, warp_vars_offsets, warp_vars_id_offsets, pb->vars_bin_offsets, 4); - RAFT_CHECK_CUDA(stream); + RAFT_CHECK_CUDA(stream.get()); streams.sync_test_all_issued(); if (!calc_slack_erase_inf_cnst_graph_created) { @@ -491,10 +491,10 @@ void load_balanced_bounds_presolve_t::calculate_constraint_slack_iter( // writes nans to constraint activities that are infeasible //-> less expensive checks for update bounds step raft::common::nvtx::range scope("act_cuda_task_graph"); - cudaGraphLaunch(calc_slack_erase_inf_cnst_exec, handle_ptr->get_stream()); + cudaGraphLaunch(calc_slack_erase_inf_cnst_exec, handle_ptr->get_stream().get()); } infeas_cnst_slack_set_to_nan = true; - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } template @@ -505,10 +505,10 @@ void load_balanced_bounds_presolve_t::calculate_constraint_slack( h_bounds_changed = 0; { raft::common::nvtx::range scope("act_cuda_task_graph"); - cudaGraphLaunch(calc_slack_exec, handle_ptr->get_stream()); + cudaGraphLaunch(calc_slack_exec, handle_ptr->get_stream().get()); } infeas_cnst_slack_set_to_nan = false; - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } template @@ -518,9 +518,9 @@ bool load_balanced_bounds_presolve_t::update_bounds_from_slack( // bounds_changed is copied to h_bounds_changed in upd_bnd_exec { raft::common::nvtx::range scope("upd_cuda_task_graph"); - cudaGraphLaunch(upd_bnd_exec, handle_ptr->get_stream()); + cudaGraphLaunch(upd_bnd_exec, handle_ptr->get_stream().get()); } - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); constexpr i_t zero = 0; return (zero < h_bounds_changed); } @@ -602,7 +602,7 @@ bool load_balanced_bounds_presolve_t::calculate_infeasible_redundant_c thrust::reduce(handle_ptr->get_thrust_policy(), detect_iter, detect_iter + pb->n_constraints); handle_ptr->sync_stream(); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } else { auto detect_iter = thrust::make_transform_iterator( thrust::make_zip_iterator(thrust::make_tuple(pb->constraint_lower_bounds.begin(), @@ -614,7 +614,7 @@ bool load_balanced_bounds_presolve_t::calculate_infeasible_redundant_c infeas_constraints_count = thrust::reduce(handle_ptr->get_thrust_policy(), detect_iter, detect_iter + pb->n_constraints); handle_ptr->sync_stream(); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } if (infeas_constraints_count > 0) { CUOPT_LOG_TRACE("LB Infeasible constraint count %d", infeas_constraints_count); diff --git a/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve.cuh b/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve.cuh index 341d9cd262..5e88f7bebf 100644 --- a/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve.cuh +++ b/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve.cuh @@ -75,7 +75,7 @@ class managed_stream_pool { void wait_issued_on_event(cudaEvent_t e) { for (int i = 0; i < end_unsycned + 1; ++i) { - cudaStreamWaitEvent(streams_[i].view(), e, 0); + cudaStreamWaitEvent(streams_[i].view().get(), e, 0); } } @@ -94,7 +94,7 @@ class managed_stream_pool { cudaEventCreate(&e); } for (int i = 0; i < end_unsycned + 1; ++i) { - cudaEventRecord(events[i], streams_[i].view()); + cudaEventRecord(events[i], streams_[i].view().get()); } return events; } @@ -103,7 +103,7 @@ class managed_stream_pool { { for (int i = 0; i < end_unsycned + 1; ++i) { streams_[i].synchronize(); - RAFT_CHECK_CUDA(streams_[i].value()); + RAFT_CHECK_CUDA(streams_[i].view().get()); } end_unsycned = -1; next_stream = 0; diff --git a/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve_helpers.cuh b/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve_helpers.cuh index 6f8a811309..7e3885b795 100644 --- a/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve_helpers.cuh +++ b/cpp/src/mip_heuristics/presolve/load_balanced_bounds_presolve_helpers.cuh @@ -151,7 +151,7 @@ void calc_activity_heavy_cnst(managed_stream_pool& streams, { if (num_blocks_heavy_cnst != 0) { auto heavy_cnst_stream = streams.get_stream(); - RAFT_CHECK_CUDA(heavy_cnst_stream); + RAFT_CHECK_CUDA(heavy_cnst_stream.get()); // TODO : Check heavy_cnst_block_segments size for profiling if (!dry_run) { auto heavy_cnst_beg_id = get_id_offset(cnst_bin_offsets, heavy_degree_cutoff); @@ -163,18 +163,18 @@ void calc_activity_heavy_cnst(managed_stream_pool& streams, heavy_degree_cutoff, view, tmp_cnst_act); - RAFT_CHECK_CUDA(heavy_cnst_stream); + RAFT_CHECK_CUDA(heavy_cnst_stream.get()); auto num_heavy_cnst = cnst_bin_offsets.back() - heavy_cnst_beg_id; if (erase_inf_cnst) { finalize_calc_act_kernel <<>>( heavy_cnst_beg_id, make_span(heavy_cnst_block_segments), tmp_cnst_act, view); - RAFT_CHECK_CUDA(heavy_cnst_stream); + RAFT_CHECK_CUDA(heavy_cnst_stream.get()); } else { finalize_calc_act_kernel <<>>( heavy_cnst_beg_id, make_span(heavy_cnst_block_segments), tmp_cnst_act, view); - RAFT_CHECK_CUDA(heavy_cnst_stream); + RAFT_CHECK_CUDA(heavy_cnst_stream.get()); } } } @@ -200,11 +200,11 @@ void calc_activity_per_block(managed_stream_pool& streams, if (erase_inf_cnst) { lb_calc_act_block_kernel <<>>(cnst_id_beg, view); - RAFT_CHECK_CUDA(block_stream); + RAFT_CHECK_CUDA(block_stream.get()); } else { lb_calc_act_block_kernel <<>>(cnst_id_beg, view); - RAFT_CHECK_CUDA(block_stream); + RAFT_CHECK_CUDA(block_stream.get()); } } } @@ -261,11 +261,11 @@ void calc_activity_sub_warp(managed_stream_pool& streams, if (erase_inf_cnst) { lb_calc_act_sub_warp_kernel <<>>(cnst_id_beg, cnst_id_end, view); - RAFT_CHECK_CUDA(sub_warp_thread); + RAFT_CHECK_CUDA(sub_warp_thread.get()); } else { lb_calc_act_sub_warp_kernel <<>>(cnst_id_beg, cnst_id_end, view); - RAFT_CHECK_CUDA(sub_warp_thread); + RAFT_CHECK_CUDA(sub_warp_thread.get()); } } } @@ -306,12 +306,12 @@ void calc_activity_sub_warp(managed_stream_pool& streams, lb_calc_act_sub_warp_kernel <<>>( view, make_span(warp_cnst_offsets), make_span(warp_cnst_id_offsets)); - RAFT_CHECK_CUDA(sub_warp_stream); + RAFT_CHECK_CUDA(sub_warp_stream.get()); } else { lb_calc_act_sub_warp_kernel <<>>( view, make_span(warp_cnst_offsets), make_span(warp_cnst_id_offsets)); - RAFT_CHECK_CUDA(sub_warp_stream); + RAFT_CHECK_CUDA(sub_warp_stream.get()); } } } diff --git a/cpp/src/mip_heuristics/presolve/load_balanced_partition_helpers.cuh b/cpp/src/mip_heuristics/presolve/load_balanced_partition_helpers.cuh index 1f2b387cc2..5ad92873c0 100644 --- a/cpp/src/mip_heuristics/presolve/load_balanced_partition_helpers.cuh +++ b/cpp/src/mip_heuristics/presolve/load_balanced_partition_helpers.cuh @@ -271,7 +271,7 @@ log_dist_t vertex_bin_t::run(rmm::device_uvector& reorganized_ver offsets_, vertex_begin_, vertex_end_, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); return log_dist_t(reorganized_vertices, bin_offsets_); } diff --git a/cpp/src/mip_heuristics/presolve/multi_probe.cu b/cpp/src/mip_heuristics/presolve/multi_probe.cu index 394d89f580..1f4bb3f945 100644 --- a/cpp/src/mip_heuristics/presolve/multi_probe.cu +++ b/cpp/src/mip_heuristics/presolve/multi_probe.cu @@ -115,14 +115,14 @@ void multi_probe_t::calculate_activity(problem_t& pb, auto& upd = skip_0 ? upd_1 : upd_0; constexpr auto n_threads = 256; calc_activity_kernel - <<get_stream()>>>(pb.view(), upd.view()); + <<get_stream().get()>>>(pb.view(), upd.view()); } else { constexpr auto n_threads = 256; calc_activity_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( pb.view(), upd_0.view(), upd_1.view()); } - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } template @@ -150,16 +150,16 @@ bool multi_probe_t::calculate_bounds_update(problem_t& pb, } else if (skip_0) { upd_1.bounds_changed.set_value_async(zero, handle_ptr->get_stream()); update_bounds_kernel - <<get_stream()>>>(pb.view(), upd_1.view()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + <<get_stream().get()>>>(pb.view(), upd_1.view()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); i_t h_bounds_changed_1 = upd_1.bounds_changed.value(handle_ptr->get_stream()); CUOPT_LOG_TRACE("Bounds changed upd 1 %d", h_bounds_changed_1); skip_1 = (h_bounds_changed_1 == zero); } else if (skip_1) { upd_0.bounds_changed.set_value_async(zero, handle_ptr->get_stream()); update_bounds_kernel - <<get_stream()>>>(pb.view(), upd_0.view()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + <<get_stream().get()>>>(pb.view(), upd_0.view()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); i_t h_bounds_changed_0 = upd_0.bounds_changed.value(handle_ptr->get_stream()); CUOPT_LOG_TRACE("Bounds changed upd 0 %d", h_bounds_changed_0); skip_0 = (h_bounds_changed_0 == zero); @@ -167,9 +167,9 @@ bool multi_probe_t::calculate_bounds_update(problem_t& pb, upd_0.bounds_changed.set_value_async(zero, handle_ptr->get_stream()); upd_1.bounds_changed.set_value_async(zero, handle_ptr->get_stream()); update_bounds_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( pb.view(), upd_0.view(), upd_1.view()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); i_t h_bounds_changed_0 = upd_0.bounds_changed.value(handle_ptr->get_stream()); CUOPT_LOG_TRACE("Bounds changed upd 0 %d", h_bounds_changed_0); i_t h_bounds_changed_1 = upd_1.bounds_changed.value(handle_ptr->get_stream()); @@ -233,7 +233,7 @@ void multi_probe_t::set_interval_bounds( }); init_changed_constraints = false; handle_ptr->sync_stream(); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } template @@ -262,7 +262,7 @@ void multi_probe_t::set_bounds( upd_1_v.ub[thrust::get<0>(t)] = thrust::get<2>(t); }); handle_ptr->sync_stream(); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } template @@ -465,7 +465,7 @@ void multi_probe_t::constraint_stats(problem_t& pb, thrust::make_tuple(0, 0, 0, 0), tuple_plus_t{}); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); if (redund_constraints_count_0 > 0) { CUOPT_LOG_TRACE("First probe: Redundant constraint count %d", redund_constraints_count_0); diff --git a/cpp/src/mip_heuristics/presolve/probing_cache.cu b/cpp/src/mip_heuristics/presolve/probing_cache.cu index d331f27f80..6f7a081489 100644 --- a/cpp/src/mip_heuristics/presolve/probing_cache.cu +++ b/cpp/src/mip_heuristics/presolve/probing_cache.cu @@ -337,7 +337,7 @@ inline std::vector compute_prioritized_integer_indices( CUOPT_LOG_DEBUG("prioritized integer_indices n_integer_vars %d", problem.n_integer_vars); // compute the min var slack compute_min_slack_per_var - <<get_stream()>>>( + <<get_stream().get()>>>( problem.view(), make_span(bound_presolve.upd.min_activity), make_span(bound_presolve.upd.max_activity), @@ -804,7 +804,7 @@ std::vector compute_priority_indices_by_implied_integers(problem_t{}, 0, - problem.handle_ptr->get_stream()); + problem.handle_ptr->get_stream().get()); rmm::device_uvector temp_storage(temp_storage_bytes, problem.handle_ptr->get_stream()); @@ -820,7 +820,7 @@ std::vector compute_priority_indices_by_implied_integers(problem_t{}, 0, - problem.handle_ptr->get_stream()); + problem.handle_ptr->get_stream().get()); // keeps the count of number of other integers that this variables shares a constraint with rmm::device_uvector count_per_variable(problem.n_variables, problem.handle_ptr->get_stream()); @@ -842,7 +842,7 @@ std::vector compute_priority_indices_by_implied_integers(problem_t{}, 0, - problem.handle_ptr->get_stream()); + problem.handle_ptr->get_stream().get()); temp_storage.resize(temp_storage_bytes, problem.handle_ptr->get_stream()); d_temp_storage = thrust::raw_pointer_cast(temp_storage.data()); @@ -857,7 +857,7 @@ std::vector compute_priority_indices_by_implied_integers(problem_t{}, 0, - problem.handle_ptr->get_stream()); + problem.handle_ptr->get_stream().get()); thrust::for_each(problem.handle_ptr->get_thrust_policy(), thrust::make_counting_iterator(0), thrust::make_counting_iterator(problem.n_variables), diff --git a/cpp/src/mip_heuristics/presolve/third_party_presolve.cpp b/cpp/src/mip_heuristics/presolve/third_party_presolve.cpp index a8f32d3e62..0292c8ff8f 100644 --- a/cpp/src/mip_heuristics/presolve/third_party_presolve.cpp +++ b/cpp/src/mip_heuristics/presolve/third_party_presolve.cpp @@ -1215,7 +1215,7 @@ void third_party_presolve_t::undo_from_device(rmm::device_uvector raft::copy(h_primal.data(), primal_solution.data(), primal_solution.size(), stream_view); raft::copy(h_dual.data(), dual_solution.data(), dual_solution.size(), stream_view); raft::copy(h_rc.data(), reduced_costs.data(), reduced_costs.size(), stream_view); - stream_view.synchronize(); + stream_view.sync(); undo(h_primal, h_dual, h_rc, category, status_to_skip, dual_postsolve); @@ -1225,7 +1225,7 @@ void third_party_presolve_t::undo_from_device(rmm::device_uvector raft::copy(primal_solution.data(), h_primal.data(), h_primal.size(), stream_view); raft::copy(dual_solution.data(), h_dual.data(), h_dual.size(), stream_view); raft::copy(reduced_costs.data(), h_rc.data(), h_rc.size(), stream_view); - stream_view.synchronize(); + stream_view.sync(); } template diff --git a/cpp/src/mip_heuristics/presolve/trivial_presolve.cuh b/cpp/src/mip_heuristics/presolve/trivial_presolve.cuh index 5b4994d325..a126fa4020 100644 --- a/cpp/src/mip_heuristics/presolve/trivial_presolve.cuh +++ b/cpp/src/mip_heuristics/presolve/trivial_presolve.cuh @@ -126,7 +126,7 @@ void update_from_csr(problem_t& pb, bool remap_cache_ids) cnst.end(), cnst.begin(), thrust::maximum{}); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); // partition coo - fixed variables reside in second partition i_t nnz_edge_count = pb.coefficients.size(); @@ -139,7 +139,7 @@ void update_from_csr(problem_t& pb, bool remap_cache_ids) coo_begin + cnst.size(), is_variable_free_t{pb.tolerances.integrality_tolerance, make_span(pb.variable_bounds)}); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); nnz_edge_count = partition_iter - coo_begin; } @@ -154,13 +154,13 @@ void update_from_csr(problem_t& pb, bool remap_cache_ids) thrust::make_constant_iterator(1) + nnz_edge_count, cnst.begin(), cnst_map.begin()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); thrust::scatter(handle_ptr->get_thrust_policy(), thrust::make_constant_iterator(1), thrust::make_constant_iterator(1) + nnz_edge_count, pb.variables.begin(), var_map.begin()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); auto unused_var_count = thrust::count(handle_ptr->get_thrust_policy(), var_map.begin(), var_map.end(), 0); @@ -197,7 +197,7 @@ void update_from_csr(problem_t& pb, bool remap_cache_ids) pb.reverse_original_ids[pb.original_ids[i]] = i; } } - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } if (nnz_edge_count != static_cast(pb.coefficients.size())) { @@ -218,7 +218,7 @@ void update_from_csr(problem_t& pb, bool remap_cache_ids) thrust::make_transform_iterator(thrust::make_counting_iterator(nnz_edge_count), mul), unused_coo_cnst.begin(), unused_coo_cnst_bound_updates.begin()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); auto unused_coo_cnst_count = iter.first - unused_coo_cnst.begin(); unused_coo_cnst.resize(unused_coo_cnst_count, handle_ptr->get_stream()); unused_coo_cnst_bound_updates.resize(unused_coo_cnst_count, handle_ptr->get_stream()); @@ -231,7 +231,7 @@ void update_from_csr(problem_t& pb, bool remap_cache_ids) make_span(unused_coo_cnst_bound_updates), make_span(pb.constraint_lower_bounds), make_span(pb.constraint_upper_bounds)}); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } // update objective_offset @@ -243,7 +243,7 @@ void update_from_csr(problem_t& pb, bool remap_cache_ids) make_span(var_map), make_span(pb.objective_coefficients), make_span(pb.variable_bounds)}, 0., thrust::plus{}); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); // create renumbering maps rmm::device_uvector cnst_renum_ids(pb.n_constraints, handle_ptr->get_stream()); diff --git a/cpp/src/mip_heuristics/problem/load_balanced_problem.cu b/cpp/src/mip_heuristics/problem/load_balanced_problem.cu index 4911a15de8..18be697125 100644 --- a/cpp/src/mip_heuristics/problem/load_balanced_problem.cu +++ b/cpp/src/mip_heuristics/problem/load_balanced_problem.cu @@ -203,19 +203,19 @@ void create_constraint_graph(const raft::handle_t* handle_ptr, handle_ptr->get_thrust_policy(), offsets.begin(), offsets.end(), offsets.begin()); // copy adjacency lists and vertex properties - constraint_data_copy<<get_stream()>>>( + constraint_data_copy<<get_stream().get()>>>( make_span(reorg_ids), make_span(offsets), make_span(coeff), make_span(edge), bounds, pb.view()); if (debug) { rmm::device_scalar errors(zero_v, handle_ptr->get_stream()); check_constraint_data - <<get_stream()>>>(make_span(reorg_ids), - make_span(offsets), - make_span(coeff), - make_span(edge), - bounds, - pb.view(), - errors.data()); + <<get_stream().get()>>>(make_span(reorg_ids), + make_span(offsets), + make_span(coeff), + make_span(edge), + bounds, + pb.view(), + errors.data()); i_t error_count = errors.value(handle_ptr->get_stream()); if (error_count != 0) { std::cerr << "adjacency list copy mismatch\n"; } } @@ -245,25 +245,25 @@ void create_variable_graph(const raft::handle_t* handle_ptr, // copy adjacency lists and vertex properties variable_data_copy - <<get_stream()>>>(make_span(reorg_ids), - make_span(offsets), - make_span(coeff), - make_span(edge), - bounds, - make_span(types), - pb.view()); + <<get_stream().get()>>>(make_span(reorg_ids), + make_span(offsets), + make_span(coeff), + make_span(edge), + bounds, + make_span(types), + pb.view()); if (debug) { rmm::device_scalar errors(zero_v, handle_ptr->get_stream()); check_variable_data - <<get_stream()>>>(make_span(reorg_ids), - make_span(offsets), - make_span(coeff), - make_span(edge), - bounds, - make_span(types), - pb.view(), - errors.data()); + <<get_stream().get()>>>(make_span(reorg_ids), + make_span(offsets), + make_span(coeff), + make_span(edge), + bounds, + make_span(types), + pb.view(), + errors.data()); i_t error_count = errors.value(handle_ptr->get_stream()); if (error_count != 0) { std::cerr << "adjacency list copy mismatch\n"; } } diff --git a/cpp/src/mip_heuristics/problem/problem.cu b/cpp/src/mip_heuristics/problem/problem.cu index 0264147781..411724ede8 100644 --- a/cpp/src/mip_heuristics/problem/problem.cu +++ b/cpp/src/mip_heuristics/problem/problem.cu @@ -454,7 +454,7 @@ void csr_to_csc_transpose(const i_t* csr_offsets, rmm::device_uvector next_pos(n_cols, stream); raft::copy(next_pos.data(), csc_offsets, n_cols, stream); - csr_to_csc_scatter_kernel<<>>( + csr_to_csc_scatter_kernel<<>>( n_rows, csr_offsets, csr_indices, csr_values, next_pos.data(), csc_indices, csc_values); RAFT_CUDA_TRY(cudaPeekAtLastError()); @@ -473,7 +473,7 @@ void csr_to_csc_transpose(const i_t* csr_offsets, n_cols, csc_offsets, csc_offsets + 1, - stream); + stream.get()); rmm::device_uvector temp_storage(temp_storage_bytes, stream); cub::DeviceSegmentedSort::SortPairs(temp_storage.data(), @@ -486,12 +486,12 @@ void csr_to_csc_transpose(const i_t* csr_offsets, n_cols, csc_offsets, csc_offsets + 1, - stream); + stream.get()); // Copy sorted results back raft::copy(csc_indices, row_ind_sorted.data(), nnz, stream); raft::copy(csc_values, val_sorted.data(), nnz, stream); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); + stream.sync(); } template @@ -500,9 +500,10 @@ void problem_t::compute_transpose_of_problem() raft::common::nvtx::range fun_scope("compute_transpose_of_problem"); csrsort_cusparse(coefficients, variables, offsets, n_constraints, n_variables, handle_ptr); RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); // Resize what is needed for LP reverse_offsets.resize(n_variables + 1, handle_ptr->get_stream()); reverse_constraints.resize(nnz, handle_ptr->get_stream()); @@ -1018,7 +1019,7 @@ void problem_t::compute_related_variables(double time_limit) related_variables.size() / (f_t)1e6); thrust::fill(handle_ptr->get_thrust_policy(), varmap.begin(), varmap.end(), 0); - compute_related_vars_unique<<<1024, 128, 0, handle_ptr->get_stream()>>>( + compute_related_vars_unique<<<1024, 128, 0, handle_ptr->get_stream().get()>>>( pb_view, slice_begin, slice_end, make_span(varmap)); // prefix sum to generate offsets @@ -1508,7 +1509,7 @@ void problem_t::substitute_variables(const std::vector& var_indic offsets.data() + 1, cuda::std::plus<>{}, initial_value, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); rmm::device_uvector temp_storage(temp_storage_bytes, handle_ptr->get_stream()); d_temp_storage = thrust::raw_pointer_cast(temp_storage.data()); @@ -1523,8 +1524,8 @@ void problem_t::substitute_variables(const std::vector& var_indic offsets.data() + 1, cuda::std::plus<>{}, initial_value, - handle_ptr->get_stream()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + handle_ptr->get_stream().get()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); thrust::for_each( handle_ptr->get_thrust_policy(), thrust::make_counting_iterator(0), @@ -1632,7 +1633,7 @@ void problem_t::fix_given_variables(problem_t& original_prob original_problem.offsets.data() + 1, cuda::std::plus<>{}, initial_value, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); rmm::device_uvector temp_storage(temp_storage_bytes, handle_ptr->get_stream()); d_temp_storage = thrust::raw_pointer_cast(temp_storage.data()); @@ -1647,8 +1648,8 @@ void problem_t::fix_given_variables(problem_t& original_prob original_problem.offsets.data() + 1, cuda::std::plus<>{}, initial_value, - handle_ptr->get_stream()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + handle_ptr->get_stream().get()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); thrust::for_each( handle_ptr->get_thrust_policy(), thrust::make_counting_iterator(0), @@ -1687,7 +1688,7 @@ problem_t problem_t::get_problem_after_fixing_vars( variable_map.resize(assignment.size() - variables_to_fix.size(), handle_ptr->get_stream()); // compute variable map to recover the assignment later // get the variable indices to gather - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); cuopt_assert( (thrust::is_sorted( handle_ptr->get_thrust_policy(), variables_to_fix.begin(), variables_to_fix.end())), @@ -1699,14 +1700,14 @@ problem_t problem_t::get_problem_after_fixing_vars( variables_to_fix.begin(), variables_to_fix.end(), variable_map.begin()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); cuopt_assert(result_end - variable_map.data() == variable_map.size(), "Size issue in set_difference"); CUOPT_LOG_DEBUG("Fixing assignment hash 0x%x, vars to fix: 0x%x", mip::compute_hash(assignment, handle_ptr->get_stream()), mip::compute_hash(variables_to_fix, handle_ptr->get_stream())); problem.fix_given_variables(*this, assignment, variables_to_fix, handle_ptr); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); problem.remove_given_variables(*this, assignment, variable_map, handle_ptr); // if we are fixing on the original problem, the variable_map is what we want in // problem.original_ids but considering the case that we are fixing some variables multiple times, @@ -1721,7 +1722,7 @@ problem_t problem_t::get_problem_after_fixing_vars( "Variable index out of bounds"); problem.reverse_original_ids[original_ids[h_variable_map[i]]] = i; } - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); auto end_time = std::chrono::high_resolution_clock::now(); double time_taken = std::chrono::duration_cast(end_time - start_time).count(); @@ -1790,9 +1791,9 @@ void problem_t::remove_given_variables(problem_t& original_p presolve_data.var_flags.resize(variable_map.size(), handle_ptr->get_stream()); const i_t TPB = 64; // compute new offsets - compute_new_offsets<<get_stream()>>>( + compute_new_offsets<<get_stream().get()>>>( original_problem.view(), view(), cuopt::make_span(variable_map)); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); thrust::exclusive_scan(handle_ptr->get_thrust_policy(), offsets.data(), offsets.data() + offsets.size(), @@ -1800,9 +1801,9 @@ void problem_t::remove_given_variables(problem_t& original_p rmm::device_uvector write_pos(n_constraints, handle_ptr->get_stream()); thrust::fill(handle_ptr->get_thrust_policy(), write_pos.begin(), write_pos.end(), 0); // compute new csr - compute_new_csr<<get_stream()>>>( + compute_new_csr<<get_stream().get()>>>( original_problem.view(), view(), cuopt::make_span(variable_map), cuopt::make_span(write_pos)); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); // assign nnz, number of variables etc. nnz = offsets.back_element(handle_ptr->get_stream()); n_variables = variable_map.size(); @@ -2150,7 +2151,7 @@ void problem_t::set_constraints_from_host_csr(const std::vector& thrust::fill( handle_ptr->get_thrust_policy(), lp_state.prev_dual.begin(), lp_state.prev_dual.end(), f_t{0}); handle_ptr->sync_stream(); - RAFT_CHECK_CUDA(stream); + RAFT_CHECK_CUDA(stream.get()); compute_transpose_of_problem(); combined_bounds.resize(n_constraints, stream); @@ -2409,7 +2410,7 @@ void problem_t::update_variable_bounds(const std::vector& var_ind variable_bounds[var_idx].y = ub_values[i]; }); handle_ptr->sync_stream(); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } #if MIP_INSTANTIATE_FLOAT || PDLP_INSTANTIATE_FLOAT diff --git a/cpp/src/mip_heuristics/problem/problem_helpers.cuh b/cpp/src/mip_heuristics/problem/problem_helpers.cuh index 388fae4ecd..00093330ca 100644 --- a/cpp/src/mip_heuristics/problem/problem_helpers.cuh +++ b/cpp/src/mip_heuristics/problem/problem_helpers.cuh @@ -143,7 +143,7 @@ static void convert_to_maximization_problem(mip::problem_t& op_problem op_problem.objective_coefficients.data(), op_problem.objective_coefficients.size(), mip::negate(), - op_problem.handle_ptr->get_stream()); + op_problem.handle_ptr->get_stream().get()); } // Negate objective scaling factor and objective offset so that primal / dual stay same sign after // negating objective coeffs @@ -219,7 +219,7 @@ static bool check_transpose_validity(const rmm::device_uvector& coefficient rmm::device_scalar failed(false_v, handle_ptr->get_stream()); kernel_check_transpose_validity - <<get_stream()>>>( + <<get_stream().get()>>>( raft::device_span(coefficients.data(), coefficients.size()), raft::device_span(offsets.data(), offsets.size()), raft::device_span(variables.data(), variables.size()), @@ -366,7 +366,7 @@ static void csrsort_cusparse(rmm::device_uvector& values, auto stream = offsets.stream(); cusparseHandle_t handle; cusparseCreate(&handle); - cusparseSetStream(handle, stream); + cusparseSetStream(handle, stream.get()); i_t nnz = values.size(); i_t m = rows; @@ -411,14 +411,14 @@ static void convert_greater_to_less(mip::problem_t& problem) constexpr i_t TPB = 256; kernel_convert_greater_to_less - <<get_stream()>>>( + <<get_stream().get()>>>( raft::device_span(problem.coefficients.data(), problem.coefficients.size()), raft::device_span(problem.offsets.data(), problem.offsets.size()), raft::device_span(problem.constraint_lower_bounds.data(), problem.constraint_lower_bounds.size()), raft::device_span(problem.constraint_upper_bounds.data(), problem.constraint_upper_bounds.size())); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); problem.compute_transpose_of_problem(); diff --git a/cpp/src/mip_heuristics/solution/feasibility_test.cuh b/cpp/src/mip_heuristics/solution/feasibility_test.cuh index 140603c763..39170798ab 100644 --- a/cpp/src/mip_heuristics/solution/feasibility_test.cuh +++ b/cpp/src/mip_heuristics/solution/feasibility_test.cuh @@ -77,7 +77,7 @@ void solution_t::test_feasibility(bool check_integer) cuopt_assert(compute_feasibility(), "Solution is not feasible!"); test_variable_bounds(check_integer); handle_ptr->sync_stream(); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } // test feasibility on @@ -86,8 +86,8 @@ void solution_t::test_absolute_feasibility() { i_t TPB = 64; i_t n_blocks = (problem_ptr->n_constraints + TPB - 1) / TPB; - test_feasibility_kernel<<get_stream()>>>(view()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + test_feasibility_kernel<<get_stream().get()>>>(view()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } template @@ -96,8 +96,8 @@ void solution_t::test_variable_bounds(bool check_integer, i_t* is_feas i_t TPB = 64; i_t n_blocks = (problem_ptr->n_variables + TPB - 1) / TPB; test_variable_bounds_kernel - <<get_stream()>>>(view(), check_integer, is_feasible); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + <<get_stream().get()>>>(view(), check_integer, is_feasible); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } } // namespace cuopt::mathematical_optimization::mip diff --git a/cpp/src/mip_heuristics/solution/solution.cu b/cpp/src/mip_heuristics/solution/solution.cu index 64cd156747..0d411a0ffc 100644 --- a/cpp/src/mip_heuristics/solution/solution.cu +++ b/cpp/src/mip_heuristics/solution/solution.cu @@ -296,8 +296,8 @@ void solution_t::compute_constraints() i_t TPB = 64; compute_constraint_values - <<n_constraints, TPB, 0, handle_ptr->get_stream()>>>(view()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + <<n_constraints, TPB, 0, handle_ptr->get_stream().get()>>>(view()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } template @@ -313,11 +313,12 @@ f_t solution_t::compute_l2_residual() upper_excess.data(), problem_ptr->n_constraints, [] __device__(f_t lower, f_t upper) -> f_t { return max(abs(lower), abs(upper)); }, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); pdlp::my_l2_norm(combined_excess, l2_residual, handle_ptr); return l2_residual.value(handle_ptr->get_stream()); } diff --git a/cpp/src/mip_heuristics/solve.cu b/cpp/src/mip_heuristics/solve.cu index 709ddc45b0..7f7fee22fc 100644 --- a/cpp/src/mip_heuristics/solve.cu +++ b/cpp/src/mip_heuristics/solve.cu @@ -78,9 +78,10 @@ static void init_handler(const raft::handle_t* handle_ptr) { // Init cuBlas / cuSparse context here to avoid having it during solving time RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } template @@ -948,7 +949,7 @@ std::unique_ptr> solve_mip( auto gpu_solution = solve_mip(*gpu_problem, settings); // Ensure all GPU work from the solve is complete before D2H copies in to_cpu_solution(), - // which uses rmm::cuda_stream_per_thread (a different stream than the solver used). + // which uses the per-thread default stream (a different stream than the solver used). stream.synchronize(); // Convert GPU solution back to CPU diff --git a/cpp/src/mip_heuristics/solver.cu b/cpp/src/mip_heuristics/solver.cu index 645de93cd6..a5201aa078 100644 --- a/cpp/src/mip_heuristics/solver.cu +++ b/cpp/src/mip_heuristics/solver.cu @@ -45,9 +45,10 @@ static void init_handler(const raft::handle_t* handle_ptr) { // Init cuBlas / cuSparse context here to avoid having it during solving time RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } template diff --git a/cpp/src/mip_heuristics/solver_solution.cu b/cpp/src/mip_heuristics/solver_solution.cu index 1997d684dc..8e89829fe3 100644 --- a/cpp/src/mip_heuristics/solver_solution.cu +++ b/cpp/src/mip_heuristics/solver_solution.cu @@ -215,8 +215,8 @@ void mip_solution_t::write_to_sol_file(std::string_view filename, auto& var_names = get_variable_names(); std::vector solution; solution.resize(solution_.size()); - raft::copy(solution.data(), solution_.data(), solution_.size(), stream_view.value()); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view.value())); + raft::copy(solution.data(), solution_.data(), solution_.size(), stream_view.get()); + stream_view.sync(); solution_writer_t::write_solution_to_sol_file( std::string(filename), status, objective_value, var_names, solution); diff --git a/cpp/src/mip_heuristics/utils.cuh b/cpp/src/mip_heuristics/utils.cuh index faf4718a5e..76e9ce8a3d 100644 --- a/cpp/src/mip_heuristics/utils.cuh +++ b/cpp/src/mip_heuristics/utils.cuh @@ -34,7 +34,7 @@ template inline uint32_t compute_hash(raft::device_span values, rmm::cuda_stream_view stream) { auto h_contents = cuopt::host_copy(values, stream); - RAFT_CHECK_CUDA(stream); + RAFT_CHECK_CUDA(stream.get()); return cuopt::compute_hash(h_contents); } @@ -42,7 +42,7 @@ template inline uint32_t compute_hash(const rmm::device_uvector& values, rmm::cuda_stream_view stream) { auto h_contents = cuopt::host_copy(values, stream); - RAFT_CHECK_CUDA(stream); + RAFT_CHECK_CUDA(stream.get()); return cuopt::compute_hash(h_contents); } @@ -333,7 +333,7 @@ static __global__ void run_lambda_kernel(F f) template static void inline run_device_lambda(const rmm::cuda_stream_view& stream, Func f) { - run_lambda_kernel<<<1, 1, 0, stream.value()>>>(f); + run_lambda_kernel<<<1, 1, 0, stream.get()>>>(f); } template diff --git a/cpp/src/pdlp/cpu_pdlp_warm_start_data.cu b/cpp/src/pdlp/cpu_pdlp_warm_start_data.cu index f2af28ba51..604c7b4377 100644 --- a/cpp/src/pdlp/cpu_pdlp_warm_start_data.cu +++ b/cpp/src/pdlp/cpu_pdlp_warm_start_data.cu @@ -22,7 +22,7 @@ std::vector device_to_host_vector(const rmm::device_uvector& device_vec, std::vector host_vec(device_vec.size()); raft::copy(host_vec.data(), device_vec.data(), device_vec.size(), stream); - stream.synchronize(); + stream.sync(); return host_vec; } @@ -35,7 +35,7 @@ rmm::device_uvector host_to_device_vector(const std::vector& host_vec, rmm::device_uvector device_vec(host_vec.size(), stream); raft::copy(device_vec.data(), host_vec.data(), host_vec.size(), stream); - stream.synchronize(); + stream.sync(); return device_vec; } diff --git a/cpp/src/pdlp/cuopt_c_internal.hpp b/cpp/src/pdlp/cuopt_c_internal.hpp index 338e998df2..104dbc07af 100644 --- a/cpp/src/pdlp/cuopt_c_internal.hpp +++ b/cpp/src/pdlp/cuopt_c_internal.hpp @@ -15,6 +15,8 @@ #include #include +#include + #include #include @@ -30,7 +32,7 @@ struct problem_and_stream_view_t { if (mem_backend == memory_backend_t::GPU) { // Use RAII locals so partial allocations are cleaned up if a later new throws std::unique_ptr sv( - new rmm::cuda_stream_view(rmm::cuda_stream_per_thread)); + new rmm::cuda_stream_view(cuda::stream_ref{cudaStreamPerThread})); std::unique_ptr h(new raft::handle_t(*sv)); std::unique_ptr> gp( new optimization_problem_t(h.get())); diff --git a/cpp/src/pdlp/cusparse_view.cu b/cpp/src/pdlp/cusparse_view.cu index d0802ae0b0..7c63c87c91 100644 --- a/cpp/src/pdlp/cusparse_view.cu +++ b/cpp/src/pdlp/cusparse_view.cu @@ -439,7 +439,7 @@ cusparse_view_t::cusparse_view_t( dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_non_transpose, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_non_transpose.resize(buffer_size_non_transpose, handle_ptr->get_stream()); size_t buffer_size_transpose = 0; @@ -453,7 +453,7 @@ cusparse_view_t::cusparse_view_t( c.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_transpose, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_transpose.resize(buffer_size_transpose, handle_ptr->get_stream()); @@ -470,7 +470,7 @@ cusparse_view_t::cusparse_view_t( batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, &buffer_size_transpose_batch, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_transpose_batch.resize(buffer_size_transpose_batch, handle_ptr->get_stream()); size_t buffer_size_non_transpose_batch = 0; @@ -485,7 +485,7 @@ cusparse_view_t::cusparse_view_t( batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, &buffer_size_non_transpose_batch, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_non_transpose_batch.resize(buffer_size_non_transpose_batch, handle_ptr->get_stream()); // In row row the buffer size may be different @@ -502,7 +502,7 @@ cusparse_view_t::cusparse_view_t( batch_current_AtYs.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &buffer_size_transpose_batch_row_row, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_transpose_batch_row_row_.resize(buffer_size_transpose_batch_row_row, handle_ptr->get_stream()); size_t buffer_size_non_transpose_batch_row_row = 0; @@ -517,7 +517,7 @@ cusparse_view_t::cusparse_view_t( batch_dual_gradients.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &buffer_size_non_transpose_batch_row_row, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_non_transpose_batch_row_row_.resize(buffer_size_non_transpose_batch_row_row, handle_ptr->get_stream()); } @@ -532,7 +532,7 @@ cusparse_view_t::cusparse_view_t( dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_non_transpose.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); my_cusparsespmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, @@ -543,7 +543,7 @@ cusparse_view_t::cusparse_view_t( c.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_transpose.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); my_cusparsespmm_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, CUSPARSE_OPERATION_NON_TRANSPOSE, @@ -554,7 +554,7 @@ cusparse_view_t::cusparse_view_t( batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, buffer_transpose_batch.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); my_cusparsespmm_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, @@ -566,7 +566,7 @@ cusparse_view_t::cusparse_view_t( batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, buffer_non_transpose_batch.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); if (batch_mode_) { my_cusparsespmm_preprocess( handle_ptr_->get_cusparse_handle(), @@ -579,7 +579,7 @@ cusparse_view_t::cusparse_view_t( batch_current_AtYs.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, buffer_transpose_batch_row_row_.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); my_cusparsespmm_preprocess( handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, @@ -591,7 +591,7 @@ cusparse_view_t::cusparse_view_t( batch_dual_gradients.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, buffer_non_transpose_batch_row_row_.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); } #endif @@ -606,13 +606,13 @@ cusparse_view_t::cusparse_view_t( A_float_.data(), op_problem_scaled.nnz, double_to_float_functor{}, - handle_ptr->get_stream().value())); + handle_ptr->get_stream().get())); RAFT_CUDA_TRY(cub::DeviceTransform::Transform(A_T_.data(), A_T_float_.data(), op_problem_scaled.nnz, double_to_float_functor{}, - handle_ptr->get_stream().value())); + handle_ptr->get_stream().get())); A_mixed_ = make_csr(op_problem_scaled.n_constraints, op_problem_scaled.n_variables, @@ -639,7 +639,7 @@ cusparse_view_t::cusparse_view_t( beta_d.data(), dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); buffer_non_transpose_mixed_.resize(buffer_size_non_transpose_mixed, handle_ptr->get_stream()); size_t buffer_size_transpose_mixed = @@ -651,7 +651,7 @@ cusparse_view_t::cusparse_view_t( beta_d.data(), c.get(), CUSPARSE_SPMV_CSR_ALG2, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); buffer_transpose_mixed_.resize(buffer_size_transpose_mixed, handle_ptr->get_stream()); #if CUDA_VER_12_4_UP @@ -664,7 +664,7 @@ cusparse_view_t::cusparse_view_t( dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_non_transpose_mixed_.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); mixed_precision_spmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, @@ -675,7 +675,7 @@ cusparse_view_t::cusparse_view_t( c.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_transpose_mixed_.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); #endif } } @@ -726,8 +726,9 @@ cusparse_view_t::cusparse_view_t( std::cout << "PDLP cusparse view init" << std::endl; #endif - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr_->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr_->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); // setup cusparse view A = make_csr(op_problem.n_constraints, @@ -796,7 +797,7 @@ cusparse_view_t::cusparse_view_t( dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_non_transpose, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_non_transpose.resize(buffer_size_non_transpose, handle_ptr->get_stream()); size_t buffer_size_transpose = 0; @@ -810,7 +811,7 @@ cusparse_view_t::cusparse_view_t( c.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_transpose, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_transpose.resize(buffer_size_transpose, handle_ptr->get_stream()); @@ -827,7 +828,7 @@ cusparse_view_t::cusparse_view_t( batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, &buffer_size_transpose_batch, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_transpose_batch.resize(buffer_size_transpose_batch, handle_ptr->get_stream()); size_t buffer_size_non_transpose_batch = 0; RAFT_CUSPARSE_TRY( @@ -841,7 +842,7 @@ cusparse_view_t::cusparse_view_t( batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, &buffer_size_non_transpose_batch, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_non_transpose_batch.resize(buffer_size_non_transpose_batch, handle_ptr->get_stream()); } @@ -855,7 +856,7 @@ cusparse_view_t::cusparse_view_t( dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_non_transpose.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); my_cusparsespmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, @@ -866,7 +867,7 @@ cusparse_view_t::cusparse_view_t( c.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_transpose.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); if (batch_mode_) { my_cusparsespmm_preprocess(handle_ptr_->get_cusparse_handle(), @@ -879,7 +880,7 @@ cusparse_view_t::cusparse_view_t( batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, buffer_non_transpose_batch.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); my_cusparsespmm_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, @@ -891,7 +892,7 @@ cusparse_view_t::cusparse_view_t( batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, buffer_transpose_batch.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); } #endif } @@ -934,8 +935,9 @@ cusparse_view_t::cusparse_view_t( std::cout << "Restart Strategy cusparse view init" << std::endl; #endif - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr_->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr_->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); // Need to reinstanciate the cuSparse views // Copying them from the existing cuSparse view is a bad practice and creates segfault post @@ -987,7 +989,7 @@ cusparse_view_t::cusparse_view_t( dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_non_transpose, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_non_transpose.resize(buffer_size_non_transpose, handle_ptr->get_stream()); size_t buffer_size_transpose = 0; @@ -1001,7 +1003,7 @@ cusparse_view_t::cusparse_view_t( c.get(), CUSPARSE_SPMV_CSR_ALG2, &buffer_size_transpose, - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); buffer_transpose.resize(buffer_size_transpose, handle_ptr->get_stream()); @@ -1015,7 +1017,7 @@ cusparse_view_t::cusparse_view_t( dual_solution.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_non_transpose.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); my_cusparsespmv_preprocess(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, @@ -1026,7 +1028,7 @@ cusparse_view_t::cusparse_view_t( c.get(), CUSPARSE_SPMV_CSR_ALG2, buffer_transpose.data(), - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); #endif } @@ -1072,15 +1074,15 @@ void cusparse_view_t::update_mixed_precision_matrices() A_float_.data(), A_.size(), double_to_float_functor{}, - handle_ptr_->get_stream().value())); + handle_ptr_->get_stream().get())); RAFT_CUDA_TRY(cub::DeviceTransform::Transform(A_T_.data(), A_T_float_.data(), A_T_.size(), double_to_float_functor{}, - handle_ptr_->get_stream().value())); + handle_ptr_->get_stream().get())); - handle_ptr_->get_stream().synchronize(); + handle_ptr_->get_stream().sync(); } } @@ -1202,7 +1204,7 @@ void cusparse_view_t::create_spmv_op_plans(bool is_reflected) #if CUOPT_CUSPARSE_VER_12_8_UP if (!is_cusparse_runtime_spmvop_supported() || !(std::is_same_v)) { return; } RAFT_CUSPARSE_TRY( - cusparseSetStream(handle_ptr_->get_cusparse_handle(), handle_ptr_->get_stream())); + cusparseSetStream(handle_ptr_->get_cusparse_handle(), handle_ptr_->get_stream().get())); // Prepare buffers for At_y SpMVOp size_t buffer_size_transpose = 0; RAFT_CUSPARSE_TRY(cusparse_spmvop_buffer_size(handle_ptr_->get_cusparse_handle(), diff --git a/cpp/src/pdlp/cusparse_view.hpp b/cpp/src/pdlp/cusparse_view.hpp index 0f109c1c0e..67f3d0c01c 100644 --- a/cpp/src/pdlp/cusparse_view.hpp +++ b/cpp/src/pdlp/cusparse_view.hpp @@ -12,7 +12,6 @@ #include -#include #include #include diff --git a/cpp/src/pdlp/distributed_pdlp/distributed_algorithms.cu b/cpp/src/pdlp/distributed_pdlp/distributed_algorithms.cu index 75646e7ce2..ba141b413f 100644 --- a/cpp/src/pdlp/distributed_pdlp/distributed_algorithms.cu +++ b/cpp/src/pdlp/distributed_pdlp/distributed_algorithms.cu @@ -300,7 +300,7 @@ f_t multi_gpu_engine_t::distributed_max_singular_value_squared(i_t n_g q[r].data(), n_owned, divide_by_device_scalar_t{norm_q[r].data()}, - s.stream.view().value()); + s.stream.view().get()); }); // atq = A^T q (fused halo-refresh of q + per-shard local SpMV). @@ -320,7 +320,7 @@ f_t multi_gpu_engine_t::distributed_max_singular_value_squared(i_t n_g q[r].data(), n_owned, residual_fma_neg_scalar_t{sigma_sq[r].data()}, - s.stream.view().value()); + s.stream.view().get()); }); // Convergence check via global residual norm. diff --git a/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.cu b/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.cu index 050c360e0c..70f32bc5b0 100644 --- a/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.cu +++ b/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.cu @@ -185,7 +185,7 @@ void multi_gpu_engine_t::halo_exchange_bufs_impl( nccl_data_type(), peer, s.comm.get(), - s.stream.view().value())); + s.stream.view().get())); } }); for_each_shard([&](auto& s, int r) { @@ -199,7 +199,7 @@ void multi_gpu_engine_t::halo_exchange_bufs_impl( nccl_data_type(), peer, s.comm.get(), - s.stream.view().value())); + s.stream.view().get())); } }); CUOPT_NCCL_TRY(ncclGroupEnd()); @@ -317,7 +317,7 @@ void multi_gpu_engine_t::allreduce_sum_inplace_bufs( nccl_data_type(), ncclSum, s.comm.get(), - s.stream.view().value())); + s.stream.view().get())); }); CUOPT_NCCL_TRY(ncclGroupEnd()); } @@ -363,7 +363,7 @@ void multi_gpu_engine_t::distributed_dot_bufs( b_bufs[r].data(), 1, out_scalars[r].data_handle(), - s.stream.view().value())); + s.stream.view().get())); }); allreduce_sum_inplace_bufs(out_scalars); @@ -394,7 +394,7 @@ void multi_gpu_engine_t::distributed_l2_norm_bufs( out_scalars[r].data_handle(), 1, [] __device__(f_t x) { return cuda::std::sqrt(x); }, - s.stream.view().value()); + s.stream.view().get()); }); } diff --git a/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.hpp b/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.hpp index fa77c3aed2..d3896f4ce7 100644 --- a/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.hpp +++ b/cpp/src/pdlp/distributed_pdlp/multi_gpu_engine.hpp @@ -140,7 +140,7 @@ struct multi_gpu_engine_t { "distributed_transform_bufs: in_tuples / outs / sizes must " "all have size == shards.size()"); for_each_shard([&](auto& s, int r) { - cub::DeviceTransform::Transform(in_tuples[r], outs[r], sizes[r], op, s.stream.view()); + cub::DeviceTransform::Transform(in_tuples[r], outs[r], sizes[r], op, s.stream.view().get()); }); } diff --git a/cpp/src/pdlp/initial_scaling_strategy/initial_scaling.cu b/cpp/src/pdlp/initial_scaling_strategy/initial_scaling.cu index 5925ec9aea..28f8b47257 100644 --- a/cpp/src/pdlp/initial_scaling_strategy/initial_scaling.cu +++ b/cpp/src/pdlp/initial_scaling_strategy/initial_scaling.cu @@ -101,10 +101,12 @@ pdlp_initial_scaling_strategy_t::pdlp_initial_scaling_strategy_t( cuopt_assert(original_batch_size_ > 0, "Original batch size must be positive"); // start with all one for scaling vectors + RAFT_CUDA_TRY(cudaMemsetAsync(iteration_constraint_matrix_scaling_.data(), + 0.0, + sizeof(f_t) * dual_size_h_, + stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync( - iteration_constraint_matrix_scaling_.data(), 0.0, sizeof(f_t) * dual_size_h_, stream_view_)); - RAFT_CUDA_TRY(cudaMemsetAsync( - iteration_variable_scaling_.data(), 0.0, sizeof(f_t) * primal_size_h_, stream_view_)); + iteration_variable_scaling_.data(), 0.0, sizeof(f_t) * primal_size_h_, stream_view_.get())); thrust::fill(handle_ptr_->get_thrust_policy(), cummulative_constraint_matrix_scaling_.begin(), cummulative_constraint_matrix_scaling_.end(), @@ -231,10 +233,12 @@ template void pdlp_initial_scaling_strategy_t::ruiz_iter_local() { // Reset the iteration_scaling vectors to all 0 + RAFT_CUDA_TRY(cudaMemsetAsync(iteration_constraint_matrix_scaling_.data(), + 0, + sizeof(f_t) * dual_size_h_, + stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync( - iteration_constraint_matrix_scaling_.data(), 0, sizeof(f_t) * dual_size_h_, stream_view_)); - RAFT_CUDA_TRY(cudaMemsetAsync( - iteration_variable_scaling_.data(), 0, sizeof(f_t) * primal_size_h_, stream_view_)); + iteration_variable_scaling_.data(), 0, sizeof(f_t) * primal_size_h_, stream_view_.get())); // Inf-norm over rows and columns. Split into two kernels so the distributed path can // touch only owned entries. @@ -243,15 +247,20 @@ void pdlp_initial_scaling_strategy_t::ruiz_iter_local() i_t number_of_blocks = op_problem_scaled_.n_constraints / block_size; if (op_problem_scaled_.n_constraints % block_size) number_of_blocks++; i_t number_of_threads = std::min(op_problem_scaled_.n_variables, (i_t)block_size); - inf_norm_row_kernel<<>>( + inf_norm_row_kernel<<>>( op_problem_scaled_.view(), this->view()); RAFT_CUDA_TRY(cudaPeekAtLastError()); i_t number_of_blocks_col = op_problem_scaled_.n_variables / block_size; if (op_problem_scaled_.n_variables % block_size) number_of_blocks_col++; i_t number_of_threads_col = std::min(op_problem_scaled_.n_constraints, (i_t)block_size); - inf_norm_col_kernel<<>>( - op_problem_scaled_.view(), this->view(), A_T_.data(), A_T_offsets_.data(), A_T_indices_.data()); + inf_norm_col_kernel + <<>>( + op_problem_scaled_.view(), + this->view(), + A_T_.data(), + A_T_offsets_.data(), + A_T_indices_.data()); RAFT_CUDA_TRY(cudaPeekAtLastError()); if (running_mip_) { reset_integer_variables(); } @@ -263,14 +272,14 @@ void pdlp_initial_scaling_strategy_t::ruiz_iter_local() iteration_constraint_matrix_scaling_.data(), dual_size_h_, a_divides_sqrt_b_bounded(), - stream_view_); + stream_view_.get()); raft::linalg::binaryOp(cummulative_variable_scaling_.data(), cummulative_variable_scaling_.data(), iteration_variable_scaling_.data(), primal_size_h_, a_divides_sqrt_b_bounded(), - stream_view_); + stream_view_.get()); } template @@ -378,10 +387,12 @@ template void pdlp_initial_scaling_strategy_t::pock_chambolle_scaling(f_t alpha) { // Reset the iteration_scaling vectors to all 0 + RAFT_CUDA_TRY(cudaMemsetAsync(iteration_constraint_matrix_scaling_.data(), + 0.0, + sizeof(f_t) * dual_size_h_, + stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync( - iteration_constraint_matrix_scaling_.data(), 0.0, sizeof(f_t) * dual_size_h_, stream_view_)); - RAFT_CUDA_TRY(cudaMemsetAsync( - iteration_variable_scaling_.data(), 0.0, sizeof(f_t) * primal_size_h_, stream_view_)); + iteration_variable_scaling_.data(), 0.0, sizeof(f_t) * primal_size_h_, stream_view_.get())); EXE_CUOPT_EXPECTS( alpha >= 0.0 && alpha <= 2.0, @@ -396,13 +407,13 @@ void pdlp_initial_scaling_strategy_t::pock_chambolle_scaling(f_t alpha constexpr i_t number_of_threads = 128; pock_chambolle_scaling_kernel_row - <<>>( + <<>>( op_problem_scaled_.view(), alpha, this->view()); RAFT_CUDA_TRY(cudaPeekAtLastError()); // Use transposed matrix instead to compute column-wise more easily pock_chambolle_scaling_kernel_col - <<>>( + <<>>( op_problem_scaled_.view(), alpha, this->view(), @@ -420,13 +431,13 @@ void pdlp_initial_scaling_strategy_t::pock_chambolle_scaling(f_t alpha iteration_constraint_matrix_scaling_.data(), dual_size_h_, a_divides_sqrt_b_bounded(), - stream_view_); + stream_view_.get()); raft::linalg::binaryOp(cummulative_variable_scaling_.data(), cummulative_variable_scaling_.data(), iteration_variable_scaling_.data(), primal_size_h_, a_divides_sqrt_b_bounded(), - stream_view_); + stream_view_.get()); } template @@ -516,10 +527,10 @@ void pdlp_initial_scaling_strategy_t::swap_context( const auto [grid_size, block_size] = kernel_config_from_batch_size(static_cast(swap_pairs.size())); scaling_swap_rescaling_kernel - <<>>(thrust::raw_pointer_cast(swap_pairs.data()), - static_cast(swap_pairs.size()), - make_span(bound_rescaling_), - make_span(objective_rescaling_)); + <<>>(thrust::raw_pointer_cast(swap_pairs.data()), + static_cast(swap_pairs.size()), + make_span(bound_rescaling_), + make_span(objective_rescaling_)); RAFT_CUDA_TRY(cudaPeekAtLastError()); for (const auto& pair : swap_pairs) { @@ -553,7 +564,7 @@ void pdlp_initial_scaling_strategy_t::apply_cummulative_scaling_to_pro i_t number_of_blocks = op_problem_scaled_.n_constraints / block_size; if (op_problem_scaled_.n_constraints % block_size) number_of_blocks++; i_t number_of_threads = std::min(op_problem_scaled_.n_variables, block_size); - scale_problem_kernel<<>>( + scale_problem_kernel<<>>( this->view(), op_problem_scaled_.view()); RAFT_CUDA_TRY(cudaPeekAtLastError()); @@ -563,7 +574,7 @@ void pdlp_initial_scaling_strategy_t::apply_cummulative_scaling_to_pro i_t number_of_threads_transposed = std::min(op_problem_scaled_.n_constraints, block_size); scale_transposed_problem_kernel - <<>>( + <<>>( this->view(), A_T_.data(), A_T_offsets_.data(), A_T_indices_.data()); RAFT_CUDA_TRY(cudaPeekAtLastError()); @@ -574,7 +585,7 @@ void pdlp_initial_scaling_strategy_t::apply_cummulative_scaling_to_pro op_problem_scaled_.objective_coefficients.data(), op_problem_scaled_.objective_coefficients.size(), cuda::std::multiplies{}, - stream_view_); + stream_view_.get()); using f_t2 = typename type_2::type; cub::DeviceTransform::Transform( @@ -583,7 +594,7 @@ void pdlp_initial_scaling_strategy_t::apply_cummulative_scaling_to_pro op_problem_scaled_.variable_bounds.data(), op_problem_scaled_.variable_bounds.size(), divide_check_zero(), - stream_view_.value()); + stream_view_.get()); if (pdhg_solver_ptr_ && pdhg_solver_ptr_->get_new_bounds_idx().size() != 0) { cub::DeviceTransform::Transform( @@ -599,7 +610,7 @@ void pdlp_initial_scaling_strategy_t::apply_cummulative_scaling_to_pro if (s != f_t(0)) { return {lower / s, upper / s}; } return {lower, upper}; }, - stream_view_); + stream_view_.get()); } cub::DeviceTransform::Transform( @@ -608,14 +619,14 @@ void pdlp_initial_scaling_strategy_t::apply_cummulative_scaling_to_pro op_problem_scaled_.constraint_lower_bounds.data(), op_problem_scaled_.constraint_lower_bounds.size(), cuda::std::multiplies{}, - stream_view_); + stream_view_.get()); cub::DeviceTransform::Transform( cuda::std::make_tuple(op_problem_scaled_.constraint_upper_bounds.data(), problem_wrap_container(cummulative_constraint_matrix_scaling_)), op_problem_scaled_.constraint_upper_bounds.data(), op_problem_scaled_.constraint_upper_bounds.size(), cuda::std::multiplies{}, - stream_view_); + stream_view_.get()); #ifdef CUPDLP_DEBUG_MODE print("constraint_lower_bound", op_problem_scaled_.constraint_lower_bounds); @@ -662,7 +673,7 @@ void pdlp_initial_scaling_strategy_t::apply_bound_objective_rescaling_ f_t bound_rescaling) -> thrust::tuple { return {constraint_lower_bound * bound_rescaling, constraint_upper_bound * bound_rescaling}; }, - stream_view_.value()); + stream_view_.get()); // In batch mode we don't scale the variable bounds (here) because they are shared across // climbers. While the variable bounds are the same across climbers, there can be different @@ -679,7 +690,7 @@ void pdlp_initial_scaling_strategy_t::apply_bound_objective_rescaling_ [bound_rescaling = bound_rescaling_.data()] __device__(f_t2 variable_bounds) -> f_t2 { return {variable_bounds.x * *bound_rescaling, variable_bounds.y * *bound_rescaling}; }, - stream_view_); + stream_view_.get()); } cub::DeviceTransform::Transform( @@ -688,7 +699,7 @@ void pdlp_initial_scaling_strategy_t::apply_bound_objective_rescaling_ op_problem_scaled_.objective_coefficients.data(), op_problem_scaled_.objective_coefficients.size(), cuda::std::multiplies{}, - stream_view_.value()); + stream_view_.get()); } template @@ -735,7 +746,7 @@ void pdlp_initial_scaling_strategy_t::scale_solutions( primal_solution.data(), primal_solution.size(), batch_safe_div(), - stream_view_); + stream_view_.get()); if (hyper_params_.bound_objective_rescaling && !running_mip_) { cub::DeviceTransform::Transform( @@ -744,7 +755,7 @@ void pdlp_initial_scaling_strategy_t::scale_solutions( primal_solution.data(), primal_solution.size(), cuda::std::multiplies{}, - stream_view_); + stream_view_.get()); } } @@ -762,7 +773,7 @@ void pdlp_initial_scaling_strategy_t::scale_solutions( dual_solution.data(), dual_solution.size(), batch_safe_div(), - stream_view_); + stream_view_.get()); if (hyper_params_.bound_objective_rescaling && !running_mip_) { cub::DeviceTransform::Transform( @@ -771,7 +782,7 @@ void pdlp_initial_scaling_strategy_t::scale_solutions( dual_solution.data(), dual_solution.size(), cuda::std::multiplies{}, - stream_view_); + stream_view_.get()); } } @@ -789,7 +800,7 @@ void pdlp_initial_scaling_strategy_t::scale_solutions( dual_slack.data(), dual_slack.size(), cuda::std::multiplies<>{}, - stream_view_); + stream_view_.get()); if (hyper_params_.bound_objective_rescaling && !running_mip_) { cub::DeviceTransform::Transform( @@ -798,7 +809,7 @@ void pdlp_initial_scaling_strategy_t::scale_solutions( dual_slack.data(), dual_slack.size(), cuda::std::multiplies{}, - stream_view_); + stream_view_.get()); } } } @@ -857,7 +868,7 @@ void pdlp_initial_scaling_strategy_t::unscale_solutions( primal_solution.data(), primal_solution.size(), cuda::std::multiplies<>{}, - stream_view_); + stream_view_.get()); if (hyper_params_.bound_objective_rescaling && !running_mip_) { cub::DeviceTransform::Transform( @@ -868,7 +879,7 @@ void pdlp_initial_scaling_strategy_t::unscale_solutions( primal_solution.data(), primal_solution.size(), cuda::std::multiplies{}, - stream_view_); + stream_view_.get()); } } @@ -887,7 +898,7 @@ void pdlp_initial_scaling_strategy_t::unscale_solutions( dual_solution.data(), dual_solution.size(), cuda::std::multiplies<>{}, - stream_view_); + stream_view_.get()); if (hyper_params_.bound_objective_rescaling && !running_mip_) { cub::DeviceTransform::Transform( cuda::std::make_tuple(dual_solution.data(), @@ -897,7 +908,7 @@ void pdlp_initial_scaling_strategy_t::unscale_solutions( dual_solution.data(), dual_solution.size(), cuda::std::multiplies{}, - stream_view_); + stream_view_.get()); } } @@ -914,7 +925,7 @@ void pdlp_initial_scaling_strategy_t::unscale_solutions( dual_slack.data(), dual_slack.size(), batch_safe_div(), - stream_view_); + stream_view_.get()); if (hyper_params_.bound_objective_rescaling && !running_mip_) { cub::DeviceTransform::Transform( cuda::std::make_tuple(dual_slack.data(), @@ -924,7 +935,7 @@ void pdlp_initial_scaling_strategy_t::unscale_solutions( dual_slack.data(), dual_slack.size(), cuda::std::multiplies{}, - stream_view_); + stream_view_.get()); } } } diff --git a/cpp/src/pdlp/optimal_batch_size_handler/optimal_batch_size_handler.cu b/cpp/src/pdlp/optimal_batch_size_handler/optimal_batch_size_handler.cu index c969d0347a..0df699ae1e 100644 --- a/cpp/src/pdlp/optimal_batch_size_handler/optimal_batch_size_handler.cu +++ b/cpp/src/pdlp/optimal_batch_size_handler/optimal_batch_size_handler.cu @@ -62,7 +62,7 @@ struct SpMM_benchmarks_context_t { y_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &buffer_size_non_transpose_batch, - stream_view)); + stream_view.get())); size_t buffer_size_transpose_batch = 0; RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsespmm_bufferSize( @@ -76,7 +76,7 @@ struct SpMM_benchmarks_context_t { x_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &buffer_size_transpose_batch, - stream_view)); + stream_view.get())); buffer_transpose_batch = rmm::device_buffer(buffer_size_transpose_batch, stream_view); buffer_non_transpose_batch = rmm::device_buffer(buffer_size_non_transpose_batch, stream_view); @@ -94,7 +94,7 @@ struct SpMM_benchmarks_context_t { x_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, buffer_transpose_batch.data(), - stream_view); + stream_view.get()); my_cusparsespmm_preprocess( handle_ptr->get_cusparse_handle(), @@ -107,7 +107,7 @@ struct SpMM_benchmarks_context_t { y_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, buffer_non_transpose_batch.data(), - stream_view); + stream_view.get()); #endif // First empty run for warm up @@ -129,7 +129,7 @@ struct SpMM_benchmarks_context_t { y_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, (f_t*)buffer_non_transpose_batch.data(), - stream_view)); + stream_view.get())); RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsespmm( handle_ptr->get_cusparse_handle(), @@ -142,7 +142,7 @@ struct SpMM_benchmarks_context_t { x_descr.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, (f_t*)buffer_transpose_batch.data(), - stream_view)); + stream_view.get())); } cusparse_dn_mat_uptr x_descr; @@ -240,7 +240,7 @@ int optimal_batch_size_handler(const optimization_problem_t& op_proble i_t dual_size = problem.n_constraints; // Sync before starting anything to make sure everything is done - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view)); + stream_view.sync(); // Evaluate current, left and right nodes to pick a direction diff --git a/cpp/src/pdlp/optimization_problem.cu b/cpp/src/pdlp/optimization_problem.cu index 95457e2556..3a8bcb0b2a 100644 --- a/cpp/src/pdlp/optimization_problem.cu +++ b/cpp/src/pdlp/optimization_problem.cu @@ -1543,7 +1543,7 @@ rmm::device_uvector gpu_cast(const rmm::device_uvector& src, rmm::cuda rmm::device_uvector dst(src.size(), stream); if (src.size() > 0) { RAFT_CUDA_TRY(cub::DeviceTransform::Transform( - src.data(), dst.data(), src.size(), cast_op{}, stream.value())); + src.data(), dst.data(), src.size(), cast_op{}, stream.get())); } return dst; } @@ -1577,43 +1577,43 @@ optimization_problem_t optimization_problem_t::convert static_cast(A_indices_.size()), A_offsets_.data(), static_cast(A_offsets_.size())); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } if (c_.size() > 0) { auto other_c = gpu_cast(c_, stream); other.set_objective_coefficients(other_c.data(), static_cast(other_c.size())); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } if (b_.size() > 0) { auto other_b = gpu_cast(b_, stream); other.set_constraint_bounds(other_b.data(), static_cast(other_b.size())); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } if (constraint_lower_bounds_.size() > 0) { auto other_clb = gpu_cast(constraint_lower_bounds_, stream); other.set_constraint_lower_bounds(other_clb.data(), static_cast(other_clb.size())); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } if (constraint_upper_bounds_.size() > 0) { auto other_cub = gpu_cast(constraint_upper_bounds_, stream); other.set_constraint_upper_bounds(other_cub.data(), static_cast(other_cub.size())); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } if (variable_lower_bounds_.size() > 0) { auto other_vlb = gpu_cast(variable_lower_bounds_, stream); other.set_variable_lower_bounds(other_vlb.data(), static_cast(other_vlb.size())); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } if (variable_upper_bounds_.size() > 0) { auto other_vub = gpu_cast(variable_upper_bounds_, stream); other.set_variable_upper_bounds(other_vub.data(), static_cast(other_vub.size())); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } if (variable_types_.size() > 0) { diff --git a/cpp/src/pdlp/pdhg.cu b/cpp/src/pdlp/pdhg.cu index 67117e14d0..7ef0a615c0 100644 --- a/cpp/src/pdlp/pdhg.cu +++ b/cpp/src/pdlp/pdhg.cu @@ -191,7 +191,7 @@ new_bounds_groups_t copy_new_bounds_to_groups( raft::copy(h_idx.data(), new_bounds_idx.data(), n_entries, stream_view); raft::copy(h_lower.data(), new_bounds_lower.data(), n_entries, stream_view); raft::copy(h_upper.data(), new_bounds_upper.data(), n_entries, stream_view); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view)); + stream_view.sync(); } new_bounds_groups_t groups(batch_size); @@ -411,7 +411,7 @@ void pdhg_solver_t::compute_next_dual_solution(rmm::device_uvector::compute_next_dual_solution(rmm::device_uvector::compute_next_dual_solution(rmm::device_uvector(dual_step_size.data()), - stream_view_.value()); + stream_view_.get()); } template @@ -460,7 +460,7 @@ void pdhg_solver_t::spmvop_At_y() cusparse_view_.dual_solution.get(), cusparse_view_.current_AtY.get(), cusparse_view_.current_AtY.get(), - stream_view_.value()); + stream_view_.get()); return; } #endif @@ -473,7 +473,7 @@ void pdhg_solver_t::spmvop_At_y() cusparse_view_.current_AtY.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_transpose.data(), - stream_view_)); + stream_view_.get())); } template @@ -488,7 +488,7 @@ void pdhg_solver_t::spmvop_A_x() cusparse_view_.reflected_primal_solution.get(), cusparse_view_.dual_gradient.get(), cusparse_view_.dual_gradient.get(), - stream_view_.value()); + stream_view_.get()); return; } #endif @@ -502,7 +502,7 @@ void pdhg_solver_t::spmvop_A_x() cusparse_view_.dual_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_non_transpose.data(), - stream_view_)); + stream_view_.get())); } template @@ -529,7 +529,7 @@ void pdhg_solver_t::compute_At_y() cusparse_view_.current_AtY.get(), CUSPARSE_SPMV_CSR_ALG2, cusparse_view_.buffer_transpose_mixed_.data(), - stream_view_); + stream_view_.get()); } else { spmvop_At_y(); } @@ -544,7 +544,7 @@ void pdhg_solver_t::compute_At_y() cusparse_view_.current_AtY.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_transpose.data(), - stream_view_)); + stream_view_.get())); } } else { RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsespmm( @@ -558,7 +558,7 @@ void pdhg_solver_t::compute_At_y() cusparse_view_.batch_current_AtYs.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, (f_t*)cusparse_view_.buffer_transpose_batch_row_row_.data(), - stream_view_)); + stream_view_.get())); } } @@ -587,7 +587,7 @@ void pdhg_solver_t::compute_A_x() cusparse_view_.dual_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, cusparse_view_.buffer_non_transpose_mixed_.data(), - stream_view_); + stream_view_.get()); } else { spmvop_A_x(); } @@ -602,7 +602,7 @@ void pdhg_solver_t::compute_A_x() cusparse_view_.dual_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_non_transpose.data(), - stream_view_)); + stream_view_.get())); } } else { RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsespmm( @@ -616,7 +616,7 @@ void pdhg_solver_t::compute_A_x() cusparse_view_.batch_dual_gradients.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, (f_t*)cusparse_view_.buffer_non_transpose_batch_row_row_.data(), - stream_view_)); + stream_view_.get())); } } @@ -636,7 +636,7 @@ void pdhg_solver_t::spmv_At_into(cusparseDnVecDescr_t in_desc, out_desc, CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_transpose.data(), - stream_view_)); + stream_view_.get())); } // out_desc = A @ in_desc, the counterpart of spmv_At_into on this shard's local A. @@ -654,7 +654,7 @@ void pdhg_solver_t::spmv_A_into(cusparseDnVecDescr_t in_desc, out_desc, CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_non_transpose.data(), - stream_view_)); + stream_view_.get())); } template @@ -678,7 +678,7 @@ void pdhg_solver_t::compute_primal_projection_with_gradient( tmp_primal_.data()), primal_size_h_, primal_projection(primal_step_size.data()), - stream_view_.value()); + stream_view_.get()); } template @@ -764,7 +764,7 @@ void pdhg_solver_t::primal_reflected_major_projection_transform( potential_next_primal_solution_.data(), dual_slack_.data(), reflected_primal_.data()), primal_size_h_, primal_reflected_major_projection(primal_step_size.data()), - stream_view_.value()); + stream_view_.get()); } template @@ -807,7 +807,7 @@ void pdhg_solver_t::primal_reflected_projection_transform( reflected_primal_.data(), primal_size_h_, primal_reflected_projection(primal_step_size.data()), - stream_view_.value()); + stream_view_.get()); } template @@ -851,7 +851,7 @@ void pdhg_solver_t::dual_reflected_major_projection_transform( thrust::make_zip_iterator(potential_next_dual_solution_.data(), reflected_dual_.data()), dual_size_h_, dual_reflected_major_projection(dual_step_size.data()), - stream_view_.value()); + stream_view_.get()); } template @@ -894,7 +894,7 @@ void pdhg_solver_t::dual_reflected_projection_transform( reflected_dual_.data(), dual_size_h_, dual_reflected_projection(dual_step_size.data()), - stream_view_.value()); + stream_view_.get()); } template @@ -1217,7 +1217,7 @@ void pdhg_solver_t::refine_initial_primal_projection( make_span(bound_rescaling), make_span(current_saddle_point_state_.get_primal_solution()), problem_ptr->n_variables}, - stream_view_.value()); + stream_view_.get()); } template @@ -1265,7 +1265,7 @@ void pdhg_solver_t::compute_next_primal_dual_solution_reflected( reflected_primal_.data(), batch_size_divisor_, problem_ptr->objective_coefficients.size() > static_cast(primal_size_h_)}, - stream_view_.value()); + stream_view_.get()); } if (new_bounds_idx_.size() != 0) { #ifdef CUPDLP_DEBUG_MODE @@ -1297,7 +1297,7 @@ void pdhg_solver_t::compute_next_primal_dual_solution_reflected( make_span(reflected_primal_), (int)climber_strategies_.size(), problem_ptr->objective_coefficients.size() > static_cast(primal_size_h_)}, - stream_view_.value()); + stream_view_.get()); } #ifdef CUPDLP_DEBUG_MODE print("potential_next_primal_solution_", potential_next_primal_solution_); @@ -1329,7 +1329,7 @@ void pdhg_solver_t::compute_next_primal_dual_solution_reflected( reflected_dual_.data(), batch_size_divisor_, problem_ptr->constraint_lower_bounds.size() > static_cast(dual_size_h_)}, - stream_view_.value()); + stream_view_.get()); } #ifdef CUPDLP_DEBUG_MODE @@ -1380,7 +1380,7 @@ void pdhg_solver_t::compute_next_primal_dual_solution_reflected( reflected_primal_.data(), (int)climber_strategies_.size(), problem_ptr->objective_coefficients.size() > static_cast(primal_size_h_)}, - stream_view_.value()); + stream_view_.get()); } if (new_bounds_idx_.size() != 0) { #ifdef CUPDLP_DEBUG_MODE @@ -1410,7 +1410,7 @@ void pdhg_solver_t::compute_next_primal_dual_solution_reflected( make_span(reflected_primal_), (int)climber_strategies_.size(), problem_ptr->objective_coefficients.size() > static_cast(primal_size_h_)}, - stream_view_.value()); + stream_view_.get()); } #ifdef CUPDLP_DEBUG_MODE print("reflected_primal_", reflected_primal_); @@ -1445,7 +1445,7 @@ void pdhg_solver_t::compute_next_primal_dual_solution_reflected( reflected_dual_.data(), (int)climber_strategies_.size(), problem_ptr->constraint_lower_bounds.size() > static_cast(dual_size_h_)}, - stream_view_.value()); + stream_view_.get()); } #ifdef CUPDLP_DEBUG_MODE print("reflected_dual_", reflected_dual_); diff --git a/cpp/src/pdlp/pdlp.cu b/cpp/src/pdlp/pdlp.cu index abc119b1e6..86a32f028e 100644 --- a/cpp/src/pdlp/pdlp.cu +++ b/cpp/src/pdlp/pdlp.cu @@ -592,7 +592,7 @@ void pdlp_solver_t::set_initial_primal_solution( initial_primal_.data(), initial_primal_.size(), cuda::std::identity{}, - stream_view_); + stream_view_.get()); } template @@ -606,7 +606,7 @@ void pdlp_solver_t::set_initial_dual_solution( initial_dual_.data(), initial_dual_.size(), cuda::std::identity{}, - stream_view_); + stream_view_.get()); } static bool time_limit_reached(const timer_t& timer) { return timer.check_time_limit(); } @@ -945,7 +945,7 @@ template optimization_problem_solution_t pdlp_solver_t::finalize_batch_return() { current_termination_strategy_.fill_gpu_terms_stats(total_pdlp_iterations_); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); current_termination_strategy_.convert_gpu_terms_stats_to_host( batch_solution_to_return_.get_additional_termination_informations()); return optimization_problem_solution_t{ @@ -1086,7 +1086,7 @@ pdlp_solver_t::check_batch_termination(const timer_t& timer) sb_view_.mark_solved(climber_strategies_[i].original_index); } } - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); return current_termination_strategy_.fill_return_problem_solution( internal_solver_iterations_, pdhg_solver_, @@ -1484,11 +1484,11 @@ static void compute_stats(const rmm::device_uvector& vec, n, cuda::minimum<>{}, std::numeric_limits::max(), - stream)); + stream.get())); RAFT_CUDA_TRY(cub::DeviceReduce::Reduce( - d_temp, bytes_2, abs_iter, d_largest.data(), n, cuda::maximum<>{}, f_t(0), stream)); + d_temp, bytes_2, abs_iter, d_largest.data(), n, cuda::maximum<>{}, f_t(0), stream.get())); RAFT_CUDA_TRY(cub::DeviceReduce::Reduce( - d_temp, bytes_3, abs_iter, d_sum.data(), n, cuda::std::plus<>{}, f_t(0), stream)); + d_temp, bytes_3, abs_iter, d_sum.data(), n, cuda::std::plus<>{}, f_t(0), stream.get())); size_t max_bytes = std::max({bytes_1, bytes_2, bytes_3}); rmm::device_buffer temp_buf(max_bytes, stream); @@ -1500,11 +1500,23 @@ static void compute_stats(const rmm::device_uvector& vec, n, cuda::minimum<>{}, std::numeric_limits::max(), - stream)); - RAFT_CUDA_TRY(cub::DeviceReduce::Reduce( - temp_buf.data(), bytes_2, abs_iter, d_largest.data(), n, cuda::maximum<>{}, f_t(0), stream)); - RAFT_CUDA_TRY(cub::DeviceReduce::Reduce( - temp_buf.data(), bytes_3, abs_iter, d_sum.data(), n, cuda::std::plus<>{}, f_t(0), stream)); + stream.get())); + RAFT_CUDA_TRY(cub::DeviceReduce::Reduce(temp_buf.data(), + bytes_2, + abs_iter, + d_largest.data(), + n, + cuda::maximum<>{}, + f_t(0), + stream.get())); + RAFT_CUDA_TRY(cub::DeviceReduce::Reduce(temp_buf.data(), + bytes_3, + abs_iter, + d_sum.data(), + n, + cuda::std::plus<>{}, + f_t(0), + stream.get())); smallest = d_smallest.value(stream); largest = d_largest.value(stream); @@ -1628,7 +1640,7 @@ void pdlp_solver_t::update_primal_dual_solutions( RAFT_CUDA_TRY(cudaMemsetAsync(saddle.get_current_AtY().data(), f_t(0.0), sizeof(f_t) * saddle.get_current_AtY().size(), - stream_view_)); + stream_view_.get())); // Scale if should compute initial step size after scaling if (!settings_.hyper_params.compute_initial_step_size_before_scaling) { @@ -1797,13 +1809,13 @@ void pdlp_solver_t::swap_context( const auto [grid_size, block_size] = kernel_config_from_batch_size(static_cast(swap_pairs.size())); pdlp_swap_device_vectors_kernel - <<>>(thrust::raw_pointer_cast(swap_pairs.data()), - static_cast(swap_pairs.size()), - make_span(primal_weight_), - make_span(best_primal_weight_), - make_span(step_size_), - make_span(primal_step_size_), - make_span(dual_step_size_)); + <<>>(thrust::raw_pointer_cast(swap_pairs.data()), + static_cast(swap_pairs.size()), + make_span(primal_weight_), + make_span(best_primal_weight_), + make_span(step_size_), + make_span(primal_step_size_), + make_span(dual_step_size_)); RAFT_CUDA_TRY(cudaPeekAtLastError()); // Swap unscaled problem's per-climber fields (COL-major blocks) if (problem_ptr->objective_coefficients.size() > static_cast(primal_size_h_)) { @@ -1862,7 +1874,7 @@ void pdlp_solver_t::swap_all_context( host_vector_swap(climber_strategies_, pair.left, pair.right); } - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } template @@ -1881,7 +1893,7 @@ void pdlp_solver_t::resize_all_context(i_t new_size) // Resize PDLP own context resize_context(new_size); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } template @@ -2025,7 +2037,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( pdhg_cusparse_view.batch_current_AtYs.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &new_buf_size, - stream_view_)); + stream_view_.get())); pdhg_cusparse_view.buffer_transpose_batch_row_row_.resize(new_buf_size, stream_view_); // PDHG row-row: A * batch_reflected_primal_solutions -> batch_dual_gradients @@ -2040,7 +2052,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( pdhg_cusparse_view.batch_dual_gradients.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, &new_buf_size, - stream_view_)); + stream_view_.get())); pdhg_cusparse_view.buffer_non_transpose_batch_row_row_.resize(new_buf_size, stream_view_); // Adaptive step size: A_T * batch_potential_next_dual_solution -> batch_next_AtYs @@ -2055,7 +2067,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( pdhg_cusparse_view.batch_next_AtYs.get(), CUSPARSE_SPMM_CSR_ALG3, &new_buf_size, - stream_view_)); + stream_view_.get())); pdhg_cusparse_view.buffer_transpose_batch.resize(new_buf_size, stream_view_); // Convergence info: A_T * batch_dual_solutions -> batch_tmp_primals @@ -2070,7 +2082,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( current_op_problem_evaluation_cusparse_view_.batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, &new_buf_size, - stream_view_)); + stream_view_.get())); current_op_problem_evaluation_cusparse_view_.buffer_transpose_batch.resize(new_buf_size, stream_view_); @@ -2086,7 +2098,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( current_op_problem_evaluation_cusparse_view_.batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, &new_buf_size, - stream_view_)); + stream_view_.get())); current_op_problem_evaluation_cusparse_view_.buffer_non_transpose_batch.resize(new_buf_size, stream_view_); } @@ -2106,7 +2118,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( pdhg_cusparse_view.batch_current_AtYs.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, pdhg_cusparse_view.buffer_transpose_batch_row_row_.data(), - stream_view_); + stream_view_.get()); my_cusparsespmm_preprocess( handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, @@ -2118,7 +2130,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( pdhg_cusparse_view.batch_dual_gradients.get(), (deterministic_batch_pdlp) ? CUSPARSE_SPMM_CSR_ALG3 : CUSPARSE_SPMM_CSR_ALG2, pdhg_cusparse_view.buffer_non_transpose_batch_row_row_.data(), - stream_view_); + stream_view_.get()); // Adaptive step size strategy SpMM preprocess my_cusparsespmm_preprocess(handle_ptr_->get_cusparse_handle(), @@ -2131,7 +2143,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( pdhg_cusparse_view.batch_next_AtYs.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)pdhg_cusparse_view.buffer_transpose_batch.data(), - stream_view_); + stream_view_.get()); // Convergence information SpMM preprocess my_cusparsespmm_preprocess( @@ -2145,7 +2157,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( current_op_problem_evaluation_cusparse_view_.batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)current_op_problem_evaluation_cusparse_view_.buffer_transpose_batch.data(), - stream_view_); + stream_view_.get()); my_cusparsespmm_preprocess( handle_ptr_->get_cusparse_handle(), @@ -2158,7 +2170,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( current_op_problem_evaluation_cusparse_view_.batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)current_op_problem_evaluation_cusparse_view_.buffer_non_transpose_batch.data(), - stream_view_); + stream_view_.get()); #endif // Set PDHG graphs to uninitialized so that next call can start a new graph. @@ -2168,7 +2180,7 @@ void pdlp_solver_t::resize_and_swap_all_context_loop( // graph_all_non_major (reflected non-major). pdhg_solver_.get_graph_all() = ping_pong_graph_t(stream_view_, true); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } // delta = reflected - current, for both primal and dual, written into the @@ -2182,13 +2194,13 @@ static void compute_primal_dual_deltas(pdhg_solver_t& pdhg, rmm::cuda_ pdhg.get_saddle_point_state().get_delta_primal().data(), pdhg.get_primal_solution().size(), cuda::std::minus{}, - stream); + stream.get()); cub::DeviceTransform::Transform( cuda::std::make_tuple(pdhg.get_reflected_dual().data(), pdhg.get_dual_solution().data()), pdhg.get_saddle_point_state().get_delta_dual().data(), pdhg.get_dual_solution().size(), cuda::std::minus{}, - stream); + stream.get()); } template @@ -2285,7 +2297,7 @@ void pdlp_solver_t::compute_fixed_error(std::vector& has_restarte } else { // Sync to make sure all previous cuSparse operations are finished before setting the // potential_next_dual_solution - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); // Make potential_next_dual_solution point towards reflected dual solution to reuse the code RAFT_CUSPARSE_TRY(cusparseDnVecSetValues(cusparse_view.potential_next_dual_solution.get(), @@ -2302,7 +2314,7 @@ void pdlp_solver_t::compute_fixed_error(std::vector& has_restarte if (batch_mode_) { const auto [grid_size, block_size] = kernel_config_from_batch_size(climber_strategies_.size()); - kernel_compute_fixed_error<<>>( + kernel_compute_fixed_error<<>>( make_span(step_size_strategy_.get_norm_squared_delta_primal()), make_span(step_size_strategy_.get_norm_squared_delta_dual()), make_span(primal_weight_), @@ -2310,7 +2322,7 @@ void pdlp_solver_t::compute_fixed_error(std::vector& has_restarte make_span(step_size_strategy_.get_interaction()), make_span(restart_strategy_.fixed_point_error_)); RAFT_CUDA_TRY(cudaStreamSynchronize( - stream_view_)); // To make sure all the data is written from device to host + stream_view_.get())); // To make sure all the data is written from device to host RAFT_CUDA_TRY(cudaPeekAtLastError()); #ifdef CUPDLP_DEBUG_MODE @@ -2327,7 +2339,7 @@ void pdlp_solver_t::compute_fixed_error(std::vector& has_restarte // Sync to make sure all previous cuSparse operations are finished before setting the // potential_next_dual_solution - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); // Put back, already done in multi-gpu side if (!is_distributed_master()) { @@ -2388,10 +2400,10 @@ void pdlp_solver_t::transpose_problem_fields(bool to_row) transposed.data(), *output_ld)); raft::copy(field.data(), transposed.data(), field.size(), stream_view_); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); }; - RAFT_CUBLAS_TRY(cublasSetStream(handle_ptr_->get_cublas_handle(), stream_view_)); + RAFT_CUBLAS_TRY(cublasSetStream(handle_ptr_->get_cublas_handle(), stream_view_.get())); // We need to swap the scaled version because they can be dynamically resized and swapped. transpose_field(op_problem_scaled_.objective_coefficients, primal_size_h_); transpose_field(op_problem_scaled_.constraint_lower_bounds, dual_size_h_); @@ -2413,7 +2425,7 @@ void pdlp_solver_t::transpose_primal_dual_to_row( rmm::device_uvector dual_slack_transposed( is_dual_slack_empty ? 0 : primal_size_h_ * climber_strategies_.size(), stream_view_); - RAFT_CUBLAS_TRY(cublasSetStream(handle_ptr_->get_cublas_handle(), stream_view_)); + RAFT_CUBLAS_TRY(cublasSetStream(handle_ptr_->get_cublas_handle(), stream_view_.get())); CUBLAS_CHECK(cublasGeam(handle_ptr_->get_cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, @@ -2476,7 +2488,7 @@ void pdlp_solver_t::transpose_primal_dual_to_row( dual_size_h_ * climber_strategies_.size(), stream_view_); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } template @@ -2492,7 +2504,7 @@ void pdlp_solver_t::transpose_primal_dual_back_to_col( rmm::device_uvector dual_slack_transposed( is_dual_slack_empty ? 0 : primal_size_h_ * climber_strategies_.size(), stream_view_); - RAFT_CUBLAS_TRY(cublasSetStream(handle_ptr_->get_cublas_handle(), stream_view_)); + RAFT_CUBLAS_TRY(cublasSetStream(handle_ptr_->get_cublas_handle(), stream_view_.get())); CUBLAS_CHECK(cublasGeam(handle_ptr_->get_cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, @@ -2556,7 +2568,7 @@ void pdlp_solver_t::transpose_primal_dual_back_to_col( dual_size_h_ * climber_strategies_.size(), stream_view_); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } template @@ -2747,7 +2759,7 @@ optimization_problem_solution_t pdlp_solver_t::run_solver(co pdhg_solver_.get_primal_solution().data(), pdhg_solver_.get_primal_solution().size(), clamp(), - stream_view_.value()); + stream_view_.get()); } else { cub::DeviceTransform::Transform( cuda::std::make_tuple(pdhg_solver_.get_primal_solution().data(), @@ -2755,7 +2767,7 @@ optimization_problem_solution_t pdlp_solver_t::run_solver(co pdhg_solver_.get_primal_solution().data(), pdhg_solver_.get_primal_solution().size(), clamp(), - stream_view_.value()); + stream_view_.get()); } pdhg_solver_.refine_initial_primal_projection( @@ -2771,7 +2783,7 @@ optimization_problem_solution_t pdlp_solver_t::run_solver(co unscaled_primal_avg_solution_.data(), primal_size_h_, clamp(), - stream_view_.value()); + stream_view_.get()); } } @@ -3191,7 +3203,7 @@ void pdlp_solver_t::halpern_update() (f_t(1.0) - reflection_coefficient) * current_primal; return weight * reflected + (f_t(1.0) - weight) * initial_primal; }, - stream_view_.value()); + stream_view_.get()); #ifdef CUPDLP_DEBUG_MODE print("pdhg_solver_.get_reflected_dual()", pdhg_solver_.get_reflected_dual()); @@ -3215,7 +3227,7 @@ void pdlp_solver_t::halpern_update() (f_t(1.0) - reflection_coefficient) * current_dual; return weight * reflected + (f_t(1.0) - weight) * initial_dual; }, - stream_view_.value()); + stream_view_.get()); #ifdef CUPDLP_DEBUG_MODE print("halpen_update current primal", @@ -3325,7 +3337,7 @@ void pdlp_solver_t::compute_initial_step_size() op_problem_scaled_.nnz, red_op, 0.0, - stream_view_); + stream_view_.get()); // Allocate temporary storage rmm::device_buffer cub_tmp{temp_storage_bytes, stream_view_}; // Run max-reduction @@ -3336,12 +3348,12 @@ void pdlp_solver_t::compute_initial_step_size() op_problem_scaled_.nnz, red_op, 0.0, - stream_view_); + stream_view_.get()); raft::linalg::eltwiseDivideCheckZero( - step_size_.data(), step_size_.data(), abs_max_element.data(), 1, stream_view_); + step_size_.data(), step_size_.data(), abs_max_element.data(), 1, stream_view_.get()); // Sync since we are using local variable - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } else { i_t m = op_problem_scaled_.n_constraints; i_t n = op_problem_scaled_.n_variables; @@ -3384,7 +3396,7 @@ void pdlp_solver_t::compute_initial_step_size() d_q.data(), d_q.size(), divide_by_device_scalar_t{norm_q.data()}, - stream_view_.value()); + stream_view_.get()); // A_t_q = A_t @ d_q RAFT_CUSPARSE_TRY( @@ -3397,7 +3409,7 @@ void pdlp_solver_t::compute_initial_step_size() vecATQ, CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_transpose.data(), - stream_view_.value())); + stream_view_.get())); // z = A @ A_t_q RAFT_CUSPARSE_TRY( @@ -3410,7 +3422,7 @@ void pdlp_solver_t::compute_initial_step_size() vecZ, CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view_.buffer_non_transpose.data(), - stream_view_.value())); + stream_view_.get())); // sigma_max_sq = dot(q, z) RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(handle_ptr_->get_cublas_handle(), m, @@ -3419,14 +3431,14 @@ void pdlp_solver_t::compute_initial_step_size() d_z.data(), primal_stride, sigma_max_sq.data(), - stream_view_.value())); + stream_view_.get())); // d_q := -sigma_max_sq * d_q + d_z cub::DeviceTransform::Transform(cuda::std::make_tuple(d_q.data(), d_z.data()), d_q.data(), d_q.size(), residual_fma_neg_scalar_t{sigma_max_sq.data()}, - stream_view_.value()); + stream_view_.get()); my_l2_norm(d_q, residual_norm, handle_ptr_); @@ -3441,7 +3453,7 @@ void pdlp_solver_t::compute_initial_step_size() handle_ptr_->get_thrust_policy(), step_size_.begin(), step_size_.end(), step_size); // Sync since we are using local variable - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); RAFT_CUSPARSE_TRY(cusparseDestroyDnVec(vecZ)); RAFT_CUSPARSE_TRY(cusparseDestroyDnVec(vecQ)); RAFT_CUSPARSE_TRY(cusparseDestroyDnVec(vecATQ)); @@ -3536,16 +3548,16 @@ void pdlp_solver_t::compute_initial_primal_weight() const auto [grid_size, block_size] = kernel_config_from_batch_size(climber_strategies_.size()); compute_weights_initial_primal_weight_from_squared_norms - <<>>(b_vec_norm.data(), - c_vec_norm.data(), - make_span(primal_weight_), - make_span(best_primal_weight_), - climber_strategies_.size(), - settings_.hyper_params); + <<>>(b_vec_norm.data(), + c_vec_norm.data(), + make_span(primal_weight_), + make_span(best_primal_weight_), + climber_strategies_.size(), + settings_.hyper_params); RAFT_CUDA_TRY(cudaPeekAtLastError()); // Sync since we are using local variable - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } template diff --git a/cpp/src/pdlp/pdlp_warm_start_data.cu b/cpp/src/pdlp/pdlp_warm_start_data.cu index a214f1a165..2ce6b0f4b5 100644 --- a/cpp/src/pdlp/pdlp_warm_start_data.cu +++ b/cpp/src/pdlp/pdlp_warm_start_data.cu @@ -11,6 +11,8 @@ #include +#include + #include #include @@ -64,16 +66,23 @@ pdlp_warm_start_data_t::pdlp_warm_start_data_t( template pdlp_warm_start_data_t::pdlp_warm_start_data_t() - : current_primal_solution_{rmm::device_uvector(0, rmm::cuda_stream_default)}, - current_dual_solution_{rmm::device_uvector(0, rmm::cuda_stream_default)}, - initial_primal_average_{rmm::device_uvector(0, rmm::cuda_stream_default)}, - initial_dual_average_{rmm::device_uvector(0, rmm::cuda_stream_default)}, - current_ATY_{rmm::device_uvector(0, rmm::cuda_stream_default)}, - sum_primal_solutions_{rmm::device_uvector(0, rmm::cuda_stream_default)}, - sum_dual_solutions_{rmm::device_uvector(0, rmm::cuda_stream_default)}, + : current_primal_solution_{rmm::device_uvector( + 0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}})}, + current_dual_solution_{ + rmm::device_uvector(0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}})}, + initial_primal_average_{ + rmm::device_uvector(0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}})}, + initial_dual_average_{ + rmm::device_uvector(0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}})}, + current_ATY_{rmm::device_uvector(0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}})}, + sum_primal_solutions_{ + rmm::device_uvector(0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}})}, + sum_dual_solutions_{ + rmm::device_uvector(0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}})}, last_restart_duality_gap_primal_solution_{ - rmm::device_uvector(0, rmm::cuda_stream_default)}, - last_restart_duality_gap_dual_solution_{rmm::device_uvector(0, rmm::cuda_stream_default)} + rmm::device_uvector(0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}})}, + last_restart_duality_gap_dual_solution_{ + rmm::device_uvector(0, cuda::stream_ref{cudaStream_t{cudaStreamDefault}})} { } diff --git a/cpp/src/pdlp/restart_strategy/localized_duality_gap_container.cu b/cpp/src/pdlp/restart_strategy/localized_duality_gap_container.cu index 7610e4f7dc..eb356022eb 100644 --- a/cpp/src/pdlp/restart_strategy/localized_duality_gap_container.cu +++ b/cpp/src/pdlp/restart_strategy/localized_duality_gap_container.cu @@ -52,11 +52,11 @@ localized_duality_gap_container_t::localized_duality_gap_container_t( RAFT_CUDA_TRY(cudaMemsetAsync(primal_solution_.data(), f_t(0.0), sizeof(f_t) * primal_solution_.size(), - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); RAFT_CUDA_TRY(cudaMemsetAsync(dual_solution_.data(), f_t(0.0), sizeof(f_t) * dual_solution_.size(), - handle_ptr->get_stream())); + handle_ptr->get_stream().get())); } template @@ -96,7 +96,7 @@ void localized_duality_gap_container_t::swap_context( const auto [grid_size, block_size] = kernel_config_from_batch_size(static_cast(swap_pairs.size())); localized_duality_gap_swap_device_vectors_kernel - <<>>( + <<>>( thrust::raw_pointer_cast(swap_pairs.data()), static_cast(swap_pairs.size()), make_span(primal_distance_traveled_), diff --git a/cpp/src/pdlp/restart_strategy/localized_duality_gap_container.hpp b/cpp/src/pdlp/restart_strategy/localized_duality_gap_container.hpp index 283d00c4db..ee2e1cf34d 100644 --- a/cpp/src/pdlp/restart_strategy/localized_duality_gap_container.hpp +++ b/cpp/src/pdlp/restart_strategy/localized_duality_gap_container.hpp @@ -13,7 +13,6 @@ #include #include -#include #include #include diff --git a/cpp/src/pdlp/restart_strategy/pdlp_restart_strategy.cu b/cpp/src/pdlp/restart_strategy/pdlp_restart_strategy.cu index 01605dfb93..06158f4d1c 100644 --- a/cpp/src/pdlp/restart_strategy/pdlp_restart_strategy.cu +++ b/cpp/src/pdlp/restart_strategy/pdlp_restart_strategy.cu @@ -216,11 +216,11 @@ pdlp_restart_strategy_t::pdlp_restart_strategy_t( RAFT_CUDA_TRY(cudaMemsetAsync(last_restart_duality_gap_.primal_solution_.data(), 0.0, sizeof(f_t) * last_restart_duality_gap_.primal_solution_.size(), - stream_view_)); + stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync(last_restart_duality_gap_.dual_solution_.data(), 0.0, sizeof(f_t) * last_restart_duality_gap_.dual_solution_.size(), - stream_view_)); + stream_view_.get())); // Trigger the costly (costly for ms instances) GetDeviceProperty only if need trust region // restart @@ -231,13 +231,13 @@ pdlp_restart_strategy_t::pdlp_restart_strategy_t( problem_ptr->constraint_upper_bounds.data(), dual_size_h_, transform_constraint_lower_bounds(), - stream_view_); + stream_view_.get()); raft::linalg::binaryOp(transformed_constraint_upper_bounds_.data(), problem_ptr->constraint_lower_bounds.data(), problem_ptr->constraint_upper_bounds.data(), dual_size_h_, transform_constraint_upper_bounds(), - stream_view_); + stream_view_.get()); // Check that device support CooperativeLaunch int dev = 0; @@ -287,7 +287,7 @@ pdlp_restart_strategy_t::pdlp_restart_strategy_t( reusable_device_scalar_1_.data(), climber_strategies_.size(), primal_size_h_, - stream_view_); + stream_view_.get()); dot_product_bytes = std::max(dot_product_bytes, byte_needed); cub::DeviceSegmentedReduce::Sum( @@ -297,7 +297,7 @@ pdlp_restart_strategy_t::pdlp_restart_strategy_t( reusable_device_scalar_1_.data(), climber_strategies_.size(), dual_size_h_, - stream_view_); + stream_view_.get()); dot_product_bytes = std::max(dot_product_bytes, byte_needed); dot_product_storage.resize(dot_product_bytes, stream_view_); @@ -351,12 +351,12 @@ bool pdlp_restart_strategy_t::run_trust_region_restart( reusable_device_scalar_value_1_.data(), primal_step_size.data(), 1, - stream_view_); + stream_view_.get()); raft::linalg::eltwiseDivideCheckZero(dual_norm_weight_.data(), reusable_device_scalar_value_1_.data(), dual_step_size.data(), 1, - stream_view_); + stream_view_.get()); i_t restart = should_do_artificial_restart(total_number_of_iterations); @@ -447,11 +447,11 @@ f_t pdlp_restart_strategy_t::compute_kkt_score( const rmm::device_uvector& gap, const rmm::device_uvector& primal_weight) { - kernel_compute_kkt_score<<<1, 1, 0, stream_view_>>>(l2_primal_residual.data(), - l2_dual_residual.data(), - gap.data(), - primal_weight.data(), - tmp_kkt_score_.data()); + kernel_compute_kkt_score<<<1, 1, 0, stream_view_.get()>>>(l2_primal_residual.data(), + l2_dual_residual.data(), + gap.data(), + primal_weight.data(), + tmp_kkt_score_.data()); return tmp_kkt_score_.value(stream_view_); } @@ -928,10 +928,10 @@ void pdlp_restart_strategy_t::cupdlpx_restart( if (batch_mode_) { const auto [grid_size, block_size] = kernel_config_from_batch_size(climber_strategies_.size()); kernel_compute_next_cupdlpx_primal_weight - <<>>(view, climber_strategies_.size()); + <<>>(view, climber_strategies_.size()); RAFT_CUDA_TRY(cudaPeekAtLastError()); RAFT_CUDA_TRY(cudaStreamSynchronize( - stream_view_)); // To make sure all the data is written from device to host + stream_view_.get())); // To make sure all the data is written from device to host #ifdef CUPDLP_DEBUG_MODE RAFT_CUDA_TRY(cudaDeviceSynchronize()); #endif @@ -1232,11 +1232,12 @@ void pdlp_restart_strategy_t::compute_new_primal_weight( cuopt_assert(!batch_mode_, "compute_new_primal_weight not supported in batch mode"); - compute_new_primal_weight_kernel<<<1, 1, 0, stream_view_>>>(duality_gap.view(), - primal_weight.data(), - step_size.data(), - primal_step_size.data(), - dual_step_size.data()); + compute_new_primal_weight_kernel + <<<1, 1, 0, stream_view_.get()>>>(duality_gap.view(), + primal_weight.data(), + step_size.data(), + primal_step_size.data(), + dual_step_size.data()); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -1262,7 +1263,7 @@ void pdlp_restart_strategy_t::distance_squared_moved_from_last_restart old_solution.data(), stride, debuga.data(), - stream_view_)); + stream_view_.get())); RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(handle_ptr_->get_cublas_handle(), size_of_solutions_h, new_solution.data(), @@ -1270,7 +1271,7 @@ void pdlp_restart_strategy_t::distance_squared_moved_from_last_restart new_solution.data(), stride, debugb.data(), - stream_view_)); + stream_view_.get())); std::cout << "Distance squared moved:\n" << " Old location=" << debuga.value(stream_view_) << "\n" << " New location=" << debugb.value(stream_view_) << std::endl; @@ -1282,7 +1283,7 @@ void pdlp_restart_strategy_t::distance_squared_moved_from_last_restart new_solution.data(), new_solution.size(), a_sub_scalar_times_b(reusable_device_scalar_value_1_.data()), - stream_view_); + stream_view_.get()); if (!batch_mode_) { RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(handle_ptr_->get_cublas_handle(), @@ -1292,7 +1293,7 @@ void pdlp_restart_strategy_t::distance_squared_moved_from_last_restart tmp.data(), stride, distance_moved.data(), - stream_view_)); + stream_view_.get())); } else { cub::DeviceSegmentedReduce::Sum( dot_product_storage.data(), @@ -1301,7 +1302,7 @@ void pdlp_restart_strategy_t::distance_squared_moved_from_last_restart distance_moved.data(), climber_strategies_.size(), size_of_solutions_h, - stream_view_); + stream_view_.get()); } } @@ -1348,7 +1349,7 @@ void pdlp_restart_strategy_t::update_last_restart_information( { raft::common::nvtx::range fun_scope("update_last_restart_information"); - compute_distance_traveled_last_restart_kernel<<<1, 1, 0, stream_view_>>>( + compute_distance_traveled_last_restart_kernel<<<1, 1, 0, stream_view_.get()>>>( duality_gap.view(), primal_weight.data(), last_restart_duality_gap_.distance_traveled_.data()); RAFT_CUDA_TRY(cudaPeekAtLastError()); @@ -1383,8 +1384,8 @@ __global__ void pick_restart_candidate_kernel( template i_t pdlp_restart_strategy_t::pick_restart_candidate() { - pick_restart_candidate_kernel - <<<1, 1, 0, stream_view_>>>(avg_duality_gap_.view(), current_duality_gap_.view(), this->view()); + pick_restart_candidate_kernel<<<1, 1, 0, stream_view_.get()>>>( + avg_duality_gap_.view(), current_duality_gap_.view(), this->view()); RAFT_CUDA_TRY(cudaPeekAtLastError()); i_t restart_to_average_h = candidate_is_avg_.value(stream_view_); @@ -1394,7 +1395,7 @@ i_t pdlp_restart_strategy_t::pick_restart_candidate() candidate_duality_gap_ = ¤t_duality_gap_; } - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); return restart_to_average_h; } @@ -1447,15 +1448,15 @@ void pdlp_restart_strategy_t::should_do_adaptive_restart_normalized_du // 2 * primal_weight + lri.dual_distance_moved_last_restart_period ^ 2 / primal_weight, compute_distance_traveled_last_restart_kernel - <<<1, 1, 0, stream_view_>>>(candidate_duality_gap.view(), - primal_weight.data(), - last_restart_duality_gap_.distance_traveled_.data()); + <<<1, 1, 0, stream_view_.get()>>>(candidate_duality_gap.view(), + primal_weight.data(), + last_restart_duality_gap_.distance_traveled_.data()); RAFT_CUDA_TRY(cudaPeekAtLastError()); bound_optimal_objective( last_restart_duality_gap_cusparse_view_, last_restart_duality_gap_, tmp_primal, tmp_dual); - adaptive_restart_triggered<<<1, 1, 0, stream_view_>>>( + adaptive_restart_triggered<<<1, 1, 0, stream_view_.get()>>>( candidate_duality_gap.view(), last_restart_duality_gap_.view(), this->view()); RAFT_CUDA_TRY(cudaPeekAtLastError()); @@ -1552,7 +1553,7 @@ void pdlp_restart_strategy_t::compute_localized_duality_gaps( current_duality_gap_cusparse_view_, current_duality_gap_, tmp_primal, tmp_dual); compute_normalized_gaps_kernel - <<<1, 1, 0, stream_view_>>>(avg_duality_gap_.view(), current_duality_gap_.view()); + <<<1, 1, 0, stream_view_.get()>>>(avg_duality_gap_.view(), current_duality_gap_.view()); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -1589,7 +1590,8 @@ void pdlp_restart_strategy_t::compute_bound(const rmm::device_uvector< #ifdef PDLP_DEBUG_MODE std::cout << "Compute bound" << std::endl; #endif - raft::linalg::eltwiseSub(tmp.data(), solution_tr.data(), solution.data(), size, stream_view_); + raft::linalg::eltwiseSub( + tmp.data(), solution_tr.data(), solution.data(), size, stream_view_.get()); RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(handle_ptr_->get_cublas_handle(), size, @@ -1598,9 +1600,9 @@ void pdlp_restart_strategy_t::compute_bound(const rmm::device_uvector< gradient.data(), stride, bound.data(), - stream_view_)); + stream_view_.get())); - raft::linalg::eltwiseAdd(bound.data(), bound.data(), lagrangian.data(), 1, stream_view_); + raft::linalg::eltwiseAdd(bound.data(), bound.data(), lagrangian.data(), 1, stream_view_.get()); } template @@ -1947,7 +1949,7 @@ void pdlp_restart_strategy_t::solve_bound_constrained_trust_region( duality_gap.dual_gradient_.data(), dual_size_h_, negate_t(), - stream_view_); + stream_view_.get()); // Use high_radius_squared_ to store objective_vector l2_norm my_l2_norm(objective_vector_, high_radius_squared_, handle_ptr_); @@ -1974,10 +1976,12 @@ void pdlp_restart_strategy_t::solve_bound_constrained_trust_region( const f_t zero_float = f_t(0.0); high_radius_squared_.set_value_async(zero_float, stream_view_); low_radius_squared_.set_value_async(zero_float, stream_view_); + RAFT_CUDA_TRY(cudaMemsetAsync(direction_full_.data(), + 0, + sizeof(f_t) * (primal_size_h_ + dual_size_h_), + stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync( - direction_full_.data(), 0, sizeof(f_t) * (primal_size_h_ + dual_size_h_), stream_view_)); - RAFT_CUDA_TRY(cudaMemsetAsync( - threshold_.data(), 0, sizeof(f_t) * (primal_size_h_ + dual_size_h_), stream_view_)); + threshold_.data(), 0, sizeof(f_t) * (primal_size_h_ + dual_size_h_), stream_view_.get())); /* ----- */ // Determine the direction which each component has moved and the threshold for when the @@ -1990,7 +1994,7 @@ void pdlp_restart_strategy_t::solve_bound_constrained_trust_region( thrust::make_zip_iterator(thrust::make_tuple(lower_bound_.data(), upper_bound_.data())), primal_size_h_, extract_bounds_t(), - stream_view_.value()); + stream_view_.get()); raft::copy(lower_bound_.data() + primal_size_h_, transformed_constraint_lower_bounds_.data(), dual_size_h_, @@ -2131,7 +2135,7 @@ void pdlp_restart_strategy_t::solve_bound_constrained_trust_region( dimBlock, kernel_args, 0, - stream_view_)); + stream_view_.get())); // Find max threshold for the join problem const f_t* max_threshold = @@ -2146,7 +2150,7 @@ void pdlp_restart_strategy_t::solve_bound_constrained_trust_region( // target_threshold which was computed before the loop in the direction_and_threshold_kernel // Otherwise use the test_threshold determined in the loop // { - target_threshold_determination_kernel<<<1, 1, 0, stream_view_>>>( + target_threshold_determination_kernel<<<1, 1, 0, stream_view_.get()>>>( this->view(), duality_gap.distance_traveled_.data(), max_threshold, max_threshold); RAFT_CUDA_TRY(cudaPeekAtLastError()); // } @@ -2160,13 +2164,13 @@ void pdlp_restart_strategy_t::solve_bound_constrained_trust_region( unsorted_direction_full_.data(), primal_size_h_, a_add_scalar_times_b(target_threshold_.data()), - stream_view_); + stream_view_.get()); raft::linalg::binaryOp(duality_gap.dual_solution_tr_.data(), duality_gap.dual_solution_.data(), unsorted_direction_full_.data() + primal_size_h_, dual_size_h_, a_add_scalar_times_b(target_threshold_.data()), - stream_view_); + stream_view_.get()); // project by max(min(x[i], upperbound[i]),lowerbound[i]) for primal part using f_t2 = typename type_2::type; cub::DeviceTransform::Transform(cuda::std::make_tuple(duality_gap.primal_solution_tr_.data(), @@ -2174,7 +2178,7 @@ void pdlp_restart_strategy_t::solve_bound_constrained_trust_region( duality_gap.primal_solution_tr_.data(), primal_size_h_, clamp(), - stream_view_.value()); + stream_view_.get()); // project by max(min(y[i], upperbound[i]),lowerbound[i]) raft::linalg::ternaryOp(duality_gap.dual_solution_tr_.data(), @@ -2183,7 +2187,7 @@ void pdlp_restart_strategy_t::solve_bound_constrained_trust_region( transformed_constraint_upper_bounds_.data(), dual_size_h_, constraint_clamp(), - stream_view_); + stream_view_.get()); // } } @@ -2245,7 +2249,7 @@ void pdlp_restart_strategy_t::compute_distance_traveled_from_last_rest // distance_traveled = primal_distance * 0.5 * primal_weight // + dual_distance * 0.5 / primal_weight - compute_distance_traveled_last_restart_kernel<<<1, 1, 0, stream_view_>>>( + compute_distance_traveled_last_restart_kernel<<<1, 1, 0, stream_view_.get()>>>( duality_gap.view(), primal_weight.data(), duality_gap.distance_traveled_.data()); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -2276,7 +2280,7 @@ void pdlp_restart_strategy_t::compute_primal_gradient( cusparse_view.primal_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_transpose.data(), - stream_view_)); + stream_view_.get())); } template @@ -2344,21 +2348,22 @@ void pdlp_restart_strategy_t::compute_dual_gradient( cusparse_view.dual_gradient.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_non_transpose.data(), - stream_view_)); + stream_view_.get())); // tmp_dual will contain the subgradient i_t number_of_blocks = dual_size_h_ / block_size; if (dual_size_h_ % block_size) number_of_blocks++; i_t number_of_threads = std::min(dual_size_h_, block_size); - compute_subgradient_kernel<<>>( - this->view(), problem_ptr->view(), duality_gap.view(), tmp_dual.data()); + compute_subgradient_kernel + <<>>( + this->view(), problem_ptr->view(), duality_gap.view(), tmp_dual.data()); // dual gradient = subgradient - primal_product (tmp_dual-dual_gradient) raft::linalg::eltwiseSub(duality_gap.dual_gradient_.data(), tmp_dual.data(), duality_gap.dual_gradient_.data(), dual_size_h_, - stream_view_); + stream_view_.get()); } template @@ -2389,7 +2394,7 @@ void pdlp_restart_strategy_t::compute_lagrangian_value( problem_ptr->objective_coefficients.data(), primal_stride, reusable_device_scalar_1_.data(), - stream_view_)); + stream_view_.get())); // third term, let beta be 0 to not add what is in tmp_primal, compute it and compute dot RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsespmv(handle_ptr_->get_cusparse_handle(), @@ -2401,7 +2406,7 @@ void pdlp_restart_strategy_t::compute_lagrangian_value( cusparse_view.tmp_primal.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_transpose.data(), - stream_view_)); + stream_view_.get())); RAFT_CUBLAS_TRY(raft::linalg::detail::cublasdot(handle_ptr_->get_cublas_handle(), primal_size_h_, @@ -2410,7 +2415,7 @@ void pdlp_restart_strategy_t::compute_lagrangian_value( tmp_primal.data(), primal_stride, reusable_device_scalar_2_.data(), - stream_view_)); + stream_view_.get())); // fourth term //tmp_dual still contains subgradient from the dual_gradient computation reusable_device_scalar_3_.set_value_to_zero_async(stream_view_); @@ -2421,19 +2426,19 @@ void pdlp_restart_strategy_t::compute_lagrangian_value( tmp_dual.data(), dual_stride, reusable_device_scalar_3_.data(), - stream_view_)); + stream_view_.get())); // subtract third term from second up raft::linalg::eltwiseSub(reusable_device_scalar_1_.data(), reusable_device_scalar_1_.data(), reusable_device_scalar_2_.data(), 1, - stream_view_); + stream_view_.get()); raft::linalg::eltwiseAdd(duality_gap.lagrangian_value_.data(), reusable_device_scalar_1_.data(), reusable_device_scalar_3_.data(), 1, - stream_view_); + stream_view_.get()); } template diff --git a/cpp/src/pdlp/restart_strategy/weighted_average_solution.cu b/cpp/src/pdlp/restart_strategy/weighted_average_solution.cu index 50ad27334b..b291df8dd0 100644 --- a/cpp/src/pdlp/restart_strategy/weighted_average_solution.cu +++ b/cpp/src/pdlp/restart_strategy/weighted_average_solution.cu @@ -33,19 +33,19 @@ weighted_average_solution_t::weighted_average_solution_t(raft::handle_ iterations_since_last_restart_{0}, graph(stream_view_, is_batch_mode) { - RAFT_CUDA_TRY( - cudaMemsetAsync(sum_primal_solutions_.data(), 0.0, sizeof(f_t) * primal_size_h_, stream_view_)); - RAFT_CUDA_TRY( - cudaMemsetAsync(sum_dual_solutions_.data(), 0.0, sizeof(f_t) * dual_size_h_, stream_view_)); + RAFT_CUDA_TRY(cudaMemsetAsync( + sum_primal_solutions_.data(), 0.0, sizeof(f_t) * primal_size_h_, stream_view_.get())); + RAFT_CUDA_TRY(cudaMemsetAsync( + sum_dual_solutions_.data(), 0.0, sizeof(f_t) * dual_size_h_, stream_view_.get())); } template void weighted_average_solution_t::reset_weighted_average_solution() { - RAFT_CUDA_TRY( - cudaMemsetAsync(sum_primal_solutions_.data(), 0.0, sizeof(f_t) * primal_size_h_, stream_view_)); - RAFT_CUDA_TRY( - cudaMemsetAsync(sum_dual_solutions_.data(), 0.0, sizeof(f_t) * dual_size_h_, stream_view_)); + RAFT_CUDA_TRY(cudaMemsetAsync( + sum_primal_solutions_.data(), 0.0, sizeof(f_t) * primal_size_h_, stream_view_.get())); + RAFT_CUDA_TRY(cudaMemsetAsync( + sum_dual_solutions_.data(), 0.0, sizeof(f_t) * dual_size_h_, stream_view_.get())); sum_primal_solution_weights_.set_value_to_zero_async(stream_view_); sum_dual_solution_weights_.set_value_to_zero_async(stream_view_); iterations_since_last_restart_ = 0; @@ -78,20 +78,20 @@ void weighted_average_solution_t::add_current_solution_to_weighted_ave sum_primal_solutions_.data(), primal_size_h_, a_add_scalar_times_b(weight.data()), - stream_view_.value()); + stream_view_.get()); cub::DeviceTransform::Transform( cuda::std::make_tuple(sum_dual_solutions_.data(), dual_solution), sum_dual_solutions_.data(), dual_size_h_, a_add_scalar_times_b(weight.data()), - stream_view_.value()); + stream_view_.get()); // update weight sums and count (add weight and +1 respectively) - add_weight_sums<<<1, 1, 0, stream_view_>>>(weight.data(), - weight.data(), - sum_primal_solution_weights_.data(), - sum_dual_solution_weights_.data()); + add_weight_sums<<<1, 1, 0, stream_view_.get()>>>(weight.data(), + weight.data(), + sum_primal_solution_weights_.data(), + sum_dual_solution_weights_.data()); }); iterations_since_last_restart_ += 1; @@ -103,10 +103,10 @@ void weighted_average_solution_t::compute_averages(rmm::device_uvector { // no iterations have added to the sum, so avg is all zero vector if (!iterations_since_last_restart_) { + RAFT_CUDA_TRY(cudaMemsetAsync( + avg_primal.data(), f_t(0.0), sizeof(f_t) * primal_size_h_, stream_view_.get())); RAFT_CUDA_TRY( - cudaMemsetAsync(avg_primal.data(), f_t(0.0), sizeof(f_t) * primal_size_h_, stream_view_)); - RAFT_CUDA_TRY( - cudaMemsetAsync(avg_dual.data(), f_t(0.0), sizeof(f_t) * dual_size_h_, stream_view_)); + cudaMemsetAsync(avg_dual.data(), f_t(0.0), sizeof(f_t) * dual_size_h_, stream_view_.get())); return; } @@ -114,19 +114,19 @@ void weighted_average_solution_t::compute_averages(rmm::device_uvector f_t sum_primal_solution_weights_h = sum_primal_solution_weights_.value(stream_view_); f_t sum_dual_solution_weights_h = sum_dual_solution_weights_.value(stream_view_); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); // compute sum_primal_solutions/primal_size raft::linalg::divideScalar(avg_primal.data(), sum_primal_solutions_.data(), sum_primal_solution_weights_h, primal_size_h_, - stream_view_); + stream_view_.get()); raft::linalg::divideScalar(avg_dual.data(), sum_dual_solutions_.data(), sum_dual_solution_weights_h, dual_size_h_, - stream_view_); + stream_view_.get()); } template diff --git a/cpp/src/pdlp/saddle_point.cu b/cpp/src/pdlp/saddle_point.cu index b92fdc2fb3..5edf3a5c66 100644 --- a/cpp/src/pdlp/saddle_point.cu +++ b/cpp/src/pdlp/saddle_point.cu @@ -47,13 +47,15 @@ saddle_point_state_t::saddle_point_state_t(raft::handle_t const* handl handle_ptr->get_thrust_policy(), dual_solution_.data(), dual_solution_.end(), f_t(0)); RAFT_CUDA_TRY(cudaMemsetAsync( - delta_primal_.data(), 0, sizeof(f_t) * delta_primal_.size(), handle_ptr->get_stream())); + delta_primal_.data(), 0, sizeof(f_t) * delta_primal_.size(), handle_ptr->get_stream().get())); RAFT_CUDA_TRY(cudaMemsetAsync( - delta_dual_.data(), 0, sizeof(f_t) * delta_dual_.size(), handle_ptr->get_stream())); + delta_dual_.data(), 0, sizeof(f_t) * delta_dual_.size(), handle_ptr->get_stream().get())); + RAFT_CUDA_TRY(cudaMemsetAsync(primal_gradient_.data(), + 0, + sizeof(f_t) * primal_gradient_.size(), + handle_ptr->get_stream().get())); RAFT_CUDA_TRY(cudaMemsetAsync( - primal_gradient_.data(), 0, sizeof(f_t) * primal_gradient_.size(), handle_ptr->get_stream())); - RAFT_CUDA_TRY(cudaMemsetAsync( - dual_gradient_.data(), 0, sizeof(f_t) * dual_gradient_.size(), handle_ptr->get_stream())); + dual_gradient_.data(), 0, sizeof(f_t) * dual_gradient_.size(), handle_ptr->get_stream().get())); // No need to 0 init current/next AtY, they are directlty written as result of SpMV } diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index 53c10b6a83..b7868c2a41 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -49,6 +49,8 @@ #include #include + +#include #include #include #include @@ -81,9 +83,10 @@ static void init_handler(const raft::handle_t* handle_ptr) { // Init cuBlas / cuSparse context here to avoid having it during solving time RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } // Corresponds to the first good general settings we found @@ -343,7 +346,7 @@ void adjust_dual_solution_and_reduced_cost(rmm::device_uvector& dual_soluti dual_solution.data(), dual_solution.size(), [] HD(f_t dual) { return -dual; }, - stream_view); + stream_view.get()); // z <- -z cub::DeviceTransform::Transform( @@ -351,7 +354,7 @@ void adjust_dual_solution_and_reduced_cost(rmm::device_uvector& dual_soluti reduced_cost.data(), reduced_cost.size(), [] HD(f_t reduced_cost) { return -reduced_cost; }, - stream_view); + stream_view.get()); } template @@ -1665,7 +1668,7 @@ optimization_problem_solution_t run_concurrent( { try { auto call_barrier_thread = [&]() { - rmm::cuda_stream_view barrier_stream = rmm::cuda_stream_per_thread; + rmm::cuda_stream_view barrier_stream = cuda::stream_ref{cudaStreamPerThread}; barrier_handle_ptr = std::make_unique(barrier_stream); run_barrier_thread(dual_simplex_problem, settings_pdlp, @@ -2424,7 +2427,7 @@ cuopt::mathematical_optimization::io::mps_data_model_t op_problem_to_m raft::copy(h_constr_lb.data(), d_constr_lb.data(), d_constr_lb.size(), stream); raft::copy(h_constr_ub.data(), d_constr_ub.data(), d_constr_ub.size(), stream); raft::copy(h_var_types_enum.data(), d_var_types.data(), d_var_types.size(), stream); - stream.synchronize(); + stream.sync(); if (!h_offsets.empty()) { mps.set_csr_constraint_matrix( @@ -2762,7 +2765,7 @@ std::unique_ptr> solve_lp( *gpu_problem, settings, problem_checking, use_pdlp_solver_mode, is_batch_mode); // Ensure all GPU work from the solve is complete before D2H copies in to_cpu_solution(), - // which uses rmm::cuda_stream_per_thread (a different stream than the solver used). + // which uses the per-thread default stream (a different stream than the solver used). stream.synchronize(); // Convert GPU solution back to CPU diff --git a/cpp/src/pdlp/solver_solution.cu b/cpp/src/pdlp/solver_solution.cu index 08e5ee00a8..0fbda1701e 100644 --- a/cpp/src/pdlp/solver_solution.cu +++ b/cpp/src/pdlp/solver_solution.cu @@ -235,11 +235,10 @@ void optimization_problem_solution_t::write_to_file(std::string_view f dual_solution.resize(dual_solution_.size()); reduced_cost.resize(reduced_cost_.size()); raft::copy( - primal_solution.data(), primal_solution_.data(), primal_solution_.size(), stream_view.value()); - raft::copy( - dual_solution.data(), dual_solution_.data(), dual_solution_.size(), stream_view.value()); - raft::copy(reduced_cost.data(), reduced_cost_.data(), reduced_cost_.size(), stream_view.value()); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view.value())); + primal_solution.data(), primal_solution_.data(), primal_solution_.size(), stream_view.get()); + raft::copy(dual_solution.data(), dual_solution_.data(), dual_solution_.size(), stream_view.get()); + raft::copy(reduced_cost.data(), reduced_cost_.data(), reduced_cost_.size(), stream_view.get()); + stream_view.sync(); myfile << "{ " << std::endl; myfile << "\t\"Termination reason\" : \"" << get_termination_status_string() << "\"," @@ -446,9 +445,8 @@ void optimization_problem_solution_t::write_to_sol_file( auto objective_value = get_objective_value(0); std::vector solution; solution.resize(primal_solution_.size()); - raft::copy( - solution.data(), primal_solution_.data(), primal_solution_.size(), stream_view.value()); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view.value())); + raft::copy(solution.data(), primal_solution_.data(), primal_solution_.size(), stream_view.get()); + stream_view.sync(); solution_writer_t::write_solution_to_sol_file( std::string(filename), status, objective_value, var_names_, solution); } diff --git a/cpp/src/pdlp/step_size_strategy/adaptive_step_size_strategy.cu b/cpp/src/pdlp/step_size_strategy/adaptive_step_size_strategy.cu index 5e1340a590..245c36f8bf 100644 --- a/cpp/src/pdlp/step_size_strategy/adaptive_step_size_strategy.cu +++ b/cpp/src/pdlp/step_size_strategy/adaptive_step_size_strategy.cu @@ -78,7 +78,7 @@ adaptive_step_size_strategy_t::adaptive_step_size_strategy_t( interaction_.data(), climber_strategies_.size(), primal_size_, - stream_view_.value())); + stream_view_.get())); dot_product_bytes = std::max(dot_product_bytes, byte_needed); RAFT_CUDA_TRY(cub::DeviceSegmentedReduce::Sum( @@ -88,7 +88,7 @@ adaptive_step_size_strategy_t::adaptive_step_size_strategy_t( norm_squared_delta_primal_.data(), climber_strategies_.size(), primal_size_, - stream_view_.value())); + stream_view_.get())); dot_product_bytes = std::max(dot_product_bytes, byte_needed); RAFT_CUDA_TRY(cub::DeviceSegmentedReduce::Sum( @@ -98,10 +98,10 @@ adaptive_step_size_strategy_t::adaptive_step_size_strategy_t( norm_squared_delta_dual_.data(), climber_strategies_.size(), dual_size_, - stream_view_.value())); + stream_view_.get())); dot_product_bytes = std::max(dot_product_bytes, byte_needed); - dot_product_storage.resize(dot_product_bytes, stream_view_.value()); + dot_product_storage.resize(dot_product_bytes, stream_view_.get()); } } @@ -141,12 +141,11 @@ void adaptive_step_size_strategy_t::swap_context( const auto [grid_size, block_size] = kernel_config_from_batch_size(static_cast(swap_pairs.size())); adaptive_step_size_swap_device_vectors_kernel - <<>>( - thrust::raw_pointer_cast(swap_pairs.data()), - static_cast(swap_pairs.size()), - make_span(interaction_), - make_span(norm_squared_delta_primal_), - make_span(norm_squared_delta_dual_)); + <<>>(thrust::raw_pointer_cast(swap_pairs.data()), + static_cast(swap_pairs.size()), + make_span(interaction_), + make_span(norm_squared_delta_primal_), + make_span(norm_squared_delta_dual_)); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -158,9 +157,9 @@ void adaptive_step_size_strategy_t::resize_context(i_t new_size) cuopt_assert(new_size > 0, "New size must be greater than 0"); cuopt_assert(new_size < batch_size, "New size must be less than batch size"); - interaction_.resize(new_size, stream_view_.value()); - norm_squared_delta_primal_.resize(new_size, stream_view_.value()); - norm_squared_delta_dual_.resize(new_size, stream_view_.value()); + interaction_.resize(new_size, stream_view_.get()); + norm_squared_delta_primal_.resize(new_size, stream_view_.get()); + norm_squared_delta_dual_.resize(new_size, stream_view_.get()); } template @@ -275,19 +274,19 @@ i_t adaptive_step_size_strategy_t::get_valid_step_size() const template f_t adaptive_step_size_strategy_t::get_interaction(i_t i) const { - return interaction_.element(i, stream_view_.value()); + return interaction_.element(i, stream_view_.get()); } template f_t adaptive_step_size_strategy_t::get_norm_squared_delta_primal(i_t i) const { - return norm_squared_delta_primal_.element(i, stream_view_.value()); + return norm_squared_delta_primal_.element(i, stream_view_.get()); } template f_t adaptive_step_size_strategy_t::get_norm_squared_delta_dual(i_t i) const { - return norm_squared_delta_dual_.element(i, stream_view_.value()); + return norm_squared_delta_dual_.element(i, stream_view_.get()); } template @@ -352,13 +351,13 @@ void adaptive_step_size_strategy_t::compute_step_sizes( pdhg_solver.get_saddle_point_state()); // Compute n_lim, n_next and decide if step size is valid compute_step_sizes_from_movement_and_interaction - <<<1, 1, 0, stream_view_.value()>>>(this->view(), - primal_step_size.data(), - dual_step_size.data(), - pdhg_solver.get_d_total_pdhg_iterations().data()); + <<<1, 1, 0, stream_view_.get()>>>(this->view(), + primal_step_size.data(), + dual_step_size.data(), + pdhg_solver.get_d_total_pdhg_iterations().data()); }); // Steam sync so that next call can see modification made to host var valid_step_size - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_.value())); + stream_view_.sync(); } template @@ -421,7 +420,7 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( cusparse_view.next_AtY.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_transpose.data(), - stream_view_.value())); + stream_view_.get())); } else { // TODO later batch mode: handle if not all restart RAFT_CUSPARSE_TRY( @@ -435,7 +434,7 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( cusparse_view.batch_next_AtYs.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)cusparse_view.buffer_transpose_batch.data(), - stream_view_.value())); + stream_view_.get())); } // Compute Ay' - Ay = next_Aty - current_Aty @@ -446,7 +445,7 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( tmp_primal.data(), tmp_primal.size(), cuda::std::minus<>{}, - stream_view_.value()); + stream_view_.get()); if (!batch_mode_) { // compute interaction (x'-x) . (A(y'-y)) @@ -458,7 +457,7 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( current_saddle_point_state.get_delta_primal().data(), primal_stride, interaction_.data(), - stream_view_.value())); + stream_view_.get())); // Compute movement // compute euclidean norm squared which is @@ -476,7 +475,7 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( current_saddle_point_state.get_delta_primal().data(), primal_stride, norm_squared_delta_primal_.data(), - stream_view_.value())); + stream_view_.get())); RAFT_CUBLAS_TRY( raft::linalg::detail::cublasdot(handle_ptr_->get_cublas_handle(), @@ -486,7 +485,7 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( current_saddle_point_state.get_delta_dual().data(), dual_stride, norm_squared_delta_dual_.data(), - stream_view_.value())); + stream_view_.get())); } else { // TODO later batch mode: remove this once you want to do per climber restart cub::DeviceSegmentedReduce::Sum( @@ -499,7 +498,7 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( interaction_.data(), climber_strategies_.size(), primal_size_, - stream_view_.value()); + stream_view_.get()); cub::DeviceSegmentedReduce::Sum( dot_product_storage.data(), @@ -509,7 +508,7 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( norm_squared_delta_primal_.data(), climber_strategies_.size(), primal_size_, - stream_view_.value()); + stream_view_.get()); cub::DeviceSegmentedReduce::Sum( dot_product_storage.data(), @@ -519,7 +518,7 @@ void adaptive_step_size_strategy_t::compute_interaction_and_movement( norm_squared_delta_dual_.data(), climber_strategies_.size(), dual_size_, - stream_view_.value()); + stream_view_.get()); } } @@ -562,10 +561,10 @@ void adaptive_step_size_strategy_t::get_primal_and_dual_stepsizes( cuopt_assert(step_size_->size() == climber_strategies_.size(), "step size must be the same size as the number of climber strategies"); compute_actual_stepsizes - <<>>(this->view(), - make_span(primal_step_size), - make_span(dual_step_size), - climber_strategies_.size()); + <<>>(this->view(), + make_span(primal_step_size), + make_span(dual_step_size), + climber_strategies_.size()); RAFT_CUDA_TRY(cudaPeekAtLastError()); } diff --git a/cpp/src/pdlp/swap_and_resize_helper.cuh b/cpp/src/pdlp/swap_and_resize_helper.cuh index e09ba4f9ed..cc73401538 100644 --- a/cpp/src/pdlp/swap_and_resize_helper.cuh +++ b/cpp/src/pdlp/swap_and_resize_helper.cuh @@ -83,7 +83,7 @@ void matrix_swap(rmm::device_uvector& matrix, [] HD(thrust::tuple values) -> thrust::tuple { return thrust::make_tuple(thrust::get<1>(values), thrust::get<0>(values)); }, - matrix.stream().value()); + matrix.stream().get()); } template diff --git a/cpp/src/pdlp/termination_strategy/convergence_information.cu b/cpp/src/pdlp/termination_strategy/convergence_information.cu index dd7cb925f6..f7e2725dc7 100644 --- a/cpp/src/pdlp/termination_strategy/convergence_information.cu +++ b/cpp/src/pdlp/termination_strategy/convergence_information.cu @@ -90,20 +90,22 @@ convergence_information_t::convergence_information_t( { // Zero-init per-climber scalars RAFT_CUDA_TRY(cudaMemsetAsync( - primal_objective_.data(), 0, sizeof(f_t) * primal_objective_.size(), stream_view_)); - RAFT_CUDA_TRY( - cudaMemsetAsync(dual_objective_.data(), 0, sizeof(f_t) * dual_objective_.size(), stream_view_)); - RAFT_CUDA_TRY(cudaMemsetAsync(gap_.data(), 0, sizeof(f_t) * gap_.size(), stream_view_)); - RAFT_CUDA_TRY( - cudaMemsetAsync(abs_objective_.data(), 0, sizeof(f_t) * abs_objective_.size(), stream_view_)); + primal_objective_.data(), 0, sizeof(f_t) * primal_objective_.size(), stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync( - l2_dual_residual_.data(), 0, sizeof(f_t) * l2_dual_residual_.size(), stream_view_)); + dual_objective_.data(), 0, sizeof(f_t) * dual_objective_.size(), stream_view_.get())); + RAFT_CUDA_TRY(cudaMemsetAsync(gap_.data(), 0, sizeof(f_t) * gap_.size(), stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync( - l2_primal_residual_.data(), 0, sizeof(f_t) * l2_primal_residual_.size(), stream_view_)); + abs_objective_.data(), 0, sizeof(f_t) * abs_objective_.size(), stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync( - linf_primal_residual_.data(), 0, sizeof(f_t) * linf_primal_residual_.size(), stream_view_)); + l2_dual_residual_.data(), 0, sizeof(f_t) * l2_dual_residual_.size(), stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync( - linf_dual_residual_.data(), 0, sizeof(f_t) * linf_dual_residual_.size(), stream_view_)); + l2_primal_residual_.data(), 0, sizeof(f_t) * l2_primal_residual_.size(), stream_view_.get())); + RAFT_CUDA_TRY(cudaMemsetAsync(linf_primal_residual_.data(), + 0, + sizeof(f_t) * linf_primal_residual_.size(), + stream_view_.get())); + RAFT_CUDA_TRY(cudaMemsetAsync( + linf_dual_residual_.data(), 0, sizeof(f_t) * linf_dual_residual_.size(), stream_view_.get())); init_objective_offsets(); init_reduction_storage(); @@ -111,9 +113,9 @@ convergence_information_t::convergence_information_t( // Zero the residual workspace (reused each iteration by compute_convergence_information). RAFT_CUDA_TRY(cudaMemsetAsync( - primal_residual_.data(), 0.0, sizeof(f_t) * primal_residual_.size(), stream_view_)); - RAFT_CUDA_TRY( - cudaMemsetAsync(dual_residual_.data(), 0.0, sizeof(f_t) * dual_residual_.size(), stream_view_)); + primal_residual_.data(), 0.0, sizeof(f_t) * primal_residual_.size(), stream_view_.get())); + RAFT_CUDA_TRY(cudaMemsetAsync( + dual_residual_.data(), 0.0, sizeof(f_t) * dual_residual_.size(), stream_view_.get())); } // --------------------------------------------------------------------------- @@ -285,7 +287,7 @@ void convergence_information_t::init_reduction_storage() bound_value_.begin(), dual_objective_.data(), dual_size_h_, - stream_view_); + stream_view_.get()); size_t temp_storage_bytes_2 = 0; cub::DeviceReduce::Sum(d_temp_storage, @@ -293,7 +295,7 @@ void convergence_information_t::init_reduction_storage() bound_value_.begin(), reduced_cost_dual_objective_.data(), primal_size_h_, - stream_view_); + stream_view_.get()); size_of_buffer_ = std::max({temp_storage_bytes_1, temp_storage_bytes_2}); this->rmm_tmp_buffer_ = rmm::device_buffer{size_of_buffer_, stream_view_}; @@ -359,21 +361,21 @@ void convergence_information_t::swap_context( const auto [grid_size, block_size] = kernel_config_from_batch_size(static_cast(swap_pairs.size())); convergence_information_swap_device_vectors_kernel - <<>>(thrust::raw_pointer_cast(swap_pairs.data()), - static_cast(swap_pairs.size()), - make_span(primal_objective_), - make_span(dual_objective_), - make_span(l2_primal_residual_), - make_span(l2_dual_residual_), - make_span(linf_primal_residual_), - make_span(linf_dual_residual_), - make_span(gap_), - make_span(abs_objective_), - make_span(dual_dot_), - make_span(sum_primal_slack_), - make_span(objective_offsets_), - make_span(l2_norm_primal_linear_objective_), - make_span(l2_norm_primal_right_hand_side_)); + <<>>(thrust::raw_pointer_cast(swap_pairs.data()), + static_cast(swap_pairs.size()), + make_span(primal_objective_), + make_span(dual_objective_), + make_span(l2_primal_residual_), + make_span(l2_dual_residual_), + make_span(linf_primal_residual_), + make_span(linf_dual_residual_), + make_span(gap_), + make_span(abs_objective_), + make_span(dual_dot_), + make_span(sum_primal_slack_), + make_span(objective_offsets_), + make_span(l2_norm_primal_linear_objective_), + make_span(l2_norm_primal_right_hand_side_)); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -413,7 +415,7 @@ void convergence_information_t::set_relative_primal_tolerance_factor( l2_norm_primal_right_hand_side_.data(), l2_norm_primal_right_hand_side_.size(), cuda::std::identity{}, - stream_view_); + stream_view_.get()); } template @@ -597,7 +599,7 @@ void convergence_information_t::compute_convergence_information( l2_primal_residual_.data(), l2_primal_residual_.size(), [] HD(f_t x) { return raft::sqrt(x); }, - stream_view_); + stream_view_.get()); } #ifdef CUPDLP_DEBUG_MODE @@ -666,7 +668,7 @@ void convergence_information_t::compute_convergence_information( l2_dual_residual_.data(), l2_dual_residual_.size(), [] HD(f_t x) { return raft::sqrt(x); }, - stream_view_); + stream_view_.get()); } #ifdef CUPDLP_DEBUG_MODE print("Absolute Dual Residual", l2_dual_residual_); @@ -690,14 +692,14 @@ void convergence_information_t::compute_convergence_information( // behaviour const auto [grid_size, block_size] = kernel_config_from_batch_size(climber_strategies_.size()); compute_remaining_stats_kernel - <<>>(this->view(), climber_strategies_.size()); + <<>>(this->view(), climber_strategies_.size()); RAFT_CUDA_TRY(cudaPeekAtLastError()); // cleanup for next termination evaluation RAFT_CUDA_TRY(cudaMemsetAsync( - primal_residual_.data(), 0.0, sizeof(f_t) * primal_residual_.size(), stream_view_)); - RAFT_CUDA_TRY( - cudaMemsetAsync(dual_residual_.data(), 0.0, sizeof(f_t) * dual_residual_.size(), stream_view_)); + primal_residual_.data(), 0.0, sizeof(f_t) * primal_residual_.size(), stream_view_.get())); + RAFT_CUDA_TRY(cudaMemsetAsync( + dual_residual_.data(), 0.0, sizeof(f_t) * dual_residual_.size(), stream_view_.get())); } template @@ -726,7 +728,7 @@ void convergence_information_t::compute_primal_residual( cusparse_view.tmp_dual.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_non_transpose.data(), - stream_view_)); + stream_view_.get())); } else { RAFT_CUSPARSE_TRY( raft::sparse::detail::cusparsespmm(handle_ptr_->get_cusparse_handle(), @@ -739,7 +741,7 @@ void convergence_information_t::compute_primal_residual( cusparse_view.batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)cusparse_view.buffer_non_transpose_batch.data(), - stream_view_)); + stream_view_.get())); } if (!hyper_params_.use_reflected_primal_dual) { @@ -754,7 +756,7 @@ void convergence_information_t::compute_primal_residual( problem_ptr->constraint_upper_bounds.data(), dual_size_h_, violation(), - stream_view_); + stream_view_.get()); } else { cuopt_assert(primal_residual_.size() == primal_slack_.size(), "Both vectors should had the same size"); @@ -774,7 +776,7 @@ void convergence_information_t::compute_primal_residual( raft::max(dual, f_t(0.0)) * finite_or_zero(lower) + raft::min(dual, f_t(0.0)) * finite_or_zero(upper)}; }, - stream_view_.value()); + stream_view_.get()); } #ifdef PDLP_DEBUG_MODE @@ -811,7 +813,7 @@ void convergence_information_t::compute_primal_objective_owned_partial problem_ptr->objective_coefficients.data(), primal_stride, primal_objective_.data(), - stream_view_)); + stream_view_.get())); } template @@ -846,7 +848,7 @@ template void convergence_information_t::apply_primal_objective_scaling_and_offset() { const auto [grid_size, block_size] = kernel_config_from_batch_size(climber_strategies_.size()); - apply_objective_scaling_and_offset<<>>( + apply_objective_scaling_and_offset<<>>( make_span(primal_objective_), problem_ptr->presolve_data.objective_scaling_factor, make_span(objective_offsets_), @@ -888,7 +890,7 @@ void convergence_information_t::compute_dual_residual( cusparse_view.tmp_primal.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_transpose.data(), - stream_view_)); + stream_view_.get())); } else { RAFT_CUSPARSE_TRY( raft::sparse::detail::cusparsespmm(handle_ptr_->get_cusparse_handle(), @@ -901,7 +903,7 @@ void convergence_information_t::compute_dual_residual( cusparse_view.batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)cusparse_view.buffer_transpose_batch.data(), - stream_view_)); + stream_view_.get())); } // Substract with the objective vector manually to avoid possible cusparse bug w/ nonzero beta and @@ -912,7 +914,7 @@ void convergence_information_t::compute_dual_residual( tmp_primal.data(), tmp_primal.size(), cuda::std::minus<>{}, - stream_view_); + stream_view_.get()); if (hyper_params_.use_reflected_primal_dual) { cuopt_assert(reduced_cost_.size() == dual_slack.size(), @@ -922,7 +924,7 @@ void convergence_information_t::compute_dual_residual( dual_residual_.data(), dual_residual_.size(), cuda::std::minus<>{}, - stream_view_.value()); + stream_view_.get()); } else { cuopt_expects(!batch_mode_, error_type_t::ValidationError, @@ -935,7 +937,7 @@ void convergence_information_t::compute_dual_residual( tmp_primal.data(), // primal_gradient reduced_cost_.data(), primal_size_h_, - stream_view_); + stream_view_.get()); } } @@ -963,7 +965,7 @@ void convergence_information_t::compute_dual_objective_owned_partial( primal_solution.data(), primal_stride, dual_dot_.data(), - stream_view_)); + stream_view_.get())); // sum_primal_slack_ = Σ primal_slack_[0:n_owned_cstr] // primal_slack_ is assumed populated for owned cstrs by a prior @@ -973,14 +975,14 @@ void convergence_information_t::compute_dual_objective_owned_partial( primal_slack_.data(), sum_primal_slack_.data(), static_cast(n_owned_cstr), - stream_view_); + stream_view_.get()); // dual_objective_ = dual_dot_ + sum_primal_slack_ (still a partial sum). cub::DeviceTransform::Transform(cuda::std::make_tuple(dual_dot_.data(), sum_primal_slack_.data()), dual_objective_.data(), 1, cuda::std::plus<>{}, - stream_view_); + stream_view_.get()); } template @@ -1008,14 +1010,14 @@ void convergence_information_t::compute_dual_objective( problem_ptr->constraint_upper_bounds.data(), dual_size_h_, constraint_bound_value_reduced_cost_product(), - stream_view_); + stream_view_.get()); cub::DeviceReduce::Sum(rmm_tmp_buffer_.data(), size_of_buffer_, bound_value_.begin(), dual_objective_.data(), dual_size_h_, - stream_view_); + stream_view_.get()); compute_reduced_costs_dual_objective_contribution(); @@ -1023,7 +1025,7 @@ void convergence_information_t::compute_dual_objective( dual_objective_.data(), reduced_cost_dual_objective_.data(), 1, - stream_view_); + stream_view_.get()); } else { // Reflected path. if (!batch_mode_) { @@ -1048,7 +1050,7 @@ void convergence_information_t::compute_dual_objective( dual_objective_.data(), dual_objective_.size(), cuda::std::plus<>{}, - stream_view_); + stream_view_.get()); } } @@ -1064,7 +1066,7 @@ template void convergence_information_t::apply_dual_objective_scaling_and_offset() { const auto [grid_size, block_size] = kernel_config_from_batch_size(climber_strategies_.size()); - apply_objective_scaling_and_offset<<>>( + apply_objective_scaling_and_offset<<>>( make_span(dual_objective_), problem_ptr->presolve_data.objective_scaling_factor, make_span(objective_offsets_), @@ -1084,7 +1086,7 @@ void convergence_information_t::compute_reduced_cost_from_primal_gradi bound_value_.data(), primal_size_h_, bound_value_gradient(), - stream_view_.value()); + stream_view_.get()); if (hyper_params_.handle_some_primal_gradients_on_finite_bounds_as_residuals) { raft::linalg::ternaryOp(reduced_cost_.data(), @@ -1093,14 +1095,14 @@ void convergence_information_t::compute_reduced_cost_from_primal_gradi primal_gradient.data(), primal_size_h_, copy_gradient_if_should_be_reduced_cost(), - stream_view_); + stream_view_.get()); } else { raft::linalg::binaryOp(reduced_cost_.data(), bound_value_.data(), primal_gradient.data(), primal_size_h_, copy_gradient_if_finite_bounds(), - stream_view_); + stream_view_.get()); } } @@ -1117,7 +1119,7 @@ void convergence_information_t::compute_reduced_costs_dual_objective_c bound_value_.data(), primal_size_h_, bound_value_reduced_cost_product(), - stream_view_.value()); + stream_view_.get()); // sum over bound_value*reduced_cost, but should be -inf if any element is -inf cub::DeviceReduce::Sum(rmm_tmp_buffer_.data(), @@ -1125,7 +1127,7 @@ void convergence_information_t::compute_reduced_costs_dual_objective_c bound_value_.begin(), reduced_cost_dual_objective_.data(), primal_size_h_, - stream_view_); + stream_view_.get()); } template diff --git a/cpp/src/pdlp/termination_strategy/infeasibility_information.cu b/cpp/src/pdlp/termination_strategy/infeasibility_information.cu index f4f3a45577..100365c366 100644 --- a/cpp/src/pdlp/termination_strategy/infeasibility_information.cu +++ b/cpp/src/pdlp/termination_strategy/infeasibility_information.cu @@ -104,11 +104,11 @@ infeasibility_information_t::infeasibility_information_t( RAFT_CUDA_TRY(cudaMemsetAsync(homogenous_primal_residual_.data(), 0.0, sizeof(f_t) * homogenous_primal_residual_.size(), - stream_view_)); + stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync(homogenous_dual_residual_.data(), 0.0, sizeof(f_t) * homogenous_dual_residual_.size(), - stream_view_)); + stream_view_.get())); // variable bounds in the homogenous primal are 0.0 if the original bound was finite, and // otherwise it is -inf for lower bounds and inf for upper bounds @@ -116,12 +116,12 @@ infeasibility_information_t::infeasibility_information_t( problem_ptr->constraint_lower_bounds.data(), dual_size_h_, zero_if_is_finite(), - stream_view_); + stream_view_.get()); raft::linalg::unaryOp(homogenous_dual_upper_bounds_.data(), problem_ptr->constraint_upper_bounds.data(), dual_size_h_, zero_if_is_finite(), - stream_view_); + stream_view_.get()); void* d_temp_storage = NULL; size_t temp_storage_bytes_1 = 0; @@ -130,7 +130,7 @@ infeasibility_information_t::infeasibility_information_t( bound_value_.begin(), dual_ray_linear_objective_.data(), dual_size_h_, - stream_view_); + stream_view_.get()); size_t temp_storage_bytes_2 = 0; cub::DeviceReduce::Sum(d_temp_storage, @@ -138,7 +138,7 @@ infeasibility_information_t::infeasibility_information_t( bound_value_.begin(), reduced_cost_dual_objective_.data(), primal_size_h_, - stream_view_); + stream_view_.get()); size_of_buffer_ = std::max({temp_storage_bytes_1, temp_storage_bytes_2}); this->rmm_tmp_buffer_ = rmm::device_buffer{size_of_buffer_, stream_view_}; @@ -146,20 +146,20 @@ infeasibility_information_t::infeasibility_information_t( RAFT_CUDA_TRY(cudaMemsetAsync(dual_ray_linear_objective_.data(), 0, sizeof(f_t) * dual_ray_linear_objective_.size(), - stream_view_)); + stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync(max_dual_ray_infeasibility_.data(), 0, sizeof(f_t) * max_dual_ray_infeasibility_.size(), - stream_view_)); + stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync(primal_ray_linear_objective_.data(), 0, sizeof(f_t) * primal_ray_linear_objective_.size(), - stream_view_)); + stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync(max_primal_ray_infeasibility_.data(), 0, sizeof(f_t) * max_primal_ray_infeasibility_.size(), - stream_view_)); + stream_view_.get())); } } @@ -263,7 +263,7 @@ void infeasibility_information_t::compute_infeasibility_information( if (isfinite(upper)) primal_to_return = cuda::std::min(primal_to_return, f_t(0.0)); return primal_to_return; }, - stream_view_); + stream_view_.get()); // Inf norm of primal ray segmented_sum_handler_.segmented_reduce_helper(primal_ray.data(), @@ -285,7 +285,7 @@ void infeasibility_information_t::compute_infeasibility_information( if (!isfinite(upper)) dual_to_return = cuda::std::max(dual_to_return, f_t(0.0)); return dual_to_return; }, - stream_view_); + stream_view_.get()); #ifdef CUPDLP_DEBUG_MODE print("delta_primal_solution after", primal_ray); @@ -308,7 +308,7 @@ void infeasibility_information_t::compute_infeasibility_information( if (primal_ray_inf_norm_value > f_t(0.0)) primal_ray_data[id] = primal_ray_data[id] / primal_ray_inf_norm_value; }, - stream_view_); + stream_view_.get()); #ifdef CUPDLP_DEBUG_MODE print("delta_primal_solution after scale", primal_ray); print("delta_dual_solution after scale", dual_ray); @@ -327,7 +327,7 @@ void infeasibility_information_t::compute_infeasibility_information( scaled_cusparse_view_.batch_tmp_duals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)scaled_cusparse_view_.buffer_non_transpose_batch.data(), - stream_view_)); + stream_view_.get())); RAFT_CUSPARSE_TRY( raft::sparse::detail::cusparsespmm(handle_ptr_->get_cusparse_handle(), CUSPARSE_OPERATION_NON_TRANSPOSE, @@ -339,7 +339,7 @@ void infeasibility_information_t::compute_infeasibility_information( scaled_cusparse_view_.batch_tmp_primals.get(), CUSPARSE_SPMM_CSR_ALG3, (f_t*)scaled_cusparse_view_.buffer_transpose_batch.data(), - stream_view_)); + stream_view_.get())); #ifdef CUPDLP_DEBUG_MODE print("primal_product", current_pdhg_solver.get_dual_tmp_resource()); @@ -375,7 +375,7 @@ void infeasibility_information_t::compute_infeasibility_information( return cuda::std::max(dual, f_t(0.0)) * finite_or_zero(lower) + cuda::std::min(dual, f_t(0.0)) * finite_or_zero(upper); }, - stream_view_); + stream_view_.get()); #ifdef CUPDLP_DEBUG_MODE print("primal_slack", primal_slack_); @@ -393,7 +393,7 @@ void infeasibility_information_t::compute_infeasibility_information( return cuda::std::max(-dual, f_t(0.0)) * finite_or_zero(lower) + cuda::std::min(-dual, f_t(0.0)) * finite_or_zero(upper); }, - stream_view_); + stream_view_.get()); #ifdef CUPDLP_DEBUG_MODE print("dual_slack", dual_slack_); @@ -426,7 +426,7 @@ void infeasibility_information_t::compute_infeasibility_information( cuda::std::max(primal, f_t(0.0)) * isfinite(upper)) * scale; }, - stream_view_); + stream_view_.get()); #ifdef CUPDLP_DEBUG_MODE print("primal_slack", primal_slack_); @@ -447,7 +447,7 @@ void infeasibility_information_t::compute_infeasibility_information( cuda::std::min(-dual, f_t(0.0)) * !isfinite(upper)) * scale; }, - stream_view_); + stream_view_.get()); #ifdef CUPDLP_DEBUG_MODE print("dual_slack", dual_slack_); #endif @@ -493,7 +493,7 @@ void infeasibility_information_t::compute_infeasibility_information( else return {f_t(0.0), f_t(0.0)}; }, - stream_view_); + stream_view_.get()); #ifdef CUPDLP_DEBUG_MODE printf("max_dual_ray_infeasibility=%lf\n", @@ -507,12 +507,12 @@ void infeasibility_information_t::compute_infeasibility_information( reusable_device_scalar_value_1_.data(), primal_ray_inf_norm_.data(), 1, - stream_view_); + stream_view_.get()); raft::linalg::eltwiseMultiply(neg_primal_ray_inf_norm_inverse_.data(), primal_ray_inf_norm_inverse_.data(), reusable_device_scalar_value_neg_1_.data(), 1, - stream_view_); + stream_view_.get()); compute_homogenous_primal_residual(op_problem_cusparse_view_, current_pdhg_solver.get_dual_tmp_resource()); @@ -531,14 +531,14 @@ void infeasibility_information_t::compute_infeasibility_information( my_inf_norm(dual_ray, dual_ray_inf_norm_, handle_ptr_); my_inf_norm(reduced_cost_, reduced_cost_inf_norm_, handle_ptr_); - compute_remaining_stats_kernel<<<1, 1, 0, stream_view_>>>(this->view()); + compute_remaining_stats_kernel<<<1, 1, 0, stream_view_.get()>>>(this->view()); RAFT_CUDA_TRY(cudaPeekAtLastError()); // reset for next round RAFT_CUDA_TRY(cudaMemsetAsync(homogenous_primal_residual_.data(), 0.0, sizeof(f_t) * homogenous_primal_residual_.size(), - stream_view_)); + stream_view_.get())); RAFT_CUDA_TRY(cudaMemsetAsync( homogenous_dual_residual_.data(), 0.0, sizeof(f_t) * homogenous_dual_residual_.size())); } @@ -558,7 +558,7 @@ void infeasibility_information_t::compute_homogenous_primal_residual( cusparse_view.tmp_dual.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_non_transpose.data(), - stream_view_)); + stream_view_.get())); raft::linalg::ternaryOp(homogenous_primal_residual_.data(), tmp_dual.data(), @@ -566,7 +566,7 @@ void infeasibility_information_t::compute_homogenous_primal_residual( homogenous_dual_upper_bounds_.data(), dual_size_h_, violation(), - stream_view_); + stream_view_.get()); } template @@ -599,14 +599,14 @@ void infeasibility_information_t::compute_homogenous_primal_objective( problem_ptr->objective_coefficients.data(), primal_stride, primal_ray_linear_objective_.data(), - stream_view_)); + stream_view_.get())); // just to scale from the primal ray scaling raft::linalg::eltwiseMultiply(primal_ray_linear_objective_.data(), primal_ray_linear_objective_.data(), primal_ray_inf_norm_inverse_.data(), 1, - stream_view_); + stream_view_.get()); } template @@ -628,7 +628,7 @@ void infeasibility_information_t::compute_homogenous_dual_residual( cusparse_view.tmp_primal.get(), CUSPARSE_SPMV_CSR_ALG2, (f_t*)cusparse_view.buffer_transpose.data(), - stream_view_)); + stream_view_.get())); compute_reduced_cost_from_primal_gradient(tmp_primal, primal_ray); // primal gradient is now in temp @@ -637,7 +637,7 @@ void infeasibility_information_t::compute_homogenous_dual_residual( tmp_primal.data(), // primal_gradient reduced_cost_.data(), primal_size_h_, - stream_view_); + stream_view_.get()); } template @@ -650,14 +650,14 @@ void infeasibility_information_t::compute_homogenous_dual_objective( problem_ptr->constraint_upper_bounds.data(), dual_size_h_, constraint_bound_value_reduced_cost_product(), - stream_view_); + stream_view_.get()); cub::DeviceReduce::Sum(rmm_tmp_buffer_.data(), size_of_buffer_, bound_value_.begin(), dual_ray_linear_objective_.data(), dual_size_h_, - stream_view_); + stream_view_.get()); #ifdef PDLP_DEBUG_MODE std::cout << "-compute_homogenous_dual_objective:\n" @@ -671,7 +671,7 @@ void infeasibility_information_t::compute_homogenous_dual_objective( dual_ray_linear_objective_.data(), reduced_cost_dual_objective_.data(), 1, - stream_view_); + stream_view_.get()); #ifdef PDLP_DEBUG_MODE std::cout << " reduced_cost_dual_objective_=" << reduced_cost_dual_objective_.value(stream_view_) << std::endl; @@ -690,7 +690,7 @@ void infeasibility_information_t::compute_reduced_cost_from_primal_gra bound_value_.data(), primal_size_h_, bound_value_gradient(), - stream_view_.value()); + stream_view_.get()); if (hyper_params_.handle_some_primal_gradients_on_finite_bounds_as_residuals) { raft::linalg::ternaryOp(reduced_cost_.data(), @@ -699,14 +699,14 @@ void infeasibility_information_t::compute_reduced_cost_from_primal_gra primal_gradient.data(), primal_size_h_, copy_gradient_if_should_be_reduced_cost(), - stream_view_); + stream_view_.get()); } else { raft::linalg::binaryOp(reduced_cost_.data(), bound_value_.data(), primal_gradient.data(), primal_size_h_, copy_gradient_if_finite_bounds(), - stream_view_); + stream_view_.get()); } } @@ -722,7 +722,7 @@ void infeasibility_information_t::compute_reduced_costs_dual_objective bound_value_.data(), primal_size_h_, bound_value_reduced_cost_product(), - stream_view_); + stream_view_.get()); // sum over bound_value*reduced_cost cub::DeviceReduce::Sum(rmm_tmp_buffer_.data(), @@ -730,7 +730,7 @@ void infeasibility_information_t::compute_reduced_costs_dual_objective bound_value_.begin(), reduced_cost_dual_objective_.data(), primal_size_h_, - stream_view_); + stream_view_.get()); } template diff --git a/cpp/src/pdlp/termination_strategy/termination_strategy.cu b/cpp/src/pdlp/termination_strategy/termination_strategy.cu index 13acee138c..9738675690 100644 --- a/cpp/src/pdlp/termination_strategy/termination_strategy.cu +++ b/cpp/src/pdlp/termination_strategy/termination_strategy.cu @@ -188,7 +188,7 @@ void pdlp_termination_strategy_t::evaluate_termination_criteria( check_termination_criteria(); // Sync to make sure the termination status is updated - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } template @@ -420,13 +420,13 @@ void pdlp_termination_strategy_t::check_termination_criteria() #endif const auto [grid_size, block_size] = kernel_config_from_batch_size(climber_strategies_.size()); check_termination_criteria_kernel - <<>>(convergence_information_.view(), - infeasibility_information_.view(), - make_span(termination_status_), - settings_.tolerances, - settings_.detect_infeasibility, - settings_.per_constraint_residual, - climber_strategies_.size()); + <<>>(convergence_information_.view(), + infeasibility_information_.view(), + make_span(termination_status_), + settings_.tolerances, + settings_.detect_infeasibility, + settings_.per_constraint_residual, + climber_strategies_.size()); RAFT_CUDA_TRY(cudaPeekAtLastError()); } @@ -499,7 +499,7 @@ void pdlp_termination_strategy_t::fill_gpu_terms_stats(i_t number_of_i const bool accept_primal_feasible = settings_.first_primal_feasible || settings_.all_primal_feasible; const auto [grid_size, block_size] = kernel_config_from_batch_size(climber_strategies_.size()); - fill_gpu_terms_stats_kernel<<>>( + fill_gpu_terms_stats_kernel<<>>( make_span(termination_status_), make_span(original_index_), gpu_batch_additional_termination_information_.view(), @@ -509,7 +509,7 @@ void pdlp_termination_strategy_t::fill_gpu_terms_stats(i_t number_of_i settings_.per_constraint_residual, force_all); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); } template @@ -641,7 +641,7 @@ pdlp_termination_strategy_t::fill_return_problem_solution( } } - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view_)); + stream_view_.sync(); if (deep_copy) { cuopt_assert( diff --git a/cpp/src/pdlp/translate.hpp b/cpp/src/pdlp/translate.hpp index d45d25ecfd..135d3168f6 100644 --- a/cpp/src/pdlp/translate.hpp +++ b/cpp/src/pdlp/translate.hpp @@ -354,14 +354,14 @@ void translate_to_crossover_problem(const mip::problem_t& problem, csr_A.j = std::vector(cuopt::host_copy(problem.variables, stream)); csr_A.row_start = std::vector(cuopt::host_copy(problem.offsets, stream)); - stream.synchronize(); + stream.sync(); CUOPT_LOG_DEBUG("Converting to compressed column"); csr_A.to_compressed_col(lp.A); CUOPT_LOG_DEBUG("Converted to compressed column"); std::vector slack(problem.n_constraints); std::vector tmp_x = cuopt::host_copy(sol.get_primal_solution(), stream); - stream.synchronize(); + stream.sync(); matrix_vector_multiply(lp.A, f_t(1.0), tmp_x, f_t(0.0), slack); CUOPT_LOG_DEBUG("Multiplied A and x"); @@ -400,7 +400,7 @@ void translate_to_crossover_problem(const mip::problem_t& problem, std::copy(lower.begin(), lower.begin() + problem.n_variables, lp.lower.begin()); std::copy(upper.begin(), upper.begin() + problem.n_variables, lp.upper.begin()); - problem.handle_ptr->get_stream().synchronize(); + problem.handle_ptr->get_stream().sync(); for (i_t i = 0; i < m; ++i) { lp.lower[problem.n_variables + i] = constraint_lower[i]; lp.upper[problem.n_variables + i] = constraint_upper[i]; @@ -420,7 +420,7 @@ void translate_to_crossover_problem(const mip::problem_t& problem, initial_solution.y = cuopt::host_copy(sol.get_dual_solution(), stream); std::vector tmp_z = cuopt::host_copy(sol.get_reduced_cost(), stream); - stream.synchronize(); + stream.sync(); std::copy(tmp_z.begin(), tmp_z.begin() + problem.n_variables, initial_solution.z.begin()); for (i_t j = problem.n_variables; j < n; ++j) { initial_solution.z[j] = initial_solution.y[j - problem.n_variables]; diff --git a/cpp/src/pdlp/utilities/cython_solve.cu b/cpp/src/pdlp/utilities/cython_solve.cu index ed77c4f722..4572c7b4b6 100644 --- a/cpp/src/pdlp/utilities/cython_solve.cu +++ b/cpp/src/pdlp/utilities/cython_solve.cu @@ -22,6 +22,8 @@ #include #include +#include + #include #include @@ -129,18 +131,20 @@ std::unique_ptr call_solve( // all returned device_buffers with a long-lived stream for safe deallocation later. auto& gpu_sols = std::get(response.lp_ret.solutions_); - gpu_sols.primal_solution_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.dual_solution_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.reduced_cost_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.current_primal_solution_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.current_dual_solution_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.initial_primal_average_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.initial_dual_average_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.current_ATY_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.sum_primal_solutions_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.sum_dual_solutions_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.last_restart_duality_gap_primal_solution_->set_stream(rmm::cuda_stream_per_thread); - gpu_sols.last_restart_duality_gap_dual_solution_->set_stream(rmm::cuda_stream_per_thread); + gpu_sols.primal_solution_->set_stream(cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.dual_solution_->set_stream(cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.reduced_cost_->set_stream(cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.current_primal_solution_->set_stream(cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.current_dual_solution_->set_stream(cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.initial_primal_average_->set_stream(cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.initial_dual_average_->set_stream(cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.current_ATY_->set_stream(cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.sum_primal_solutions_->set_stream(cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.sum_dual_solutions_->set_stream(cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.last_restart_duality_gap_primal_solution_->set_stream( + cuda::stream_ref{cudaStreamPerThread}); + gpu_sols.last_restart_duality_gap_dual_solution_->set_stream( + cuda::stream_ref{cudaStreamPerThread}); } else { // MIP solve @@ -153,7 +157,7 @@ std::unique_ptr call_solve( // Same stream reassociation as the LP path above. auto& gpu_sol = std::get(response.mip_ret.solution_); - gpu_sol->set_stream(rmm::cuda_stream_per_thread); + gpu_sol->set_stream(cuda::stream_ref{cudaStreamPerThread}); } // Reset warmstart data streams in solver_settings (skip in batch mode to avoid data race @@ -161,17 +165,17 @@ std::unique_ptr call_solve( if (!is_batch_mode) { auto& warmstart_data = solver_settings->get_pdlp_settings().get_pdlp_warm_start_data(); if (warmstart_data.current_primal_solution_.size() > 0) { - warmstart_data.current_primal_solution_.set_stream(rmm::cuda_stream_per_thread); - warmstart_data.current_dual_solution_.set_stream(rmm::cuda_stream_per_thread); - warmstart_data.initial_primal_average_.set_stream(rmm::cuda_stream_per_thread); - warmstart_data.initial_dual_average_.set_stream(rmm::cuda_stream_per_thread); - warmstart_data.current_ATY_.set_stream(rmm::cuda_stream_per_thread); - warmstart_data.sum_primal_solutions_.set_stream(rmm::cuda_stream_per_thread); - warmstart_data.sum_dual_solutions_.set_stream(rmm::cuda_stream_per_thread); + warmstart_data.current_primal_solution_.set_stream(cuda::stream_ref{cudaStreamPerThread}); + warmstart_data.current_dual_solution_.set_stream(cuda::stream_ref{cudaStreamPerThread}); + warmstart_data.initial_primal_average_.set_stream(cuda::stream_ref{cudaStreamPerThread}); + warmstart_data.initial_dual_average_.set_stream(cuda::stream_ref{cudaStreamPerThread}); + warmstart_data.current_ATY_.set_stream(cuda::stream_ref{cudaStreamPerThread}); + warmstart_data.sum_primal_solutions_.set_stream(cuda::stream_ref{cudaStreamPerThread}); + warmstart_data.sum_dual_solutions_.set_stream(cuda::stream_ref{cudaStreamPerThread}); warmstart_data.last_restart_duality_gap_primal_solution_.set_stream( - rmm::cuda_stream_per_thread); + cuda::stream_ref{cudaStreamPerThread}); warmstart_data.last_restart_duality_gap_dual_solution_.set_stream( - rmm::cuda_stream_per_thread); + cuda::stream_ref{cudaStreamPerThread}); } } diff --git a/cpp/src/pdlp/utils.cuh b/cpp/src/pdlp/utils.cuh index 25cd790a48..c0a39faa73 100644 --- a/cpp/src/pdlp/utils.cuh +++ b/cpp/src/pdlp/utils.cuh @@ -329,7 +329,7 @@ void inline combine_constraint_bounds(const mip::problem_t& op_problem combined_bounds.data(), combined_bounds.size(), combine_finite_abs_bounds(), - op_problem.handle_ptr->get_stream()); + op_problem.handle_ptr->get_stream().get()); } // Same as compute_sum_bounds, but without the fused sqrt. @@ -356,7 +356,7 @@ void inline compute_sum_bounds_squared(const rmm::device_uvector& constrain cuda::std::plus<>{}, rhs_sum_of_squares_t{}, f_t(0), - stream_view); + stream_view.get()); d_temp_storage.resize(bytes, stream_view); @@ -369,8 +369,8 @@ void inline compute_sum_bounds_squared(const rmm::device_uvector& constrain cuda::std::plus<>{}, rhs_sum_of_squares_t{}, f_t(0), - stream_view); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view)); + stream_view.get()); + stream_view.sync(); } // Weighted sum of squares of the first n entries of `values` (no fused sqrt). @@ -394,7 +394,7 @@ void inline compute_sum_weighted_squares(const rmm::device_uvector& values, cuda::std::plus<>{}, weighted_square_op{weight}, f_t(0), - stream_view); + stream_view.get()); d_temp_storage.resize(bytes, stream_view); @@ -406,8 +406,8 @@ void inline compute_sum_weighted_squares(const rmm::device_uvector& values, cuda::std::plus<>{}, weighted_square_op{weight}, f_t(0), - stream_view); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view)); + stream_view.get()); + stream_view.sync(); } // Like compute_sum_bounds_squared, but writes sqrt(sum of squares) (the L2 norm). @@ -428,7 +428,7 @@ void inline compute_sum_bounds(const rmm::device_uvector& constraint_lower_ cuda::std::plus<>{}, rhs_sum_of_squares_t{}, f_t(0), - stream_view); + stream_view.get()); d_temp_storage.resize(bytes, stream_view); @@ -441,8 +441,8 @@ void inline compute_sum_bounds(const rmm::device_uvector& constraint_lower_ cuda::std::plus<>{}, rhs_sum_of_squares_t{}, f_t(0), - stream_view); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view)); + stream_view.get()); + stream_view.sync(); } template @@ -688,7 +688,7 @@ void inline my_l2_norm(const f_t* in, f_t* out, size_t size, raft::handle_t cons { constexpr int stride = 1; RAFT_CUBLAS_TRY(raft::linalg::detail::cublasnrm2( - handle_ptr->get_cublas_handle(), size, in, stride, out, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), size, in, stride, out, handle_ptr->get_stream().get())); } template @@ -721,7 +721,7 @@ void inline my_l2_weighted_norm(const f_t* input_vector, (i_t)size, 1, f_t(0.0), - stream, + stream.get(), false, main_op, raft::Sum(), @@ -779,10 +779,10 @@ void inline my_inf_norm(const rmm::device_uvector& input_vector, void* d_temp = nullptr; size_t temp_bytes = 0; - cub::DeviceReduce::Max(d_temp, temp_bytes, abs_iter, result, n, stream); + cub::DeviceReduce::Max(d_temp, temp_bytes, abs_iter, result, n, stream.get()); rmm::device_buffer temp_buf(temp_bytes, stream); - cub::DeviceReduce::Max(temp_buf.data(), temp_bytes, abs_iter, result, n, stream); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); + cub::DeviceReduce::Max(temp_buf.data(), temp_bytes, abs_iter, result, n, stream.get()); + stream.sync(); } template diff --git a/cpp/src/routing/adapters/adapted_sol.cuh b/cpp/src/routing/adapters/adapted_sol.cuh index e94e401202..b152da3d95 100644 --- a/cpp/src/routing/adapters/adapted_sol.cuh +++ b/cpp/src/routing/adapters/adapted_sol.cuh @@ -351,7 +351,7 @@ struct adapted_sol_t { i_t id_to_remove = routes_to_remove[i] - i; remove_host_route(id_to_remove); } - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); populate_host_data(false, true); check_device_host_coherence(); } diff --git a/cpp/src/routing/adapters/assignment_adapter.cuh b/cpp/src/routing/adapters/assignment_adapter.cuh index c41c3e161e..9e823d8464 100644 --- a/cpp/src/routing/adapters/assignment_adapter.cuh +++ b/cpp/src/routing/adapters/assignment_adapter.cuh @@ -27,7 +27,7 @@ assignment_t ges_solver_t::get_ges_assignment( // the stream should be the initial handle stream and not the sol_handle stream as this data will // be exported auto stream = problem.handle_ptr->get_stream(); - stream.synchronize(); + stream.sync(); const auto& problem = *sol.problem_ptr; i_t n_output_nodes = sol.get_n_routes() * 2 + sol.get_num_depot_excluded_orders() + @@ -39,7 +39,7 @@ assignment_t ges_solver_t::get_ges_assignment( rmm::device_uvector route_locations_out(0, stream); rmm::device_uvector node_types_out(0, stream); auto accepted_out = cuopt::device_copy(accepted, stream); - stream.synchronize(); + stream.sync(); std::vector node_types_out_h(n_output_nodes); std::vector route_out_h(n_output_nodes); std::vector truck_id_out_h(n_output_nodes); @@ -150,7 +150,7 @@ assignment_t ges_solver_t::get_ges_assignment( auto unserviced_nodes_h = sol.get_unserviced_nodes(); auto unserviced_nodes = cuopt::device_copy(unserviced_nodes_h, stream); - stream.synchronize(); + stream.sync(); std::map objective_values; for (int i = 0; i < (int)objective_t::SIZE; ++i) { diff --git a/cpp/src/routing/adapters/solution_adapter.cuh b/cpp/src/routing/adapters/solution_adapter.cuh index 5571f4b3b3..91e3510a82 100644 --- a/cpp/src/routing/adapters/solution_adapter.cuh +++ b/cpp/src/routing/adapters/solution_adapter.cuh @@ -32,7 +32,7 @@ void fill_routes_data(solution_t& sol, auto h_node_types = cuopt::host_copy(assignment.get_node_types(), stream); sol.sol_handle->sync_stream(); - assignment.get_truck_id().stream().synchronize(); + assignment.get_truck_id().stream().sync(); i_t route_id = -1; NodeInfo curr_node; // depot is counter only once diff --git a/cpp/src/routing/assignment.cu b/cpp/src/routing/assignment.cu index be40bda183..14862914ef 100644 --- a/cpp/src/routing/assignment.cu +++ b/cpp/src/routing/assignment.cu @@ -196,10 +196,9 @@ void assignment_t::to_csv(std::string_view filename, rmm::cuda_stream_view route.resize(route_.size()); arrival_stamp.resize(arrival_stamp_.size()); truck_id.resize(truck_id_.size()); - raft::copy(route.data(), route_.data(), route_.size(), stream_view.value()); - raft::copy( - arrival_stamp.data(), arrival_stamp_.data(), arrival_stamp_.size(), stream_view.value()); - raft::copy(truck_id.data(), truck_id_.data(), truck_id_.size(), stream_view.value()); + raft::copy(route.data(), route_.data(), route_.size(), stream_view.get()); + raft::copy(arrival_stamp.data(), arrival_stamp_.data(), arrival_stamp_.size(), stream_view.get()); + raft::copy(truck_id.data(), truck_id_.data(), truck_id_.size(), stream_view.get()); std::ofstream myfile(filename.data()); std::cout << "truck_id,\troute,\tarrival_time\n"; for (size_t i = 0; i < route.size(); i++) diff --git a/cpp/src/routing/cpu_routing_problem.cu b/cpp/src/routing/cpu_routing_problem.cu index fb61c7f8cf..f48b10a592 100644 --- a/cpp/src/routing/cpu_routing_problem.cu +++ b/cpp/src/routing/cpu_routing_problem.cu @@ -87,7 +87,7 @@ std::unique_ptr> copy_u8_as_bool(std::vector // as_bool is a local temporary and the H2D copy above is async; drain the // stream before it goes out of scope so the copy does not read freed host // memory. - stream.synchronize(); + stream.sync(); return d; } @@ -302,7 +302,7 @@ cpu_routing_problem_t::to_device(raft::handle_t* handle) const data->init_types = copy_vector(types, stream); // types is a local temporary feeding an async H2D copy; drain before it // goes out of scope. - stream.synchronize(); + stream.sync(); int32_t n_nodes = static_cast(initial_solutions.routes.size()); int32_t n_sols = static_cast(initial_solutions.sol_offsets.size()); diff --git a/cpp/src/routing/crossovers/optimal_eax_cycles.cu b/cpp/src/routing/crossovers/optimal_eax_cycles.cu index d5547d2c21..3d64a4a9f0 100644 --- a/cpp/src/routing/crossovers/optimal_eax_cycles.cu +++ b/cpp/src/routing/crossovers/optimal_eax_cycles.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2023-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -151,7 +151,7 @@ void optimal_cycles_t::get_min_delta_and_index( eax_cycle_delta.data(), index_delta_pair.data(), num_items, - sol.sol.sol_handle->get_stream()); + sol.sol.sol_handle->get_stream().get()); // Allocate temporary storage if (d_cub_storage_bytes.size() < temp_storage_bytes) { d_cub_storage_bytes.resize(temp_storage_bytes, sol.sol.sol_handle->get_stream()); @@ -162,7 +162,7 @@ void optimal_cycles_t::get_min_delta_and_index( eax_cycle_delta.data(), index_delta_pair.data(), num_items, - sol.sol.sol_handle->get_stream()); + sol.sol.sol_handle->get_stream().get()); } template @@ -179,8 +179,9 @@ bool optimal_cycles_t::insert_cycle_to_found_position( return false; } // prepare the rotations once and copy them to respective device arrays - insert_optimal_rotation_kernel<<<1, TPB, sh_size, solution.sol_handle->get_stream()>>>( - solution.view(), index_delta_pair.data(), eax_fragment.view(), n_rotations); + insert_optimal_rotation_kernel + <<<1, TPB, sh_size, solution.sol_handle->get_stream().get()>>>( + solution.view(), index_delta_pair.data(), eax_fragment.view(), n_rotations); solution.compute_route_id_per_node(); solution.compute_cost(); return true; @@ -216,19 +217,20 @@ bool optimal_cycles_t::add_cycles_request( constexpr i_t TPB = 128; // prepare the rotations once and copy them to respective device arrays - create_rotations_kernel<<<1, TPB, 0, solution.sol_handle->get_stream()>>>( + create_rotations_kernel<<<1, TPB, 0, solution.sol_handle->get_stream().get()>>>( solution.view(), raft::device_span>(d_cycle.data(), d_cycle.size()), eax_fragment.view(), n_rotations); i_t n_blocks = (n_rotations * n_positions + TPB - 1) / TPB; - find_optimal_position_kernel<<get_stream()>>>( - solution.view(), - resource.ls.move_candidates.view(), - eax_fragment.view(), - n_rotations, - raft::device_span(eax_cycle_delta.data(), eax_cycle_delta.size())); + find_optimal_position_kernel + <<get_stream().get()>>>( + solution.view(), + resource.ls.move_candidates.view(), + eax_fragment.view(), + n_rotations, + raft::device_span(eax_cycle_delta.data(), eax_cycle_delta.size())); get_min_delta_and_index(sol, n_rotations * n_positions); bool success = insert_cycle_to_found_position(sol, n_rotations); diff --git a/cpp/src/routing/crossovers/ox_recombiner.cuh b/cpp/src/routing/crossovers/ox_recombiner.cuh index cefbd8df15..f16f9d2a11 100644 --- a/cpp/src/routing/crossovers/ox_recombiner.cuh +++ b/cpp/src/routing/crossovers/ox_recombiner.cuh @@ -592,7 +592,7 @@ struct OX { num_segments, row_offsets.data(), row_offsets.data() + 1, - stream_view); + stream_view.get()); d_tmp_storage_bytes.resize(tmp_storage_bytes, stream_view); cub::DeviceSegmentedSort::SortPairs(d_tmp_storage_bytes.data(), tmp_storage_bytes, @@ -604,8 +604,8 @@ struct OX { num_segments, row_offsets.data(), row_offsets.data() + 1, - stream_view); - RAFT_CHECK_CUDA(stream_view); + stream_view.get()); + RAFT_CHECK_CUDA(stream_view.get()); thrust::gather(policy, val_map.begin(), val_map.end(), graph.buckets.data(), gather_int.data()); thrust::gather( @@ -622,9 +622,9 @@ struct OX { auto const n_blocks = n_buckets * d_graph.get_num_vertices(); transpose_graph.reset(A.sol.sol_handle); - transpose_graph_kernel<<get_stream()>>>( + transpose_graph_kernel<<get_stream().get()>>>( d_graph.view(), transpose_graph.view(), max_route_len); - RAFT_CHECK_CUDA(A.sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(A.sol.sol_handle->get_stream().get()); sort_graph_edges(A, transpose_graph); } @@ -646,11 +646,11 @@ struct OX { async_fill(d_path_cost, std::numeric_limits::max(), A.sol.sol_handle->get_stream()); async_fill(d_predecessor, -1, A.sol.sol_handle->get_stream()); async_fill(d_predecessor_vehicle, -1, A.sol.sol_handle->get_stream()); - bellman_ford_init<<<1, 1, 0, A.sol.sol_handle->get_stream()>>>( + bellman_ford_init<<<1, 1, 0, A.sol.sol_handle->get_stream().get()>>>( raft::device_span(d_path_cost.data(), d_path_cost.size()), raft::device_span(d_predecessor.data(), d_predecessor.size()), raft::device_span(d_predecessor_vehicle.data(), d_predecessor_vehicle.size())); - RAFT_CHECK_CUDA(A.sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(A.sol.sol_handle->get_stream().get()); constexpr auto const TPB = 128; auto min_cost_of_last_column = std::numeric_limits::max(); @@ -665,7 +665,7 @@ struct OX { // routes number exceeds num nodes. Stop the search here if (n_blocks == 0) { break; } bellman_ford_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( A.sol.view(), transpose_graph.view(), raft::device_span(d_path_cost.data(), d_path_cost.size()), @@ -675,7 +675,7 @@ struct OX { row_size, i, run_heuristic); - RAFT_CHECK_CUDA(A.sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(A.sol.sol_handle->get_stream().get()); if (optimal_routes_search) { raft::copy(&cost_of_last_column, @@ -971,14 +971,14 @@ struct OX { return; } calculate_edge_costs_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( A.sol.view(), d_graph.view(), raft::device_span(d_offspring.data(), d_offspring.size()), raft::device_span(d_vehicle_id_per_bucket.data(), d_vehicle_id_per_bucket.size()), max_route_len, gpu_weight); - RAFT_CHECK_CUDA(A.sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(A.sol.sol_handle->get_stream().get()); A.sol.sol_handle->sync_stream(); if (A.problem->data_view_ptr->get_vehicle_locations().first == nullptr) { diff --git a/cpp/src/routing/cuda_graph.cuh b/cpp/src/routing/cuda_graph.cuh index 1fb2425d2c..40e1b95f49 100644 --- a/cpp/src/routing/cuda_graph.cuh +++ b/cpp/src/routing/cuda_graph.cuh @@ -22,7 +22,7 @@ struct cuda_graph_t { { // Use ThreadLocal mode to allow multi-threaded batch execution // Global mode blocks other streams from performing operations during capture - cudaStreamBeginCapture(stream, cudaStreamCaptureModeThreadLocal); + cudaStreamBeginCapture(stream.get(), cudaStreamCaptureModeThreadLocal); capture_started = true; } @@ -30,7 +30,7 @@ struct cuda_graph_t { { cuopt_assert(capture_started, "start_capture was not called before end_capture!"); cuopt_expects(capture_started, error_type_t::RuntimeError, "A runtime error occurred!"); - cudaStreamEndCapture(stream, &graph); + cudaStreamEndCapture(stream.get(), &graph); capture_started = false; if (graph_created) { // If the graph fails to update, errorNode will be set to the @@ -52,7 +52,7 @@ struct cuda_graph_t { cudaGraphDestroy(graph); } - void launch_graph(rmm::cuda_stream_view stream) { cudaGraphLaunch(instance, stream); } + void launch_graph(rmm::cuda_stream_view stream) { cudaGraphLaunch(instance, stream.get()); } bool graph_created = false; bool capture_started = false; diff --git a/cpp/src/routing/distance_engine/waypoint_matrix.cpp b/cpp/src/routing/distance_engine/waypoint_matrix.cpp index 030c8790ea..e02d2ce970 100644 --- a/cpp/src/routing/distance_engine/waypoint_matrix.cpp +++ b/cpp/src/routing/distance_engine/waypoint_matrix.cpp @@ -248,7 +248,7 @@ void waypoint_matrix_t::compute_cost_matrix(f_t* d_cost_matrix, std::vector cost_matrix = mpsp(target_locations, n_target_locations); raft::copy(d_cost_matrix, cost_matrix.data(), cost_matrix.size(), stream_view_); - stream_view_.synchronize(); + stream_view_.sync(); } // Location values are greater or equal to n_target_locations @@ -293,7 +293,7 @@ waypoint_matrix_t::compute_waypoint_sequence(i_t const* target_locatio std::vector h_locations(n_locations); raft::copy(h_locations.data(), locations, n_locations, stream_view_); - stream_view_.synchronize(); + stream_view_.sync(); // Locations validity checks check_locations(h_locations.data(), n_locations, n_target_locations); @@ -321,7 +321,7 @@ waypoint_matrix_t::compute_waypoint_sequence(i_t const* target_locatio raft::copy(paths_offsets_out.data(), paths_offsets.data(), paths_offsets.size(), stream_view_); raft::copy(paths_list_out.data(), paths_list.data(), paths_list.size(), stream_view_); - stream_view_.synchronize(); + stream_view_.sync(); return {std::make_unique(paths_offsets_out.release()), std::make_unique(paths_list_out.release())}; @@ -406,7 +406,7 @@ void waypoint_matrix_t::compute_shortest_path_costs(f_t* d_custom_matr raft::copy( d_custom_matrix, shortest_path_matrix.data(), shortest_path_matrix.size(), stream_view_); - stream_view_.synchronize(); + stream_view_.sync(); } template class CUOPT_EXPORT waypoint_matrix_t; diff --git a/cpp/src/routing/fleet_info.cu b/cpp/src/routing/fleet_info.cu index 317191f51f..71997db103 100644 --- a/cpp/src/routing/fleet_info.cu +++ b/cpp/src/routing/fleet_info.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2023-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -153,9 +153,9 @@ void populate_fleet_info(data_model_view_t const& data_model, if (auto [start_locations, return_locations] = data_model.get_vehicle_locations(); start_locations != nullptr) { raft::copy( - fleet_info_.v_start_locations_.data(), start_locations, fleet_size, stream_view.value()); + fleet_info_.v_start_locations_.data(), start_locations, fleet_size, stream_view.get()); raft::copy( - fleet_info_.v_return_locations_.data(), return_locations, fleet_size, stream_view.value()); + fleet_info_.v_return_locations_.data(), return_locations, fleet_size, stream_view.get()); is_homogenous = is_homogenous && all_entries_are_equal(handle_ptr_, fleet_info_.v_start_locations_.data(), fleet_size); @@ -176,7 +176,7 @@ void populate_fleet_info(data_model_view_t const& data_model, if (auto drop_return_trip = data_model.get_drop_return_trips(); drop_return_trip) { raft::copy( - fleet_info_.v_drop_return_trip_.data(), drop_return_trip, fleet_size, stream_view.value()); + fleet_info_.v_drop_return_trip_.data(), drop_return_trip, fleet_size, stream_view.get()); is_homogenous = is_homogenous && all_entries_are_equal(handle_ptr_, fleet_info_.v_drop_return_trip_.data(), fleet_size); @@ -189,7 +189,7 @@ void populate_fleet_info(data_model_view_t const& data_model, if (auto skip_first_trip = data_model.get_skip_first_trips(); skip_first_trip) { raft::copy( - fleet_info_.v_skip_first_trip_.data(), skip_first_trip, fleet_size, stream_view.value()); + fleet_info_.v_skip_first_trip_.data(), skip_first_trip, fleet_size, stream_view.get()); is_homogenous = is_homogenous && all_entries_are_equal(handle_ptr_, fleet_info_.v_skip_first_trip_.data(), fleet_size); diff --git a/cpp/src/routing/generator/generator.cu b/cpp/src/routing/generator/generator.cu index 587792ef11..d9042e19cd 100644 --- a/cpp/src/routing/generator/generator.cu +++ b/cpp/src/routing/generator/generator.cu @@ -119,7 +119,7 @@ detail::fleet_order_constraints_t generate_fleet_order_constraints( n_orders - 1, params.min_service_time, params.max_service_time + 1, - handle.get_stream()); + handle.get_stream().get()); } return fleet_order_constraints; } @@ -188,7 +188,7 @@ coordinates_t generate_coordinates(raft::handle_t& handle, params.n_locations, n_cols, n_clusters, - handle.get_stream(), + handle.get_stream().get(), false, (f_t*)nullptr, (f_t*)nullptr, @@ -228,13 +228,13 @@ d_mdarray_t generate_matrices(raft::handle_t& handle, rmm::device_uvector v_rands(params.n_locations * params.n_locations, handle.get_stream()); detail::build_cost_matrix - <<>>(cost_matrix.data(), - std::get<0>(coordinates).data(), - std::get<1>(coordinates).data(), - params.n_locations, - params.asymmetric, - asymmetry_scalar); - RAFT_CHECK_CUDA(handle.get_stream()); + <<>>(cost_matrix.data(), + std::get<0>(coordinates).data(), + std::get<1>(coordinates).data(), + params.n_locations, + params.asymmetric, + asymmetry_scalar); + RAFT_CHECK_CUDA(handle.get_stream().get()); auto seed = params.seed; auto matrices = detail::create_device_mdarray( @@ -248,7 +248,7 @@ d_mdarray_t generate_matrices(raft::handle_t& handle, v_rands.size(), static_cast(1.1), static_cast(1.5), - handle.get_stream()); + handle.get_stream().get()); auto matrix_span = matrices.get_cost_matrix(vehicle_type, matrix_type); @@ -309,7 +309,7 @@ rmm::device_uvector generate_vehicle_capacities(raft::handle_t& handle, fleet_size, static_cast(h_min_capacities[i]), static_cast(h_max_capacities[i] + 1), - handle.get_stream()); + handle.get_stream().get()); } return capacities; } @@ -334,7 +334,7 @@ rmm::device_uvector generate_demands(raft::handle_t& handle, params.n_locations - 1, static_cast(h_min_demand[i]), static_cast(h_max_demand[i] + 1), - handle.get_stream()); + handle.get_stream().get()); } return demands; } @@ -467,7 +467,7 @@ rmm ::device_uvector create_service_time(raft::handle_t& handle, v_service_time.size() - 1, params.min_service_time, params.max_service_time + 1, - handle.get_stream()); + handle.get_stream().get()); return v_service_time; } @@ -488,13 +488,13 @@ time_window_t generate_time_windows(raft::handle_t& handle, auto time_matrix = matrices.get_time_matrix(0); auto v_service_time = create_service_time(handle, params); detail::fill_time_windows - <<>>(time_matrix, - v_earliest_time.data(), - v_latest_time.data(), - params.tw_tightness, - params.n_locations); + <<>>(time_matrix, + v_earliest_time.data(), + v_latest_time.data(), + params.tw_tightness, + params.n_locations); handle.sync_stream(); - RAFT_CHECK_CUDA(handle.get_stream()); + RAFT_CHECK_CUDA(handle.get_stream().get()); return std::make_tuple( std::move(v_earliest_time), std::move(v_latest_time), std::move(v_service_time)); diff --git a/cpp/src/routing/ges/compute_fragment_ejections.cu b/cpp/src/routing/ges/compute_fragment_ejections.cu index de5cd14020..46db0c0cbb 100644 --- a/cpp/src/routing/ges/compute_fragment_ejections.cu +++ b/cpp/src/routing/ges/compute_fragment_ejections.cu @@ -130,7 +130,7 @@ void launch_kernel_get_best_insertion_ejection_solution( blocks, kernel_args, shmem_bytes, - stream)); + stream.get())); } #define CUOPT_INSTANTIATE_GET_BEST_INSERTION_EJECTION(BLOCK_SIZE, REQ) \ diff --git a/cpp/src/routing/ges/eject_until_feasible.cu b/cpp/src/routing/ges/eject_until_feasible.cu index 5a05bde062..b5cc4dbd93 100644 --- a/cpp/src/routing/ges/eject_until_feasible.cu +++ b/cpp/src/routing/ges/eject_until_feasible.cu @@ -365,7 +365,7 @@ void solution_t::eject_until_feasible(bool add_slack_to_sol) bool is_set = set_shmem_of_kernel(eject_until_feasible_kernel, sh_size); cuopt_assert(is_set, "Not enough shared memory on device for get_all_feasible_insertion!"); cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); - eject_until_feasible_kernel<<>>( + eject_until_feasible_kernel<<>>( view(), add_slack_to_sol, problem_ptr->seed_gen.get_seed()); compute_cost(); global_runtime_checks(false, true, "eject_until_feasible"); @@ -381,9 +381,9 @@ void solution_t::populate_ep_with_unserved( rmm::device_scalar ep_index_out(EP.index_, stream); const i_t TPB = 256; populate_ep_with_unserved_kernel - <<<1, TPB, 0, stream>>>(view(), EP.view(), ep_index_out.data()); + <<<1, TPB, 0, stream.get()>>>(view(), EP.view(), ep_index_out.data()); EP.index_ = ep_index_out.value(stream); - stream.synchronize(); + stream.sync(); if (EP.size() > 1) { thrust::default_random_engine g(problem_ptr->seed_gen.get_seed()); thrust::shuffle( @@ -404,11 +404,11 @@ void solution_t::populate_ep_with_selected_unserved( auto unserviced_view = raft::device_span(unserviced_device.data(), unserviced_device.size()); - populate_ep_with_selected_unserved_kernel<<<1, TPB, 0, stream>>>( + populate_ep_with_selected_unserved_kernel<<<1, TPB, 0, stream.get()>>>( view(), unserviced_view, EP.view(), ep_index_out.data(), problem_ptr->seed_gen.get_seed()); - RAFT_CHECK_CUDA(stream); + RAFT_CHECK_CUDA(stream.get()); EP.index_ = ep_index_out.value(stream); - stream.synchronize(); + stream.sync(); } template void solution_t::eject_until_feasible(bool); diff --git a/cpp/src/routing/ges/ejection_pool.cuh b/cpp/src/routing/ges/ejection_pool.cuh index afd566f475..b07160accd 100644 --- a/cpp/src/routing/ges/ejection_pool.cuh +++ b/cpp/src/routing/ges/ejection_pool.cuh @@ -63,7 +63,7 @@ struct ejection_pool_t { // replace with thrust shuffle // how to get sol_handle::get_thrust_policy? if (size() > 1) - device_random_shuffle<<<1, 1, 0, stream_>>>(stack_.data(), size(), seed); + device_random_shuffle<<<1, 1, 0, stream_.get()>>>(stack_.data(), size(), seed); } bool empty() const diff --git a/cpp/src/routing/ges/execute_insertion.cu b/cpp/src/routing/ges/execute_insertion.cu index ddec22acee..8dace9832d 100644 --- a/cpp/src/routing/ges/execute_insertion.cu +++ b/cpp/src/routing/ges/execute_insertion.cu @@ -261,7 +261,7 @@ bool guided_ejection_search_t::execute_best_insertion_ejectio for (; bit_cast(feasible_candidates_data_.front_element( solution_ptr->sol_handle->get_stream())) == unset_val && !time_stop_condition_reached() && fragment_size + fragment_step <= max_fragment_size;) { - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); // Increment here and not in for loop to not have it incremented if conditions are not met fragment_size += fragment_step; shared_for_delete_array = @@ -294,7 +294,7 @@ bool guided_ejection_search_t::execute_best_insertion_ejectio args, solution_ptr->sol_handle->get_stream()); } - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); } // Didn't manage to insert even with deleting @@ -308,13 +308,13 @@ bool guided_ejection_search_t::execute_best_insertion_ejectio <<<1, 1024, shared_for_delete_array + shared_for_tmp_route, - solution_ptr->sol_handle->get_stream()>>>(solution_ptr->view(), - d_request, - (uint64_t*)feasible_candidates_data_.data(), - EP.view(), - fragment_step, - fragment_size); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + solution_ptr->sol_handle->get_stream().get()>>>(solution_ptr->view(), + d_request, + (uint64_t*)feasible_candidates_data_.data(), + EP.view(), + fragment_step, + fragment_size); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); // Update EP index, route_id contains the amount we deleted found_sol_t selected_move = feasible_candidates_data_.element(0, solution_ptr->sol_handle->get_stream()); @@ -345,7 +345,7 @@ found_sol_t select_random_initialized(rmm::device_uvector& feasible } if (!updated) { *output_ptr = data; } }); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); return random_selected_candidate.value(solution_ptr->sol_handle->get_stream()); } @@ -365,9 +365,9 @@ bool guided_ejection_search_t::perform_insertion( } execute_feasible_insert - <<<1, 1024, shared_for_tmp_route, solution_ptr->sol_handle->get_stream()>>>( + <<<1, 1024, shared_for_tmp_route, solution_ptr->sol_handle->get_stream().get()>>>( solution_ptr->view(), request, selected_candidate); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); return true; } @@ -398,7 +398,7 @@ i_t guided_ejection_search_t::find_single_insertion( <<sol_handle->get_stream()>>>( + solution_ptr->sol_handle->get_stream().get()>>>( solution_ptr->view(), request, feasible_move_t(cuopt::make_span(feasible_candidates_data_), @@ -408,7 +408,7 @@ i_t guided_ejection_search_t::find_single_insertion( solution_ptr->get_n_routes()), solution_ptr->problem_ptr->seed_gen.get_seed()); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); return feasible_candidates_size_.value(solution_ptr->sol_handle->get_stream()); } diff --git a/cpp/src/routing/ges/guided_ejection_search.cu b/cpp/src/routing/ges/guided_ejection_search.cu index 1e88375a92..bc127d2462 100644 --- a/cpp/src/routing/ges/guided_ejection_search.cu +++ b/cpp/src/routing/ges/guided_ejection_search.cu @@ -270,10 +270,10 @@ bool guided_ejection_search_t::guided_ejection_search_loop(i_ } // Increase penalty counter for this request - incr_p_scores<<<1, 1, 0, solution_ptr->sol_handle->get_stream()>>>( + incr_p_scores<<<1, 1, 0, solution_ptr->sol_handle->get_stream().get()>>>( request, p_scores_.data(), depot_included); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); bool move_executed = config.frag_eject_first ? execute_best_insertion_ejection_solution(request, counter) : run_lexicographic_search(request); @@ -306,7 +306,7 @@ bool guided_ejection_search_t::guided_ejection_search_loop(i_ return false; } - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); // reinsert the request and increase the ejection failure counter EP.push_back_last(); consecutive_ejection_failure++; @@ -522,9 +522,9 @@ void guided_ejection_search_t::route_minimizer_loop() std::tie(vehicle_id, random_route_id) = next_route_id(); if (random_route_id < 0) { break; } // Save solution state before ges loop in case of route restoration - stream.synchronize(); + stream.sync(); ges_loop_save_state.copy_device_solution(*solution_ptr); - stream.synchronize(); + stream.sync(); solution_ptr->remove_routes(EP, std::vector{random_route_id}); // Routes can be empty when number of vehicles is more than number of requests @@ -532,9 +532,9 @@ void guided_ejection_search_t::route_minimizer_loop() // If ges loop left early, restore state if (!guided_ejection_search_loop(counter, true)) { - stream.synchronize(); + stream.sync(); solution_ptr->copy_device_solution(ges_loop_save_state); - stream.synchronize(); + stream.sync(); } solution_ptr->global_runtime_checks(true, true, "route_minimizer_loop"); } diff --git a/cpp/src/routing/ges/guided_ejection_search.cuh b/cpp/src/routing/ges/guided_ejection_search.cuh index a8c163939e..5026f5c9dc 100644 --- a/cpp/src/routing/ges/guided_ejection_search.cuh +++ b/cpp/src/routing/ges/guided_ejection_search.cuh @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2022-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -12,7 +12,6 @@ #include "ejection_pool.cuh" #include "found_solution.cuh" -#include #include #include diff --git a/cpp/src/routing/ges/lexicographic_search/brute_force_lexico.cu b/cpp/src/routing/ges/lexicographic_search/brute_force_lexico.cu index 020c5e89ab..2d0241103b 100644 --- a/cpp/src/routing/ges/lexicographic_search/brute_force_lexico.cu +++ b/cpp/src/routing/ges/lexicographic_search/brute_force_lexico.cu @@ -202,15 +202,15 @@ std::vector guided_ejection_search_t::brute_force_lexico size_t shared_size = shared_size_for_route + shared_size_for_intra_indices; i_t n_blocks = combinations.size(); brute_force_lexico_kernel - <<>>(d_combinations.data(), - sol.view(), - route.view(), - n_ejections, - req, - global_min_p.data(), - global_sequence.data(), - EP.view(), - p_scores_.data()); + <<>>(d_combinations.data(), + sol.view(), + route.view(), + n_ejections, + req, + global_min_p.data(), + global_sequence.data(), + EP.view(), + p_scores_.data()); // copy the best result and keep it here sol.sol_handle->sync_stream(); } @@ -219,7 +219,7 @@ std::vector guided_ejection_search_t::brute_force_lexico std::vector sequence(global_sequence.element(0, stream) + 3); // copy including pickup and delivery raft::copy(sequence.data(), global_sequence.data() + 1, sequence.size(), stream); - stream.synchronize(); + stream.sync(); return sequence; } return std::vector{}; diff --git a/cpp/src/routing/ges/lexicographic_search/lexicographic_search.cu b/cpp/src/routing/ges/lexicographic_search/lexicographic_search.cu index 8be74cd348..504d7408e3 100644 --- a/cpp/src/routing/ges/lexicographic_search/lexicographic_search.cu +++ b/cpp/src/routing/ges/lexicographic_search/lexicographic_search.cu @@ -43,7 +43,7 @@ bool compare_lexico_results(guided_ejection_search_t& ges, std::vector lexico_sequence(2 * k_max + 1); raft::update_host( lexico_sequence.data(), ges.global_sequence_.data() + 2, 2 * k_max + 1, stream); - stream.synchronize(); + stream.sync(); p_val_seq_t p_val(0, 0); memcpy((uint32_t*)&p_val, &h_global_min, sizeof(uint32_t)); cuopt_assert(p_val.p_val == brute_force_sequence[0], "p scores don't match"); @@ -668,7 +668,7 @@ bool guided_ejection_search_t::run_lexicographic_search( request_info_t* __restrict__ request_id) { auto stream = solution_ptr->sol_handle->get_stream(); - RAFT_CHECK_CUDA(stream); + RAFT_CHECK_CUDA(stream.get()); i_t average_route_size = solution_ptr->get_num_orders() / solution_ptr->n_routes; @@ -713,15 +713,16 @@ bool guided_ejection_search_t::run_lexicographic_search( solution_ptr->d_lock.set_value_async(zero, stream); global_random_counter_.set_value_async(zero, stream); lexicographic_search - <<>>(solution_ptr->view(), - k_max, - request_id, - p_scores_.data(), - global_min_p_.data(), - global_sequence_.data(), - global_random_counter_.data()); + <<>>( + solution_ptr->view(), + k_max, + request_id, + p_scores_.data(), + global_min_p_.data(), + global_sequence_.data(), + global_random_counter_.data()); solution_ptr->sol_handle->sync_stream(); - RAFT_CHECK_CUDA(stream); + RAFT_CHECK_CUDA(stream.get()); // If global_min_p_ != max do the move if (global_min_p_.value(stream) != max) { // cuopt_assert(compare_lexico_results(*this, solution, request_id, EP, k_max), ""); @@ -731,13 +732,13 @@ bool guided_ejection_search_t::run_lexicographic_search( return false; } execute_lexico_move - <<<1, threads_per_block_lexico, shared_for_tmp_route, stream>>>(solution_ptr->view(), - request_id, - global_min_p_.data(), - global_sequence_.data(), - EP.view(), - p_scores_.data()); - RAFT_CHECK_CUDA(stream); + <<<1, threads_per_block_lexico, shared_for_tmp_route, stream.get()>>>(solution_ptr->view(), + request_id, + global_min_p_.data(), + global_sequence_.data(), + EP.view(), + p_scores_.data()); + RAFT_CHECK_CUDA(stream.get()); i_t removed_size = global_sequence_.element(1, stream); if constexpr (REQUEST == request_t::PDP) { removed_size = (removed_size - 1) / 2; } EP.index_ += removed_size; diff --git a/cpp/src/routing/ges/squeeze.cu b/cpp/src/routing/ges/squeeze.cu index 5de35d153a..e104d6c1ac 100644 --- a/cpp/src/routing/ges/squeeze.cu +++ b/cpp/src/routing/ges/squeeze.cu @@ -37,9 +37,9 @@ bool guided_ejection_search_t::repair_empty_routes() // reset the best move stored best_move.set_value_async(uninit_cand, solution_ptr->sol_handle->get_stream()); find_best_empty_route_move - <<sol_handle->get_stream()>>>( + <<sol_handle->get_stream().get()>>>( solution_ptr->view(), best_move.data(), include_objective, default_weights, excess_limit); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); // If unable to find feasible moves, switch to least excess moves cand_t best_move_h = best_move.value(solution_ptr->sol_handle->get_stream()); @@ -51,9 +51,9 @@ bool guided_ejection_search_t::repair_empty_routes() if (!set_shmem_of_kernel(execute_best_empty_route_move, sh_route)) { break; } execute_best_empty_route_move - <<<1, TPB, sh_route, solution_ptr->sol_handle->get_stream()>>>(solution_ptr->view(), - best_move.data()); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + <<<1, TPB, sh_route, solution_ptr->sol_handle->get_stream().get()>>>(solution_ptr->view(), + best_move.data()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); ++counter; } solution_ptr->sol_handle->sync_stream(); @@ -95,24 +95,24 @@ i_t guided_ejection_search_t::try_multiple_insert(i_t n_inser cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); // insert the request greedily to a position that will generate the least excess find_all_squeeze_pos - <<>>(solution_ptr->view(), - EP.view(), - cuopt::make_span(best_squeeze_per_cand), - cuopt::make_span(best_squeeze_per_route), - include_objective, - weights, - excess_limit, - n_insertions, - inserted_requests.data()); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + <<>>(solution_ptr->view(), + EP.view(), + cuopt::make_span(best_squeeze_per_cand), + cuopt::make_span(best_squeeze_per_route), + include_objective, + weights, + excess_limit, + n_insertions, + inserted_requests.data()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); if constexpr (squeeze_mode) { size_t move_blocks = solution_ptr->get_num_requests(); extract_best_per_route - <<>>(solution_ptr->view(), - cuopt::make_span(best_squeeze_per_cand), - cuopt::make_span(best_squeeze_per_route)); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + <<>>(solution_ptr->view(), + cuopt::make_span(best_squeeze_per_cand), + cuopt::make_span(best_squeeze_per_route)); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); } size_t move_blocks = solution_ptr->get_n_routes(); @@ -121,19 +121,20 @@ i_t guided_ejection_search_t::try_multiple_insert(i_t n_inser cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); // execute squeeze moves execute_all_move - <<>>(solution_ptr->view(), - cuopt::make_span(best_squeeze_per_cand), - cuopt::make_span(best_squeeze_per_route), - inserted_requests.data(), - number_of_inserted.data()); - RAFT_CHECK_CUDA(stream); + <<>>( + solution_ptr->view(), + cuopt::make_span(best_squeeze_per_cand), + cuopt::make_span(best_squeeze_per_route), + inserted_requests.data(), + number_of_inserted.data()); + RAFT_CHECK_CUDA(stream.get()); auto n_inserted = number_of_inserted.value(stream); if (n_inserted == 0) { // Some of the attempted requests could not be inserted in this call or following ones // after perturbations - increase_multiple_p_scores - <<<1, 64, 0, stream>>>(EP.view(), p_scores_.data(), inserted_requests.data(), n_insertions); + increase_multiple_p_scores<<<1, 64, 0, stream.get()>>>( + EP.view(), p_scores_.data(), inserted_requests.data(), n_insertions); break; } counter += n_inserted; @@ -141,7 +142,7 @@ i_t guided_ejection_search_t::try_multiple_insert(i_t n_inser solution_ptr->compute_cost(); solution_ptr->global_runtime_checks(false, false, "try_multiple_insert_end"); - stream.synchronize(); + stream.sync(); return counter; } @@ -171,9 +172,10 @@ i_t guided_ejection_search_t::try_multiple_feasible_insertion i_t successful_insertions = try_multiple_insert( n_insertions, default_weights, std::numeric_limits::epsilon(), include_objective); - eject_inserted_requests<<<1, 32, 0, solution_ptr->sol_handle->get_stream()>>>( - EP.view(), inserted_requests.data(), n_insertions); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + eject_inserted_requests + <<<1, 32, 0, solution_ptr->sol_handle->get_stream().get()>>>( + EP.view(), inserted_requests.data(), n_insertions); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); // Index is not updated in device view EP.index_ -= successful_insertions; @@ -210,9 +212,9 @@ void guided_ejection_search_t::squeeze_all_ep() if (successful_insertions == 0) { run_batches = false; } eject_inserted_requests - <<<1, 32, 0, solution_ptr->sol_handle->get_stream()>>>( + <<<1, 32, 0, solution_ptr->sol_handle->get_stream().get()>>>( EP.view(), inserted_requests.data(), batch_size); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); // Index is not updated in device view EP.index_ -= successful_insertions; @@ -299,29 +301,30 @@ void guided_ejection_search_t::squeeze( i_t route_id = dist_candidate(gen_candidate) % solution_ptr->get_n_routes(); // insert the request greedily to a position that will generate the least excess find_best_squeeze_pos - <<<1, TPB, sh_size, stream>>>(solution_ptr->view(), - request, - best_move.data(), - include_objective, - local_search_ptr_->move_candidates.weights, - route_id); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + <<<1, TPB, sh_size, stream.get()>>>(solution_ptr->view(), + request, + best_move.data(), + include_objective, + local_search_ptr_->move_candidates.weights, + route_id); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); } else { find_best_squeeze_pos - <<>>(solution_ptr->view(), - request, - best_move.data(), - include_objective, - local_search_ptr_->move_candidates.weights); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + <<>>(solution_ptr->view(), + request, + best_move.data(), + include_objective, + local_search_ptr_->move_candidates.weights); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); } cuopt_assert(best_move.value(stream).cost_counter.cost != std::numeric_limits::max(), "At least a move should be found in squeeze"); // execute squeeze - execute_move<<<1, 1, 0, stream>>>(solution_ptr->view(), request, best_move.data()); + execute_move + <<<1, 1, 0, stream.get()>>>(solution_ptr->view(), request, best_move.data()); solution_ptr->compute_cost(); solution_ptr->global_runtime_checks(false, false, "squeeze"); - stream.synchronize(); + stream.sync(); } template @@ -378,9 +381,9 @@ void guided_ejection_search_t::squeeze_breaks() return; } - squeeze_breaks_kernel<<>>( + squeeze_breaks_kernel<<>>( solution_ptr->view(), false, local_search_ptr_->move_candidates.weights); - RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution_ptr->sol_handle->get_stream().get()); solution_ptr->compute_cost(); solution_ptr->global_runtime_checks(false, false, "squeeze_breaks_end"); return; diff --git a/cpp/src/routing/local_search/breaks_insertion.cu b/cpp/src/routing/local_search/breaks_insertion.cu index 8fd06d83f1..0361cac80f 100644 --- a/cpp/src/routing/local_search/breaks_insertion.cu +++ b/cpp/src/routing/local_search/breaks_insertion.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2023-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -167,12 +167,12 @@ void find_break_insertions(solution_t& sol, } find_break_insertions_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.include_objective, move_candidates.weights, move_candidates.breaks_move_candidates.view()); - RAFT_CUDA_TRY(cudaStreamSynchronize(sol.sol_handle->get_stream())); + sol.sol_handle->get_stream().sync(); } } @@ -254,9 +254,9 @@ bool local_search_t::perform_break_moves(solution_t, shared_size)) { return false; } execute_break_moves - <<get_stream()>>>(sol.view(), - move_candidates.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + <<get_stream().get()>>>(sol.view(), + move_candidates.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); sol.compute_cost(); sol.sol_handle->sync_stream(); diff --git a/cpp/src/routing/local_search/compute_compatible.cu b/cpp/src/routing/local_search/compute_compatible.cu index 457e970632..8a168d4493 100644 --- a/cpp/src/routing/local_search/compute_compatible.cu +++ b/cpp/src/routing/local_search/compute_compatible.cu @@ -448,11 +448,11 @@ void local_search_t::calculate_route_compatibility( i_t TPB = 128; i_t n_blocks = sol.n_routes * sol.get_num_requests(); calculate_route_compatibility_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.route_compatibility.data(), move_candidates.viables.compatibility_matrix.data()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); } // sort the viable matrix according to the distance after the insertion @@ -635,42 +635,44 @@ void initialize_incompatible(problem_t& problem, solution_t - <<get_stream()>>>( + <<get_stream().get()>>>( problem.view(), viables.compatibility_matrix.data(), sol_view); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_ptr->get_stream())); + handle_ptr->get_stream().sync(); n_blocks = (problem.get_num_orders() * problem.get_num_orders() - 1 + TPB) / TPB; initialize_viable_kernel - <<get_stream()>>>(problem.view(), - viables.viable_to_pickups.data(), - viables.viable_from_pickups.data(), - viables.n_viable_to_pickups.data(), - viables.n_viable_from_pickups.data(), - viables.viable_to_deliveries.data(), - viables.viable_from_deliveries.data(), - viables.n_viable_to_deliveries.data(), - viables.n_viable_from_deliveries.data(), - sol_view, - is_problem_run); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_ptr->get_stream())); + <<get_stream().get()>>>( + problem.view(), + viables.viable_to_pickups.data(), + viables.viable_from_pickups.data(), + viables.n_viable_to_pickups.data(), + viables.n_viable_from_pickups.data(), + viables.viable_to_deliveries.data(), + viables.viable_from_deliveries.data(), + viables.n_viable_to_deliveries.data(), + viables.n_viable_from_deliveries.data(), + sol_view, + is_problem_run); + handle_ptr->get_stream().sync(); } else { initialize_incompatible_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( problem.view(), viables.compatibility_matrix.data(), sol_view); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_ptr->get_stream())); + handle_ptr->get_stream().sync(); n_blocks = (problem.get_num_orders() * problem.get_num_orders() - 1 + TPB) / TPB; initialize_viable_kernel - <<get_stream()>>>(problem.view(), - viables.viable_to_pickups.data(), - viables.viable_from_pickups.data(), - viables.n_viable_to_pickups.data(), - viables.n_viable_from_pickups.data(), - viables.viable_to_deliveries.data(), - viables.viable_from_deliveries.data(), - viables.n_viable_to_deliveries.data(), - viables.n_viable_from_deliveries.data(), - sol_view, - is_problem_run); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_ptr->get_stream())); + <<get_stream().get()>>>( + problem.view(), + viables.viable_to_pickups.data(), + viables.viable_from_pickups.data(), + viables.n_viable_to_pickups.data(), + viables.n_viable_from_pickups.data(), + viables.viable_to_deliveries.data(), + viables.viable_from_deliveries.data(), + viables.n_viable_to_deliveries.data(), + viables.n_viable_from_deliveries.data(), + sol_view, + is_problem_run); + handle_ptr->get_stream().sync(); } problem.sort_viable_matrix(viables.viable_to_pickups, viables.viable_from_pickups); problem.sort_viable_matrix(viables.viable_to_deliveries, viables.viable_from_deliveries); diff --git a/cpp/src/routing/local_search/compute_insertions.cu b/cpp/src/routing/local_search/compute_insertions.cu index 1f69065446..f35d0dc528 100644 --- a/cpp/src/routing/local_search/compute_insertions.cu +++ b/cpp/src/routing/local_search/compute_insertions.cu @@ -830,7 +830,7 @@ void find_insertions(solution_t& sol, "Not enough shared memory on device for computing local search insertions!"); cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); find_insertions_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), sol.problem_ptr->seed_gen.get_seed()); } else { // for cross the load-balance factor is always 4 @@ -846,7 +846,7 @@ void find_insertions(solution_t& sol, "Not enough shared memory on device for computing local search insertions!"); cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); find_insertions_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), sol.problem_ptr->seed_gen.get_seed()); } else if (search_type == search_type_t::RANDOM) { // we don't search for relocates in random. @@ -858,11 +858,11 @@ void find_insertions(solution_t& sol, "Not enough shared memory on device for computing local search insertions!"); cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); find_insertions_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), sol.problem_ptr->seed_gen.get_seed()); } } - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); sol.sol_handle->sync_stream(); } @@ -891,9 +891,9 @@ void find_unserviced_insertions(solution_t& sol, cuopt_assert(is_set, "Not enough shared memory on device for computing local search insertions!"); cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); find_insertions_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), sol.problem_ptr->seed_gen.get_seed()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); sol.sol_handle->sync_stream(); } diff --git a/cpp/src/routing/local_search/cycle_finder/cycle_finder.cu b/cpp/src/routing/local_search/cycle_finder/cycle_finder.cu index 65d654b06b..c9994f672c 100644 --- a/cpp/src/routing/local_search/cycle_finder/cycle_finder.cu +++ b/cpp/src/routing/local_search/cycle_finder/cycle_finder.cu @@ -33,12 +33,14 @@ bool ExactCycleFinder::call_init(graph_t& graph) bool is_set = set_shmem_of_kernel(init_kernel, sh_size); if (!is_set) { return false; } - init_kernel<<get_stream()>>>( - graph.view(), d_valid_paths.subspan(level)); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + init_kernel + <<get_stream().get()>>>( + graph.view(), d_valid_paths.subspan(level)); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); // we have a safe-guard in the kernel for the global array stores // do the safe guard here for the occupied size - clamp_occupied<<<1, 1, 0, handle_ptr->get_stream()>>>(d_valid_paths.subspan(level)); + clamp_occupied + <<<1, 1, 0, handle_ptr->get_stream().get()>>>(d_valid_paths.subspan(level)); return true; } @@ -79,7 +81,7 @@ void ExactCycleFinder::sort_cycle_costs_by_key(int n_items n_items, begin_bit, end_bit, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); // Allocate temporary storage if (d_cub_storage_bytes.size() < temp_storage_bytes) { @@ -95,7 +97,7 @@ void ExactCycleFinder::sort_cycle_costs_by_key(int n_items n_items, begin_bit, end_bit, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); } template @@ -112,7 +114,7 @@ bool ExactCycleFinder::call_find(graph_t& graph, if (last_level) { if (!set_shmem_of_kernel(find_kernel, sh_size)) { return false; } find_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( level, graph.view(), d_valid_paths.subspan(level - 1), @@ -122,7 +124,7 @@ bool ExactCycleFinder::call_find(graph_t& graph, } else { if (!set_shmem_of_kernel(find_kernel, sh_size)) { return false; } find_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( level, graph.view(), d_valid_paths.subspan(level - 1), @@ -131,7 +133,7 @@ bool ExactCycleFinder::call_find(graph_t& graph, depot_included); } - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); return true; } @@ -141,8 +143,8 @@ void detail::device_map_t::clear(rmm::cuda_stream_view strea auto max_vals = max_level * max_available; auto n_threads = 256; auto n_blocks = std::min((max_vals + n_threads - 1) / n_threads, max_blocks); - clear_map<<>>(this->view()); - RAFT_CHECK_CUDA(stream); + clear_map<<>>(this->view()); + RAFT_CHECK_CUDA(stream.get()); } template @@ -152,8 +154,8 @@ bool test_empty(typename detail::device_map_t, double>::view_t auto max_vals = map_view.max_available; auto n_threads = 256; auto n_blocks = (max_vals + n_threads - 1) / n_threads; - test_empty, double><<>>(map_view); - RAFT_CHECK_CUDA(stream); + test_empty, double><<>>(map_view); + RAFT_CHECK_CUDA(stream.get()); return true; } @@ -187,25 +189,25 @@ void ExactCycleFinder::get_cycle(graph_t& graph, cuopt_func_call(d_ret.total_cycle_cost = 0.); for (i_t cycle_id = 0; cycle_id < n_cycles; ++cycle_id) { init_cycle - <<<1, 1, 0, handle_ptr->get_stream()>>>(d_ret.view(), best_cycles.subspan(cycle_id)); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + <<<1, 1, 0, handle_ptr->get_stream().get()>>>(d_ret.view(), best_cycles.subspan(cycle_id)); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); i_t level = level_vec[cycle_id]; for (int i = level; i > 0; --i) { extend_cycle - <<get_stream()>>>(graph.view(), - d_valid_paths.subspan(i), - best_cycles.subspan(cycle_id), - d_ret.view(), - i, - (level + 1) - i); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + <<get_stream().get()>>>(graph.view(), + d_valid_paths.subspan(i), + best_cycles.subspan(cycle_id), + d_ret.view(), + i, + (level + 1) - i); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } - close_cycle<<<1, 1, 0, handle_ptr->get_stream()>>>( + close_cycle<<<1, 1, 0, handle_ptr->get_stream().get()>>>( d_ret.view(), best_cycles.subspan(cycle_id), level + 1); cuopt_func_call(d_ret.total_cycle_cost += best_cycles.cost_ptr.element(cycle_id, handle_ptr->get_stream())); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } } @@ -300,7 +302,7 @@ void ExactCycleFinder::sort_occupied(int level, curr_map.occupied_indices.data(), curr_level_occupied, [] __device__(int2 a, int2 b) -> bool { return a.y < b.y; }, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); // Allocate temporary storage if (d_cub_storage_bytes.size() < temp_storage_bytes) { d_cub_storage_bytes.resize(temp_storage_bytes, handle_ptr->get_stream()); @@ -312,7 +314,7 @@ void ExactCycleFinder::sort_occupied(int level, curr_map.occupied_indices.data(), curr_level_occupied, [] __device__(int2 a, int2 b) -> bool { return a.y < b.y; }, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); // do an exclusive scan for the offsets of heads, this will be used in kernels temp_storage_bytes = 0; @@ -321,7 +323,7 @@ void ExactCycleFinder::sort_occupied(int level, curr_map.size_per_head.data(), curr_map.size_per_head.data(), graph.get_num_vertices() + 1, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); // Allocate temporary storage if (d_cub_storage_bytes.size() < temp_storage_bytes) { d_cub_storage_bytes.resize(temp_storage_bytes, handle_ptr->get_stream()); @@ -332,7 +334,7 @@ void ExactCycleFinder::sort_occupied(int level, curr_map.size_per_head.data(), curr_map.size_per_head.data(), graph.get_num_vertices() + 1, - handle_ptr->get_stream()); + handle_ptr->get_stream().get()); } template @@ -349,11 +351,11 @@ void ExactCycleFinder::find_best_cycles( sort_cycle_costs_by_key(cycle_candidates.size * cycle_candidates.n_paths); // record best cycles record_best_cycles - <<<1, 1, 0, handle_ptr->get_stream()>>>(cycle_candidates.size * cycle_candidates.n_paths, - graph.view(), - cycle_candidates.view(), - best_cycles.view(), - sorted_key_indices.data()); + <<<1, 1, 0, handle_ptr->get_stream().get()>>>(cycle_candidates.size * cycle_candidates.n_paths, + graph.view(), + cycle_candidates.view(), + best_cycles.view(), + sorted_key_indices.data()); get_cycle(graph, ret); cuopt_assert(check_cycle(graph, ret), "Recomputed cost mismatch"); } diff --git a/cpp/src/routing/local_search/cycle_finder/cycle_finder.hpp b/cpp/src/routing/local_search/cycle_finder/cycle_finder.hpp index 73a334ffd6..7db5d0c417 100644 --- a/cpp/src/routing/local_search/cycle_finder/cycle_finder.hpp +++ b/cpp/src/routing/local_search/cycle_finder/cycle_finder.hpp @@ -66,7 +66,8 @@ struct path_t { all_found.set_value_to_zero_async(stream); // device_bitset_t is all zeros when cleared; memset avoids a host-source copy, which // is not capturable into a CUDA graph on CUDA 13. - RAFT_CUDA_TRY(cudaMemsetAsync(all_mask.data(), 0, sizeof(device_bitset_t), stream)); + RAFT_CUDA_TRY( + cudaMemsetAsync(all_mask.data(), 0, sizeof(device_bitset_t), stream.get())); } struct view_t { diff --git a/cpp/src/routing/local_search/fill_gpu_graph.cu b/cpp/src/routing/local_search/fill_gpu_graph.cu index 5cb0e6c81e..036b84fcfa 100644 --- a/cpp/src/routing/local_search/fill_gpu_graph.cu +++ b/cpp/src/routing/local_search/fill_gpu_graph.cu @@ -158,13 +158,13 @@ void local_search_t::fill_gpu_graph(solution_tsync_stream(); const auto stream = solution.sol_handle->get_stream(); move_candidates.graph.special_index = solution.get_num_orders() + solution.n_routes; - fill_intra_candidates<<>>( + fill_intra_candidates<<>>( solution.view(), move_candidates.view(), solution.problem_ptr->seed_gen.get_seed()); // +1 for special node i_t n_blocks = solution.get_num_requests() + 1; fill_graph_kernel - <<>>(solution.view(), move_candidates.view()); - stream.synchronize(); + <<>>(solution.view(), move_candidates.view()); + stream.sync(); } template void local_search_t::fill_gpu_graph( solution_t&); diff --git a/cpp/src/routing/local_search/hvrp/vehicle_assignment.cu b/cpp/src/routing/local_search/hvrp/vehicle_assignment.cu index 7767ec9cdd..f2782704bc 100644 --- a/cpp/src/routing/local_search/hvrp/vehicle_assignment.cu +++ b/cpp/src/routing/local_search/hvrp/vehicle_assignment.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2024-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2024-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -24,9 +24,9 @@ auto compute_route_costs(solution_t& sol, if (!is_set) { return false; } compute_route_costs_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), vehicle_assignment.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); return true; } @@ -44,8 +44,9 @@ auto compute_route_cost_differences(solution_t& sol, if (!is_set) { return false; } compute_route_cost_differences_kernel - <<get_stream()>>>(sol.view(), vehicle_assignment.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + <<get_stream().get()>>>(sol.view(), + vehicle_assignment.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); return true; } @@ -61,8 +62,9 @@ auto compute_route_vehicle_assignments(solution_t& sol, if (!is_set) { return false; } compute_route_vehicle_assignments_kernel - <<get_stream()>>>(sol.view(), vehicle_assignment.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + <<get_stream().get()>>>(sol.view(), + vehicle_assignment.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); return true; } @@ -77,9 +79,10 @@ auto update_assignment(solution_t& sol, if (!is_set) { return false; } auto k_iter = vehicle_assignment.get_k_regrets() - 1; - update_assignment_kernel<<get_stream()>>>( - sol.view(), move_candidates.view(), vehicle_assignment.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + update_assignment_kernel + <<get_stream().get()>>>( + sol.view(), move_candidates.view(), vehicle_assignment.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); return true; } @@ -91,8 +94,8 @@ void reset_vehicle_availability(solution_t& sol, async_fill(vehicle_assignment.vehicle_availability, -1, sol.sol_handle->get_stream()); auto k_iter = vehicle_assignment.get_k_regrets() - 1; reset_vehicle_availability_kernel - <<get_stream()>>>(sol.view(), vehicle_assignment.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + <<get_stream().get()>>>(sol.view(), vehicle_assignment.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); } template @@ -142,8 +145,8 @@ auto find_best_assignment(solution_t& sol, bool is_set = set_shmem_of_kernel(find_best_assignment_kernel, shmem); if (!is_set) { return false; } find_best_assignment_kernel - <<<1, TPB, shmem, sol.sol_handle->get_stream()>>>(sol.view(), vehicle_assignment.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + <<<1, TPB, shmem, sol.sol_handle->get_stream().get()>>>(sol.view(), vehicle_assignment.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); return true; } @@ -159,9 +162,9 @@ auto update_solution(solution_t& sol, bool is_set = set_shmem_of_kernel(update_solution_kernel, shmem); if (!is_set) { return false; } update_solution_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), vehicle_assignment.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); sol.compute_cost(); sol.sol_handle->sync_stream(); diff --git a/cpp/src/routing/local_search/local_search.cu b/cpp/src/routing/local_search/local_search.cu index a774e3f82f..08f72970b7 100644 --- a/cpp/src/routing/local_search/local_search.cu +++ b/cpp/src/routing/local_search/local_search.cu @@ -290,7 +290,7 @@ void local_search_t::run_best_local_search(solution_t(sol, move_candidates, search_type_t::IMPROVE); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); sol.sol_handle->sync_stream(); fill_gpu_graph(sol); @@ -348,7 +348,7 @@ void local_search_t::run_random_local_search(solution_t(sol, move_candidates, search_type_t::RANDOM); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); sol.sol_handle->sync_stream(); populate_random_moves(sol); diff --git a/cpp/src/routing/local_search/perform_moves.cu b/cpp/src/routing/local_search/perform_moves.cu index d4c1144256..4cedba85da 100644 --- a/cpp/src/routing/local_search/perform_moves.cu +++ b/cpp/src/routing/local_search/perform_moves.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2022-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -419,7 +419,7 @@ bool local_search_t::populate_cross_moves( return false; } populate_cross_list_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( solution.view(), move_candidates.view()); sh_size = sizeof(i_t) * (solution.n_routes + 1) * solution.n_routes; @@ -428,8 +428,8 @@ bool local_search_t::populate_cross_moves( return false; } populate_cross_moves_kernel - <<<1, TPB, sh_size, solution.sol_handle->get_stream()>>>(solution.view(), - move_candidates.view()); + <<<1, TPB, sh_size, solution.sol_handle->get_stream().get()>>>(solution.view(), + move_candidates.view()); solution.sol_handle->sync_stream(); return true; } @@ -442,11 +442,12 @@ void local_search_t::populate_move_path( auto n_cycles = move_candidates.cycles.n_cycles_.value(solution.sol_handle->get_stream()); if (n_cycles) { populate_move_path_kernel - <<get_stream()>>>(solution.view(), - move_candidates.view()); + <<get_stream().get()>>>(solution.view(), + move_candidates.view()); } populate_intra_candidates - <<<1, 128, 0, solution.sol_handle->get_stream()>>>(solution.view(), move_candidates.view()); + <<<1, 128, 0, solution.sol_handle->get_stream().get()>>>(solution.view(), + move_candidates.view()); } template @@ -464,7 +465,7 @@ void local_search_t::perform_moves(solution_t - <<>>(solution.view(), move_candidates.view()); + <<>>(solution.view(), move_candidates.view()); solution.compute_route_id_per_node(); solution.compute_cost(); solution.global_runtime_checks(false, false, "perform_moves_end"); diff --git a/cpp/src/routing/local_search/prize_collection.cu b/cpp/src/routing/local_search/prize_collection.cu index 6d10d310c2..a47a28b5a1 100644 --- a/cpp/src/routing/local_search/prize_collection.cu +++ b/cpp/src/routing/local_search/prize_collection.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2023-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -228,18 +228,19 @@ bool local_search_t::perform_prize_collection(solution_t - <<get_stream()>>>(sol.view(), - move_candidates.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + <<get_stream().get()>>>(sol.view(), + move_candidates.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); if (!move_candidates.prize_move_candidates.has_improving_routes(sol.sol_handle)) { return false; } n_blocks = sol.get_n_routes(); shared_size = sol.check_routes_can_insert_and_get_sh_size(request_info_t::size()); if (!set_shmem_of_kernel(execute_moves, shared_size)) { return false; } - execute_moves<<get_stream()>>>( - sol.view(), move_candidates.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + execute_moves + <<get_stream().get()>>>(sol.view(), + move_candidates.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); sol.compute_cost(); diff --git a/cpp/src/routing/local_search/random_cross.cu b/cpp/src/routing/local_search/random_cross.cu index 7d90c96eb6..0769e28c5f 100644 --- a/cpp/src/routing/local_search/random_cross.cu +++ b/cpp/src/routing/local_search/random_cross.cu @@ -203,9 +203,9 @@ void select_random_route_pairs(solution_t& sol, return; } select_random_route_pairs_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), sol.problem_ptr->seed_gen.get_seed()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); } template @@ -216,9 +216,9 @@ void pick_random_move_per_route_pair(solution_t& sol, i_t n_route_pair = sol.n_routes * sol.n_routes; auto nblocks = (n_route_pair + nthreads - 1) / nthreads; pick_random_move_per_route_pair_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), sol.problem_ptr->seed_gen.get_seed()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); } template @@ -228,9 +228,10 @@ void get_offsets_of_route_pairs(solution_t& sol, { constexpr i_t nthreads = 256; auto nblocks = ((n_random_moves + 1) + nthreads - 1) / nthreads; - extract_offsets_kernel<<get_stream()>>>( - sol.view(), move_candidates.view(), n_random_moves); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + extract_offsets_kernel + <<get_stream().get()>>>( + sol.view(), move_candidates.view(), n_random_moves); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); } template @@ -248,7 +249,7 @@ i_t sort_random_moves_by_route_pair_idx(solution_t& sol, random_candidates.moves_per_route_pair.data(), n_random_moves, [] __device__(int2 a, int2 b) -> bool { return a.y < b.y; }, - sol.sol_handle->get_stream()); + sol.sol_handle->get_stream().get()); // Allocate temporary storage if (random_candidates.d_cub_storage_bytes.size() < temp_storage_bytes) { random_candidates.d_cub_storage_bytes.resize(temp_storage_bytes, sol.sol_handle->get_stream()); @@ -260,7 +261,7 @@ i_t sort_random_moves_by_route_pair_idx(solution_t& sol, random_candidates.moves_per_route_pair.data(), n_random_moves, [] __device__(int2 a, int2 b) -> bool { return a.y < b.y; }, - sol.sol_handle->get_stream()); + sol.sol_handle->get_stream().get()); return n_random_moves; } @@ -272,8 +273,9 @@ void local_search_t::populate_random_moves(solution_t - <<get_stream()>>>(sol.view(), move_candidates.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + <<get_stream().get()>>>(sol.view(), + move_candidates.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); // sort valid moves by route pair index i_t n_random_moves = sort_random_moves_by_route_pair_idx(sol, move_candidates); if (n_random_moves == 0) return; diff --git a/cpp/src/routing/local_search/sliding_tsp.cu b/cpp/src/routing/local_search/sliding_tsp.cu index bf206018b5..8ae80d46a4 100644 --- a/cpp/src/routing/local_search/sliding_tsp.cu +++ b/cpp/src/routing/local_search/sliding_tsp.cu @@ -427,7 +427,7 @@ void resize_temp_storage(solution_t& sol, distances_ptr, distances_ptr, n_nodes + 1, - sol.sol_handle->get_stream()); + sol.sol_handle->get_stream().get()); if (temp_storage_bytes > 0) { move_candidates.temp_storage.resize(temp_storage_bytes, sol.sol_handle->get_stream()); @@ -446,12 +446,12 @@ void compute_cumulative_distances(solution_t& sol, auto n_fill_blocks = (sol.get_num_orders() + n_threads - 1) / n_threads; if (reverse) { fill_reverse_distances_kernel - <<get_stream()>>>(sol.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + <<get_stream().get()>>>(sol.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); } else { fill_forward_distances_kernel - <<get_stream()>>>(sol.view()); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + <<get_stream().get()>>>(sol.view()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); } size_t n_temp_storage_bytes = 0; @@ -460,7 +460,7 @@ void compute_cumulative_distances(solution_t& sol, distances_ptr, distances_ptr, n_nodes + 2, - sol.sol_handle->get_stream()); + sol.sol_handle->get_stream().get()); if (n_temp_storage_bytes > 0) { cuopt_expects(n_temp_storage_bytes == temp_storage_bytes, @@ -473,7 +473,7 @@ void compute_cumulative_distances(solution_t& sol, distances_ptr, distances_ptr, n_nodes + 2, - sol.sol_handle->get_stream()); + sol.sol_handle->get_stream().get()); } template @@ -510,12 +510,12 @@ bool local_search_t::perform_sliding_tsp( if (!set_shmem_of_kernel(find_sliding_moves_tsp, sh_size)) { return false; } find_sliding_moves_tsp - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), cuopt::make_span(sampled_tsp_data_), cuopt::make_span(locks_)); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); n_moves_found = thrust::count_if(rmm::exec_policy(sol.sol_handle->get_stream()), sampled_tsp_data_.begin(), @@ -526,9 +526,9 @@ bool local_search_t::perform_sliding_tsp( async_fill(moved_region_node_infos_, NodeInfo{}, sol.sol_handle->get_stream()); set_moved_regions_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), cuopt::make_span(moved_region_node_infos_)); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); cuopt_func_call( move_candidates.debug_delta.set_value_to_zero_async(sol.sol_handle->get_stream())); @@ -548,12 +548,12 @@ bool local_search_t::perform_sliding_tsp( }); execute_sliding_moves_tsp - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), cuopt::make_span(sampled_tsp_data_), cuopt::make_span(moved_region_node_infos_)); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); compute_cumulative_distances( sol, move_candidates, n_nodes, n_threads, temp_storage_bytes); diff --git a/cpp/src/routing/local_search/sliding_window.cu b/cpp/src/routing/local_search/sliding_window.cu index 2d676d9b38..1a26b46ef6 100644 --- a/cpp/src/routing/local_search/sliding_window.cu +++ b/cpp/src/routing/local_search/sliding_window.cu @@ -1065,25 +1065,25 @@ bool local_search_t::perform_sliding_window( <<get_stream()>>>(solution.view(), - found_sliding_solution_data_.data(), - move_candidates.view(), - locks_.data(), - blocks_per_node); + solution.sol_handle->get_stream().get()>>>(solution.view(), + found_sliding_solution_data_.data(), + move_candidates.view(), + locks_.data(), + blocks_per_node); } else { kernel_perform_sliding_window <<get_stream()>>>(solution.view(), - found_sliding_solution_data_.data(), - move_candidates.view(), - locks_.data(), - blocks_per_node); + solution.sol_handle->get_stream().get()>>>(solution.view(), + found_sliding_solution_data_.data(), + move_candidates.view(), + locks_.data(), + blocks_per_node); } sliding_cuda_graph.end_capture(solution.sol_handle->get_stream()); sliding_cuda_graph.launch_graph(solution.sol_handle->get_stream()); - RAFT_CHECK_CUDA(solution.sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution.sol_handle->get_stream().get()); n_moves_found = thrust::count_if(solution.sol_handle->get_thrust_policy(), found_sliding_solution_data_.begin(), found_sliding_solution_data_.end(), @@ -1104,12 +1104,12 @@ bool local_search_t::perform_sliding_window( // One block for each found route execute_sliding_move - <<get_stream()>>>( + <<get_stream().get()>>>( solution.view(), found_sliding_solution_data_.data(), move_candidates.view(), move_candidates.debug_delta.data()); - RAFT_CHECK_CUDA(solution.sol_handle->get_stream()); + RAFT_CHECK_CUDA(solution.sol_handle->get_stream().get()); cuopt_func_call(solution.compute_cost()); cuopt_func_call(cost_after = solution.get_cost(move_candidates.include_objective, move_candidates.weights)); diff --git a/cpp/src/routing/local_search/two_opt.cu b/cpp/src/routing/local_search/two_opt.cu index abe6e8a928..ab88e1c146 100644 --- a/cpp/src/routing/local_search/two_opt.cu +++ b/cpp/src/routing/local_search/two_opt.cu @@ -393,13 +393,13 @@ bool local_search_t::perform_two_opt( if (!set_shmem_of_kernel(find_two_opt_moves, sh_size)) { return false; } find_two_opt_moves - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), cuopt::make_span(two_opt_cand_data_), cuopt::make_span(sampled_nodes_data_), cuopt::make_span(locks_)); - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); n_moves_found = thrust::count_if(sol.sol_handle->get_thrust_policy(), sampled_nodes_data_.begin(), @@ -434,7 +434,7 @@ bool local_search_t::perform_two_opt( sol.sol_handle->get_stream()); async_fill(moved_regions_, 0, sol.sol_handle->get_stream()); execute_recycle - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), cuopt::make_span(sampled_nodes_data_), @@ -442,13 +442,13 @@ bool local_search_t::perform_two_opt( } else { if (!set_shmem_of_kernel(execute_two_opt_moves, sh_size)) { return false; } execute_two_opt_moves - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), cuopt::make_span(two_opt_cand_data_), cuopt::make_span(moved_regions_)); } - RAFT_CHECK_CUDA(sol.sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol.sol_handle->get_stream().get()); cuopt_func_call(sol.compute_cost()); cuopt_func_call(cost_after = diff --git a/cpp/src/routing/local_search/vrp/nodes_to_search.cu b/cpp/src/routing/local_search/vrp/nodes_to_search.cu index f1e8b708d7..a3969cebe4 100644 --- a/cpp/src/routing/local_search/vrp/nodes_to_search.cu +++ b/cpp/src/routing/local_search/vrp/nodes_to_search.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2023-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -56,7 +56,7 @@ void run_extract_kernel(solution_t& sol, i_t TPB = 256; i_t n_blocks = sol.get_n_routes(); extract_nodes_to_search_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), nodes_to_search.view(), restore_phase); } diff --git a/cpp/src/routing/local_search/vrp/vrp_execute.cu b/cpp/src/routing/local_search/vrp/vrp_execute.cu index d65ec4fb36..2b6c14f5fe 100644 --- a/cpp/src/routing/local_search/vrp/vrp_execute.cu +++ b/cpp/src/routing/local_search/vrp/vrp_execute.cu @@ -380,8 +380,8 @@ i_t extract_non_overlapping_moves(solution_t& sol, i_t TPB = 128; i_t n_blocks_for_compact = (sol.n_routes * sol.n_routes + TPB - 1) / TPB; compact_best_route_pair_moves - <<get_stream()>>>(sol.view(), - move_candidates.view()); + <<get_stream().get()>>>(sol.view(), + move_candidates.view()); i_t n_best_route_pair_moves = move_candidates.vrp_move_candidates.n_best_route_pair_moves.value(sol.sol_handle->get_stream()); n_best_route_pair_moves = std::min(n_best_route_pair_moves, max_n_best_route_pair_moves); @@ -393,7 +393,7 @@ i_t extract_non_overlapping_moves(solution_t& sol, "Not enough shared memory on device for extract_non_overlapping_moves_kernel!"); cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); extract_non_overlapping_moves_kernel - <<<1, TPB, sh_size, sol.sol_handle->get_stream()>>>( + <<<1, TPB, sh_size, sol.sol_handle->get_stream().get()>>>( sol.view(), move_candidates.view(), sol.problem_ptr->seed_gen.get_seed()); return move_candidates.vrp_move_candidates.n_of_selected_moves.value( sol.sol_handle->get_stream()); @@ -407,7 +407,7 @@ void find_max_added_size(solution_t& sol, i_t TPB = 32; i_t n_blocks = n_moves_found; find_max_added_size_kernel - <<get_stream()>>>(sol.view(), move_candidates.view()); + <<get_stream().get()>>>(sol.view(), move_candidates.view()); } template @@ -454,7 +454,7 @@ bool execute_vrp_moves(solution_t& sol, dimBlock, kernelArgs, sh_size, - sol.sol_handle->get_stream()); + sol.sol_handle->get_stream().get()); sol.compute_route_id_per_node(); sol.compute_cost(); // move_candidates.vrp_execute_graph.end_capture(sol.sol_handle->get_stream()); diff --git a/cpp/src/routing/local_search/vrp/vrp_search.cu b/cpp/src/routing/local_search/vrp/vrp_search.cu index 1f71458856..11804f5fc2 100644 --- a/cpp/src/routing/local_search/vrp/vrp_search.cu +++ b/cpp/src/routing/local_search/vrp/vrp_search.cu @@ -652,7 +652,7 @@ bool find_vrp_moves(solution_t& sol, if (sol.problem_ptr->is_cvrp()) { compute_reverse_distances - <<get_stream()>>>(sol.view()); + <<get_stream().get()>>>(sol.view()); } i_t TPB = std::min(max_n_neighbors, sol.problem_ptr->get_num_orders()); size_t size_of_frag = dimensions_route_t::get_shared_size( @@ -672,7 +672,7 @@ bool find_vrp_moves(solution_t& sol, move_candidates.vrp_move_candidates.find_kernel_graph.start_capture(sol.sol_handle->get_stream()); move_candidates.vrp_move_candidates.reset(sol.sol_handle); find_vrp_moves_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( sol.view(), move_candidates.view(), recycle); move_candidates.vrp_move_candidates.find_kernel_graph.end_capture(sol.sol_handle->get_stream()); move_candidates.vrp_move_candidates.find_kernel_graph.launch_graph(sol.sol_handle->get_stream()); diff --git a/cpp/src/routing/order_info.cu b/cpp/src/routing/order_info.cu index 1d7e4de236..8766517f32 100644 --- a/cpp/src/routing/order_info.cu +++ b/cpp/src/routing/order_info.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2023-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -36,9 +36,9 @@ void populate_time_windows(data_model_view_t const& data_model, raft::copy(order_info_.v_earliest_time_.data(), earliest, order_info_.get_num_orders(), - stream_view.value()); + stream_view.get()); raft::copy( - order_info_.v_latest_time_.data(), latest, order_info_.get_num_orders(), stream_view.value()); + order_info_.v_latest_time_.data(), latest, order_info_.get_num_orders(), stream_view.get()); } else { // subtract -1 to ensure that we can set max values for service times // in vehicle order match @@ -113,7 +113,7 @@ void check_depot_times(data_model_view_t const& data_model) i_t depot_earliest, depot_latest; raft::copy(&depot_earliest, earliest, 1, handle_ptr->get_stream()); raft::copy(&depot_latest, latest, 1, handle_ptr->get_stream()); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_ptr->get_stream())); + handle_ptr->get_stream().sync(); rmm::device_uvector v_latest_time(n_orders, handle_ptr->get_stream()); rmm::device_uvector v_earliest_time(n_orders, handle_ptr->get_stream()); @@ -195,7 +195,7 @@ void populate_order_info(data_model_view_t const& data_model, thrust::max_element(handle_ptr_->get_thrust_policy(), temp_abs.begin(), temp_abs.end()); i_t h_max_element; raft::copy(&h_max_element, max_element_ptr, 1, stream); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); + stream.sync(); cuopt_expects(norders - 1 == h_max_element, error_type_t::ValidationError, "Index given is too big or an index in the delivery pickup pairs is missing!"); diff --git a/cpp/src/routing/problem/problem.cu b/cpp/src/routing/problem/problem.cu index 6868736fc3..dfb02d3f49 100644 --- a/cpp/src/routing/problem/problem.cu +++ b/cpp/src/routing/problem/problem.cu @@ -720,7 +720,7 @@ void problem_t::populate_special_nodes() special_nodes.earliest_time = cuopt::device_copy(node_earliest_h, handle_ptr->get_stream()); special_nodes.latest_time = cuopt::device_copy(node_latest_h, handle_ptr->get_stream()); special_nodes.break_loc_to_idx = cuopt::device_copy(break_loc_to_idx_h, handle_ptr->get_stream()); - RAFT_CHECK_CUDA(handle_ptr->get_stream()); + RAFT_CHECK_CUDA(handle_ptr->get_stream().get()); } template diff --git a/cpp/src/routing/route/capacity_route.cuh b/cpp/src/routing/route/capacity_route.cuh index 3ee61c2c85..776262a497 100644 --- a/cpp/src/routing/route/capacity_route.cuh +++ b/cpp/src/routing/route/capacity_route.cuh @@ -72,7 +72,7 @@ class capacity_route_t { std::min(old_stride, new_stride) * sizeof(i_t), n_dims, cudaMemcpyDeviceToDevice, - stream.value())); + stream.get())); } vec = std::move(new_vec); }; diff --git a/cpp/src/routing/solution/pool_allocator.cuh b/cpp/src/routing/solution/pool_allocator.cuh index d78df69517..393740c351 100644 --- a/cpp/src/routing/solution/pool_allocator.cuh +++ b/cpp/src/routing/solution/pool_allocator.cuh @@ -70,7 +70,7 @@ class pool_allocator_t { } } - void sync_all_streams() const { stream.synchronize(); } + void sync_all_streams() const { stream.sync(); } // problem description rmm::cuda_stream_view stream; diff --git a/cpp/src/routing/solution/solution.cu b/cpp/src/routing/solution/solution.cu index cbf7ed9384..db6b7ff4b8 100644 --- a/cpp/src/routing/solution/solution.cu +++ b/cpp/src/routing/solution/solution.cu @@ -171,8 +171,9 @@ void solution_t::add_nodes_to_route( bool is_set = set_shmem_of_kernel(insert_nodes_to_route_kernel, sh_size); cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); i_t TPB = 256; - insert_nodes_to_route_kernel<<<1, TPB, sh_size, sol_handle->get_stream()>>>( - view(), route_id, intra_idx, n_nodes_to_insert, temp_nodes.data()); + insert_nodes_to_route_kernel + <<<1, TPB, sh_size, sol_handle->get_stream().get()>>>( + view(), route_id, intra_idx, n_nodes_to_insert, temp_nodes.data()); thrust::fill(sol_handle->get_thrust_policy(), routes_to_search.data() + route_id, routes_to_search.data() + route_id + 1, @@ -193,7 +194,8 @@ void solution_t::add_nodes_to_best( bool is_set = set_shmem_of_kernel(insert_node_to_best_kernel, sh_size); cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); insert_node_to_best_kernel - <<<1, TPB, sh_size, sol_handle->get_stream()>>>(view(), node, include_objective, weights); + <<<1, TPB, sh_size, sol_handle->get_stream().get()>>>( + view(), node, include_objective, weights); sol_handle->sync_stream(); } this->global_runtime_checks(false, false, "add_nodes_to_best"); @@ -214,7 +216,7 @@ bool solution_t::remove_nodes(const std::vector>& cuopt_assert(is_set, "Not enough shared memory on device for remove_nodes!"); cuopt_expects(is_set, error_type_t::OutOfMemoryError, "Not enough shared memory on device"); i_t TPB = 256; - remove_nodes_kernel<<<1, TPB, sh_size, sol_handle->get_stream()>>>( + remove_nodes_kernel<<<1, TPB, sh_size, sol_handle->get_stream().get()>>>( view(), n_nodes_to_eject, temp_nodes.data(), empty_route_produced.data()); sol_handle->sync_stream(); return !empty_route_produced.value(sol_handle->get_stream()); @@ -323,7 +325,7 @@ void solution_t::random_init_routes() { raft::common::nvtx::range fun_scope("random_init_routes"); auto stream = sol_handle->get_stream(); - stream.synchronize(); + stream.sync(); const i_t one = 1; d_sol_found.set_value_async(one, stream); std::vector indices(get_num_requests()); @@ -343,7 +345,7 @@ void solution_t::random_init_routes() } } set_initial_nodes(d_indices, n_routes); - stream.synchronize(); + stream.sync(); } template @@ -542,8 +544,8 @@ void solution_t::copy_device_solution(solution_t - <<get_stream()>>>(view(), src_sol.view()); - RAFT_CHECK_CUDA(sol_handle->get_stream()); + <<get_stream().get()>>>(view(), src_sol.view()); + RAFT_CHECK_CUDA(sol_handle->get_stream().get()); cuopt_assert(route_node_map.intra_route_idx_per_node.size() == (size_t)get_num_orders(), "Intra route size mismatch!"); @@ -569,7 +571,7 @@ void solution_t::copy_device_solution(solution_tget_stream()); unset_routes_to_copy(); sol_handle->sync_stream(); - RAFT_CHECK_CUDA(sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol_handle->get_stream().get()); } template @@ -585,7 +587,8 @@ void solution_t::compute_cost() objective_cost.set_value_async(zero_obj, sol_handle->get_stream()); n_infeasible_routes.set_value_to_zero_async(sol_handle->get_stream()); if (get_n_routes() < 1) return; - compute_cost_kernel<<get_stream()>>>(view()); + compute_cost_kernel + <<get_stream().get()>>>(view()); } template @@ -627,12 +630,12 @@ void solution_t::shift_move_routes( if (n_blocks > 0) { // Decrement route_id_per_node for this route remap_route_nodes - <<get_stream()>>>( + <<get_stream().get()>>>( routes_view.data(), route_node_map.view(), route_ids_device_copy.data(), route_ids.size()); - RAFT_CHECK_CUDA(sol_handle->get_stream()); - shift_routes_kernel<<<1, 1, 0, sol_handle->get_stream()>>>( + RAFT_CHECK_CUDA(sol_handle->get_stream().get()); + shift_routes_kernel<<<1, 1, 0, sol_handle->get_stream().get()>>>( view(), route_ids_device_copy.data(), route_ids.size()); - RAFT_CHECK_CUDA(sol_handle->get_stream()); + RAFT_CHECK_CUDA(sol_handle->get_stream().get()); } sol_handle->sync_stream(); n_routes -= route_ids.size(); @@ -679,7 +682,7 @@ void solution_t::remove_routes( cuopt_assert(ejection_pool.index_ >= 0, "Index should be at least 0"); set_deleted_routes_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( view(), cuopt::make_span(routes_view), cuopt::make_span(temp_int_vector), @@ -706,7 +709,7 @@ void solution_t::remove_routes(const std::vector& routes "route to remove should be in range"); } set_deleted_routes_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( view(), cuopt::make_span(routes_view), cuopt::make_span(temp_int_vector)); shift_move_routes(routes_to_remove, temp_int_vector); } @@ -732,7 +735,8 @@ i_t solution_t::compute_max_active() { raft::common::nvtx::range fun_scope("compute_max_active"); i_t TPB = 1024; - compute_max_active_kernel<<<1, TPB, 0, sol_handle->get_stream()>>>(view()); + compute_max_active_kernel + <<<1, TPB, 0, sol_handle->get_stream().get()>>>(view()); max_active_nodes = max_active_nodes_for_all_routes.value(sol_handle->get_stream()); return max_active_nodes; } @@ -742,8 +746,8 @@ void solution_t::compute_route_id_per_node() { raft::common::nvtx::range fun_scope("compute_route_id_per_node"); i_t TPB = 256; - compute_route_id_kernel - <<get_stream()>>>(routes_view.data(), route_node_map.view()); + compute_route_id_kernel<<get_stream().get()>>>( + routes_view.data(), route_node_map.view()); global_runtime_checks(false, false, "compute_route_id_per_node"); } diff --git a/cpp/src/routing/solution/solution_handle.cuh b/cpp/src/routing/solution/solution_handle.cuh index 2a74ac7341..38675021b5 100644 --- a/cpp/src/routing/solution/solution_handle.cuh +++ b/cpp/src/routing/solution/solution_handle.cuh @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2022-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -44,7 +44,7 @@ class solution_handle_t { rmm::exec_policy& get_thrust_policy() const noexcept { return *thrust_policy_; } rmm::cuda_stream_view get_stream() const noexcept { return stream_view_; } i_t get_device() const { return dev_id_; } - void sync_stream() const { stream_view_.synchronize(); }; + void sync_stream() const { stream_view_.sync(); }; const cudaDeviceProp& get_device_properties() const { diff --git a/cpp/src/routing/util_kernels/compute_backward_forward.cu b/cpp/src/routing/util_kernels/compute_backward_forward.cu index bdde6336f4..4d8faf4095 100644 --- a/cpp/src/routing/util_kernels/compute_backward_forward.cu +++ b/cpp/src/routing/util_kernels/compute_backward_forward.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2022-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -46,7 +46,7 @@ void solution_t::compute_backward_forward() constexpr i_t TPB = 32; if (n_routes) { compute_backward_forward_kernel - <<get_stream()>>>(view().routes); + <<get_stream().get()>>>(view().routes); sol_handle->sync_stream(); } } @@ -58,7 +58,7 @@ void solution_t::compute_actual_arrival_times() constexpr i_t TPB = 32; if (n_routes && problem_ptr->dimensions_info.has_dimension(dim_t::TIME)) compute_actual_arrival_kernel - <<get_stream()>>>(view().routes); + <<get_stream().get()>>>(view().routes); } template void solution_t::compute_backward_forward(); diff --git a/cpp/src/routing/util_kernels/runtime_checks.cu b/cpp/src/routing/util_kernels/runtime_checks.cu index b1142f2e04..b9f28f6a18 100644 --- a/cpp/src/routing/util_kernels/runtime_checks.cu +++ b/cpp/src/routing/util_kernels/runtime_checks.cu @@ -255,12 +255,12 @@ bool global_runtime_checks_(solution_t& solution, solution.run_coherence_check(); async_fill(solution.runtime_check_histo, 0, solution.sol_handle->get_stream()); - fill_histo<<>>( + fill_histo<<>>( solution.view(), solution.runtime_check_histo.data()); const bool depot_included = solution.problem_ptr->order_info.depot_included_; check_histogram - <<<(solution.get_num_depot_excluded_orders() + 32 - 1) / 32, 32, 0, stream>>>( + <<<(solution.get_num_depot_excluded_orders() + 32 - 1) / 32, 32, 0, stream.get()>>>( solution.runtime_check_histo.data(), solution.get_num_orders(), all_nodes_should_be_served, @@ -268,7 +268,7 @@ bool global_runtime_checks_(solution_t& solution, if (solution.problem_ptr->get_max_break_dimensions() > 0) { auto sh_size = solution.problem_ptr->get_max_break_dimensions() * sizeof(i_t); - check_breaks<<>>( + check_breaks<<>>( solution.view(), all_nodes_should_be_served); } @@ -304,14 +304,14 @@ template void solution_t::run_feasibility_check() { cuopt_func_call((feasibility_check - <<get_stream()>>>(view()))); + <<get_stream().get()>>>(view()))); } template void solution_t::run_coherence_check() { cuopt_func_call((node_global_coherence_check - <<get_stream()>>>(view()))); + <<get_stream().get()>>>(view()))); } template void solution_t::global_runtime_checks( diff --git a/cpp/src/routing/util_kernels/set_initial_nodes.cu b/cpp/src/routing/util_kernels/set_initial_nodes.cu index 675357d0a3..75e72e6260 100644 --- a/cpp/src/routing/util_kernels/set_initial_nodes.cu +++ b/cpp/src/routing/util_kernels/set_initial_nodes.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2021-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2021-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -227,10 +227,10 @@ void solution_t::set_initial_nodes(const rmm::device_uvector< -1); constexpr i_t TPB = 32; i_t n_blocks = (desired_n_routes + TPB - 1) / TPB; - set_initial_nodes_kernel - <<get_stream()>>>(view(), problem_ptr->view(), d_indices.data()); + set_initial_nodes_kernel<<get_stream().get()>>>( + view(), problem_ptr->view(), d_indices.data()); - sol_handle->get_stream().synchronize(); + sol_handle->get_stream().sync(); } template @@ -239,7 +239,7 @@ void solution_t::set_nodes_data_of_solution() constexpr i_t TPB = 32; i_t n_blocks = n_routes; set_nodes_data_of_solution_kernel - <<get_stream()>>>(view(), problem_ptr->view()); + <<get_stream().get()>>>(view(), problem_ptr->view()); } template @@ -247,7 +247,7 @@ void solution_t::set_nodes_data_of_route(i_t route_id) { constexpr i_t TPB = 32; set_nodes_data_of_route_kernel - <<<1, TPB, 0, sol_handle->get_stream()>>>(view(), problem_ptr->view(), route_id); + <<<1, TPB, 0, sol_handle->get_stream().get()>>>(view(), problem_ptr->view(), route_id); } template @@ -257,7 +257,7 @@ void solution_t::set_nodes_data_of_new_routes(i_t added_route constexpr i_t TPB = 32; i_t starting_route_id = prev_route_size; set_nodes_data_of_new_routes_kernel - <<get_stream()>>>( + <<get_stream().get()>>>( view(), problem_ptr->view(), starting_route_id); } diff --git a/cpp/src/routing/utilities/check_input.cu b/cpp/src/routing/utilities/check_input.cu index eccc3179bb..f8d58645a1 100644 --- a/cpp/src/routing/utilities/check_input.cu +++ b/cpp/src/routing/utilities/check_input.cu @@ -39,7 +39,7 @@ void transform_absolute(rmm::device_uvector& v, rmm::cuda_stream_view stream_ rmm::exec_policy(stream_view), v.begin(), v.end(), v.begin(), [] __device__(T x) -> T { return x < 0 ? -x : x; }); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view.value())); + stream_view.sync(); } /** @@ -70,7 +70,7 @@ bool check_pickup_tw(const i_t* pickup_indices, zip_iterator, zip_iterator + n_requests, [] __device__(const auto& x) -> bool { return thrust::get<0>(x) > thrust::get<1>(x); }); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view.value())); + stream_view.sync(); return !violates_sanity; } @@ -98,7 +98,7 @@ bool check_pickup_demands(const i_t* pickup_indices, zip_iterator, zip_iterator + n_requests, [] __device__(const auto& x) -> bool { return thrust::get<0>(x) != -thrust::get<1>(x); }); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view.value())); + stream_view.sync(); return !violates_sanity; } @@ -117,7 +117,7 @@ bool check_pdp_values(const i_t* pickup_indices, zip_iterator, zip_iterator + n_requests, [] __device__(const auto& x) -> bool { return thrust::get<0>(x) != thrust::get<1>(x); }); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view.value())); + stream_view.sync(); return !violates_sanity; } @@ -138,8 +138,8 @@ bool is_symmetric_matrix(f_t const* matrix, i_t width, raft::handle_t const* han transposed_matrix.data_handle(), width, width, - handle_ptr->get_stream()); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_ptr->get_stream())); + handle_ptr->get_stream().get()); + handle_ptr->get_stream().sync(); return thrust::equal(handle_ptr->get_thrust_policy(), matrix, @@ -163,8 +163,8 @@ bool check_min_latest_with_depot(rmm::device_uvector& v_latest_time, i_t min_latest; i_t* min_latest_ptr = thrust::min_element( rmm::exec_policy(stream_view), v_latest_time.begin() + 1, v_latest_time.end()); - raft::copy(&min_latest, min_latest_ptr, 1, stream_view.value()); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view.value())); + raft::copy(&min_latest, min_latest_ptr, 1, stream_view.get()); + stream_view.sync(); return min_latest >= depot_earliest; } @@ -183,8 +183,8 @@ bool check_max_earliest_with_depot(rmm::device_uvector& v_earliest_time, i_t max_earliest; i_t* max_earliest_ptr = thrust::max_element( rmm::exec_policy(stream_view), v_earliest_time.begin() + 1, v_earliest_time.end()); - raft::copy(&max_earliest, max_earliest_ptr, 1, stream_view.value()); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view.value())); + raft::copy(&max_earliest, max_earliest_ptr, 1, stream_view.get()); + stream_view.sync(); return max_earliest <= depot_latest; } @@ -225,9 +225,9 @@ bool check_min_max_values(const T* ptr, T min, max; thrust::pair pair = thrust::minmax_element(rmm::exec_policy(stream_view), ptr, ptr + size); - raft::copy(&min, pair.first, 1, stream_view.value()); - raft::copy(&max, pair.second, 1, stream_view.value()); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream_view.value())); + raft::copy(&min, pair.first, 1, stream_view.get()); + raft::copy(&max, pair.second, 1, stream_view.get()); + stream_view.sync(); return (min >= static_cast(min_value)) && (max <= static_cast(max_value)); } @@ -262,7 +262,7 @@ void check_guess(i_t const* guess_id, d_int_drop_return_trip.data(), id); raft::update_host( - h_drop_return_trip.data(), d_int_drop_return_trip.data(), fleet_size, stream_view.value()); + h_drop_return_trip.data(), d_int_drop_return_trip.data(), fleet_size, stream_view.get()); thrust::transform(rmm::exec_policy(stream_view), skip_first_trip, @@ -270,7 +270,7 @@ void check_guess(i_t const* guess_id, d_int_skip_first_trip.data(), id); raft::update_host( - h_skip_first_trip.data(), d_int_skip_first_trip.data(), fleet_size, stream_view.value()); + h_skip_first_trip.data(), d_int_skip_first_trip.data(), fleet_size, stream_view.get()); raft::update_host(h_guess_id.data(), guess_id, size, stream_view); raft::update_host(h_truck_id.data(), truck_id, size, stream_view); diff --git a/cpp/src/routing/utilities/cython.cu b/cpp/src/routing/utilities/cython.cu index 5a0b9bf6b2..7c1e0170e3 100644 --- a/cpp/src/routing/utilities/cython.cu +++ b/cpp/src/routing/utilities/cython.cu @@ -125,7 +125,7 @@ std::vector> call_batch_solve( auto routing_solution = cuopt::routing::solve(*data_models[i], *settings); // Make sure current solve is finished - stream_pool.get_stream(i).synchronize(); + stream_pool.get_stream(i).sync(); // Create buffers and reassociate them with the original stream so they // outlive the local stream which will be destroyed at end of loop iteration @@ -152,7 +152,7 @@ std::vector> call_batch_solve( // Restore the old stream raft::resource::set_cuda_stream(*(data_models[i]->get_handle_ptr()), old_stream); - old_stream.synchronize(); + old_stream.sync(); } return list; diff --git a/cpp/src/utilities/copy_helpers.hpp b/cpp/src/utilities/copy_helpers.hpp index 6aa9efbab8..211fd4552a 100644 --- a/cpp/src/utilities/copy_helpers.hpp +++ b/cpp/src/utilities/copy_helpers.hpp @@ -124,7 +124,7 @@ auto host_copy(T const* device_ptr, size_t size, rmm::cuda_stream_view stream_vi if (!device_ptr) return std::vector{}; std::vector host_vec(size); raft::copy(host_vec.data(), device_ptr, size, stream_view); - stream_view.synchronize(); + stream_view.sync(); return host_vec; } @@ -150,7 +150,7 @@ inline auto host_copy(bool const* device_ptr, size_t size, rmm::cuda_stream_view for (size_t i = 0; i < h_int_vec.size(); ++i) { h_bool_vec[i] = static_cast(h_int_vec[i]); } - stream_view.synchronize(); + stream_view.sync(); return h_bool_vec; } @@ -167,7 +167,7 @@ auto host_copy(rmm::device_uvector const& device_vec, rmm::cuda_stream_view s { std::vector host_vec(device_vec.size()); raft::copy(host_vec.data(), device_vec.data(), device_vec.size(), stream_view); - stream_view.synchronize(); + stream_view.sync(); return host_vec; } diff --git a/cpp/src/utilities/event_handler.cuh b/cpp/src/utilities/event_handler.cuh index 452fe37804..17f6bd787c 100644 --- a/cpp/src/utilities/event_handler.cuh +++ b/cpp/src/utilities/event_handler.cuh @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2024-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2024-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -23,17 +23,17 @@ class event_handler_t { void record(rmm::cuda_stream_view stream_view) { - RAFT_CUDA_TRY(cudaEventRecord(event_, stream_view)); + RAFT_CUDA_TRY(cudaEventRecord(event_, stream_view.get())); } void record_with_flags(rmm::cuda_stream_view stream_view, int flags) { - RAFT_CUDA_TRY(cudaEventRecordWithFlags(event_, stream_view, flags)); + RAFT_CUDA_TRY(cudaEventRecordWithFlags(event_, stream_view.get(), flags)); } void stream_wait(rmm::cuda_stream_view stream_view) { - RAFT_CUDA_TRY(cudaStreamWaitEvent(stream_view, event_)); + RAFT_CUDA_TRY(cudaStreamWaitEvent(stream_view.get(), event_)); } float elapsed_time_since_ms(const event_handler_t& start) diff --git a/cpp/src/utilities/manual_cuda_graph.cuh b/cpp/src/utilities/manual_cuda_graph.cuh index d61cf04af8..bdc5ba9fd4 100644 --- a/cpp/src/utilities/manual_cuda_graph.cuh +++ b/cpp/src/utilities/manual_cuda_graph.cuh @@ -71,15 +71,15 @@ class manual_cuda_graph_t { void run(rmm::cuda_stream_view stream, F&& work) { if (instance_ != nullptr) { - RAFT_CUDA_TRY(cudaGraphLaunch(instance_, stream.value())); + RAFT_CUDA_TRY(cudaGraphLaunch(instance_, stream.get())); return; } // RAII: if user code throws mid-capture, end capture so the stream isn't // left in capture state. Errors are swallowed -- we're already unwinding. - capture_guard_t guard{stream.value()}; + capture_guard_t guard{stream.get()}; - RAFT_CUDA_TRY(cudaStreamBeginCapture(stream.value(), cudaStreamCaptureModeThreadLocal)); + RAFT_CUDA_TRY(cudaStreamBeginCapture(stream.get(), cudaStreamCaptureModeThreadLocal)); guard.capture_active = true; cudaGraph_t captured = nullptr; @@ -92,7 +92,7 @@ class manual_cuda_graph_t { // call). End the capture and let its status disambiguate: if the capture was // invalidated the recorded work was never issued, so recover by re-running // `work` eagerly; otherwise the error is genuine and is rethrown. - cudaError_t catch_end_err = cudaStreamEndCapture(stream.value(), &captured); + cudaError_t catch_end_err = cudaStreamEndCapture(stream.get(), &captured); guard.capture_active = false; if (catch_end_err == cudaErrorStreamCaptureInvalidated) { cudaGetLastError(); @@ -103,7 +103,7 @@ class manual_cuda_graph_t { throw; } - cudaError_t end_err = cudaStreamEndCapture(stream.value(), &captured); + cudaError_t end_err = cudaStreamEndCapture(stream.get(), &captured); guard.capture_active = false; if (end_err == cudaErrorStreamCaptureInvalidated) { @@ -124,7 +124,7 @@ class manual_cuda_graph_t { RAFT_CUDA_TRY_NO_THROW(cudaGraphDestroy(captured)); RAFT_CUDA_TRY(inst_err); - RAFT_CUDA_TRY(cudaGraphLaunch(instance_, stream.value())); + RAFT_CUDA_TRY(cudaGraphLaunch(instance_, stream.get())); } bool is_initialized() const noexcept { return instance_ != nullptr; } diff --git a/cpp/src/utilities/vector_helpers.cuh b/cpp/src/utilities/vector_helpers.cuh index b2c6cabbac..91f35c34fc 100644 --- a/cpp/src/utilities/vector_helpers.cuh +++ b/cpp/src/utilities/vector_helpers.cuh @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2022-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -43,7 +43,7 @@ void async_fill(rmm::device_uvector& vec, T item, rmm::cuda_stream_view strea { constexpr size_t TPB = 256; size_t n_blocks = (vec.size() + TPB - 1) / TPB; - fill_kernel<<>>(vec.data(), item, vec.size()); + fill_kernel<<>>(vec.data(), item, vec.size()); } template @@ -51,7 +51,7 @@ void async_fill(T* vec, T item, size_t size, rmm::cuda_stream_view stream) { constexpr size_t TPB = 256; size_t n_blocks = (size + TPB - 1) / TPB; - fill_kernel<<>>(vec, item, size); + fill_kernel<<>>(vec, item, size); } template @@ -59,7 +59,7 @@ void async_sequence(rmm::device_uvector& vec, rmm::cuda_stream_view stream) { constexpr size_t TPB = 256; size_t n_blocks = (vec.size() + TPB - 1) / TPB; - sequence_kernel<<>>(vec.data(), vec.size()); + sequence_kernel<<>>(vec.data(), vec.size()); } template @@ -69,7 +69,7 @@ void async_sequence_with_multiplier(rmm::device_uvector& vec, { constexpr size_t TPB = 256; size_t n_blocks = (vec.size() + TPB - 1) / TPB; - sequence_with_multiplier_kernel<<>>(vec.data(), mult, vec.size()); + sequence_with_multiplier_kernel<<>>(vec.data(), mult, vec.size()); } template diff --git a/cpp/tests/distance_engine/waypoint_matrix_test.cpp b/cpp/tests/distance_engine/waypoint_matrix_test.cpp index 88d4c53229..fdcada6544 100644 --- a/cpp/tests/distance_engine/waypoint_matrix_test.cpp +++ b/cpp/tests/distance_engine/waypoint_matrix_test.cpp @@ -59,7 +59,7 @@ class waypoint_matrix_waypoints_sequence_test_t std::vector h_cost_matrix(this->target_locations.size() * this->target_locations.size()); raft::copy(h_cost_matrix.data(), d_cost_matrix.data(), h_cost_matrix.size(), stream); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); + stream.sync(); for (size_t i = 0; i != h_cost_matrix.size(); ++i) EXPECT_EQ(h_cost_matrix[i], expected_cost_matrix[i]); @@ -78,7 +78,7 @@ class waypoint_matrix_waypoints_sequence_test_t h_sequence_offsets.size(), stream); raft::copy(h_full_path.data(), (i_t*)d_full_path.get()->data(), h_full_path.size(), stream); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); + stream.sync(); for (size_t i = 0; i != h_sequence_offsets.size(); ++i) EXPECT_EQ(h_sequence_offsets[i], expected_sequence_offsets[i]); @@ -154,7 +154,7 @@ class waypoint_matrix_shortest_path_cost_t std::vector h_custom_matrix(this->target_locations.size() * this->target_locations.size()); raft::copy(h_custom_matrix.data(), d_custom_matrix.data(), h_custom_matrix.size(), stream); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); + stream.sync(); for (size_t i = 0; i != h_custom_matrix.size(); ++i) EXPECT_EQ(h_custom_matrix[i], ref_custom_matrix[i]); @@ -207,7 +207,7 @@ class waypoint_matrix_cost_matrix_test_t std::vector h_cost_matrix(this->target_locations.size() * this->target_locations.size()); raft::copy(h_cost_matrix.data(), d_cost_matrix.data(), h_cost_matrix.size(), stream); - RAFT_CUDA_TRY(cudaStreamSynchronize(stream)); + stream.sync(); for (size_t i = 0; i != h_cost_matrix.size(); ++i) EXPECT_NEAR(h_cost_matrix[i], this->ref_cost_matrix[i], 0.001f); diff --git a/cpp/tests/dual_simplex/unit_tests/solve_barrier.cu b/cpp/tests/dual_simplex/unit_tests/solve_barrier.cu index 16640c6c60..de2f65bbfb 100644 --- a/cpp/tests/dual_simplex/unit_tests/solve_barrier.cu +++ b/cpp/tests/dual_simplex/unit_tests/solve_barrier.cu @@ -34,9 +34,10 @@ static void init_handler(const raft::handle_t* handle_ptr) { // Init cuBlas / cuSparse context here to avoid having it during solving time RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } TEST(barrier, chess_set) diff --git a/cpp/tests/linear_programming/pdlp_test.cu b/cpp/tests/linear_programming/pdlp_test.cu index d707082168..76f4e01044 100644 --- a/cpp/tests/linear_programming/pdlp_test.cu +++ b/cpp/tests/linear_programming/pdlp_test.cu @@ -503,7 +503,7 @@ TEST(pdlp_class, initial_solution_test) solver_settings); auto pdlp_timer = timer_t(solver_settings.time_limit); solver.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_NEAR(initial_step_size_afiro, solver.get_step_size_h(0), factor_tolerance); EXPECT_NEAR(initial_primal_weight_afiro, solver.get_primal_weight_h(0), factor_tolerance); } @@ -518,7 +518,7 @@ TEST(pdlp_class, initial_solution_test) auto d_initial_primal = device_copy(initial_primal, handle_.get_stream()); solver.set_initial_primal_solution(d_initial_primal); solver.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_NEAR(initial_step_size_afiro, solver.get_step_size_h(0), factor_tolerance); EXPECT_NEAR(initial_primal_weight_afiro, solver.get_primal_weight_h(0), factor_tolerance); } @@ -530,7 +530,7 @@ TEST(pdlp_class, initial_solution_test) auto d_initial_dual = device_copy(initial_dual, handle_.get_stream()); solver.set_initial_dual_solution(d_initial_dual); solver.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_NEAR(initial_step_size_afiro, solver.get_step_size_h(0), factor_tolerance); EXPECT_NEAR(initial_primal_weight_afiro, solver.get_primal_weight_h(0), factor_tolerance); } @@ -545,7 +545,7 @@ TEST(pdlp_class, initial_solution_test) auto d_initial_dual = device_copy(initial_dual, handle_.get_stream()); solver.set_initial_dual_solution(d_initial_dual); solver.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_NEAR(initial_step_size_afiro, solver.get_step_size_h(0), factor_tolerance); EXPECT_NEAR(initial_primal_weight_afiro, solver.get_primal_weight_h(0), factor_tolerance); } @@ -557,7 +557,7 @@ TEST(pdlp_class, initial_solution_test) auto pdlp_timer = timer_t(solver_settings.time_limit); solver_settings.hyper_params.update_step_size_on_initial_solution = true; solver.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_NEAR(initial_step_size_afiro, solver.get_step_size_h(0), factor_tolerance); EXPECT_NEAR(initial_primal_weight_afiro, solver.get_primal_weight_h(0), factor_tolerance); solver_settings.hyper_params.update_step_size_on_initial_solution = false; @@ -568,7 +568,7 @@ TEST(pdlp_class, initial_solution_test) auto pdlp_timer = timer_t(solver_settings.time_limit); solver_settings.hyper_params.update_primal_weight_on_initial_solution = true; solver.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_NEAR(initial_step_size_afiro, solver.get_step_size_h(0), factor_tolerance); EXPECT_NEAR(initial_primal_weight_afiro, solver.get_primal_weight_h(0), factor_tolerance); solver_settings.hyper_params.update_primal_weight_on_initial_solution = false; @@ -580,7 +580,7 @@ TEST(pdlp_class, initial_solution_test) solver_settings.hyper_params.update_primal_weight_on_initial_solution = true; solver_settings.hyper_params.update_step_size_on_initial_solution = true; solver.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_NEAR(initial_step_size_afiro, solver.get_step_size_h(0), factor_tolerance); EXPECT_NEAR(initial_primal_weight_afiro, solver.get_primal_weight_h(0), factor_tolerance); solver_settings.hyper_params.update_primal_weight_on_initial_solution = false; @@ -598,7 +598,7 @@ TEST(pdlp_class, initial_solution_test) auto d_initial_primal = device_copy(initial_primal, handle_.get_stream()); solver.set_initial_primal_solution(d_initial_primal); solver.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_NEAR(initial_step_size_afiro, solver.get_step_size_h(0), factor_tolerance); EXPECT_NEAR(initial_primal_weight_afiro, solver.get_primal_weight_h(0), factor_tolerance); solver_settings.hyper_params.update_step_size_on_initial_solution = false; @@ -612,7 +612,7 @@ TEST(pdlp_class, initial_solution_test) auto d_initial_dual = device_copy(initial_dual, handle_.get_stream()); solver.set_initial_dual_solution(d_initial_dual); solver.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_NEAR(initial_step_size_afiro, solver.get_step_size_h(0), factor_tolerance); EXPECT_NEAR(initial_primal_weight_afiro, solver.get_primal_weight_h(0), factor_tolerance); solver_settings.hyper_params.update_step_size_on_initial_solution = false; @@ -799,7 +799,7 @@ TEST(pdlp_class, initial_primal_weight_step_size_test) solver.set_initial_primal_weight(test_initial_primal_weight); solver.set_initial_step_size(test_initial_step_size); solver.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_EQ(test_initial_step_size, solver.get_step_size_h(0)); EXPECT_EQ(test_initial_primal_weight, solver.get_primal_weight_h(0)); } @@ -834,7 +834,7 @@ TEST(pdlp_class, initial_primal_weight_step_size_test) solver2.set_initial_primal_solution(d_initial_primal); solver2.set_initial_dual_solution(d_initial_dual); solver2.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); const double sovler2_step_size = solver2.get_step_size_h(0); const double sovler2_primal_weight = solver2.get_primal_weight_h(0); EXPECT_NOT_NEAR(previous_step_size, sovler2_step_size, factor_tolerance); @@ -851,7 +851,7 @@ TEST(pdlp_class, initial_primal_weight_step_size_test) solver3.set_initial_dual_solution(d_initial_dual); solver3.set_initial_dual_solution(d_initial_dual); solver3.run_solver(pdlp_timer); - RAFT_CUDA_TRY(cudaStreamSynchronize(handle_.get_stream())); + handle_.get_stream().sync(); EXPECT_NOT_NEAR(sovler2_step_size, solver3.get_step_size_h(0), factor_tolerance); EXPECT_NEAR(sovler2_primal_weight, solver3.get_primal_weight_h(0), factor_tolerance); } diff --git a/cpp/tests/linear_programming/unit_tests/solution_interface_test.cu b/cpp/tests/linear_programming/unit_tests/solution_interface_test.cu index 7d1fc6b21b..616e0d62f1 100644 --- a/cpp/tests/linear_programming/unit_tests/solution_interface_test.cu +++ b/cpp/tests/linear_programming/unit_tests/solution_interface_test.cu @@ -28,6 +28,8 @@ #include +#include + #include #include @@ -145,7 +147,7 @@ static std::unique_ptr> make_cpu_mip_solution() // Build a gpu_lp_solution_t with known device data (no solver needed) static gpu_lp_solution_t make_gpu_lp_solution() { - auto stream = rmm::cuda_stream_per_thread; + auto stream = cuda::stream_ref{cudaStreamPerThread}; rmm::device_uvector primal(kNVars, stream); rmm::device_uvector dual(kNCons, stream); @@ -180,7 +182,7 @@ static gpu_lp_solution_t make_gpu_lp_solution() // Build a gpu_mip_solution_t with known device data (no solver needed) static gpu_mip_solution_t make_gpu_mip_solution() { - auto stream = rmm::cuda_stream_per_thread; + auto stream = cuda::stream_ref{cudaStreamPerThread}; rmm::device_uvector sol(kNVars, stream); std::vector h_sol = {1.0, 0.0, 1.0}; diff --git a/cpp/tests/mip/bounds_standardization_test.cu b/cpp/tests/mip/bounds_standardization_test.cu index fffaec4989..aa920f5c0d 100644 --- a/cpp/tests/mip/bounds_standardization_test.cu +++ b/cpp/tests/mip/bounds_standardization_test.cu @@ -35,9 +35,10 @@ static void init_handler(const raft::handle_t* handle_ptr) { // Init cuBlas / cuSparse context here to avoid having it during solving time RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } void test_bounds_standardization_test(std::string test_instance) diff --git a/cpp/tests/mip/elim_var_remap_test.cu b/cpp/tests/mip/elim_var_remap_test.cu index 1cbc1cc60f..56521d614a 100644 --- a/cpp/tests/mip/elim_var_remap_test.cu +++ b/cpp/tests/mip/elim_var_remap_test.cu @@ -38,9 +38,10 @@ static void init_handler(const raft::handle_t* handle_ptr) { // Init cuBlas / cuSparse context here to avoid having it during solving time RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } std::vector select_k_random(int population_size, int sample_size) diff --git a/cpp/tests/mip/feasibility_jump_tests.cu b/cpp/tests/mip/feasibility_jump_tests.cu index c0c7800cad..6e1a9d0f25 100644 --- a/cpp/tests/mip/feasibility_jump_tests.cu +++ b/cpp/tests/mip/feasibility_jump_tests.cu @@ -41,9 +41,10 @@ void init_handler(const raft::handle_t* handle_ptr) { // Init cuBlas / cuSparse context here to avoid having it during solving time RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } struct fj_tweaks_t { diff --git a/cpp/tests/mip/load_balancing_test.cu b/cpp/tests/mip/load_balancing_test.cu index c686388aad..f05befefd4 100644 --- a/cpp/tests/mip/load_balancing_test.cu +++ b/cpp/tests/mip/load_balancing_test.cu @@ -38,9 +38,10 @@ void init_handler(const raft::handle_t* handle_ptr) { // Init cuBlas / cuSparse context here to avoid having it during solving time RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } std::tuple, std::vector, std::vector> select_k_random( diff --git a/cpp/tests/mip/multi_probe_test.cu b/cpp/tests/mip/multi_probe_test.cu index 9438bf6183..61be30e15c 100644 --- a/cpp/tests/mip/multi_probe_test.cu +++ b/cpp/tests/mip/multi_probe_test.cu @@ -37,9 +37,10 @@ static void init_handler(const raft::handle_t* handle_ptr) { // Init cuBlas / cuSparse context here to avoid having it during solving time RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } std::tuple, std::vector, std::vector> select_k_random( diff --git a/cpp/tests/routing/level0/l0_routing_test.cu b/cpp/tests/routing/level0/l0_routing_test.cu index 28bd8db9c7..40e076f542 100644 --- a/cpp/tests/routing/level0/l0_routing_test.cu +++ b/cpp/tests/routing/level0/l0_routing_test.cu @@ -372,11 +372,11 @@ class routing_retail_test_t : public base_test_t, raft::copy(this->vehicle_earliest_d.data(), this->vehicle_earliest_h.data(), input_.n_vehicles, - this->stream_view_.value()); + this->stream_view_.get()); raft::copy(this->vehicle_latest_d.data(), this->vehicle_latest_h.data(), input_.n_vehicles, - this->stream_view_.value()); + this->stream_view_.get()); data_model.set_vehicle_time_windows(this->vehicle_earliest_d.data(), this->vehicle_latest_d.data()); } @@ -392,7 +392,7 @@ class routing_retail_test_t : public base_test_t, raft::copy(d_int_drop_return_trip.data(), this->drop_return_trips_h.data(), input_.n_vehicles, - this->stream_view_.value()); + this->stream_view_.get()); thrust::transform(this->handle_.get_thrust_policy(), d_int_drop_return_trip.begin(), d_int_drop_return_trip.end(), @@ -402,13 +402,13 @@ class routing_retail_test_t : public base_test_t, raft::copy(d_int_skip_first_trip.data(), this->skip_first_trips_h.data(), input_.n_vehicles, - this->stream_view_.value()); + this->stream_view_.get()); thrust::transform(this->handle_.get_thrust_policy(), d_int_skip_first_trip.begin(), d_int_skip_first_trip.end(), d_skip_first_trip.begin(), id); - RAFT_CUDA_TRY(cudaStreamSynchronize(this->stream_view_.value())); + this->stream_view_.sync(); data_model.set_drop_return_trips(d_drop_return_trip.data()); data_model.set_skip_first_trips(d_skip_first_trip.data()); } @@ -423,11 +423,11 @@ class routing_retail_test_t : public base_test_t, raft::copy(this->random_demand_d.data(), shuffled_vec.data(), this->n_orders, - this->stream_view_.value()); + this->stream_view_.get()); raft::copy(this->mixed_capacity_d.data(), input_.mixed_capacity_h.data(), this->n_vehicles, - this->stream_view_.value()); + this->stream_view_.get()); data_model.add_capacity_dimension( "random", this->random_demand_d.data(), this->mixed_capacity_d.data()); } diff --git a/cpp/tests/routing/level0/l0_vehicle_order_match.cu b/cpp/tests/routing/level0/l0_vehicle_order_match.cu index f99d1a33df..0f0e6390aa 100644 --- a/cpp/tests/routing/level0/l0_vehicle_order_match.cu +++ b/cpp/tests/routing/level0/l0_vehicle_order_match.cu @@ -58,7 +58,7 @@ class vehicle_order_test_t : public base_test_t, public ::testing::Tes d_int_vec.end(), d_drop_return_trip.begin(), cuda::std::identity{}); - RAFT_CUDA_TRY(cudaStreamSynchronize(this->stream_view_.value())); + this->stream_view_.sync(); } data_model.set_drop_return_trips(d_drop_return_trip.data()); diff --git a/cpp/tests/routing/unit_tests/local_search_cand_test.cu b/cpp/tests/routing/unit_tests/local_search_cand_test.cu index e865a789b3..40997d8497 100644 --- a/cpp/tests/routing/unit_tests/local_search_cand_test.cu +++ b/cpp/tests/routing/unit_tests/local_search_cand_test.cu @@ -1,6 +1,6 @@ /* clang-format off */ /* - * SPDX-FileCopyrightText: Copyright (c) 2021-2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-FileCopyrightText: Copyright (c) 2021-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. * SPDX-License-Identifier: Apache-2.0 */ /* clang-format on */ @@ -357,7 +357,7 @@ class routing_ges_test_t : public ::testing::TestWithParamtest_type == test_t::INFEASIBLE) { double w[] = {100., 10000., 100., 100., 100.}; introduce_infeasibility - <<<1, 1, 0, sol.sol_handle->get_stream()>>>(sol.view()); + <<<1, 1, 0, sol.sol_handle->get_stream().get()>>>(sol.view()); sol.set_nodes_data_of_solution(); sol.compute_initial_data(); f_t old_cost = sol.get_total_cost(w); diff --git a/cpp/tests/routing/unit_tests/top_k.cu b/cpp/tests/routing/unit_tests/top_k.cu index c6d377a63f..6a4e48b790 100644 --- a/cpp/tests/routing/unit_tests/top_k.cu +++ b/cpp/tests/routing/unit_tests/top_k.cu @@ -98,7 +98,7 @@ class top_cand_test_t : public routing_test_t, public ::testing::TestW raft::copy(d_input_cost.data(), h_input_cost.data(), h_input_cost.size(), this->stream_view_); - this->stream_view_.synchronize(); + this->stream_view_.sync(); call_top_k(d_input_cost, d_output_cost, d_out_index); verify_top_k(h_input_cost, d_output_cost, d_out_index); @@ -159,11 +159,12 @@ class top_cand_test_t : public routing_test_t, public ::testing::TestW rmm::device_uvector& out_index) { constexpr int TPB = 128; - top_k_indices<<stream_view_>>>(width, - cuopt::make_span(input_cost), - cuopt::make_span(output_cost), - cuopt::make_span(out_index)); - this->stream_view_.synchronize(); + top_k_indices + <<stream_view_.get()>>>(width, + cuopt::make_span(input_cost), + cuopt::make_span(output_cost), + cuopt::make_span(out_index)); + this->stream_view_.sync(); RAFT_CUDA_TRY(cudaGetLastError()); } @@ -171,15 +172,15 @@ class top_cand_test_t : public routing_test_t, public ::testing::TestW rmm::device_uvector& d_output_cost, rmm::device_uvector& d_out_index) { - this->stream_view_.synchronize(); + this->stream_view_.sync(); std::vector h_output_cost(d_output_cost.size()); raft::copy( h_output_cost.data(), d_output_cost.data(), d_output_cost.size(), this->stream_view_); - this->stream_view_.synchronize(); + this->stream_view_.sync(); std::vector h_sorted_index(d_out_index.size()); raft::copy(h_sorted_index.data(), d_out_index.data(), d_out_index.size(), this->stream_view_); - this->stream_view_.synchronize(); + this->stream_view_.sync(); std::vector sorted_data(width); for (int i = 0; i < width; ++i) { // copy row i @@ -234,11 +235,11 @@ class top_cand_test_t : public routing_test_t, public ::testing::TestW num_segments, segment_marker.data(), segment_marker.data() + 1, - this->stream_view_); + this->stream_view_.get()); rmm::device_uvector d_cub_storage_bytes(0, this->stream_view_); d_cub_storage_bytes.resize(tmp_storage_bytes, this->stream_view_); double elapsed_ms; - this->stream_view_.synchronize(); + this->stream_view_.sync(); { time_it t(&elapsed_ms); for (int i = 0; i < iter; ++i) { @@ -252,9 +253,9 @@ class top_cand_test_t : public routing_test_t, public ::testing::TestW num_segments, segment_marker.data(), segment_marker.data() + 1, - this->stream_view_); + this->stream_view_.get()); } - this->stream_view_.synchronize(); + this->stream_view_.sync(); } return elapsed_ms; } @@ -268,13 +269,13 @@ class top_cand_test_t : public routing_test_t, public ::testing::TestW raft::copy(d_input_cost.data(), input_cost.data(), input_cost.size(), this->stream_view_); double elapsed_ms; - this->stream_view_.synchronize(); + this->stream_view_.sync(); { time_it t(&elapsed_ms); for (int i = 0; i < iter; ++i) { call_top_k(d_input_cost, d_output_cost, d_out_index); } - this->stream_view_.synchronize(); + this->stream_view_.sync(); } return elapsed_ms; } diff --git a/cpp/tests/routing/utilities/check_constraints.cu b/cpp/tests/routing/utilities/check_constraints.cu index e5debd4440..068182d7ba 100644 --- a/cpp/tests/routing/utilities/check_constraints.cu +++ b/cpp/tests/routing/utilities/check_constraints.cu @@ -10,8 +10,6 @@ #include -#include - #include #include diff --git a/cpp/tests/socp/general_quadratic_test.cu b/cpp/tests/socp/general_quadratic_test.cu index b2a5afeafb..182c978eef 100644 --- a/cpp/tests/socp/general_quadratic_test.cu +++ b/cpp/tests/socp/general_quadratic_test.cu @@ -40,9 +40,10 @@ using qc_t = optimization_problem_interface_t::quadratic_constraint_t; static void init_handler(const raft::handle_t* handle_ptr) { RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } // Test: general convex quadratic constraint with dense PD Q matrix. diff --git a/cpp/tests/socp/second_order_cone_kernels.cu b/cpp/tests/socp/second_order_cone_kernels.cu index 4193838272..8081fdc97c 100644 --- a/cpp/tests/socp/second_order_cone_kernels.cu +++ b/cpp/tests/socp/second_order_cone_kernels.cu @@ -7,6 +7,8 @@ #include +#include + #include #include @@ -20,7 +22,7 @@ namespace cuopt::mathematical_optimization::barrier::test { TEST(second_order_cone_kernels, topology_and_scratch_layout) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{3, 2, 5}; rmm::device_uvector x(10, stream); @@ -62,7 +64,7 @@ TEST(second_order_cone_kernels, topology_and_scratch_layout) TEST(second_order_cone_kernels, segmented_sum_uses_all_cone_size_buckets) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{65, 3, 66, 32769}; rmm::device_uvector x(32903, stream); @@ -93,7 +95,7 @@ TEST(second_order_cone_kernels, segmented_sum_uses_all_cone_size_buckets) TEST(second_order_cone_kernels, nt_scaling_matches_host_reference) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{3, 65, 32769}; std::size_t n_cone_entries = 0; @@ -204,7 +206,7 @@ TEST(second_order_cone_kernels, nt_scaling_matches_host_reference) TEST(second_order_cone_kernels, nt_scaling_many_small_one_medium_cone) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; // chainsing-like topology: many small cones plus one dim-1000 medium cone (warp_cone_dim=64). std::vector cone_dimensions; @@ -309,7 +311,7 @@ TEST(second_order_cone_kernels, nt_scaling_many_small_one_medium_cone) TEST(second_order_cone_kernels, cone_step_length_many_small_one_sparse_medium_cone) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions; for (int i = 0; i < 20; ++i) { @@ -367,7 +369,7 @@ TEST(second_order_cone_kernels, cone_step_length_many_small_one_sparse_medium_co TEST(second_order_cone_kernels, cone_step_length_keeps_iterate_in_cone) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{3, 65, 32769, 40000}; std::size_t n_cone_entries = 0; @@ -521,7 +523,7 @@ TEST(second_order_cone_kernels, cone_step_length_keeps_iterate_in_cone) TEST(second_order_cone_kernels, scaling_operators_match_host_reference) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{3, 65, 32769}; std::size_t n_cone_entries = 0; @@ -650,7 +652,7 @@ TEST(second_order_cone_kernels, scaling_operators_match_host_reference) TEST(second_order_cone_kernels, combined_cone_rhs_matches_host_reference) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{3, 65, 32769}; std::size_t n_cone_entries = 0; @@ -812,7 +814,7 @@ TEST(second_order_cone_kernels, combined_cone_rhs_matches_host_reference) TEST(second_order_cone_kernels, sparse_cone_classification) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; // threshold=5: cones of dim 3,2 are dense; dim 6,32769 are sparse std::vector cone_dimensions{3, 6, 2, 32769}; @@ -866,7 +868,7 @@ sparse_scaling_head_t sparse_scaling_head_reference(double w0) TEST(second_order_cone_kernels, update_scaling_sparse_matches_reference) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{3, 6}; rmm::device_uvector x(9, stream); @@ -924,7 +926,7 @@ TEST(second_order_cone_kernels, update_scaling_sparse_matches_reference) TEST(second_order_cone_kernels, update_scaling_sparse_two_cones) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; // sparse cones: dim 6 (offset 3) and dim 5 (offset 9) std::vector cone_dimensions{3, 6, 5}; diff --git a/cpp/tests/socp/solve_barrier_socp.cu b/cpp/tests/socp/solve_barrier_socp.cu index 68e2cb2d31..59fa339904 100644 --- a/cpp/tests/socp/solve_barrier_socp.cu +++ b/cpp/tests/socp/solve_barrier_socp.cu @@ -27,9 +27,10 @@ static void init_handler(const raft::handle_t* handle_ptr) { // Init cuBlas / cuSparse context here to avoid having it during solving time RAFT_CUBLAS_TRY(raft::linalg::detail::cublassetpointermode( - handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream())); - RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode( - handle_ptr->get_cusparse_handle(), CUSPARSE_POINTER_MODE_DEVICE, handle_ptr->get_stream())); + handle_ptr->get_cublas_handle(), CUBLAS_POINTER_MODE_DEVICE, handle_ptr->get_stream().get())); + RAFT_CUSPARSE_TRY(raft::sparse::detail::cusparsesetpointermode(handle_ptr->get_cusparse_handle(), + CUSPARSE_POINTER_MODE_DEVICE, + handle_ptr->get_stream().get())); } TEST(barrier, cone_metadata_reindexed_when_slack_is_inserted_before_cones) diff --git a/cpp/tests/socp/sparse_augmented_kkt_test.cu b/cpp/tests/socp/sparse_augmented_kkt_test.cu index b4dae0591b..533b29cceb 100644 --- a/cpp/tests/socp/sparse_augmented_kkt_test.cu +++ b/cpp/tests/socp/sparse_augmented_kkt_test.cu @@ -9,6 +9,8 @@ #include #include +#include + #include #include @@ -49,7 +51,7 @@ std::vector expected_Hs_diag(const cone_data_t& cones, TEST(sparse_augmented_kkt, cone_counts_and_expansion_size) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{3, 6, 5}; rmm::device_uvector x(14, stream); @@ -66,7 +68,7 @@ TEST(sparse_augmented_kkt, cone_counts_and_expansion_size) TEST(sparse_augmented_kkt, scatter_sparse_hessian_into_augmented) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; // Two sparse cones so the fused entry-parallel kernel has more than one sparse-cone // boundary to get right. @@ -177,7 +179,7 @@ TEST(sparse_augmented_kkt, scatter_sparse_hessian_into_augmented) TEST(sparse_augmented_kkt, sparse_augmented_matvec) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{6}; rmm::device_uvector x(6, stream); @@ -256,7 +258,7 @@ TEST(sparse_augmented_kkt, sparse_augmented_matvec) TEST(sparse_augmented_kkt, update_scaling_sparse_dim_1000) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{1000}; rmm::device_uvector x(1000, stream); @@ -350,7 +352,7 @@ TEST(sparse_augmented_kkt, update_scaling_sparse_dim_1000) TEST(sparse_augmented_kkt, gpu_augmented_csr_metadata_matches_host) { - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; std::vector cone_dimensions{3, 6, 5}; rmm::device_uvector x(14, stream); @@ -411,7 +413,7 @@ TEST(sparse_augmented_kkt, augmented_csr_indices_mixed_dense_sparse_qp) // and the right-hand side b are irrelevant and left unset. using i_t = int; using f_t = double; - auto stream = rmm::cuda_stream_default; + auto stream = cuda::stream_ref{cudaStream_t{cudaStreamDefault}}; // Layout: 1 linear var, dense Q^3 cone (cols [1,4)), sparse Q^4 cone (cols // [4,8)), 2 constraints. Factorization size = n + m + p = 8 + 2 + 2 = 12. From 0f79f11a6c3904fae097ec26d5a7d05f4e35a48a Mon Sep 17 00:00:00 2001 From: Trevor McKay Date: Tue, 8 Sep 2026 12:17:23 -0400 Subject: [PATCH 043/113] move server LPData transform into a standalone function (#1848) This is one of a series of changes to refactor the cuopt http server so that a new proxy server can be added that shares the same data conversion code but delegates solves to the gRPC server. It will run as a sidecar. Here, we are moving an LPData transform function out of a class into a standalone function. It converts lists to numpy arrays and "inf" and "ninf" to numpy infinity values, etc. Authors: - Trevor McKay (https://github.com/tmckayus) Approvers: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) URL: https://github.com/NVIDIA/cuopt/pull/1848 --- .../tests/test_lp_data_transformation.py | 191 ++++++++++++++++++ .../cuopt_server/utils/job_queue.py | 131 +----------- .../linear_programming/data_transformation.py | 139 +++++++++++++ .../cuopt_server/cuopt_server/utils/utils.py | 12 +- 4 files changed, 339 insertions(+), 134 deletions(-) create mode 100644 python/cuopt_server/cuopt_server/tests/test_lp_data_transformation.py create mode 100644 python/cuopt_server/cuopt_server/utils/linear_programming/data_transformation.py diff --git a/python/cuopt_server/cuopt_server/tests/test_lp_data_transformation.py b/python/cuopt_server/cuopt_server/tests/test_lp_data_transformation.py new file mode 100644 index 0000000000..fd895b5628 --- /dev/null +++ b/python/cuopt_server/cuopt_server/tests/test_lp_data_transformation.py @@ -0,0 +1,191 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import copy + +import numpy as np + +from cuopt_server.utils.linear_programming.data_definition import LPData +from cuopt_server.utils.linear_programming.data_transformation import ( + transform_lp_data, +) + + +def _sample_lp_dict(): + return { + "csr_constraint_matrix": { + "offsets": [0, 2], + "indices": [0, 1], + "values": [1.0, 1.0], + }, + "constraint_bounds": { + "upper_bounds": [5000.0], + "lower_bounds": ["ninf"], + "types": ["L"], + }, + "objective_data": { + "coefficients": [1.2, 1.7], + "scalability_factor": 1.0, + "offset": 0.0, + }, + "variable_bounds": { + "upper_bounds": ["inf", "inf"], + "lower_bounds": [0.0, 0.0], + }, + "initial_solution": { + "primal": [1.0, 2.0], + "dual": [0.5], + }, + "variable_types": ["C", "I"], + "variable_names": ["x", "y"], + "maximize": False, + } + + +def test_transform_lp_data_converts_dict_lists_to_numpy_dtypes(): + data = _sample_lp_dict() + + transform_lp_data(data) + + csr = data["csr_constraint_matrix"] + assert csr["offsets"].dtype == np.int32 + assert csr["indices"].dtype == np.int32 + assert csr["values"].dtype == np.float64 + np.testing.assert_array_equal( + csr["offsets"], np.array([0, 2], dtype=np.int32) + ) + np.testing.assert_array_equal( + csr["indices"], np.array([0, 1], dtype=np.int32) + ) + np.testing.assert_array_equal( + csr["values"], np.array([1.0, 1.0], dtype=np.float64) + ) + + assert data["constraint_bounds"]["upper_bounds"].dtype == np.float64 + assert data["constraint_bounds"]["types"].dtype == np.dtype("U1") + np.testing.assert_array_equal( + data["constraint_bounds"]["types"], np.array(["L"], dtype="U1") + ) + + assert data["objective_data"]["coefficients"].dtype == np.float64 + assert data["initial_solution"]["primal"].dtype == np.float64 + assert data["initial_solution"]["dual"].dtype == np.float64 + assert data["variable_types"].dtype == np.dtype("U1") + np.testing.assert_array_equal( + data["variable_types"], np.array(["C", "I"], dtype="U1") + ) + + # Unmapped fields must keep their original Python types. + assert data["variable_names"] == ["x", "y"] + assert data["maximize"] is False + assert data["objective_data"]["offset"] == 0.0 + + +def test_transform_lp_data_converts_inf_and_ninf_in_dict(): + data = _sample_lp_dict() + + transform_lp_data(data) + + np.testing.assert_array_equal( + data["constraint_bounds"]["lower_bounds"], + np.array([-np.inf], dtype=np.float64), + ) + np.testing.assert_array_equal( + data["variable_bounds"]["upper_bounds"], + np.array([np.inf, np.inf], dtype=np.float64), + ) + assert data["variable_bounds"]["lower_bounds"].dtype == np.float64 + + +def test_transform_lp_data_converts_lpdata_lists_to_numpy_dtypes(): + lp_data = LPData.parse_obj(_sample_lp_dict()) + + transform_lp_data(lp_data) + + csr = lp_data.csr_constraint_matrix + assert csr.offsets.dtype == np.int32 + assert csr.indices.dtype == np.int32 + assert csr.values.dtype == np.float64 + np.testing.assert_array_equal( + csr.offsets, np.array([0, 2], dtype=np.int32) + ) + + assert lp_data.objective_data.coefficients.dtype == np.float64 + assert lp_data.initial_solution.primal.dtype == np.float64 + assert lp_data.initial_solution.dual.dtype == np.float64 + assert lp_data.variable_bounds.lower_bounds.dtype == np.float64 + + # LPData path omits dtype for types / variable_types (numpy default). + assert isinstance(lp_data.constraint_bounds.types, np.ndarray) + assert isinstance(lp_data.variable_types, np.ndarray) + np.testing.assert_array_equal(lp_data.constraint_bounds.types, ["L"]) + np.testing.assert_array_equal(lp_data.variable_types, ["C", "I"]) + + +def test_transform_lp_data_converts_inf_and_ninf_in_lpdata(): + lp_data = LPData.parse_obj(_sample_lp_dict()) + + transform_lp_data(lp_data) + + np.testing.assert_array_equal( + lp_data.constraint_bounds.lower_bounds, + np.array([-np.inf], dtype=np.float64), + ) + np.testing.assert_array_equal( + lp_data.variable_bounds.upper_bounds, + np.array([np.inf, np.inf], dtype=np.float64), + ) + + +def test_transform_lp_data_leaves_non_list_values_unchanged(): + existing = np.array([1.0, 2.0], dtype=np.float32) + data = { + "objective_data": {"coefficients": existing}, + "variable_names": ["x", "y"], + } + + transform_lp_data(data) + + assert data["objective_data"]["coefficients"] is existing + assert data["variable_names"] == ["x", "y"] + + +def test_transform_lp_data_mutates_in_place(): + data = _sample_lp_dict() + original = data + nested = data["csr_constraint_matrix"] + + transform_lp_data(data) + + assert data is original + assert data["csr_constraint_matrix"] is nested + assert isinstance(nested["offsets"], np.ndarray) + + +def test_transform_lp_data_mixed_inf_and_finite_values(): + data = { + "variable_bounds": { + "upper_bounds": [1.5, "inf", 3.0, "ninf"], + "lower_bounds": ["ninf", 0.0, "inf"], + } + } + + transform_lp_data(data) + + np.testing.assert_array_equal( + data["variable_bounds"]["upper_bounds"], + np.array([1.5, np.inf, 3.0, -np.inf], dtype=np.float64), + ) + np.testing.assert_array_equal( + data["variable_bounds"]["lower_bounds"], + np.array([-np.inf, 0.0, np.inf], dtype=np.float64), + ) + + +def test_transform_lp_data_does_not_touch_unrelated_dict_keys(): + data = copy.deepcopy(_sample_lp_dict()) + data["solver_config"] = {"time_limit": 5} + + transform_lp_data(data) + + assert data["solver_config"] == {"time_limit": 5} diff --git a/python/cuopt_server/cuopt_server/utils/job_queue.py b/python/cuopt_server/cuopt_server/utils/job_queue.py index ac067e5e0d..53557c69fb 100644 --- a/python/cuopt_server/cuopt_server/utils/job_queue.py +++ b/python/cuopt_server/cuopt_server/utils/job_queue.py @@ -37,6 +37,9 @@ exception_handler, http_exception_handler, ) +from cuopt_server.utils.linear_programming.data_transformation import ( + transform_lp_data, +) from cuopt_server.utils.logutil import message from cuopt_server.utils.routing.initial_solution import add_initial_sol @@ -884,133 +887,7 @@ def get_data(self): return self.LP_data def _transform(self, data): - np = numpy - tmap = { - "csr_constraint_matrix": { - "offsets": (True, np.int32), - "indices": (True, np.int32), - "values": (True, np.float64), - }, - "constraint_bounds": { - "bounds": (True, np.float64), - "upper_bounds": (True, np.float64), - "lower_bounds": (True, np.float64), - "types": (True, "U1"), - }, - "initial_solution": { - "primal": (True, np.float64), - "dual": (True, np.float64), - }, - "objective_data": { - "coefficients": (True, np.float64), - }, - "variable_bounds": { - "upper_bounds": (True, np.float64), - "lower_bounds": (True, np.float64), - }, - "variable_types": (True, "U1"), - } - - def modify(value, key, dtype=None, indent=""): - if isinstance(value, list): - if "inf" in value or "ninf" in value: - value = [ - np.inf if x == "inf" else -np.inf if x == "ninf" else x - for x in value - ] - if dtype is None: - return np.array(value) - return np.array(value, dtype) - return value - - def apply(data, tmap, indent=""): - for key, value in data.items(): - try: - if isinstance(value, dict) and key in tmap: - apply(value, tmap[key], indent + " ") - elif key in tmap and tmap[key][0]: - data[key] = modify(value, key, tmap[key][1], indent) - except Exception as e: - logging.debug(e) - logging.debug( - f"{indent}exception key is {key} value is {value}" - ) - raise - - def apply_LPData(data): - data.csr_constraint_matrix.indices = modify( - data.csr_constraint_matrix.indices, - "csr_constraint_matrix.indices", - np.int32, - ) - data.csr_constraint_matrix.offsets = modify( - data.csr_constraint_matrix.offsets, - "csr_constraint_matrix.offsets", - np.int32, - ) - data.csr_constraint_matrix.values = modify( - data.csr_constraint_matrix.values, - "csr_constraint_matrix.values", - np.float64, - ) - - data.constraint_bounds.bounds = modify( - data.constraint_bounds.bounds, - "constraint_bounds.bounds", - np.float64, - ) - data.constraint_bounds.upper_bounds = modify( - data.constraint_bounds.upper_bounds, - "constraint_bounds.upper_bounds", - np.float64, - ) - data.constraint_bounds.lower_bounds = modify( - data.constraint_bounds.lower_bounds, - "constraint_bounds.lower_bounds", - np.float64, - ) - data.constraint_bounds.types = modify( - data.constraint_bounds.types, - "constraint_bounds.types", - ) - - data.initial_solution.primal = modify( - data.initial_solution.primal, - "initial_solution.primal", - np.float64, - ) - - data.initial_solution.dual = modify( - data.initial_solution.dual, "initial_solution.dual", np.float64 - ) - - data.objective_data.coefficients = modify( - data.objective_data.coefficients, - "objective_data.coefficients", - np.float64, - ) - - data.variable_bounds.upper_bounds = modify( - data.variable_bounds.upper_bounds, - "variable_bounds.upper_bounds", - np.float64, - ) - data.variable_bounds.lower_bounds = modify( - data.variable_bounds.lower_bounds, - "variable_bounds.lower_bounds", - np.float64, - ) - data.variable_types = modify( - data.variable_types, - "variable_types", - ) - - then = time.time() - if isinstance(data, LPData): - apply_LPData(data) - else: - apply(data, tmap) - logging.info(f"transform time {time.time() - then}") + transform_lp_data(data) def _load_data(self): if not self.transformed: diff --git a/python/cuopt_server/cuopt_server/utils/linear_programming/data_transformation.py b/python/cuopt_server/cuopt_server/utils/linear_programming/data_transformation.py new file mode 100644 index 0000000000..4b76e7f550 --- /dev/null +++ b/python/cuopt_server/cuopt_server/utils/linear_programming/data_transformation.py @@ -0,0 +1,139 @@ +# SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import logging +import time + +import numpy + +from cuopt_server.utils.linear_programming.data_definition import LPData + + +def transform_lp_data(data): + np = numpy + tmap = { + "csr_constraint_matrix": { + "offsets": (True, np.int32), + "indices": (True, np.int32), + "values": (True, np.float64), + }, + "constraint_bounds": { + "bounds": (True, np.float64), + "upper_bounds": (True, np.float64), + "lower_bounds": (True, np.float64), + "types": (True, "U1"), + }, + "initial_solution": { + "primal": (True, np.float64), + "dual": (True, np.float64), + }, + "objective_data": { + "coefficients": (True, np.float64), + }, + "variable_bounds": { + "upper_bounds": (True, np.float64), + "lower_bounds": (True, np.float64), + }, + "variable_types": (True, "U1"), + } + + def modify(value, key, dtype=None, indent=""): + if isinstance(value, list): + if "inf" in value or "ninf" in value: + value = [ + np.inf if x == "inf" else -np.inf if x == "ninf" else x + for x in value + ] + if dtype is None: + return np.array(value) + return np.array(value, dtype) + return value + + def apply(data, tmap, indent=""): + for key, value in data.items(): + try: + if isinstance(value, dict) and key in tmap: + apply(value, tmap[key], indent + " ") + elif key in tmap and tmap[key][0]: + data[key] = modify(value, key, tmap[key][1], indent) + except Exception as e: + logging.debug(e) + logging.debug( + f"{indent}exception key is {key} value is {value}" + ) + raise + + def apply_LPData(data): + data.csr_constraint_matrix.indices = modify( + data.csr_constraint_matrix.indices, + "csr_constraint_matrix.indices", + np.int32, + ) + data.csr_constraint_matrix.offsets = modify( + data.csr_constraint_matrix.offsets, + "csr_constraint_matrix.offsets", + np.int32, + ) + data.csr_constraint_matrix.values = modify( + data.csr_constraint_matrix.values, + "csr_constraint_matrix.values", + np.float64, + ) + + data.constraint_bounds.bounds = modify( + data.constraint_bounds.bounds, + "constraint_bounds.bounds", + np.float64, + ) + data.constraint_bounds.upper_bounds = modify( + data.constraint_bounds.upper_bounds, + "constraint_bounds.upper_bounds", + np.float64, + ) + data.constraint_bounds.lower_bounds = modify( + data.constraint_bounds.lower_bounds, + "constraint_bounds.lower_bounds", + np.float64, + ) + data.constraint_bounds.types = modify( + data.constraint_bounds.types, + "constraint_bounds.types", + ) + + data.initial_solution.primal = modify( + data.initial_solution.primal, + "initial_solution.primal", + np.float64, + ) + + data.initial_solution.dual = modify( + data.initial_solution.dual, "initial_solution.dual", np.float64 + ) + + data.objective_data.coefficients = modify( + data.objective_data.coefficients, + "objective_data.coefficients", + np.float64, + ) + + data.variable_bounds.upper_bounds = modify( + data.variable_bounds.upper_bounds, + "variable_bounds.upper_bounds", + np.float64, + ) + data.variable_bounds.lower_bounds = modify( + data.variable_bounds.lower_bounds, + "variable_bounds.lower_bounds", + np.float64, + ) + data.variable_types = modify( + data.variable_types, + "variable_types", + ) + + then = time.time() + if isinstance(data, LPData): + apply_LPData(data) + else: + apply(data, tmap) + logging.info(f"transform time {time.time() - then}") diff --git a/python/cuopt_server/cuopt_server/utils/utils.py b/python/cuopt_server/cuopt_server/utils/utils.py index 3c7bb012a7..8d5c837509 100644 --- a/python/cuopt_server/cuopt_server/utils/utils.py +++ b/python/cuopt_server/cuopt_server/utils/utils.py @@ -1,11 +1,13 @@ -# SPDX-FileCopyrightText: Copyright (c) 2024-2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-FileCopyrightText: Copyright (c) 2024-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 import json import os -from cuopt_server.utils.job_queue import SolverLPJob from cuopt_server.utils.linear_programming.data_definition import LPData +from cuopt_server.utils.linear_programming.data_transformation import ( + transform_lp_data, +) from cuopt_server.utils.linear_programming.solver import ( create_data_model as lp_create_data_model, create_solver as lp_create_solver, @@ -74,12 +76,8 @@ def build_lp_datamodel_from_json(data): "requires json input" ) - stub_id = 9999 - stub_warnings = [] - job = SolverLPJob(stub_id, data, None, stub_warnings) # transform data into digestible format - job._transform(job.LP_data) - data = job.get_data() + transform_lp_data(data) _, data_model = lp_create_data_model(data) _, solver_settings = lp_create_solver(data, None) From 34ac7fcc3244709ced926f1d98e3849295686478 Mon Sep 17 00:00:00 2001 From: Ramakrishna Prabhu <42624703+ramakrishnap-nv@users.noreply.github.com> Date: Tue, 8 Sep 2026 12:56:12 -0500 Subject: [PATCH 044/113] refactor: make to_optimization_problem a free function (#1802) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `to_optimization_problem()` was a pure virtual on `optimization_problem_interface_t`, so it occupied a slot in **every** implementer's vtable — including `cpu_optimization_problem_t`, whose vtable then held an entry only `libcuopt` can define. Vtable relocations resolve **eagerly at load time**, so this cannot be deferred or hidden behind lazy binding: any library carrying that vtable is unloadable without `libcuopt.so`. That blocks the CUDA-free client library (#1804). It is now a free function declared in `optimization_problem.hpp`, defined in `cpu_optimization_problem_to_gpu.cpp`, dispatching on the concrete type: ```diff - auto gpu = problem->to_optimization_problem(&handle); + auto gpu = to_optimization_problem(*problem, &handle); ``` Semantics are unchanged — the GPU override was a one-line `return nullptr`, so a GPU-backed problem still yields `nullptr`. Unrecognised implementations now throw instead of returning `nullptr`, since the documented fallback `static_cast`s the reference and would otherwise be UB. 5 call sites updated. **Breaking:** removes a pure virtual from an installed public header. Out-of-tree implementers should delete their override; callers switch to the free function as above. 2 of 4 toward a CUDA-free client library (#1801 merged, #1803, #1804 follow). 🤖 Generated with [Claude Code](https://claude.com/claude-code) --------- Signed-off-by: Ramakrishna Prabhu Co-authored-by: Claude Opus 5 --- .../cpu_optimization_problem.hpp | 18 +-- .../optimization_problem.hpp | 28 +++- .../optimization_problem_interface.hpp | 23 +-- cpp/src/grpc/server/grpc_worker.cpp | 4 +- cpp/src/mip_heuristics/solve.cu | 2 +- cpp/src/pdlp/CMakeLists.txt | 1 + cpp/src/pdlp/cpu_optimization_problem.cpp | 95 ----------- .../pdlp/cpu_optimization_problem_to_gpu.cpp | 149 ++++++++++++++++++ cpp/src/pdlp/optimization_problem.cu | 8 - cpp/src/pdlp/solve.cu | 2 +- .../unit_tests/solution_interface_test.cu | 4 +- 11 files changed, 191 insertions(+), 143 deletions(-) create mode 100644 cpp/src/pdlp/cpu_optimization_problem_to_gpu.cpp diff --git a/cpp/include/cuopt/mathematical_optimization/cpu_optimization_problem.hpp b/cpp/include/cuopt/mathematical_optimization/cpu_optimization_problem.hpp index 28aa91a82f..191eac62e8 100644 --- a/cpp/include/cuopt/mathematical_optimization/cpu_optimization_problem.hpp +++ b/cpp/include/cuopt/mathematical_optimization/cpu_optimization_problem.hpp @@ -30,6 +30,7 @@ class mps_data_model_t; // Forward declarations template class optimization_problem_t; + template class pdlp_solver_settings_t; template @@ -166,17 +167,6 @@ class cpu_optimization_problem_t : public optimization_problem_interface_t get_row_types_host() const override; std::vector get_variable_types_host() const override; - /** - * @brief Convert this CPU optimization problem to an optimization_problem_t - * by copying CPU data to GPU (requires GPU memory transfer). - * - * @param handle_ptr RAFT handle with CUDA resources for GPU memory allocation. - * @return unique_ptr to new optimization_problem_t with all data copied to GPU - * @throws std::runtime_error if handle_ptr is null - */ - std::unique_ptr> to_optimization_problem( - raft::handle_t const* handle_ptr = nullptr) override; - /** * @brief Write the optimization problem to an MPS file. * @param[in] mps_file_path Path to the output MPS file @@ -207,6 +197,12 @@ class cpu_optimization_problem_t : public optimization_problem_interface_t + friend std::unique_ptr> to_optimization_problem( + optimization_problem_interface_t&, raft::handle_t const*); + problem_category_t problem_category_ = problem_category_t::LP; bool maximize_{false}; i_t n_vars_{0}; diff --git a/cpp/include/cuopt/mathematical_optimization/optimization_problem.hpp b/cpp/include/cuopt/mathematical_optimization/optimization_problem.hpp index bdfc2ffbd4..5363cfe812 100644 --- a/cpp/include/cuopt/mathematical_optimization/optimization_problem.hpp +++ b/cpp/include/cuopt/mathematical_optimization/optimization_problem.hpp @@ -352,13 +352,6 @@ class optimization_problem_t : public optimization_problem_interface_t template optimization_problem_t convert_to_other_prec(rmm::cuda_stream_view stream) const; - /** - * @brief Returns nullptr since this is already a GPU problem. - * @return nullptr - */ - std::unique_ptr> to_optimization_problem( - raft::handle_t const* handle_ptr = nullptr) override; - // ============================================================================ // C API support: Copy to host (polymorphic) // ============================================================================ @@ -427,5 +420,26 @@ class optimization_problem_t : public optimization_problem_interface_t std::vector row_names_{}; }; +/** + * @brief Convert a problem to a GPU-backed optimization_problem_t. + * + * For optimization_problem_t (GPU): returns nullptr (already is one). + * For cpu_optimization_problem_t: creates a new GPU problem, copies data, returns it. + * + * Usage pattern: + * auto temp = to_optimization_problem(problem, &handle); + * optimization_problem_t& op = temp ? *temp : static_cast(problem); + * + * A free function rather than a virtual member so that cpu_optimization_problem_t's vtable + * carries no GPU-defined entry; see optimization_problem_interface.hpp. + * + * @param problem The problem to convert. + * @param handle_ptr RAFT handle with CUDA resources. Required for CPU->GPU conversion. + * @return unique_ptr to a new GPU problem, or nullptr if it already is one. + */ +template +std::unique_ptr> to_optimization_problem( + optimization_problem_interface_t& problem, raft::handle_t const* handle_ptr = nullptr); + } // namespace CUOPT_EXPORT mathematical_optimization } // namespace cuopt diff --git a/cpp/include/cuopt/mathematical_optimization/optimization_problem_interface.hpp b/cpp/include/cuopt/mathematical_optimization/optimization_problem_interface.hpp index 5927703f03..51796aa60d 100644 --- a/cpp/include/cuopt/mathematical_optimization/optimization_problem_interface.hpp +++ b/cpp/include/cuopt/mathematical_optimization/optimization_problem_interface.hpp @@ -478,22 +478,13 @@ class optimization_problem_interface_t { // Conversion // ============================================================================ - /** - * @brief Convert to a GPU-backed optimization_problem_t. - * - * For optimization_problem_t (GPU): returns nullptr (already is one). - * For cpu_optimization_problem_t: creates new GPU problem, copies data, returns owned pointer. - * - * Usage pattern: - * auto temp = problem_interface->to_optimization_problem(&handle); - * optimization_problem_t& op = temp ? *temp : static_cast(*this); - * - * @param handle_ptr RAFT handle with CUDA resources for GPU memory allocation. - * Required for CPU->GPU conversion. Ignored for GPU problems. - * @return unique_ptr to new GPU problem, or nullptr if already a GPU problem - */ - virtual std::unique_ptr> to_optimization_problem( - raft::handle_t const* handle_ptr = nullptr) = 0; + // NOTE: CPU -> GPU conversion is deliberately NOT a virtual member here. + // + // As a virtual, it occupied a slot in cpu_optimization_problem_t's vtable, and vtable + // relocations are resolved eagerly at load time. That made every library containing + // the vtable -- including the CUDA-free cuopt_client -- unable to load without + // libcuopt.so present. It is now the free function to_optimization_problem() declared + // in optimization_problem.hpp, which lives in libcuopt where the GPU types do. }; } // namespace cuopt::mathematical_optimization diff --git a/cpp/src/grpc/server/grpc_worker.cpp b/cpp/src/grpc/server/grpc_worker.cpp index 250b640031..aa048f34bb 100644 --- a/cpp/src/grpc/server/grpc_worker.cpp +++ b/cpp/src/grpc/server/grpc_worker.cpp @@ -425,7 +425,7 @@ static SolveResult run_mip_solve(DeserializedJob& dj, } SERVER_LOG_INFO("[Worker] Converting CPU problem to GPU problem..."); - auto gpu_problem = dj.problem.to_optimization_problem(&handle); + auto gpu_problem = to_optimization_problem(dj.problem, &handle); SERVER_LOG_INFO("[Worker] Calling solve_mip..."); auto gpu_solution = cuopt::mathematical_optimization::solve_mip(*gpu_problem, dj.mip_settings); @@ -486,7 +486,7 @@ static SolveResult run_lp_solve(DeserializedJob& dj, dj.lp_settings.log_to_console = config.log_to_console; SERVER_LOG_INFO("[Worker] Converting CPU problem to GPU problem..."); - auto gpu_problem = dj.problem.to_optimization_problem(&handle); + auto gpu_problem = to_optimization_problem(dj.problem, &handle); SERVER_LOG_INFO("[Worker] Calling solve_lp..."); auto gpu_solution = cuopt::mathematical_optimization::solve_lp(*gpu_problem, dj.lp_settings); diff --git a/cpp/src/mip_heuristics/solve.cu b/cpp/src/mip_heuristics/solve.cu index 7f7fee22fc..eb5b70b319 100644 --- a/cpp/src/mip_heuristics/solve.cu +++ b/cpp/src/mip_heuristics/solve.cu @@ -940,7 +940,7 @@ std::unique_ptr> solve_mip( raft::handle_t handle(stream); // Convert CPU problem to GPU problem - auto gpu_problem = cpu_problem.to_optimization_problem(&handle); + auto gpu_problem = to_optimization_problem(cpu_problem, &handle); // Synchronize before solving to ensure conversion is complete stream.synchronize(); diff --git a/cpp/src/pdlp/CMakeLists.txt b/cpp/src/pdlp/CMakeLists.txt index 44dced14bc..b6a1f8a46d 100644 --- a/cpp/src/pdlp/CMakeLists.txt +++ b/cpp/src/pdlp/CMakeLists.txt @@ -8,6 +8,7 @@ set(LP_CORE_FILES ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings.cu ${CMAKE_CURRENT_SOURCE_DIR}/optimization_problem.cu ${CMAKE_CURRENT_SOURCE_DIR}/cpu_optimization_problem.cpp + ${CMAKE_CURRENT_SOURCE_DIR}/cpu_optimization_problem_to_gpu.cpp ${CMAKE_CURRENT_SOURCE_DIR}/backend_selection.cpp ${CMAKE_CURRENT_SOURCE_DIR}/utilities/problem_checking.cu ${CMAKE_CURRENT_SOURCE_DIR}/solve.cu diff --git a/cpp/src/pdlp/cpu_optimization_problem.cpp b/cpp/src/pdlp/cpu_optimization_problem.cpp index 4b970eb6ec..8e310b287b 100644 --- a/cpp/src/pdlp/cpu_optimization_problem.cpp +++ b/cpp/src/pdlp/cpu_optimization_problem.cpp @@ -10,7 +10,6 @@ #include #include #include -#include #include #include @@ -634,100 +633,6 @@ std::vector cpu_optimization_problem_t::get_variable_types_host return variable_types_; } -// ============================================================================== -// Conversion to optimization_problem_t -// ============================================================================== - -template -std::unique_ptr> -cpu_optimization_problem_t::to_optimization_problem(raft::handle_t const* handle_ptr) -{ - if (handle_ptr == nullptr) { - throw std::runtime_error( - "cpu_optimization_problem_t::to_optimization_problem(): " - "handle_ptr is null. A RAFT handle with CUDA resources is required to convert " - "a CPU-backed problem to a GPU-backed optimization_problem_t."); - } - - auto gpu_problem = std::make_unique>(handle_ptr); - - // Set scalar values - gpu_problem->set_maximize(maximize_); - gpu_problem->set_objective_scaling_factor(objective_scaling_factor_); - gpu_problem->set_objective_offset(objective_offset_); - gpu_problem->set_problem_category(problem_category_); - - // Set string values - if (!objective_name_.empty()) gpu_problem->set_objective_name(objective_name_); - if (!problem_name_.empty()) gpu_problem->set_problem_name(problem_name_); - if (!var_names_.empty()) gpu_problem->set_variable_names(var_names_); - if (!row_names_.empty()) gpu_problem->set_row_names(row_names_); - - // Set CSR constraint matrix (data will be copied to GPU by optimization_problem_t setters) - // Use A_offsets_ presence as the guard: a valid CSR can have zero non-zeros but still - // needs row offsets to define the number of constraints. - if (!A_offsets_.empty()) { - gpu_problem->set_csr_constraint_matrix(A_.data(), - A_.size(), - A_indices_.data(), - A_indices_.size(), - A_offsets_.data(), - A_offsets_.size()); - } - - // Set constraint bounds - if (!b_.empty()) { gpu_problem->set_constraint_bounds(b_.data(), b_.size()); } - - // Set objective coefficients - if (!c_.empty()) { gpu_problem->set_objective_coefficients(c_.data(), c_.size()); } - - // Set quadratic objective if present (GPU setter symmetrizes once: H = Q + Q^T) - if (!Q_values_.empty()) { - gpu_problem->set_quadratic_objective_matrix(Q_values_.data(), - Q_values_.size(), - Q_indices_.data(), - Q_indices_.size(), - Q_offsets_.data(), - Q_offsets_.size()); - } - - if (!quadratic_constraints_.empty()) { - gpu_problem->set_quadratic_constraints( - std::vector::quadratic_constraint_t>( - quadratic_constraints_)); - } - - // Set variable bounds - if (!variable_lower_bounds_.empty()) { - gpu_problem->set_variable_lower_bounds(variable_lower_bounds_.data(), - variable_lower_bounds_.size()); - } - if (!variable_upper_bounds_.empty()) { - gpu_problem->set_variable_upper_bounds(variable_upper_bounds_.data(), - variable_upper_bounds_.size()); - } - - // Set variable types - if (!variable_types_.empty()) { - gpu_problem->set_variable_types(variable_types_.data(), variable_types_.size()); - } - - // Set constraint bounds - if (!constraint_lower_bounds_.empty()) { - gpu_problem->set_constraint_lower_bounds(constraint_lower_bounds_.data(), - constraint_lower_bounds_.size()); - } - if (!constraint_upper_bounds_.empty()) { - gpu_problem->set_constraint_upper_bounds(constraint_upper_bounds_.data(), - constraint_upper_bounds_.size()); - } - - // Set row types - if (!row_types_.empty()) { gpu_problem->set_row_types(row_types_.data(), row_types_.size()); } - - return gpu_problem; -} - // ============================================================================== // File I/O // ============================================================================== diff --git a/cpp/src/pdlp/cpu_optimization_problem_to_gpu.cpp b/cpp/src/pdlp/cpu_optimization_problem_to_gpu.cpp new file mode 100644 index 0000000000..af6ecd6ef7 --- /dev/null +++ b/cpp/src/pdlp/cpu_optimization_problem_to_gpu.cpp @@ -0,0 +1,149 @@ +/* clang-format off */ +/* + * SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +/* clang-format on */ + +// CPU -> GPU conversion for cpu_optimization_problem_t. +// +// Split out of cpu_optimization_problem.cpp: this is the only part of that class needing +// and a raft handle, so keeping it here leaves the rest as pure +// host code. + +#include +#include +#include +#include + +// Required: without it MIP_INSTANTIATE_* are undefined and this TU emits no symbols. +#include + +#include +#include +#include + +namespace cuopt::mathematical_optimization { + +// Dispatches on the concrete type; a GPU-backed problem yields nullptr, matching the +// previous override. Unrecognised implementations throw rather than returning nullptr: +// the documented fallback static_casts the reference to optimization_problem_t&, which +// would be undefined behaviour for any other type. +template +std::unique_ptr> to_optimization_problem( + optimization_problem_interface_t& problem, raft::handle_t const* handle_ptr) +{ + auto* cpu_problem = dynamic_cast*>(&problem); + if (cpu_problem == nullptr) { + cuopt_expects(dynamic_cast*>(&problem) != nullptr, + error_type_t::ValidationError, + "to_optimization_problem(): unsupported optimization_problem_interface_t " + "implementation. Only optimization_problem_t and cpu_optimization_problem_t " + "are supported."); + // Already a GPU-backed problem; nothing to convert. + return nullptr; + } + auto& self = *cpu_problem; + + if (handle_ptr == nullptr) { + throw std::runtime_error( + "to_optimization_problem(): " + "handle_ptr is null. A RAFT handle with CUDA resources is required to convert " + "a CPU-backed problem to a GPU-backed optimization_problem_t."); + } + + auto gpu_problem = std::make_unique>(handle_ptr); + + // Set scalar values + gpu_problem->set_maximize(self.maximize_); + gpu_problem->set_objective_scaling_factor(self.objective_scaling_factor_); + gpu_problem->set_objective_offset(self.objective_offset_); + gpu_problem->set_problem_category(self.problem_category_); + + // Set string values + if (!self.objective_name_.empty()) gpu_problem->set_objective_name(self.objective_name_); + if (!self.problem_name_.empty()) gpu_problem->set_problem_name(self.problem_name_); + if (!self.var_names_.empty()) gpu_problem->set_variable_names(self.var_names_); + if (!self.row_names_.empty()) gpu_problem->set_row_names(self.row_names_); + + // Set CSR constraint matrix (data will be copied to GPU by optimization_problem_t setters) + // Use A_offsets_ presence as the guard: a valid CSR can have zero non-zeros but still + // needs row offsets to define the number of constraints. + if (!self.A_offsets_.empty()) { + gpu_problem->set_csr_constraint_matrix(self.A_.data(), + self.A_.size(), + self.A_indices_.data(), + self.A_indices_.size(), + self.A_offsets_.data(), + self.A_offsets_.size()); + } + + // Set constraint bounds + if (!self.b_.empty()) { gpu_problem->set_constraint_bounds(self.b_.data(), self.b_.size()); } + + // Set objective coefficients + if (!self.c_.empty()) { gpu_problem->set_objective_coefficients(self.c_.data(), self.c_.size()); } + + // Set quadratic objective if present (GPU setter symmetrizes once: H = Q + Q^T) + if (!self.Q_values_.empty()) { + gpu_problem->set_quadratic_objective_matrix(self.Q_values_.data(), + self.Q_values_.size(), + self.Q_indices_.data(), + self.Q_indices_.size(), + self.Q_offsets_.data(), + self.Q_offsets_.size()); + } + + if (!self.quadratic_constraints_.empty()) { + gpu_problem->set_quadratic_constraints( + std::vector::quadratic_constraint_t>( + self.quadratic_constraints_)); + } + + // Set variable bounds + if (!self.variable_lower_bounds_.empty()) { + gpu_problem->set_variable_lower_bounds(self.variable_lower_bounds_.data(), + self.variable_lower_bounds_.size()); + } + if (!self.variable_upper_bounds_.empty()) { + gpu_problem->set_variable_upper_bounds(self.variable_upper_bounds_.data(), + self.variable_upper_bounds_.size()); + } + + // Set variable types + if (!self.variable_types_.empty()) { + gpu_problem->set_variable_types(self.variable_types_.data(), self.variable_types_.size()); + } + + // Set constraint bounds + if (!self.constraint_lower_bounds_.empty()) { + gpu_problem->set_constraint_lower_bounds(self.constraint_lower_bounds_.data(), + self.constraint_lower_bounds_.size()); + } + if (!self.constraint_upper_bounds_.empty()) { + gpu_problem->set_constraint_upper_bounds(self.constraint_upper_bounds_.data(), + self.constraint_upper_bounds_.size()); + } + + // Set row types + if (!self.row_types_.empty()) { + gpu_problem->set_row_types(self.row_types_.data(), self.row_types_.size()); + } + + return gpu_problem; +} + +// ============================================================================== +// Template instantiations matching cpu_optimization_problem.cpp +// ============================================================================== + +#if MIP_INSTANTIATE_FLOAT +template CUOPT_EXPORT std::unique_ptr> +to_optimization_problem(optimization_problem_interface_t&, raft::handle_t const*); +#endif +#if MIP_INSTANTIATE_DOUBLE +template CUOPT_EXPORT std::unique_ptr> +to_optimization_problem(optimization_problem_interface_t&, raft::handle_t const*); +#endif + +} // namespace cuopt::mathematical_optimization diff --git a/cpp/src/pdlp/optimization_problem.cu b/cpp/src/pdlp/optimization_problem.cu index 3a8bcb0b2a..87fe438ca4 100644 --- a/cpp/src/pdlp/optimization_problem.cu +++ b/cpp/src/pdlp/optimization_problem.cu @@ -639,14 +639,6 @@ raft::handle_t const* optimization_problem_t::get_handle_ptr() const n // Conversion // ============================================================================== -template -std::unique_ptr> -optimization_problem_t::to_optimization_problem(raft::handle_t const* /*handle_ptr*/) -{ - // Already a GPU problem, return nullptr - return nullptr; -} - // ============================================================================== // Host Getters (copy from GPU to CPU) // ============================================================================== diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index b7868c2a41..63b2f9681e 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -2755,7 +2755,7 @@ std::unique_ptr> solve_lp( raft::handle_t handle(stream); // Convert CPU problem to GPU problem - auto gpu_problem = cpu_problem.to_optimization_problem(&handle); + auto gpu_problem = to_optimization_problem(cpu_problem, &handle); // Synchronize before solving to ensure conversion is complete stream.synchronize(); diff --git a/cpp/tests/linear_programming/unit_tests/solution_interface_test.cu b/cpp/tests/linear_programming/unit_tests/solution_interface_test.cu index 616e0d62f1..608ebcecd5 100644 --- a/cpp/tests/linear_programming/unit_tests/solution_interface_test.cu +++ b/cpp/tests/linear_programming/unit_tests/solution_interface_test.cu @@ -307,7 +307,7 @@ TEST_F(SolutionInterfaceTest, gpu_problem_to_optimization_problem) EXPECT_EQ(problem->get_n_constraints(), kNCons); // GPU problem's to_optimization_problem() returns nullptr (already a GPU problem) - auto concrete = problem->to_optimization_problem(&handle); + auto concrete = to_optimization_problem(*problem, &handle); EXPECT_EQ(concrete, nullptr); // Verify the data is still accessible directly on the problem @@ -342,7 +342,7 @@ TEST_F(SolutionInterfaceTest, cpu_problem_to_optimization_problem) EXPECT_EQ(problem->get_n_variables(), kNVars); EXPECT_EQ(problem->get_n_constraints(), kNCons); - auto concrete = problem->to_optimization_problem(&handle); + auto concrete = to_optimization_problem(*problem, &handle); ASSERT_NE(concrete, nullptr); EXPECT_EQ(concrete->get_n_variables(), kNVars); EXPECT_EQ(concrete->get_n_constraints(), kNCons); From 7e50453e322dabd7be81be39fdc474d6cab81967 Mon Sep 17 00:00:00 2001 From: Trevor McKay Date: Tue, 8 Sep 2026 14:02:00 -0400 Subject: [PATCH 045/113] Extract LP create_data_model/create_solver into conversion.py (#1849) This change moves LP data model creation routines into a new module out of linear programming solver.py so that they can be used by a new proxy server that will delegate solves to gRPC. Authors: - Trevor McKay (https://github.com/tmckayus) Approvers: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) URL: https://github.com/NVIDIA/cuopt/pull/1849 --- .../cuopt_server/tests/test_lp_conversion.py | 88 +++++++++++ .../utils/linear_programming/conversion.py | 139 ++++++++++++++++ .../utils/linear_programming/solver.py | 149 ++---------------- .../cuopt_server/cuopt_server/utils/utils.py | 9 +- 4 files changed, 244 insertions(+), 141 deletions(-) create mode 100644 python/cuopt_server/cuopt_server/tests/test_lp_conversion.py create mode 100644 python/cuopt_server/cuopt_server/utils/linear_programming/conversion.py diff --git a/python/cuopt_server/cuopt_server/tests/test_lp_conversion.py b/python/cuopt_server/cuopt_server/tests/test_lp_conversion.py new file mode 100644 index 0000000000..ee2f89699b --- /dev/null +++ b/python/cuopt_server/cuopt_server/tests/test_lp_conversion.py @@ -0,0 +1,88 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from cuopt_server.utils.linear_programming import conversion +from cuopt_server.utils.linear_programming.data_definition import LPData +from cuopt_server.utils.utils import build_lp_datamodel_from_json + + +def get_lp_json(): + return { + "csr_constraint_matrix": { + "offsets": [0, 2], + "indices": [0, 1], + "values": [1.0, 1.0], + }, + "constraint_bounds": {"upper_bounds": [5000.0], "lower_bounds": [0.0]}, + "objective_data": { + "coefficients": [1.2, 1.7], + "scalability_factor": 1.0, + "offset": 0.5, + }, + "variable_bounds": { + "upper_bounds": [3000.0, 5000.0], + "lower_bounds": [0.0, 0.0], + }, + "maximize": True, + "variable_names": ["x", "y"], + "solver_config": {"time_limit": 5, "iteration_limit": 100}, + } + + +def get_lp_data(): + return LPData.parse_obj(get_lp_json()) + + +def test_create_data_model(): + warnings, data_model = conversion.create_data_model(get_lp_data()) + + assert warnings == [] + assert data_model.get_constraint_matrix_values().tolist() == [1.0, 1.0] + assert data_model.get_constraint_matrix_indices().tolist() == [0, 1] + assert data_model.get_constraint_matrix_offsets().tolist() == [0, 2] + assert data_model.get_constraint_lower_bounds().tolist() == [0.0] + assert data_model.get_constraint_upper_bounds().tolist() == [5000.0] + assert data_model.get_objective_coefficients().tolist() == [1.2, 1.7] + assert data_model.get_objective_scaling_factor() == 1.0 + assert data_model.get_objective_offset() == 0.5 + assert data_model.get_variable_lower_bounds().tolist() == [0.0, 0.0] + assert data_model.get_variable_upper_bounds().tolist() == [3000.0, 5000.0] + assert data_model.get_variable_names() == ["x", "y"] + + +def test_create_solver_limits(): + warnings, solver_settings = conversion.create_solver(get_lp_data(), None) + + assert warnings == [] + assert float(solver_settings.get_parameter("time_limit")) == 5.0 + assert int(solver_settings.get_parameter("iteration_limit")) == 100 + + +def test_create_solver_limits_clamped_by_environment(monkeypatch): + monkeypatch.setenv("CUOPT_LP_TIME_LIMIT_SEC", "2") + monkeypatch.setenv("CUOPT_LP_ITERATION_LIMIT", "10") + + _, solver_settings = conversion.create_solver(get_lp_data(), None) + + assert float(solver_settings.get_parameter("time_limit")) == 2.0 + assert int(solver_settings.get_parameter("iteration_limit")) == 10 + + +def test_create_solver_warns_on_ignored_fields(): + data = get_lp_json() + data["solver_config"]["user_problem_file"] = "problem.mps" + data["solver_config"]["solution_file"] = "solution.txt" + + warnings, _ = conversion.create_solver(LPData.parse_obj(data), None) + + assert warnings == [ + conversion.ignored_warning("user_problem_file"), + conversion.ignored_warning("solution_file"), + ] + + +def test_build_lp_datamodel_from_json(): + data_model, solver_settings = build_lp_datamodel_from_json(get_lp_json()) + + assert data_model.get_objective_coefficients().tolist() == [1.2, 1.7] + assert float(solver_settings.get_parameter("time_limit")) == 5.0 diff --git a/python/cuopt_server/cuopt_server/utils/linear_programming/conversion.py b/python/cuopt_server/cuopt_server/utils/linear_programming/conversion.py new file mode 100644 index 0000000000..752579a3f1 --- /dev/null +++ b/python/cuopt_server/cuopt_server/utils/linear_programming/conversion.py @@ -0,0 +1,139 @@ +# SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import logging +import os + +from cuopt import linear_programming +from cuopt.linear_programming.solver.solver_parameters import solver_params + + +def ignored_warning(field): + return f"solver config {field} ignored in the cuopt service" + + +def create_data_model(LP_data): + warnings = [] + + # Create data model object + data_model = linear_programming.DataModel() + + csr_constraint_matrix = LP_data.csr_constraint_matrix + data_model.set_csr_constraint_matrix( + csr_constraint_matrix.values, + csr_constraint_matrix.indices, + csr_constraint_matrix.offsets, + ) + + constraint_bounds = LP_data.constraint_bounds + if constraint_bounds.bounds is not None: + data_model.set_constraint_bounds(constraint_bounds.bounds) + if constraint_bounds.types is not None: + if len(constraint_bounds.types): + data_model.set_row_types(constraint_bounds.types) + if constraint_bounds.upper_bounds is not None: + if len(constraint_bounds.upper_bounds): + data_model.set_constraint_upper_bounds( + constraint_bounds.upper_bounds + ) + if constraint_bounds.lower_bounds is not None: + if len(constraint_bounds.lower_bounds): + data_model.set_constraint_lower_bounds( + constraint_bounds.lower_bounds + ) + + objective_data = LP_data.objective_data + if objective_data.coefficients is not None: + data_model.set_objective_coefficients(objective_data.coefficients) + if objective_data.scalability_factor is not None: + data_model.set_objective_scaling_factor( + objective_data.scalability_factor + ) + if objective_data.offset is not None: + data_model.set_objective_offset(objective_data.offset) + + variable_bounds = LP_data.variable_bounds + if variable_bounds.upper_bounds is not None: + data_model.set_variable_upper_bounds(variable_bounds.upper_bounds) + if variable_bounds.lower_bounds is not None: + data_model.set_variable_lower_bounds(variable_bounds.lower_bounds) + + initial_sol = LP_data.initial_solution + if initial_sol is not None: + if initial_sol.primal is not None: + data_model.set_initial_primal_solution(initial_sol.primal) + if initial_sol.dual is not None: + data_model.set_initial_dual_solution(initial_sol.dual) + + if LP_data.maximize is not None: + data_model.set_maximize(LP_data.maximize) + + if LP_data.variable_types is not None: + data_model.set_variable_types(LP_data.variable_types) + + if LP_data.variable_names is not None: + data_model.set_variable_names(LP_data.variable_names) + + return warnings, data_model + + +def create_solver(LP_data, warmstart_data): + warnings = [] + solver_settings = linear_programming.SolverSettings() + + if LP_data.solver_config is not None: + solver_config = LP_data.solver_config + for param in solver_params: + param_value = None + if param.endswith("tolerance"): + param_value = getattr(solver_config.tolerances, param, None) + else: + param_value = getattr(solver_config, param, None) + if param_value is not None and param_value != "": + solver_settings.set_parameter(param, param_value) + + if LP_data.solver_config is not None: + solver_config = LP_data.solver_config + + try: + lp_time_limit = float(os.environ.get("CUOPT_LP_TIME_LIMIT_SEC")) + except Exception: + lp_time_limit = None + if solver_config.time_limit is None: + time_limit = lp_time_limit + elif lp_time_limit: + time_limit = min(solver_config.time_limit, lp_time_limit) + else: + time_limit = solver_config.time_limit + if time_limit is not None: + logging.debug(f"setting LP time limit to {time_limit}sec") + solver_settings.set_parameter("time_limit", time_limit) + + try: + lp_iteration_limit = int( + os.environ.get("CUOPT_LP_ITERATION_LIMIT") + ) + except Exception: + lp_iteration_limit = None + if solver_config.iteration_limit is None: + iteration_limit = lp_iteration_limit + elif lp_iteration_limit: + iteration_limit = min( + solver_config.iteration_limit, lp_iteration_limit + ) + else: + iteration_limit = solver_config.iteration_limit + if iteration_limit is not None: + logging.debug(f"setting LP iteration limit to {iteration_limit}") + solver_settings.set_parameter("iteration_limit", iteration_limit) + + if warmstart_data is not None: + solver_settings.set_pdlp_warm_start_data(warmstart_data) + + if solver_config.user_problem_file != "": + warnings.append(ignored_warning("user_problem_file")) + + if solver_config.solution_file != "": + warnings.append(ignored_warning("solution_file")) + + return warnings, solver_settings diff --git a/python/cuopt_server/cuopt_server/utils/linear_programming/solver.py b/python/cuopt_server/cuopt_server/utils/linear_programming/solver.py index e05e9fed5d..07a134a2a9 100644 --- a/python/cuopt_server/cuopt_server/utils/linear_programming/solver.py +++ b/python/cuopt_server/cuopt_server/utils/linear_programming/solver.py @@ -1,8 +1,6 @@ # SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -import logging -import os import time from fastapi import HTTPException @@ -12,7 +10,6 @@ GetSolutionCallback, SetSolutionCallback, ) -from cuopt.linear_programming.solver.solver_parameters import solver_params from cuopt.linear_programming.solver.solver_wrapper import ( ErrorStatus, LPTerminationStatus, @@ -24,6 +21,15 @@ OutOfMemoryError, ) +# Conversion of request data into cuopt data models and solver settings lives +# in conversion.py. The names below are re-exported so existing importers of +# this module keep working. +from cuopt_server.utils.linear_programming.conversion import ( # noqa: F401 + create_data_model, + create_solver, + ignored_warning, +) + def dep_warning(field): return ( @@ -32,8 +38,9 @@ def dep_warning(field): ) -def ignored_warning(field): - return f"solver config {field} ignored in the cuopt service" +def warn_on_objectives(solver_config): + warnings = [] + return warnings, solver_config class CustomGetSolutionCallback(GetSolutionCallback): @@ -80,138 +87,6 @@ def set_solution(self, solution, solution_cost, solution_bound, user_data): solution_cost[0] = float(self.get_callback.solutions[-1]["cost"]) -def warn_on_objectives(solver_config): - warnings = [] - return warnings, solver_config - - -def create_data_model(LP_data): - warnings = [] - - # Create data model object - data_model = linear_programming.DataModel() - - csr_constraint_matrix = LP_data.csr_constraint_matrix - data_model.set_csr_constraint_matrix( - csr_constraint_matrix.values, - csr_constraint_matrix.indices, - csr_constraint_matrix.offsets, - ) - - constraint_bounds = LP_data.constraint_bounds - if constraint_bounds.bounds is not None: - data_model.set_constraint_bounds(constraint_bounds.bounds) - if constraint_bounds.types is not None: - if len(constraint_bounds.types): - data_model.set_row_types(constraint_bounds.types) - if constraint_bounds.upper_bounds is not None: - if len(constraint_bounds.upper_bounds): - data_model.set_constraint_upper_bounds( - constraint_bounds.upper_bounds - ) - if constraint_bounds.lower_bounds is not None: - if len(constraint_bounds.lower_bounds): - data_model.set_constraint_lower_bounds( - constraint_bounds.lower_bounds - ) - - objective_data = LP_data.objective_data - if objective_data.coefficients is not None: - data_model.set_objective_coefficients(objective_data.coefficients) - if objective_data.scalability_factor is not None: - data_model.set_objective_scaling_factor( - objective_data.scalability_factor - ) - if objective_data.offset is not None: - data_model.set_objective_offset(objective_data.offset) - - variable_bounds = LP_data.variable_bounds - if variable_bounds.upper_bounds is not None: - data_model.set_variable_upper_bounds(variable_bounds.upper_bounds) - if variable_bounds.lower_bounds is not None: - data_model.set_variable_lower_bounds(variable_bounds.lower_bounds) - - initial_sol = LP_data.initial_solution - if initial_sol is not None: - if initial_sol.primal is not None: - data_model.set_initial_primal_solution(initial_sol.primal) - if initial_sol.dual is not None: - data_model.set_initial_dual_solution(initial_sol.dual) - - if LP_data.maximize is not None: - data_model.set_maximize(LP_data.maximize) - - if LP_data.variable_types is not None: - data_model.set_variable_types(LP_data.variable_types) - - if LP_data.variable_names is not None: - data_model.set_variable_names(LP_data.variable_names) - - return warnings, data_model - - -def create_solver(LP_data, warmstart_data): - warnings = [] - solver_settings = linear_programming.SolverSettings() - - if LP_data.solver_config is not None: - solver_config = LP_data.solver_config - for param in solver_params: - param_value = None - if param.endswith("tolerance"): - param_value = getattr(solver_config.tolerances, param, None) - else: - param_value = getattr(solver_config, param, None) - if param_value is not None and param_value != "": - solver_settings.set_parameter(param, param_value) - - if LP_data.solver_config is not None: - solver_config = LP_data.solver_config - - try: - lp_time_limit = float(os.environ.get("CUOPT_LP_TIME_LIMIT_SEC")) - except Exception: - lp_time_limit = None - if solver_config.time_limit is None: - time_limit = lp_time_limit - elif lp_time_limit: - time_limit = min(solver_config.time_limit, lp_time_limit) - else: - time_limit = solver_config.time_limit - if time_limit is not None: - logging.debug(f"setting LP time limit to {time_limit}sec") - solver_settings.set_parameter("time_limit", time_limit) - - try: - lp_iteration_limit = int( - os.environ.get("CUOPT_LP_ITERATION_LIMIT") - ) - except Exception: - lp_iteration_limit = None - if solver_config.iteration_limit is None: - iteration_limit = lp_iteration_limit - elif lp_iteration_limit: - iteration_limit = min( - solver_config.iteration_limit, lp_iteration_limit - ) - else: - iteration_limit = solver_config.iteration_limit - if iteration_limit is not None: - logging.debug(f"setting LP iteration limit to {iteration_limit}") - solver_settings.set_parameter("iteration_limit", iteration_limit) - - if warmstart_data is not None: - solver_settings.set_pdlp_warm_start_data(warmstart_data) - - if solver_config.user_problem_file != "": - warnings.append(ignored_warning("user_problem_file")) - - if solver_config.solution_file != "": - warnings.append(ignored_warning("solution_file")) - - return warnings, solver_settings - - def get_solver_exception_type(status, message): msg = f"error_status: {status}, msg: {message}" diff --git a/python/cuopt_server/cuopt_server/utils/utils.py b/python/cuopt_server/cuopt_server/utils/utils.py index 8d5c837509..eeadab6dfe 100644 --- a/python/cuopt_server/cuopt_server/utils/utils.py +++ b/python/cuopt_server/cuopt_server/utils/utils.py @@ -4,14 +4,15 @@ import json import os +from cuopt_server.utils.linear_programming.conversion import ( + create_data_model as lp_create_data_model, + create_solver as lp_create_solver, +) from cuopt_server.utils.linear_programming.data_definition import LPData from cuopt_server.utils.linear_programming.data_transformation import ( transform_lp_data, ) -from cuopt_server.utils.linear_programming.solver import ( - create_data_model as lp_create_data_model, - create_solver as lp_create_solver, -) + from cuopt_server.utils.routing.data_definition import OptimizedRoutingData from cuopt_server.utils.routing.solver import ( create_data_model as routing_create_data_model, From 447aa22f1f2749912d3288eef74072a6d5e742f5 Mon Sep 17 00:00:00 2001 From: Ramakrishna Prabhu <42624703+ramakrishnap-nv@users.noreply.github.com> Date: Tue, 8 Sep 2026 14:14:47 -0500 Subject: [PATCH 046/113] Bump cuopt-java compile target from Java 11 to Java 17 (#1865) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary - `java/cuopt/pom.xml` compiled against `maven.compiler.release=11`, but the JNI bindings were hand-written specifically to support Java 17 (see the rationale comment already in `cuopt_jni.cpp`, which also notes moving to the FFM API and Java 22 is tracked in #1794). This aligns the actual compile target with that documented intent. - Updates the CI conda `openjdk` pin (`dependencies.yaml`, `java` file-key) from 11 to 17, and the `JAVA_HOME`/error-message references in `java_home.sh`, `README.md`, `TESTS.md`, and `docs/cuopt/source/cuopt-java/quick-start.rst`. - cuopt-java is still experimental/unpublished (no Maven Central release yet), so this is a build-target change only — no external consumers are affected today. Note: this raises the floor — a JVM cannot run bytecode compiled at a higher `--release` than itself, so this drops support for Java 11 runtimes in favor of Java 17+. No downstream consumer requiring <17 was found referenced anywhere in the repo. ## Test plan - [x] `mvn -DskipTests compile` succeeds with `JAVA_HOME` pointed at a JDK 17 install - [x] Verified compiled class file major version is 61 (Java 17) via `javap -verbose` - [ ] CI: `java-build` / `conda-java-tests` workflows pass with the updated `openjdk=17.*` pin Authors: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) Approvers: - Trevor McKay (https://github.com/tmckayus) URL: https://github.com/NVIDIA/cuopt/pull/1865 --- dependencies.yaml | 2 +- docs/cuopt/source/cuopt-java/quick-start.rst | 8 ++++---- java/cuopt/README.md | 2 +- java/cuopt/TESTS.md | 2 +- java/cuopt/pom.xml | 2 +- java/cuopt/scripts/java_home.sh | 21 +++++++++++++++++++- 6 files changed, 28 insertions(+), 9 deletions(-) diff --git a/dependencies.yaml b/dependencies.yaml index db63d8e896..e2e0548b0d 100644 --- a/dependencies.yaml +++ b/dependencies.yaml @@ -295,7 +295,7 @@ dependencies: - output_types: conda packages: - maven - - openjdk=11.* + - openjdk=17.* test_cpp: common: - output_types: [conda] diff --git a/docs/cuopt/source/cuopt-java/quick-start.rst b/docs/cuopt/source/cuopt-java/quick-start.rst index b3f125e21c..7563540809 100644 --- a/docs/cuopt/source/cuopt-java/quick-start.rst +++ b/docs/cuopt/source/cuopt-java/quick-start.rst @@ -11,7 +11,7 @@ Requirements The Java module requires: -* Java 11 or newer, with ``JAVA_HOME`` pointing to a JDK; +* Java 17 or newer, with ``JAVA_HOME`` pointing to a JDK; * a C++20 compiler; * an existing cuOpt installation containing ``libcuopt.so``; and * a CUDA-enabled runtime for solving problems. @@ -24,7 +24,7 @@ the JNI library. The standalone native build links to .. code-block:: bash cd /path/to/cuopt/java/cuopt - export JAVA_HOME=/path/to/jdk-11 + export JAVA_HOME=/path/to/jdk-17 export CUOPT_PREFIX=/path/to/cuopt/conda/environment bash scripts/build_native.sh @@ -45,7 +45,7 @@ at the directory containing the built native library: .. code-block:: bash cd java/cuopt - export JAVA_HOME=/usr/lib/jvm/java-11-openjdk-amd64 + export JAVA_HOME=/usr/lib/jvm/java-17-openjdk-amd64 export CUOPT_PREFIX=/path/to/cuopt/conda/environment export LD_LIBRARY_PATH=$CUOPT_PREFIX/targets/x86_64-linux/lib:$CUOPT_PREFIX/lib:build/native mvn test -Dcuopt.native.dir=build/native @@ -55,7 +55,7 @@ The helper script combines the native build and Maven test steps: .. code-block:: bash cd /path/to/cuopt/java/cuopt - export JAVA_HOME=/path/to/jdk-11 + export JAVA_HOME=/path/to/jdk-17 export CUOPT_PREFIX=/path/to/cuopt/conda/environment bash scripts/test.sh diff --git a/java/cuopt/README.md b/java/cuopt/README.md index 32dd0f5502..d105ffde30 100644 --- a/java/cuopt/README.md +++ b/java/cuopt/README.md @@ -30,7 +30,7 @@ CUOPT_PREFIX=/path/to/cuopt/conda/environment bash scripts/test.sh ``` `build_native.sh` builds `libcuopt_jni.so` in `build/native`. `test.sh` builds -that library and runs the Maven tests. Java 11 or newer and a C++20 compiler +that library and runs the Maven tests. Java 17 or newer and a C++20 compiler are required. Native solve tests require a CUDA driver and skip automatically when one is unavailable. diff --git a/java/cuopt/TESTS.md b/java/cuopt/TESTS.md index f83e7266dd..57e9dd1ed6 100644 --- a/java/cuopt/TESTS.md +++ b/java/cuopt/TESTS.md @@ -20,7 +20,7 @@ Build the JNI library and run all Java tests with: ```bash cd /path/to/cuopt/java/cuopt -export JAVA_HOME=/path/to/jdk-11 +export JAVA_HOME=/path/to/jdk-17 export CUOPT_PREFIX=/path/to/cuopt/conda/environment bash scripts/test.sh ``` diff --git a/java/cuopt/pom.xml b/java/cuopt/pom.xml index d214cf603c..33bf0225a2 100644 --- a/java/cuopt/pom.xml +++ b/java/cuopt/pom.xml @@ -40,7 +40,7 @@ SPDX-License-Identifier: Apache-2.0 - 11 + 17 UTF-8 5.11.4 diff --git a/java/cuopt/scripts/java_home.sh b/java/cuopt/scripts/java_home.sh index 85f5940ee5..27e788746e 100644 --- a/java/cuopt/scripts/java_home.sh +++ b/java/cuopt/scripts/java_home.sh @@ -4,6 +4,7 @@ cuopt_java_setup_home() { local required_binary="${1:-javac}" + local required_major_version=17 if [[ -z "${JAVA_HOME:-}" ]]; then local javac_path @@ -15,7 +16,25 @@ cuopt_java_setup_home() { fi if [[ ! -x "${JAVA_HOME:-}/bin/${required_binary}" ]]; then - echo "JAVA_HOME must point to a JDK containing bin/${required_binary} (Java 11 is required)." >&2 + echo "JAVA_HOME must point to a JDK containing bin/${required_binary} (Java ${required_major_version} or newer is required)." >&2 + exit 1 + fi + + # java -version prints "17.0.20" or the older "1.8.0_292" scheme; normalize + # both to a bare major version before comparing. + local version_line version_string major_version + version_line="$("${JAVA_HOME}/bin/java" -version 2>&1 | head -n1)" + version_string="$(printf '%s\n' "${version_line}" | sed -n 's/.*version "\([^"]*\)".*/\1/p')" + if [[ "${version_string}" == 1.* ]]; then + major_version="${version_string#1.}" + major_version="${major_version%%.*}" + else + major_version="${version_string%%.*}" + major_version="${major_version%%-*}" + fi + + if [[ ! "${major_version}" =~ ^[0-9]+$ ]] || (( major_version < required_major_version )); then + echo "JAVA_HOME (${JAVA_HOME}) reports '${version_line}', but Java ${required_major_version} or newer is required." >&2 exit 1 fi } From 49f6d48b8ccce5ec97889935c49a15a7c94a01f9 Mon Sep 17 00:00:00 2001 From: Trevor McKay Date: Tue, 8 Sep 2026 15:55:44 -0400 Subject: [PATCH 047/113] extract routing server data creation routines into standalone module (#1850) Move routing data creation routines out of routing solver.py so that they can be used in a proxy server that delegates solves to the gRPC server. Authors: - Trevor McKay (https://github.com/tmckayus) Approvers: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) URL: https://github.com/NVIDIA/cuopt/pull/1850 --- .../tests/test_routing_conversion.py | 42 ++ .../cuopt_server/utils/routing/conversion.py | 501 ++++++++++++++++++ .../cuopt_server/utils/routing/solver.py | 352 +----------- .../cuopt_server/cuopt_server/utils/solver.py | 150 +----- .../cuopt_server/cuopt_server/utils/utils.py | 13 +- 5 files changed, 561 insertions(+), 497 deletions(-) create mode 100644 python/cuopt_server/cuopt_server/tests/test_routing_conversion.py create mode 100644 python/cuopt_server/cuopt_server/utils/routing/conversion.py diff --git a/python/cuopt_server/cuopt_server/tests/test_routing_conversion.py b/python/cuopt_server/cuopt_server/tests/test_routing_conversion.py new file mode 100644 index 0000000000..5e3369f3ad --- /dev/null +++ b/python/cuopt_server/cuopt_server/tests/test_routing_conversion.py @@ -0,0 +1,42 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from cuopt_server.utils.routing import conversion +from cuopt_server.utils.routing.data_definition import ( + CostMatrices, + FleetData, + SolverSettingsConfig, + TaskData, +) + + +def test_pydantic_request_converts_and_prepares_cost_matrix(): + optimization_data = conversion.populate_optimization_data( + cost_matrix_data=CostMatrices(data={0: [[0, 1], [1, 0]]}), + fleet_data=FleetData(vehicle_locations=[[0, 0]]), + task_data=TaskData(task_locations=[1]), + solver_config=SolverSettingsConfig(time_limit=1), + ) + + prepared, cost_matrix, travel_time_matrix, waypoint_graph = ( + conversion.prep_optimization_data(optimization_data) + ) + + assert prepared is optimization_data + assert list(cost_matrix) == [0] + assert cost_matrix[0].shape == (2, 2) + assert travel_time_matrix is None + assert waypoint_graph == {} + + +def test_default_solver_time_limit(): + solver_config = SolverSettingsConfig() + optimization_data = conversion.populate_optimization_data( + cost_matrix_data=CostMatrices(data={0: [[0, 1], [1, 0]]}), + fleet_data=FleetData(vehicle_locations=[[0, 0]]), + task_data=TaskData(task_locations=[1]), + solver_config=solver_config, + ) + + assert solver_config.time_limit == 10 + 1 / 6 + assert optimization_data.solver_config["time_limit"] == 10 + 1 / 6 diff --git a/python/cuopt_server/cuopt_server/utils/routing/conversion.py b/python/cuopt_server/cuopt_server/utils/routing/conversion.py new file mode 100644 index 0000000000..dd027791d5 --- /dev/null +++ b/python/cuopt_server/cuopt_server/utils/routing/conversion.py @@ -0,0 +1,501 @@ +# SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import logging +from typing import List, Optional + +import numpy as np +from fastapi import HTTPException + +import cudf +from cuopt import distance_engine, routing + +from cuopt_server.utils.data_definition import ( + CostMatrices, + FleetData, + InitialSolution, + SolverSettingsConfig, + TaskData, + WaypointGraphData, +) +from cuopt_server.utils.routing.initial_solution import parse_initial_sol +from cuopt_server.utils.routing.optimization_data_model import ( + OptimizationDataModel, +) + + +# Return exception if validation fails +def check_valid(is_valid): + if not is_valid[0]: + raise HTTPException(status_code=400, detail=f"{is_valid[1]}") + + +def warn_on_objectives(solver_config): + warnings = [] + return warnings, solver_config + + +# Standard solve time for VRP +def std_solver_time_calc(num_tasks): + return 10 + num_tasks / 6 + + +def populate_optimization_data( + cost_waypoint_graph_data: Optional[WaypointGraphData] = None, + travel_time_waypoint_graph_data: Optional[WaypointGraphData] = None, + cost_matrix_data: Optional[CostMatrices] = None, + travel_time_matrix_data: Optional[CostMatrices] = None, + fleet_data: Optional[FleetData] = None, + task_data: Optional[TaskData] = None, + # Use the update data structure for the sync endpoint because + # it makes the time_limit value Optional + initial_solution: Optional[List[InitialSolution]] = None, + solver_config: Optional[SolverSettingsConfig] = None, + warnings=[], +): + optimization_data = OptimizationDataModel() + + if ( + not cost_waypoint_graph_data + or not cost_waypoint_graph_data.waypoint_graph + ) and (not cost_matrix_data or not cost_matrix_data.data): + raise HTTPException( + status_code=400, + detail="cost_matrix/waypoint_graph needs to be provided to find any route", # noqa + ) + + if ( + cost_waypoint_graph_data and cost_waypoint_graph_data.waypoint_graph + ) and (cost_matrix_data and cost_matrix_data.data): + raise HTTPException( + status_code=400, + detail="only one of cost_matrix or waypoint_graph needs to be provided, not both", # noqa + ) + + if (travel_time_matrix_data and travel_time_matrix_data.data) and ( + travel_time_waypoint_graph_data + and travel_time_waypoint_graph_data.waypoint_graph + ): + raise HTTPException( + status_code=400, + detail="only one of travel_time_matrix_data or travel_time_waypoint_graph_data needs to be provided, not both", # noqa + ) + + if cost_waypoint_graph_data and cost_waypoint_graph_data.waypoint_graph: + check_valid( + optimization_data.set_cost_waypoint_graph( + cost_waypoint_graph_data.waypoint_graph + ) + ) + elif cost_matrix_data and cost_matrix_data.data: + check_valid(optimization_data.set_cost_matrix(cost_matrix_data.data)) + + if ( + travel_time_waypoint_graph_data + and travel_time_waypoint_graph_data.waypoint_graph + ): + check_valid( + optimization_data.set_travel_time_waypoint_graph( + travel_time_waypoint_graph_data.waypoint_graph + ) + ) + elif travel_time_matrix_data and travel_time_matrix_data.data: + check_valid( + optimization_data.set_travel_time_matrix( + travel_time_matrix_data.data + ) + ) + + if fleet_data is not None: + check_valid( + optimization_data.set_fleet_data( + fleet_data.vehicle_ids, + fleet_data.vehicle_locations, + fleet_data.capacities, + fleet_data.vehicle_time_windows, + fleet_data.vehicle_breaks, + fleet_data.vehicle_break_time_windows, + fleet_data.vehicle_break_durations, + fleet_data.vehicle_break_locations, + fleet_data.vehicle_types, + fleet_data.vehicle_order_match, + fleet_data.skip_first_trips, + fleet_data.drop_return_trips, + fleet_data.min_vehicles, + fleet_data.vehicle_max_costs, + fleet_data.vehicle_max_times, + fleet_data.vehicle_fixed_costs, + ) + ) + + if task_data is not None: + check_valid( + optimization_data.set_task_data( + task_data.task_ids, + task_data.task_locations, + task_data.demand, + task_data.pickup_and_delivery_pairs, + task_data.task_time_windows, + task_data.service_times, + task_data.prizes, + task_data.order_vehicle_match, + ) + ) + + if initial_solution is not None: + check_valid(optimization_data.set_initial_solution(initial_solution)) + + if solver_config is not None: + if solver_config.time_limit is None: + num_tasks = len(task_data.task_locations) + solver_config.time_limit = std_solver_time_calc(num_tasks) + logging.debug( + "Solver time limit not specified, " + f"setting to {solver_config.time_limit}" + ) + else: + logging.debug( + f"Using specified solver time {solver_config.time_limit}" + ) + owarn, solver_config = warn_on_objectives(solver_config) + warnings.extend(owarn) + check_valid( + optimization_data.set_solver_config( + solver_config.time_limit, + solver_config.objectives, + solver_config.config_file, + solver_config.verbose_mode, + solver_config.error_logging, + ) + ) + + return optimization_data + + +def create_data_model( + optimization_data: OptimizationDataModel, + cost_matrix: Optional[dict] = None, + travel_time_matrix: Optional[dict] = None, +): + warnings = [] + # Make sure that we are using pool memory allocator + import rmm + + assert isinstance( + rmm.mr.get_current_device_resource(), rmm.mr.StatisticsResourceAdaptor + ) or isinstance( + rmm.mr.get_current_device_resource(), rmm.mr.PoolMemoryResource + ) + + n_fleet = len(optimization_data.fleet_data["vehicle_locations"]) + + n_locations = list(cost_matrix.values())[0].shape[0] + + locations = cudf.Series( + list(range(len(optimization_data.locations))), + index=optimization_data.locations, + ) + + n_orders = len(optimization_data.task_data["task_locations"]) + + # Create data model object + data_model = routing.DataModel(n_locations, n_fleet, n_orders) + + for key, value in cost_matrix.items(): + data_model.add_cost_matrix(value, key) + if travel_time_matrix is not None: + for key, value in travel_time_matrix.items(): + data_model.add_transit_time_matrix(value, key) + + if optimization_data.fleet_data["vehicle_locations"] is not None: + if len(optimization_data.locations) > 0: + start_location_id = locations.loc[ + optimization_data.fleet_data["vehicle_locations"][ + "start_location" + ] + ] + end_location_id = locations.loc[ + optimization_data.fleet_data["vehicle_locations"][ + "end_location" + ] + ] + data_model.set_vehicle_locations( + start_location_id, end_location_id + ) + else: + data_model.set_vehicle_locations( + optimization_data.fleet_data["vehicle_locations"][ + "start_location" + ], + optimization_data.fleet_data["vehicle_locations"][ + "end_location" + ], + ) + + if optimization_data.fleet_data["vehicle_time_windows"] is not None: + v_time_windows = optimization_data.fleet_data["vehicle_time_windows"] + data_model.set_vehicle_time_windows( + v_time_windows["earliest"], v_time_windows["latest"] + ) + + if optimization_data.fleet_data["skip_first_trips"] is not None: + data_model.set_skip_first_trips( + optimization_data.fleet_data["skip_first_trips"] + ) + + if ( + optimization_data.fleet_data["vehicle_break_time_windows"] is not None + and optimization_data.fleet_data["vehicle_break_durations"] is not None + ): + for index in range( + len(optimization_data.fleet_data["vehicle_break_time_windows"]) + ): + v_break_time_windows = optimization_data.fleet_data[ + "vehicle_break_time_windows" + ][index] + v_break_durations = optimization_data.fleet_data[ + "vehicle_break_durations" + ][index] + data_model.add_break_dimension( + v_break_time_windows["earliest"], + v_break_time_windows["latest"], + v_break_durations, + ) + + if optimization_data.fleet_data["vehicle_break_locations"] is not None: + if len(optimization_data.locations) > 0: + break_location_id = locations.loc[ + optimization_data.fleet_data["vehicle_break_locations"] + ] + data_model.set_break_locations(break_location_id) + else: + data_model.set_break_locations( + optimization_data.fleet_data["vehicle_break_locations"] + ) + + if optimization_data.fleet_data["vehicle_types"] is not None: + data_model.set_vehicle_types( + optimization_data.fleet_data["vehicle_types"] + ) + + if optimization_data.fleet_data["vehicle_breaks"] is not None: + for data in optimization_data.fleet_data["vehicle_breaks"]: + data_model.add_vehicle_break( + data["vehicle_id"], + data["earliest"], + data["latest"], + data["duration"], + cudf.Series(data["locations"]), + ) + + if optimization_data.fleet_data["vehicle_order_match"] is not None: + for data in optimization_data.fleet_data["vehicle_order_match"]: + data_model.add_vehicle_order_match( + data["vehicle_id"], cudf.Series(data["order_ids"]) + ) + + if optimization_data.fleet_data["drop_return_trips"] is not None: + data_model.set_drop_return_trips( + optimization_data.fleet_data["drop_return_trips"] + ) + + if optimization_data.fleet_data["vehicle_max_costs"] is not None: + data_model.set_vehicle_max_costs( + optimization_data.fleet_data["vehicle_max_costs"] + ) + + if optimization_data.fleet_data["vehicle_max_times"] is not None: + data_model.set_vehicle_max_times( + optimization_data.fleet_data["vehicle_max_times"] + ) + + if optimization_data.fleet_data["vehicle_fixed_costs"] is not None: + data_model.set_vehicle_fixed_costs( + optimization_data.fleet_data["vehicle_fixed_costs"] + ) + + if optimization_data.fleet_data["min_vehicles"] is not None: + data_model.set_min_vehicles( + optimization_data.fleet_data["min_vehicles"] + ) + + if optimization_data.task_data["task_locations"] is not None: + if len(optimization_data.locations) > 0: + task_index = locations.loc[ + optimization_data.task_data["task_locations"] + ] + data_model.set_order_locations(task_index) + else: + data_model.set_order_locations( + optimization_data.task_data["task_locations"] + ) + + if optimization_data.task_data["pickup_and_delivery_pairs"] is not None: + pickup_delivery = optimization_data.task_data[ + "pickup_and_delivery_pairs" + ] + data_model.set_pickup_delivery_pairs( + pickup_delivery["pickup_ind"], pickup_delivery["delivery_ind"] + ) + + if ( + optimization_data.task_data["demand"] is not None + and optimization_data.fleet_data["capacities"] is not None + ): + if ( + optimization_data.task_data["demand"].shape[1] + != optimization_data.fleet_data["capacities"].shape[1] + ): + demand_dim = optimization_data.task_data["demand"].shape[1] + cap_dim = optimization_data.fleet_data["capacities"].shape[1] + raise HTTPException( + status_code=400, + detail=( + f"Mismatch in Capacity and Demand dimension, (capacity_dim) {cap_dim} != (demand_dim) {demand_dim}" # noqa + ), + ) + for col in optimization_data.task_data["demand"].columns: + demand_name = "demand_" + str(col) + demand = optimization_data.task_data["demand"][col] + capacities = optimization_data.fleet_data["capacities"][col] + data_model.add_capacity_dimension(demand_name, demand, capacities) + + if optimization_data.task_data["task_time_windows"] is not None: + t_time_windows = optimization_data.task_data["task_time_windows"] + + data_model.set_order_time_windows( + t_time_windows["earliest"], t_time_windows["latest"] + ) + + if optimization_data.task_data["service_times"] is not None: + service_times = optimization_data.task_data["service_times"] + + if service_times is not None: + if type(service_times) is dict: + for v_id, service_time in service_times.items(): + data_model.set_order_service_times( + cudf.Series(service_time, dtype=np.int32), int(v_id) + ) + else: + data_model.set_order_service_times( + cudf.Series(service_times, dtype=np.int32) + ) + + if optimization_data.solver_config["objectives"] is not None: + data_model.set_objective_function( + optimization_data.solver_config["objectives"], + optimization_data.solver_config["objective_weights"], + ) + + if optimization_data.task_data["prizes"] is not None: + data_model.set_order_prizes(optimization_data.task_data["prizes"]) + + if optimization_data.task_data["order_vehicle_match"] is not None: + for data in optimization_data.task_data["order_vehicle_match"]: + data_model.add_order_vehicle_match( + data["order_id"], cudf.Series(data["vehicle_ids"]) + ) + + if optimization_data.initial_solution is not None: + vehicle_ids, routes, types, sol_offsets = parse_initial_sol( + optimization_data.initial_solution + ) + data_model.add_initial_solutions( + cudf.Series(vehicle_ids), + cudf.Series(routes), + cudf.Series(types), + cudf.Series(sol_offsets), + ) + return warnings, data_model + + +def create_solver(optimization_data: OptimizationDataModel): + warnings = [] + solver_settings = routing.SolverSettings() + + if optimization_data.solver_config["time_limit"] is not None: + solver_settings.set_time_limit( + optimization_data.solver_config["time_limit"] + ) + + if optimization_data.solver_config["config_file"] is not None: + solver_settings.dump_config_file( + optimization_data.solver_config["config_file"] + ) + if optimization_data.solver_config["verbose_mode"] is not None: + solver_settings.set_verbose_mode( + optimization_data.solver_config["verbose_mode"] + ) + if optimization_data.solver_config["error_logging"] is not None: + solver_settings.set_error_logging_mode( + optimization_data.solver_config["error_logging"] + ) + + return warnings, solver_settings + + +def prep_optimization_data(optimization_data): + if optimization_data.task_data["task_locations"] is None: + raise ValueError("task location is None") + elif optimization_data.fleet_data["vehicle_locations"] is None: + raise ValueError("vehicle location is None") + + cost_matrix = {} + cost_waypoint_graph = {} + travel_time_matrix = {} + travel_time_waypoint_graph = {} + + if len(optimization_data.cost_matrix) != 0: + cost_matrix = optimization_data.cost_matrix + elif len(optimization_data.waypoint_graph) != 0: + optimization_data.locations = np.append( + optimization_data.task_data["task_locations"].to_numpy(), + optimization_data.fleet_data["vehicle_locations"] + .to_numpy() + .flatten(), + ) + + if optimization_data.fleet_data["vehicle_break_locations"] is not None: + optimization_data.locations = np.append( + optimization_data.locations, + optimization_data.fleet_data[ + "vehicle_break_locations" + ].to_numpy(), + ) + optimization_data.locations = np.unique(optimization_data.locations) + + for v_type, graph in optimization_data.waypoint_graph.items(): + cost_waypoint_graph[v_type] = distance_engine.WaypointMatrix( + graph["offsets"], graph["edges"], graph["weights"] + ) + + cost_matrix[v_type] = cost_waypoint_graph[ + v_type + ].compute_cost_matrix(optimization_data.locations) + else: + raise ValueError("No cost matrix or way point graph provided") + + if len(optimization_data.travel_time_matrix) != 0: + travel_time_matrix = optimization_data.travel_time_matrix + elif len(optimization_data.travel_time_waypoint_graph) != 0: + for ( + v_type, + graph, + ) in optimization_data.travel_time_waypoint_graph.items(): + travel_time_waypoint_graph[v_type] = ( + distance_engine.WaypointMatrix( + graph["offsets"], graph["edges"], graph["weights"] + ) + ) + travel_time_matrix[v_type] = travel_time_waypoint_graph[ + v_type + ].compute_cost_matrix(optimization_data.locations) + else: + travel_time_matrix = None + + return ( + optimization_data, + cost_matrix, + travel_time_matrix, + cost_waypoint_graph, + ) diff --git a/python/cuopt_server/cuopt_server/utils/routing/solver.py b/python/cuopt_server/cuopt_server/utils/routing/solver.py index 2281488d30..bd6b9c276a 100644 --- a/python/cuopt_server/cuopt_server/utils/routing/solver.py +++ b/python/cuopt_server/cuopt_server/utils/routing/solver.py @@ -1,14 +1,11 @@ -# SPDX-FileCopyrightText: Copyright (c) 2022-2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 import time -from typing import Optional -import numpy as np from fastapi import HTTPException -import cudf -from cuopt import distance_engine, routing +from cuopt import routing from cuopt.routing import ErrorStatus from cuopt.utilities import ( InputRuntimeError, @@ -16,21 +13,17 @@ OutOfMemoryError, ) -from cuopt_server.utils.routing.initial_solution import parse_initial_sol +from cuopt_server.utils.routing.conversion import ( # noqa: F401 + create_data_model, + create_solver, + prep_optimization_data, + warn_on_objectives, +) from cuopt_server.utils.routing.optimization_data_model import ( OptimizationDataModel, objective_names, ) -dep_warning = ( - "{field} is deprecated and will be removed in the next release. Ignored." -) - - -def warn_on_objectives(solver_config): - warnings = [] - return warnings, solver_config - # Create routes as waypoint sequence from sequence of task locations def create_waypoint_sequence_routes( @@ -110,335 +103,6 @@ def create_waypoint_sequence_routes( return routes -def create_data_model( - optimization_data: OptimizationDataModel, - cost_matrix: Optional[dict] = None, - travel_time_matrix: Optional[dict] = None, -): - warnings = [] - # Make sure that we are using pool memory allocator - import rmm - - assert isinstance( - rmm.mr.get_current_device_resource(), rmm.mr.StatisticsResourceAdaptor - ) or isinstance( - rmm.mr.get_current_device_resource(), rmm.mr.PoolMemoryResource - ) - - n_fleet = len(optimization_data.fleet_data["vehicle_locations"]) - - n_locations = list(cost_matrix.values())[0].shape[0] - - locations = cudf.Series( - list(range(len(optimization_data.locations))), - index=optimization_data.locations, - ) - - n_orders = len(optimization_data.task_data["task_locations"]) - - # Create data model object - data_model = routing.DataModel(n_locations, n_fleet, n_orders) - - for key, value in cost_matrix.items(): - data_model.add_cost_matrix(value, key) - if travel_time_matrix is not None: - for key, value in travel_time_matrix.items(): - data_model.add_transit_time_matrix(value, key) - - if optimization_data.fleet_data["vehicle_locations"] is not None: - if len(optimization_data.locations) > 0: - start_location_id = locations.loc[ - optimization_data.fleet_data["vehicle_locations"][ - "start_location" - ] - ] - end_location_id = locations.loc[ - optimization_data.fleet_data["vehicle_locations"][ - "end_location" - ] - ] - data_model.set_vehicle_locations( - start_location_id, end_location_id - ) - else: - data_model.set_vehicle_locations( - optimization_data.fleet_data["vehicle_locations"][ - "start_location" - ], - optimization_data.fleet_data["vehicle_locations"][ - "end_location" - ], - ) - - if optimization_data.fleet_data["vehicle_time_windows"] is not None: - v_time_windows = optimization_data.fleet_data["vehicle_time_windows"] - data_model.set_vehicle_time_windows( - v_time_windows["earliest"], v_time_windows["latest"] - ) - - if optimization_data.fleet_data["skip_first_trips"] is not None: - data_model.set_skip_first_trips( - optimization_data.fleet_data["skip_first_trips"] - ) - - if ( - optimization_data.fleet_data["vehicle_break_time_windows"] is not None - and optimization_data.fleet_data["vehicle_break_durations"] is not None - ): - for index in range( - len(optimization_data.fleet_data["vehicle_break_time_windows"]) - ): - v_break_time_windows = optimization_data.fleet_data[ - "vehicle_break_time_windows" - ][index] - v_break_durations = optimization_data.fleet_data[ - "vehicle_break_durations" - ][index] - data_model.add_break_dimension( - v_break_time_windows["earliest"], - v_break_time_windows["latest"], - v_break_durations, - ) - - if optimization_data.fleet_data["vehicle_break_locations"] is not None: - if len(optimization_data.locations) > 0: - break_location_id = locations.loc[ - optimization_data.fleet_data["vehicle_break_locations"] - ] - data_model.set_break_locations(break_location_id) - else: - data_model.set_break_locations( - optimization_data.fleet_data["vehicle_break_locations"] - ) - - if optimization_data.fleet_data["vehicle_types"] is not None: - data_model.set_vehicle_types( - optimization_data.fleet_data["vehicle_types"] - ) - - if optimization_data.fleet_data["vehicle_breaks"] is not None: - for data in optimization_data.fleet_data["vehicle_breaks"]: - data_model.add_vehicle_break( - data["vehicle_id"], - data["earliest"], - data["latest"], - data["duration"], - cudf.Series(data["locations"]), - ) - - if optimization_data.fleet_data["vehicle_order_match"] is not None: - for data in optimization_data.fleet_data["vehicle_order_match"]: - data_model.add_vehicle_order_match( - data["vehicle_id"], cudf.Series(data["order_ids"]) - ) - - if optimization_data.fleet_data["drop_return_trips"] is not None: - data_model.set_drop_return_trips( - optimization_data.fleet_data["drop_return_trips"] - ) - - if optimization_data.fleet_data["vehicle_max_costs"] is not None: - data_model.set_vehicle_max_costs( - optimization_data.fleet_data["vehicle_max_costs"] - ) - - if optimization_data.fleet_data["vehicle_max_times"] is not None: - data_model.set_vehicle_max_times( - optimization_data.fleet_data["vehicle_max_times"] - ) - - if optimization_data.fleet_data["vehicle_fixed_costs"] is not None: - data_model.set_vehicle_fixed_costs( - optimization_data.fleet_data["vehicle_fixed_costs"] - ) - - if optimization_data.fleet_data["min_vehicles"] is not None: - data_model.set_min_vehicles( - optimization_data.fleet_data["min_vehicles"] - ) - - if optimization_data.task_data["task_locations"] is not None: - if len(optimization_data.locations) > 0: - task_index = locations.loc[ - optimization_data.task_data["task_locations"] - ] - data_model.set_order_locations(task_index) - else: - data_model.set_order_locations( - optimization_data.task_data["task_locations"] - ) - - if optimization_data.task_data["pickup_and_delivery_pairs"] is not None: - pickup_delivery = optimization_data.task_data[ - "pickup_and_delivery_pairs" - ] - data_model.set_pickup_delivery_pairs( - pickup_delivery["pickup_ind"], pickup_delivery["delivery_ind"] - ) - - if ( - optimization_data.task_data["demand"] is not None - and optimization_data.fleet_data["capacities"] is not None - ): - if ( - optimization_data.task_data["demand"].shape[1] - != optimization_data.fleet_data["capacities"].shape[1] - ): - demand_dim = optimization_data.task_data["demand"].shape[1] - cap_dim = optimization_data.fleet_data["capacities"].shape[1] - raise HTTPException( - status_code=400, - detail=( - f"Mismatch in Capacity and Demand dimension, (capacity_dim) {cap_dim} != (demand_dim) {demand_dim}" # noqa - ), - ) - for col in optimization_data.task_data["demand"].columns: - demand_name = "demand_" + str(col) - demand = optimization_data.task_data["demand"][col] - capacities = optimization_data.fleet_data["capacities"][col] - data_model.add_capacity_dimension(demand_name, demand, capacities) - - if optimization_data.task_data["task_time_windows"] is not None: - t_time_windows = optimization_data.task_data["task_time_windows"] - - data_model.set_order_time_windows( - t_time_windows["earliest"], t_time_windows["latest"] - ) - - if optimization_data.task_data["service_times"] is not None: - service_times = optimization_data.task_data["service_times"] - - if service_times is not None: - if type(service_times) is dict: - for v_id, service_time in service_times.items(): - data_model.set_order_service_times( - cudf.Series(service_time, dtype=np.int32), int(v_id) - ) - else: - data_model.set_order_service_times( - cudf.Series(service_times, dtype=np.int32) - ) - - if optimization_data.solver_config["objectives"] is not None: - data_model.set_objective_function( - optimization_data.solver_config["objectives"], - optimization_data.solver_config["objective_weights"], - ) - - if optimization_data.task_data["prizes"] is not None: - data_model.set_order_prizes(optimization_data.task_data["prizes"]) - - if optimization_data.task_data["order_vehicle_match"] is not None: - for data in optimization_data.task_data["order_vehicle_match"]: - data_model.add_order_vehicle_match( - data["order_id"], cudf.Series(data["vehicle_ids"]) - ) - - if optimization_data.initial_solution is not None: - vehicle_ids, routes, types, sol_offsets = parse_initial_sol( - optimization_data.initial_solution - ) - data_model.add_initial_solutions( - cudf.Series(vehicle_ids), - cudf.Series(routes), - cudf.Series(types), - cudf.Series(sol_offsets), - ) - return warnings, data_model - - -def create_solver(optimization_data: OptimizationDataModel): - warnings = [] - solver_settings = routing.SolverSettings() - - if optimization_data.solver_config["time_limit"] is not None: - solver_settings.set_time_limit( - optimization_data.solver_config["time_limit"] - ) - - if optimization_data.solver_config["config_file"] is not None: - solver_settings.dump_config_file( - optimization_data.solver_config["config_file"] - ) - if optimization_data.solver_config["verbose_mode"] is not None: - solver_settings.set_verbose_mode( - optimization_data.solver_config["verbose_mode"] - ) - if optimization_data.solver_config["error_logging"] is not None: - solver_settings.set_error_logging_mode( - optimization_data.solver_config["error_logging"] - ) - - return warnings, solver_settings - - -def prep_optimization_data(optimization_data): - if optimization_data.task_data["task_locations"] is None: - raise ValueError("task location is None") - elif optimization_data.fleet_data["vehicle_locations"] is None: - raise ValueError("vehicle location is None") - - cost_matrix = {} - cost_waypoint_graph = {} - travel_time_matrix = {} - travel_time_waypoint_graph = {} - - if len(optimization_data.cost_matrix) != 0: - cost_matrix = optimization_data.cost_matrix - elif len(optimization_data.waypoint_graph) != 0: - optimization_data.locations = np.append( - optimization_data.task_data["task_locations"].to_numpy(), - optimization_data.fleet_data["vehicle_locations"] - .to_numpy() - .flatten(), - ) - - if optimization_data.fleet_data["vehicle_break_locations"] is not None: - optimization_data.locations = np.append( - optimization_data.locations, - optimization_data.fleet_data[ - "vehicle_break_locations" - ].to_numpy(), - ) - optimization_data.locations = np.unique(optimization_data.locations) - - for v_type, graph in optimization_data.waypoint_graph.items(): - cost_waypoint_graph[v_type] = distance_engine.WaypointMatrix( - graph["offsets"], graph["edges"], graph["weights"] - ) - - cost_matrix[v_type] = cost_waypoint_graph[ - v_type - ].compute_cost_matrix(optimization_data.locations) - else: - raise ValueError("No cost matrix or way point graph provided") - - if len(optimization_data.travel_time_matrix) != 0: - travel_time_matrix = optimization_data.travel_time_matrix - elif len(optimization_data.travel_time_waypoint_graph) != 0: - for ( - v_type, - graph, - ) in optimization_data.travel_time_waypoint_graph.items(): - travel_time_waypoint_graph[v_type] = ( - distance_engine.WaypointMatrix( - graph["offsets"], graph["edges"], graph["weights"] - ) - ) - travel_time_matrix[v_type] = travel_time_waypoint_graph[ - v_type - ].compute_cost_matrix(optimization_data.locations) - else: - travel_time_matrix = None - - return ( - optimization_data, - cost_matrix, - travel_time_matrix, - cost_waypoint_graph, - ) - - def get_solver_exception_type(status, message): msg = f"error_status: {status}, msg: {message}" diff --git a/python/cuopt_server/cuopt_server/utils/solver.py b/python/cuopt_server/cuopt_server/utils/solver.py index 7f4e7896ff..a539f6c23d 100644 --- a/python/cuopt_server/cuopt_server/utils/solver.py +++ b/python/cuopt_server/cuopt_server/utils/solver.py @@ -11,7 +11,6 @@ from fastapi.exceptions import RequestValidationError from pydantic import ValidationError -import cuopt_server.utils.request_filter as request_filter import cuopt_server.utils.settings as settings from cuopt_server.utils.data_definition import ( CostMatrices, @@ -33,12 +32,10 @@ SolverIntermediateResponse, ) from cuopt_server.utils.logutil import set_ncaid, set_requestid, set_solverid - - -# Return exception if validation fails -def check_valid(is_valid): - if not is_valid[0]: - raise HTTPException(status_code=400, detail=f"{is_valid[1]}") +from cuopt_server.utils.routing.conversion import ( + check_valid as check_valid, + populate_optimization_data, +) # Wrap the solver response in a dictionary with a "response" @@ -130,145 +127,6 @@ def solve_LP_sync( return full_response, etl_time, solve_time -def populate_optimization_data( - cost_waypoint_graph_data: Optional[WaypointGraphData] = None, - travel_time_waypoint_graph_data: Optional[WaypointGraphData] = None, - cost_matrix_data: Optional[CostMatrices] = None, - travel_time_matrix_data: Optional[CostMatrices] = None, - fleet_data: Optional[FleetData] = None, - task_data: Optional[TaskData] = None, - # Use the update data structure for the sync endpoint because - # it makes the time_limit value Optional - initial_solution: Optional[List[InitialSolution]] = None, - solver_config: Optional[SolverSettingsConfig] = None, - warnings=[], -): - from cuopt_server.utils.routing.optimization_data_model import ( - OptimizationDataModel, - ) - from cuopt_server.utils.routing.solver import warn_on_objectives - - optimization_data = OptimizationDataModel() - - if ( - not cost_waypoint_graph_data - or not cost_waypoint_graph_data.waypoint_graph - ) and (not cost_matrix_data or not cost_matrix_data.data): - raise HTTPException( - status_code=400, - detail="cost_matrix/waypoint_graph needs to be provided to find any route", # noqa - ) - - if ( - cost_waypoint_graph_data and cost_waypoint_graph_data.waypoint_graph - ) and (cost_matrix_data and cost_matrix_data.data): - raise HTTPException( - status_code=400, - detail="only one of cost_matrix or waypoint_graph needs to be provided, not both", # noqa - ) - - if (travel_time_matrix_data and travel_time_matrix_data.data) and ( - travel_time_waypoint_graph_data - and travel_time_waypoint_graph_data.waypoint_graph - ): - raise HTTPException( - status_code=400, - detail="only one of travel_time_matrix_data or travel_time_waypoint_graph_data needs to be provided, not both", # noqa - ) - - if cost_waypoint_graph_data and cost_waypoint_graph_data.waypoint_graph: - check_valid( - optimization_data.set_cost_waypoint_graph( - cost_waypoint_graph_data.waypoint_graph - ) - ) - elif cost_matrix_data and cost_matrix_data.data: - check_valid(optimization_data.set_cost_matrix(cost_matrix_data.data)) - - if ( - travel_time_waypoint_graph_data - and travel_time_waypoint_graph_data.waypoint_graph - ): - check_valid( - optimization_data.set_travel_time_waypoint_graph( - travel_time_waypoint_graph_data.waypoint_graph - ) - ) - elif travel_time_matrix_data and travel_time_matrix_data.data: - check_valid( - optimization_data.set_travel_time_matrix( - travel_time_matrix_data.data - ) - ) - - if fleet_data is not None: - check_valid( - optimization_data.set_fleet_data( - fleet_data.vehicle_ids, - fleet_data.vehicle_locations, - fleet_data.capacities, - fleet_data.vehicle_time_windows, - fleet_data.vehicle_breaks, - fleet_data.vehicle_break_time_windows, - fleet_data.vehicle_break_durations, - fleet_data.vehicle_break_locations, - fleet_data.vehicle_types, - fleet_data.vehicle_order_match, - fleet_data.skip_first_trips, - fleet_data.drop_return_trips, - fleet_data.min_vehicles, - fleet_data.vehicle_max_costs, - fleet_data.vehicle_max_times, - fleet_data.vehicle_fixed_costs, - ) - ) - - if task_data is not None: - check_valid( - optimization_data.set_task_data( - task_data.task_ids, - task_data.task_locations, - task_data.demand, - task_data.pickup_and_delivery_pairs, - task_data.task_time_windows, - task_data.service_times, - task_data.prizes, - task_data.order_vehicle_match, - ) - ) - - if initial_solution is not None: - check_valid(optimization_data.set_initial_solution(initial_solution)) - - if solver_config is not None: - if solver_config.time_limit is None: - num_tasks = len(task_data.task_locations) - solver_config.time_limit = request_filter.std_solver_time_calc( - num_tasks - ) - logging.debug( - "Solver time limit not specified, " - f"setting to {solver_config.time_limit}" - ) - else: - logging.debug( - f"Using specified solver time {solver_config.time_limit}" - ) - owarn, solver_config = warn_on_objectives(solver_config) - warnings.extend(owarn) - check_valid( - optimization_data.set_solver_config( - solver_config.time_limit, - solver_config.objectives, - solver_config.config_file, - solver_config.verbose_mode, - solver_config.error_logging, - ) - ) - - return optimization_data - - def solve_optimized_routes_sync( cost_waypoint_graph_data: Optional[WaypointGraphData] = None, travel_time_waypoint_graph_data: Optional[WaypointGraphData] = None, diff --git a/python/cuopt_server/cuopt_server/utils/utils.py b/python/cuopt_server/cuopt_server/utils/utils.py index eeadab6dfe..01a9ae32a0 100644 --- a/python/cuopt_server/cuopt_server/utils/utils.py +++ b/python/cuopt_server/cuopt_server/utils/utils.py @@ -8,18 +8,17 @@ create_data_model as lp_create_data_model, create_solver as lp_create_solver, ) +from cuopt_server.utils.routing.conversion import ( + create_data_model as routing_create_data_model, + create_solver as routing_create_solver, + populate_optimization_data, + prep_optimization_data as routing_prep_optimization_data, +) from cuopt_server.utils.linear_programming.data_definition import LPData from cuopt_server.utils.linear_programming.data_transformation import ( transform_lp_data, ) - from cuopt_server.utils.routing.data_definition import OptimizedRoutingData -from cuopt_server.utils.routing.solver import ( - create_data_model as routing_create_data_model, - create_solver as routing_create_solver, - prep_optimization_data as routing_prep_optimization_data, -) -from cuopt_server.utils.solver import populate_optimization_data def build_routing_datamodel_from_json(data): From e1238c3a9164891c86563da89eb272de03ac9f3b Mon Sep 17 00:00:00 2001 From: Trevor McKay Date: Tue, 8 Sep 2026 18:21:03 -0400 Subject: [PATCH 048/113] move server data compression handling into a standalone module (#1851) Move the server compressed data handling logic out of job_queue.py into a standalone module so that it can be used by a proxy server which delegates solves to the gRPC server. Authors: - Trevor McKay (https://github.com/tmckayus) Approvers: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) URL: https://github.com/NVIDIA/cuopt/pull/1851 --- .../cuopt_server/tests/test_http_codec.py | 169 ++++++++++++++++++ .../cuopt_server/utils/http_codec.py | 164 +++++++++++++++++ .../cuopt_server/utils/job_queue.py | 104 ++--------- python/cuopt_server/cuopt_server/webserver.py | 60 +------ 4 files changed, 359 insertions(+), 138 deletions(-) create mode 100644 python/cuopt_server/cuopt_server/tests/test_http_codec.py create mode 100644 python/cuopt_server/cuopt_server/utils/http_codec.py diff --git a/python/cuopt_server/cuopt_server/tests/test_http_codec.py b/python/cuopt_server/cuopt_server/tests/test_http_codec.py new file mode 100644 index 0000000000..d0dbb55c41 --- /dev/null +++ b/python/cuopt_server/cuopt_server/tests/test_http_codec.py @@ -0,0 +1,169 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import json +import pickle +import zlib + +import msgpack +import numpy as np +import pytest +from fastapi import HTTPException +from fastapi.responses import JSONResponse, Response + +from cuopt_server.utils.http_codec import ( + PickleForbidden, + decode, + deserialize, + encode, + encode_bytes, + get_format, + mime_json, + mime_msgpack, + mime_pickle, + mime_wild, + mime_zlib, +) + +body_mime_types = [mime_json, mime_msgpack, mime_zlib] + +sample_data = { + "cost_matrix_data": {"data": {"1": [[0, 1], [1, 0]]}}, + "task_data": {"task_locations": [1, 1]}, + "solver_config": {"time_limit": 1.0}, +} + + +@pytest.mark.parametrize("mime_type", body_mime_types) +def test_round_trip(mime_type): + d = encode_bytes(sample_data, mime_type) + assert isinstance(d, bytes) + assert decode(mime_type, d) == sample_data + assert deserialize(mime_type, d) == sample_data + + +def test_encode_bytes_wire_format(): + assert json.loads(encode_bytes(sample_data, mime_json)) == sample_data + assert ( + json.loads(zlib.decompress(encode_bytes(sample_data, mime_zlib))) + == sample_data + ) + assert ( + msgpack.loads( + encode_bytes(sample_data, mime_msgpack), strict_map_key=False + ) + == sample_data + ) + + +def test_decode_unknown_content_type_is_msgpack(): + # Anything that is not json, zlib, or pickle is decoded as msgpack + assert decode("application/unknown", msgpack.dumps(sample_data)) == ( + sample_data + ) + + +def test_msgpack_numpy_round_trip(): + arrays = { + "primal": np.array([1.5, 2.5]), + "offsets": np.array([0, 2], dtype=np.int32), + } + result = decode(mime_msgpack, encode_bytes(arrays, mime_msgpack)) + for key, value in arrays.items(): + assert np.array_equal(result[key], value) + assert result[key].dtype == value.dtype + + +@pytest.mark.parametrize("mime_type", body_mime_types) +def test_deserialize_bad_data(mime_type): + with pytest.raises(HTTPException) as e: + deserialize(mime_type, b"this is not valid in any format") + assert e.value.status_code == 422 + assert "unable to load optimization data stream" in e.value.detail + + +def test_encode_json_returns_result(): + assert encode(sample_data, mime_json) == sample_data + + +def test_encode_unsupported_accept_returns_json(): + assert encode(sample_data, "application/unknown") == sample_data + + +@pytest.mark.parametrize("accept", [mime_msgpack, mime_zlib]) +def test_encode_binary_accept(accept): + r = encode(sample_data, accept) + assert isinstance(r, Response) + assert r.media_type == accept + assert r.status_code == 200 + assert decode(accept, r.body) == sample_data + + +@pytest.mark.parametrize("accept", mime_wild) +def test_encode_wildcard_accept_is_msgpack(accept): + # Callers resolve wildcards before encoding, an unresolved + # wildcard falls through to msgpack + r = encode(sample_data, accept) + assert r.media_type == mime_msgpack + assert decode(mime_msgpack, r.body) == sample_data + + +@pytest.mark.parametrize("accept", body_mime_types) +def test_encode_error_result(accept): + expected = {"error": "something failed", "error_result": True} + r = encode(JSONResponse({"error": "something failed"}, 500), accept, True) + assert r.status_code == 500 + if accept == mime_json: + assert isinstance(r, JSONResponse) + assert json.loads(r.body) == expected + else: + assert r.media_type == accept + assert decode(accept, r.body) == expected + + +def test_get_format(): + formats = [ + get_format(m) + for m in [mime_json, mime_zlib, mime_msgpack, mime_pickle] + ] + assert formats == ["json", "zlib", "msgpack", "pickle"] + + +@pytest.mark.parametrize("mime_type", body_mime_types) +def test_job_queue_uses_codec(mime_type): + # job_queue re-exports the shared mime types and defers to the codec + from cuopt_server.utils import job_queue + + assert job_queue.mime_json == mime_json + assert job_queue.mime_msgpack == mime_msgpack + assert job_queue.mime_pickle == mime_pickle + assert job_queue.mime_wild == mime_wild + assert job_queue.mime_zlib == mime_zlib + assert ( + job_queue.deserialize(mime_type, encode_bytes(sample_data, mime_type)) + == sample_data + ) + + +def test_pickle_round_trip(): + encoded = pickle.dumps(sample_data) + assert decode(mime_pickle, encoded) == sample_data + assert deserialize(mime_pickle, encoded) == sample_data + + +def test_pickle_forbidden_class(): + encoded = pickle.dumps({"obj": object()}) + with pytest.raises(PickleForbidden): + decode(mime_pickle, encoded) + with pytest.raises(HTTPException) as e: + deserialize(mime_pickle, encoded) + assert e.value.status_code == 422 + + +def test_job_queue_pickle_uses_codec(): + from cuopt_server.utils import job_queue + from cuopt_server.utils import http_codec as codec + + assert job_queue.deserialize is codec.deserialize + assert job_queue.SafeUnpickler is codec.SafeUnpickler + assert job_queue.cuopt_pickle_load is codec.cuopt_pickle_load diff --git a/python/cuopt_server/cuopt_server/utils/http_codec.py b/python/cuopt_server/cuopt_server/utils/http_codec.py new file mode 100644 index 0000000000..ffeacbd0d2 --- /dev/null +++ b/python/cuopt_server/cuopt_server/utils/http_codec.py @@ -0,0 +1,164 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Shared codec for cuOpt HTTP request and response bodies. +# This module holds the mime types cuOpt speaks and the JSON, msgpack, zlib, +# and pickle serialization used by every HTTP entry point. It deliberately +# knows nothing about jobs, results, caches, shared memory, or files so that +# any HTTP front end can use it. + +import io +import json +import logging +import pickle +import time +import zlib + +import msgpack +import msgpack_numpy +import numpy +import numpy.core.multiarray +from fastapi import HTTPException +from fastapi.responses import JSONResponse, Response + +msgpack_numpy.patch() + + +mime_json = "application/json" +mime_msgpack = "application/vnd.msgpack" +mime_zlib = "application/zlib" +mime_pickle = "application/octet-stream" +mime_wild = ["application/*", "*/*"] + + +class PickleForbidden(Exception): + pass + + +class SafeUnpickler(pickle.Unpickler): + def __init__(self, file, kind, allowed={}): + self.allowed = allowed + self.kind = kind + super().__init__(file) + + def find_class(self, module, name): + if ( + module not in self.allowed + or name not in self.allowed[module]["names"] + ): + raise PickleForbidden( + f"{module}.{name} is forbidden " + f"in a cuopt {self.kind}pickle file" + ) + else: + return getattr(self.allowed[module]["mod"], name) + + +# LP pickle allow is superset of VRP, so allow the kind +# to be set to "" for messaging and this routine to be +# used when we don't pre-know the problem type +def cuopt_pickle_load(s, kind="LP "): + allowed_LP = { + "numpy.core.multiarray": { + "names": ["_reconstruct"], + "mod": numpy.core.multiarray, + }, + "numpy": {"names": ["ndarray", "dtype"], "mod": numpy}, + } + + return SafeUnpickler(io.BytesIO(s), kind, allowed_LP).load() + + +def cuopt_pickle_load_VRP(s): + return SafeUnpickler(io.BytesIO(s), "VRP ").load() + + +def get_format(mime_type): + f = { + mime_json: "json", + mime_zlib: "zlib", + mime_msgpack: "msgpack", + mime_pickle: "pickle", + } + return f[mime_type] + + +def decode(ctype, buf): + # Any content type other than json, zlib, or pickle is treated as + # msgpack, matching the behavior of the original request handling + if ctype == mime_json: + logging.debug("decode as json") + data = json.loads(buf) + elif ctype == mime_zlib: + logging.debug("decode as zlib compressed json") + data = json.loads(zlib.decompress(buf)) + elif ctype == mime_pickle: + logging.debug("decode as pickle") + data = cuopt_pickle_load(buf, kind="") + else: + logging.debug("decode as msgpack") + data = msgpack.loads(buf, strict_map_key=False) + return data + + +def deserialize(ctype, buf): + # decode with a client error on bad data + try: + data = decode(ctype, buf) + except Exception as e: + raise HTTPException( + status_code=422, + detail="unable to load optimization data stream, %s" % (str(e)), + ) + return data + + +def encode_bytes(data, mime_type): + # Write data to a byte array based on mime type + if mime_type in [mime_json, mime_zlib]: + d = bytes(json.dumps(data), encoding="utf-8") + if mime_type == mime_zlib: + now = time.time() + d = zlib.compress(d, zlib.Z_BEST_SPEED) + logging.debug( + f"Time for zlib compression of result {time.time() - now}" + ) + else: + d = msgpack.dumps(data) + return d + + +def encode(result, accept, job_result=False): + if accept not in [mime_json, mime_msgpack, mime_zlib] + mime_wild: + accept = mime_json + + # This is an exception packaged up elsewhere + if isinstance(result, JSONResponse): + status_code = result.status_code + result = json.loads(result.body) + result["error_result"] = job_result + if accept == mime_json: + return JSONResponse(result, status_code) + else: + status_code = 200 + + # Expect a dictionary at this point + if accept == mime_json: + logging.debug("job_result returning json") + r = result + elif accept == mime_zlib: + logging.debug("job_result returning zlib") + d = bytes(json.dumps(result), encoding="utf-8") + r = Response( + content=zlib.compress(d, zlib.Z_BEST_SPEED), + media_type=mime_zlib, + status_code=status_code, + ) + else: + logging.debug("job_result returning msgpack") + r = Response( + content=msgpack.dumps(result), + media_type=mime_msgpack, + status_code=status_code, + ) + return r diff --git a/python/cuopt_server/cuopt_server/utils/job_queue.py b/python/cuopt_server/cuopt_server/utils/job_queue.py index 53557c69fb..006bd87704 100644 --- a/python/cuopt_server/cuopt_server/utils/job_queue.py +++ b/python/cuopt_server/cuopt_server/utils/job_queue.py @@ -1,12 +1,10 @@ # SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -import io import json import logging import multiprocessing import os -import pickle import time import uuid import zlib @@ -16,8 +14,6 @@ import msgpack import msgpack_numpy -import numpy -import numpy.core.multiarray from fastapi import HTTPException from fastapi.responses import JSONResponse @@ -37,6 +33,22 @@ exception_handler, http_exception_handler, ) + +# Re-exported so existing imports from this module keep working +from cuopt_server.utils.http_codec import ( # noqa: F401 + PickleForbidden, + SafeUnpickler, + cuopt_pickle_load, + cuopt_pickle_load_VRP, + deserialize, + encode_bytes, + get_format, + mime_json, + mime_msgpack, + mime_pickle, + mime_wild, + mime_zlib, +) from cuopt_server.utils.linear_programming.data_transformation import ( transform_lp_data, ) @@ -44,10 +56,6 @@ from cuopt_server.utils.routing.initial_solution import add_initial_sol -class PickleForbidden(Exception): - pass - - msgpack_numpy.patch() @@ -85,44 +93,6 @@ def get_solver_response(response): return response["solver_infeasible_response"] -class SafeUnpickler(pickle.Unpickler): - def __init__(self, file, kind, allowed={}): - self.allowed = allowed - self.kind = kind - super().__init__(file) - - def find_class(self, module, name): - if ( - module not in self.allowed - or name not in self.allowed[module]["names"] - ): - raise PickleForbidden( - f"{module}.{name} is forbidden " - f"in a cuopt {self.kind}pickle file" - ) - else: - return getattr(self.allowed[module]["mod"], name) - - -# LP pickle allow is superset of VRP, so allow the kind -# to be set to "" for messaging and this routine to be -# used when we don't pre-know the problem type -def cuopt_pickle_load(s, kind="LP "): - allowed_LP = { - "numpy.core.multiarray": { - "names": ["_reconstruct"], - "mod": numpy.core.multiarray, - }, - "numpy": {"names": ["ndarray", "dtype"], "mod": numpy}, - } - - return SafeUnpickler(io.BytesIO(s), kind, allowed_LP).load() - - -def cuopt_pickle_load_VRP(s): - return SafeUnpickler(io.BytesIO(s), "VRP ").load() - - all_jobs_marked_done = multiprocessing.Event() # storage for job results keyed by id @@ -142,13 +112,6 @@ def cuopt_pickle_load_VRP(s): results_lock = Lock() -mime_json = "application/json" -mime_msgpack = "application/vnd.msgpack" -mime_zlib = "application/zlib" -mime_pickle = "application/octet-stream" -mime_wild = ["application/*", "*/*"] - - def add_cache_entry(id, content_type): if all_jobs_marked_done.is_set(): return None @@ -919,28 +882,6 @@ def solve(self, intermediate_sender): return ans, self.initial_etl_time + etl, slv -def deserialize(ctype, buf): - try: - if ctype == mime_json: - logging.debug("decode as json") - data = json.loads(buf) - elif ctype == mime_zlib: - logging.debug("decode as zlib compressed json") - data = json.loads(zlib.decompress(buf)) - elif ctype == mime_pickle: - logging.debug("decode as pickle") - data = cuopt_pickle_load(buf, kind="") - else: - logging.debug("decode as msgpack") - data = msgpack.loads(buf, strict_map_key=False) - except Exception as e: - raise HTTPException( - status_code=422, - detail="unable to load optimization data stream, %s" % (str(e)), - ) - return data - - def wrapper_fields(data, do_raise=True): action = data.get("action", "cuOpt_Solver") if action not in get_valid_actions(): @@ -1473,18 +1414,7 @@ def data_to_byte(data, result_mime_type): # Write data to a byte array based on result mime type # Note that notes and warnings are serialized here before # they are popped, so the answer still has them - now = time.time() - if result_mime_type in [mime_json, mime_zlib]: - d = bytes(json.dumps(data), encoding="utf-8") - if result_mime_type == mime_zlib: - now = time.time() - d = zlib.compress(d, zlib.Z_BEST_SPEED) - logging.debug( - "Time for zlib compression of " - f"result {time.time() - now}" - ) - else: - d = msgpack.dumps(data) + d = encode_bytes(data, result_mime_type) self.size = len(d) return d diff --git a/python/cuopt_server/cuopt_server/webserver.py b/python/cuopt_server/cuopt_server/webserver.py index 4c5bde3e9e..a0874d3ad7 100644 --- a/python/cuopt_server/cuopt_server/webserver.py +++ b/python/cuopt_server/cuopt_server/webserver.py @@ -69,6 +69,15 @@ http_exception_handler, validation_exception_handler, ) +from cuopt_server.utils.http_codec import ( + encode, + get_format, + mime_json, + mime_msgpack, + mime_pickle, + mime_wild, + mime_zlib, +) from cuopt_server.utils.job_queue import ( BaseResult, BinaryJobResult, @@ -85,11 +94,6 @@ get_incumbents_for_id, get_solution_for_id, get_warmstart_data_for_id, - mime_json, - mime_msgpack, - mime_pickle, - mime_wild, - mime_zlib, status_by_id, update_cache_entry, ) @@ -271,52 +275,6 @@ def __init__(self, response): self.response = response -def encode(result, accept, job_result=False): - if accept not in [mime_json, mime_msgpack, mime_zlib] + mime_wild: - accept = mime_json - - # This is an exception packaged up elsewhere - if isinstance(result, JSONResponse): - status_code = result.status_code - result = json.loads(result.body) - result["error_result"] = job_result - if accept == mime_json: - return JSONResponse(result, status_code) - else: - status_code = 200 - - # Expect a dictionary at this point - if accept == mime_json: - logging.debug("job_result returning json") - r = result - elif accept == mime_zlib: - logging.debug("job_result returning zlib") - d = bytes(json.dumps(result), encoding="utf-8") - r = Response( - content=zlib.compress(d, zlib.Z_BEST_SPEED), - media_type=mime_zlib, - status_code=status_code, - ) - else: - logging.debug("job_result returning msgpack") - r = Response( - content=msgpack.dumps(result), - media_type=mime_msgpack, - status_code=status_code, - ) - return r - - -def get_format(mime_type): - f = { - mime_json: "json", - mime_zlib: "zlib", - mime_msgpack: "msgpack", - mime_pickle: "pickle", - } - return f[mime_type] - - @app.get( "/cuopt/log/{id}", description="Note: This is for self-hosted. " From 70b0955847b81f58b138a1874fe42b7c770afe82 Mon Sep 17 00:00:00 2001 From: Ramakrishna Prabhu <42624703+ramakrishnap-nv@users.noreply.github.com> Date: Wed, 9 Sep 2026 09:01:36 -0500 Subject: [PATCH 049/113] fix(build): pin RAPIDS_BRANCH and shared-workflows to release/26.10 (#1870) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RAPIDS main moved to 26.12 after burndown while cuOpt stays on 26.10. `RAPIDS_BRANCH` was still `main`, so CMake fetched rapids-cmake from RAPIDS main (26.12) and the shared-workflows CI containers resolved 26.12 packages, while the test environments install 26.10. rmm uses versioned inline namespaces, so the mismatch did not fail the build — it failed at test time, once the freshly built `libcuopt.so` was loaded against the installed rmm: ``` libcuopt.so: undefined symbol: _ZTIN3rmm10_RMM_26_129bad_allocE (typeinfo for rmm::_RMM_26_12::bad_alloc) ``` Every `conda-cpp-tests`, `conda-python-tests`, `wheel-tests-*`, `docs-build`, `java-build` and `multi-gpu-cpp-tests` job failed on this while all `conda-cpp-build` and `wheel-build-*` jobs passed, which is what made it look unrelated to dependencies at first glance. Pins both halves to `release/26.10`: - `RAPIDS_BRANCH` — read by `cmake/rapids_config.cmake` to pick the rapids-cmake branch - 38 `rapidsai/shared-workflows` refs across the 5 workflow files that use them `rapidsai/shared-actions` stays at `@main`: it has no release branches. This is the same fix as the 26.08 burndown, #1599 (`RAPIDS_BRANCH`) plus d81ce448 (shared-workflows), and touches the same set of files. `ci/release/update-version.sh` resets `RAPIDS_BRANCH` to `main` when run in the "main" context, which is how it reverted after #1666. Verified `release/26.10` exists in both `rapidsai/rapids-cmake` and `rapidsai/shared-workflows`, and that both report `VERSION` 26.10.00 while their `main` reports 26.12.00. All 16 workflow files still parse, and pre-commit passes. Authors: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) Approvers: - Mike Sarahan (https://github.com/msarahan) - Trevor McKay (https://github.com/tmckayus) URL: https://github.com/NVIDIA/cuopt/pull/1870 --- .github/workflows/build.yaml | 28 ++++++++-------- .github/workflows/multi_gpu_cpp_test.yaml | 2 +- .github/workflows/pr.yaml | 32 +++++++++---------- .github/workflows/test.yaml | 12 +++---- .../trigger-breaking-change-alert.yaml | 2 +- RAPIDS_BRANCH | 2 +- 6 files changed, 39 insertions(+), 39 deletions(-) diff --git a/.github/workflows/build.yaml b/.github/workflows/build.yaml index 264fb5398c..4423d41da5 100644 --- a/.github/workflows/build.yaml +++ b/.github/workflows/build.yaml @@ -46,7 +46,7 @@ permissions: {} jobs: build-details: - uses: rapidsai/shared-workflows/.github/workflows/compute-build-details.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/compute-build-details.yaml@release/26.10 cpp-build: permissions: actions: read @@ -56,7 +56,7 @@ jobs: pull-requests: read needs: [build-details] secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/conda-cpp-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/conda-cpp-build.yaml@release/26.10 with: build_type: ${{ inputs.build_type || 'branch' }} build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -73,7 +73,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@release/26.10 with: build_type: ${{ inputs.build_type || 'branch' }} branch: ${{ inputs.branch }} @@ -94,7 +94,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/conda-python-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/conda-python-build.yaml@release/26.10 with: build_type: ${{ inputs.build_type || 'branch' }} build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -112,7 +112,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/conda-upload-packages.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/conda-upload-packages.yaml@release/26.10 secrets: CONDA_RAPIDSAI_NIGHTLY_TOKEN: ${{ secrets.CONDA_RAPIDSAI_NIGHTLY_TOKEN }} CONDA_RAPIDSAI_TOKEN: ${{ secrets.CONDA_RAPIDSAI_TOKEN }} @@ -130,7 +130,7 @@ jobs: pull-requests: read needs: [build-details] secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@release/26.10 with: build_type: ${{ inputs.build_type || 'branch' }} build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -149,7 +149,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/wheels-publish.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-publish.yaml@release/26.10 secrets: CONDA_RAPIDSAI_WHEELS_NIGHTLY_TOKEN: ${{ secrets.CONDA_RAPIDSAI_WHEELS_NIGHTLY_TOKEN }} RAPIDSAI_PYPI_TOKEN: ${{ secrets.RAPIDSAI_PYPI_TOKEN }} @@ -170,7 +170,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@release/26.10 with: build_type: ${{ inputs.build_type || 'branch' }} build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -190,7 +190,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/wheels-publish.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-publish.yaml@release/26.10 secrets: CONDA_RAPIDSAI_WHEELS_NIGHTLY_TOKEN: ${{ secrets.CONDA_RAPIDSAI_WHEELS_NIGHTLY_TOKEN }} RAPIDSAI_PYPI_TOKEN: ${{ secrets.RAPIDSAI_PYPI_TOKEN }} @@ -211,7 +211,7 @@ jobs: pull-requests: read needs: [build-details] secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@release/26.10 with: build_type: ${{ inputs.build_type || 'branch' }} build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -232,7 +232,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/wheels-publish.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-publish.yaml@release/26.10 secrets: CONDA_RAPIDSAI_WHEELS_NIGHTLY_TOKEN: ${{ secrets.CONDA_RAPIDSAI_WHEELS_NIGHTLY_TOKEN }} RAPIDSAI_PYPI_TOKEN: ${{ secrets.RAPIDSAI_PYPI_TOKEN }} @@ -253,7 +253,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@release/26.10 with: build_type: ${{ inputs.build_type || 'branch' }} node_type: "gpu-l4-latest-1" @@ -274,7 +274,7 @@ jobs: pull-requests: read needs: [build-details] secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@release/26.10 with: build_type: ${{ inputs.build_type || 'branch' }} build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -296,7 +296,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/wheels-publish.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-publish.yaml@release/26.10 secrets: CONDA_RAPIDSAI_WHEELS_NIGHTLY_TOKEN: ${{ secrets.CONDA_RAPIDSAI_WHEELS_NIGHTLY_TOKEN }} RAPIDSAI_PYPI_TOKEN: ${{ secrets.RAPIDSAI_PYPI_TOKEN }} diff --git a/.github/workflows/multi_gpu_cpp_test.yaml b/.github/workflows/multi_gpu_cpp_test.yaml index 5d35c65c90..fdc99c56b8 100644 --- a/.github/workflows/multi_gpu_cpp_test.yaml +++ b/.github/workflows/multi_gpu_cpp_test.yaml @@ -61,7 +61,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@release/26.10 with: build_type: ${{ inputs.build_type }} branch: ${{ inputs.branch }} diff --git a/.github/workflows/pr.yaml b/.github/workflows/pr.yaml index ba50abcbce..dd59daeec5 100644 --- a/.github/workflows/pr.yaml +++ b/.github/workflows/pr.yaml @@ -37,12 +37,12 @@ jobs: - test-self-hosted-server permissions: contents: read - uses: rapidsai/shared-workflows/.github/workflows/pr-builder.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/pr-builder.yaml@release/26.10 if: always() with: needs: ${{ toJSON(needs) }} build-details: - uses: rapidsai/shared-workflows/.github/workflows/compute-build-details.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/compute-build-details.yaml@release/26.10 compute-matrix-filters: permissions: contents: read @@ -69,7 +69,7 @@ jobs: contents: read packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/changed-files.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/changed-files.yaml@release/26.10 with: files_yaml: | build_docs: @@ -368,7 +368,7 @@ jobs: checks: permissions: contents: read - uses: rapidsai/shared-workflows/.github/workflows/checks.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/checks.yaml@release/26.10 with: enable_check_generated_files: false ignored_pr_jobs: "pr-test-summary" @@ -387,7 +387,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/conda-cpp-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/conda-cpp-build.yaml@release/26.10 with: build_type: pull-request build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -400,7 +400,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/conda-cpp-tests.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/conda-cpp-tests.yaml@release/26.10 if: fromJSON(needs.changed-files.outputs.changed_file_groups).test_cpp with: build_type: pull-request @@ -437,7 +437,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/conda-python-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/conda-python-build.yaml@release/26.10 with: build_type: pull-request build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -452,7 +452,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/conda-python-tests.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/conda-python-tests.yaml@release/26.10 if: fromJSON(needs.changed-files.outputs.changed_file_groups).test_python_conda with: build_type: pull-request @@ -467,7 +467,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@release/26.10 if: fromJSON(needs.changed-files.outputs.changed_file_groups).build_docs with: build_type: pull-request @@ -486,7 +486,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@release/26.10 if: >- fromJSON(needs.changed-files.outputs.changed_file_groups).test_java || fromJSON(needs.changed-files.outputs.changed_file_groups).test_cpp @@ -508,7 +508,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@release/26.10 with: # build for every combination of arch and CUDA version, but only for the latest Python matrix_filter: ${{ needs.compute-matrix-filters.outputs.libcuopt_filter }} @@ -527,7 +527,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@release/26.10 with: build_type: pull-request build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -544,7 +544,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/wheels-test.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-test.yaml@release/26.10 if: fromJSON(needs.changed-files.outputs.changed_file_groups).test_python_wheels with: build_type: pull-request @@ -566,7 +566,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@release/26.10 with: build_type: pull-request build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -586,7 +586,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-build.yaml@release/26.10 with: build_type: pull-request build-datetime: ${{ needs.build-details.outputs.build-datetime }} @@ -605,7 +605,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/wheels-test.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-test.yaml@release/26.10 if: fromJSON(needs.changed-files.outputs.changed_file_groups).test_python_wheels with: build_type: pull-request diff --git a/.github/workflows/test.yaml b/.github/workflows/test.yaml index b287bcb990..eb328bcdb4 100644 --- a/.github/workflows/test.yaml +++ b/.github/workflows/test.yaml @@ -35,7 +35,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/conda-cpp-tests.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/conda-cpp-tests.yaml@release/26.10 with: build_type: ${{ inputs.build_type }} branch: ${{ inputs.branch }} @@ -57,7 +57,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/conda-python-tests.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/conda-python-tests.yaml@release/26.10 with: run_codecov: false build_type: ${{ inputs.build_type }} @@ -81,7 +81,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@release/26.10 with: build_type: ${{ inputs.build_type }} branch: ${{ inputs.branch }} @@ -99,7 +99,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/wheels-test.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-test.yaml@release/26.10 with: build_type: ${{ inputs.build_type }} branch: ${{ inputs.branch }} @@ -121,7 +121,7 @@ jobs: id-token: write packages: read pull-requests: read - uses: rapidsai/shared-workflows/.github/workflows/wheels-test.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/wheels-test.yaml@release/26.10 with: build_type: ${{ inputs.build_type }} branch: ${{ inputs.branch }} @@ -144,7 +144,7 @@ jobs: packages: read pull-requests: read secrets: inherit # zizmor: ignore[secrets-inherit] - uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/custom-job.yaml@release/26.10 with: build_type: ${{ inputs.build_type }} branch: ${{ inputs.branch }} diff --git a/.github/workflows/trigger-breaking-change-alert.yaml b/.github/workflows/trigger-breaking-change-alert.yaml index 7a8cf1b8e8..4b578114de 100644 --- a/.github/workflows/trigger-breaking-change-alert.yaml +++ b/.github/workflows/trigger-breaking-change-alert.yaml @@ -21,7 +21,7 @@ jobs: if: contains(github.event.pull_request.labels.*.name, 'breaking') permissions: contents: read - uses: rapidsai/shared-workflows/.github/workflows/breaking-change-alert.yaml@main + uses: rapidsai/shared-workflows/.github/workflows/breaking-change-alert.yaml@release/26.10 secrets: slack-webhook-url: ${{ secrets.NV_SLACK_BREAKING_CHANGE_NOTIFIER_APP }} with: diff --git a/RAPIDS_BRANCH b/RAPIDS_BRANCH index ba2906d066..b735ab15de 100644 --- a/RAPIDS_BRANCH +++ b/RAPIDS_BRANCH @@ -1 +1 @@ -main +release/26.10 From 4def469a1b8a368c614121d1e19ab102dba76d7e Mon Sep 17 00:00:00 2001 From: Trevor McKay Date: Wed, 9 Sep 2026 10:47:46 -0400 Subject: [PATCH 050/113] change Python grpc client so wait loop happens in Python (#1857) If we use the C++ wait_result() method, the GIL is locked and no other threads run. Run the wait loop in Python and call only the Cython status method to determine if the job has completed. Authors: - Trevor McKay (https://github.com/tmckayus) Approvers: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) URL: https://github.com/NVIDIA/cuopt/pull/1857 --- docs/cuopt/source/cuopt-grpc/routing.rst | 1 + .../cuopt/cuopt/grpc/client/grpc_client.pyx | 85 +++++-- .../linear_programming/test_grpc_client.py | 216 +++++++++++++++++- 3 files changed, 285 insertions(+), 17 deletions(-) diff --git a/docs/cuopt/source/cuopt-grpc/routing.rst b/docs/cuopt/source/cuopt-grpc/routing.rst index 2827301058..e911054bf9 100644 --- a/docs/cuopt/source/cuopt-grpc/routing.rst +++ b/docs/cuopt/source/cuopt-grpc/routing.rst @@ -134,6 +134,7 @@ Import path: ``cuopt.grpc.routing``. .. autoclass:: cuopt.grpc.routing.RoutingClient :members: :undoc-members: + :exclude-members: _status .. autoexception:: cuopt.grpc.routing.RoutingSolveError :members: diff --git a/python/cuopt/cuopt/grpc/client/grpc_client.pyx b/python/cuopt/cuopt/grpc/client/grpc_client.pyx index 3f9cdcb0e5..ba75aeda53 100644 --- a/python/cuopt/cuopt/grpc/client/grpc_client.pyx +++ b/python/cuopt/cuopt/grpc/client/grpc_client.pyx @@ -93,6 +93,43 @@ class JobNotReadyError(GrpcError): pass +# Matches the previous C++ wait() poll cadence. Keep this in Python +# (time.sleep) so the GIL is released between short status RPCs. +_WAIT_POLL_INTERVAL_S = 1.0 + + +def _wait_poll_loop( + get_status, + job_id, + timeout_seconds, + error_cls, + poll_interval_s=_WAIT_POLL_INTERVAL_S, +): + """Poll ``get_status(job_id)`` until the job is terminal or the timeout. + + The loop and ``time.sleep`` run in Python so the GIL is released between + short status RPCs. Concurrent incumbent/log stream threads can therefore + run during ``Client.wait``. ``timeout_seconds == 0`` waits indefinitely. + Negative values raise ``error_cls``. + """ + if timeout_seconds < 0: + raise error_cls("timeout_seconds must be non-negative") + deadline = ( + time.monotonic() + timeout_seconds if timeout_seconds > 0 else None + ) + while True: + status = get_status(job_id) + if status not in (JobStatus.QUEUED, JobStatus.PROCESSING): + return status + if deadline is None: + time.sleep(poll_interval_s) + continue + remaining = deadline - time.monotonic() + if remaining <= 0: + raise error_cls("Timeout waiting for job completion") + time.sleep(min(poll_interval_s, remaining)) + + cdef int _invoke_log_callback( const char* line, size_t line_len, @@ -315,19 +352,25 @@ cdef class Client: Block until ``job_id`` reaches a terminal state and return its :class:`JobStatus`. - ``timeout`` is in whole seconds. ``None`` waits indefinitely. - Non-``None`` values are converted with ``int(timeout)`` (so ``0.5`` - becomes ``0`` and waits indefinitely). Positive timeouts poll about - once per second and raise :class:`GrpcError` if the deadline expires - (they do not return a non-terminal :class:`JobStatus`). + ``timeout`` is in whole seconds. ``None`` or ``0`` waits indefinitely. + Negative values raise :class:`GrpcError`. Non-``None`` values are + converted with ``int(timeout)`` (so ``0.5`` becomes ``0`` and waits + indefinitely). Positive timeouts poll about once per second and raise + :class:`GrpcError` if the deadline expires (they do not return a + non-terminal :class:`JobStatus`). + + The wait loop runs in Python and only calls :meth:`status` for each + poll, so the GIL is released between checks. Concurrent + :meth:`start_incumbent_stream` and + :meth:`start_log_stream` threads can therefore make progress during + the wait. Each :meth:`status` call is a unary RPC with a separate + 60-second hang deadline; a hung poll can therefore outlast a short + ``timeout``. """ - cdef int timeout_seconds = 0 if timeout is None else int(timeout) - cdef grpc_status_result_t wait_result = self._client.get().wait( - job_id.encode("utf-8"), timeout_seconds + timeout_seconds = 0 if timeout is None else int(timeout) + return _wait_poll_loop( + self.status, job_id, timeout_seconds, GrpcError ) - if not wait_result.success: - raise GrpcError(wait_result.error_message.decode("utf-8")) - return JobStatus(wait_result.status) def cancel(self, str job_id): """ @@ -1095,18 +1138,28 @@ cdef class RoutingClient: raise RoutingSolveError(sub.error_message.decode("utf-8")) return sub.job_id.decode("utf-8") + def _status(self, str job_id): + cdef grpc_status_result_t st = self._client.get().status( + job_id.encode("utf-8") + ) + if not st.success: + raise RoutingSolveError(st.error_message.decode("utf-8")) + return JobStatus(st.status) + def wait(self, str job_id, int timeout=0): """Block until the job finishes; return the terminal status int. Raises ``RoutingSolveError`` if the wait itself fails (e.g. transport error or unknown job), mirroring the LP/MILP client. + + Polls job status from Python so the GIL is released between short + status RPCs. ``timeout == 0`` waits indefinitely; negative values + raise :class:`RoutingSolveError`. Each status poll is a unary RPC + with a separate 60-second hang deadline. """ - cdef grpc_status_result_t st = self._client.get().wait( - job_id.encode("utf-8"), timeout + return _wait_poll_loop( + self._status, job_id, timeout, RoutingSolveError ) - if not st.success: - raise RoutingSolveError(st.error_message.decode("utf-8")) - return st.status def result(self, str job_id): """Fetch and parse the routing solution for a completed job. diff --git a/python/cuopt/cuopt/tests/linear_programming/test_grpc_client.py b/python/cuopt/cuopt/tests/linear_programming/test_grpc_client.py index 7807c757ff..311c548435 100644 --- a/python/cuopt/cuopt/tests/linear_programming/test_grpc_client.py +++ b/python/cuopt/cuopt/tests/linear_programming/test_grpc_client.py @@ -2,10 +2,15 @@ # SPDX-License-Identifier: Apache-2.0 import os +import threading import time import pytest +from cuopt.grpc.client.grpc_client import ( + _WAIT_POLL_INTERVAL_S, + _wait_poll_loop, +) from cuopt.grpc.linear_programming import ( Client, GrpcError, @@ -75,6 +80,151 @@ def _assert_demo_lp_solution(client): client.delete(job_id) +class TestWaitPollLoop: + def test_returns_immediately_when_already_terminal(self, monkeypatch): + sleeps = [] + monkeypatch.setattr(time, "sleep", sleeps.append) + status = _wait_poll_loop( + lambda job_id: JobStatus.COMPLETED, "job", 0, GrpcError + ) + assert status is JobStatus.COMPLETED + assert sleeps == [] + + def test_sleeps_between_in_flight_polls(self, monkeypatch): + polls = [] + sleeps = [] + + def get_status(job_id): + polls.append(job_id) + if len(polls) < 3: + return JobStatus.PROCESSING + return JobStatus.CANCELLED + + monkeypatch.setattr(time, "sleep", sleeps.append) + status = _wait_poll_loop(get_status, "abc", 0, GrpcError) + assert status is JobStatus.CANCELLED + assert polls == ["abc", "abc", "abc"] + assert sleeps == [_WAIT_POLL_INTERVAL_S, _WAIT_POLL_INTERVAL_S] + + def test_timeout_raises_after_deadline(self, monkeypatch): + ticks = iter([100.0, 100.0, 101.0]) + monkeypatch.setattr(time, "monotonic", lambda: next(ticks)) + monkeypatch.setattr(time, "sleep", lambda seconds: None) + with pytest.raises( + GrpcError, match="Timeout waiting for job completion" + ): + _wait_poll_loop( + lambda job_id: JobStatus.QUEUED, "job", 1, GrpcError + ) + + def test_sleep_is_capped_to_remaining_deadline(self, monkeypatch): + ticks = iter([100.0, 100.6]) + sleeps = [] + monkeypatch.setattr(time, "monotonic", lambda: next(ticks)) + monkeypatch.setattr(time, "sleep", sleeps.append) + + polls = {"n": 0} + + def get_status(job_id): + polls["n"] += 1 + if polls["n"] == 1: + return JobStatus.PROCESSING + return JobStatus.COMPLETED + + status = _wait_poll_loop(get_status, "job", 1, GrpcError) + assert status is JobStatus.COMPLETED + assert sleeps == [pytest.approx(0.4)] + + def test_rejects_negative_timeout(self): + with pytest.raises( + GrpcError, match="timeout_seconds must be non-negative" + ): + _wait_poll_loop( + lambda job_id: JobStatus.PROCESSING, "job", -1, GrpcError + ) + + def test_other_thread_runs_during_sleep(self): + started = threading.Event() + progressed = threading.Event() + polls = {"n": 0} + + def get_status(job_id): + polls["n"] += 1 + if polls["n"] == 1: + started.set() + return JobStatus.PROCESSING + assert progressed.wait(timeout=2) + return JobStatus.COMPLETED + + def worker(): + assert started.wait(timeout=2) + progressed.set() + + thread = threading.Thread(target=worker) + thread.start() + status = _wait_poll_loop( + get_status, + "job", + 0, + GrpcError, + poll_interval_s=0.05, + ) + thread.join(timeout=2) + assert status is JobStatus.COMPLETED + assert progressed.is_set() + + def test_client_wait_releases_gil(self, monkeypatch): + """A GIL-bound spinner must progress during Client.wait's poll sleep. + + Hits are counted only inside the sleep, so this fails if wait() still + called the C++ WaitForCompletion/sleep path that holds the GIL. + """ + import cuopt.grpc.client.grpc_client as grpc_mod + + polls = {"n": 0} + + class StatusStubClient(Client): + def status(self, job_id): + polls["n"] += 1 + if polls["n"] == 1: + return JobStatus.PROCESSING + return JobStatus.COMPLETED + + client = StatusStubClient.__new__(StatusStubClient) + + hits = {"n": 0} + during_sleep = [] + stop = threading.Event() + + def spinner(): + while not stop.is_set(): + hits["n"] += 1 + + real_sleep = time.sleep + + def tracking_sleep(seconds): + assert seconds == _WAIT_POLL_INTERVAL_S + before = hits["n"] + real_sleep(0.15) + during_sleep.append(hits["n"] - before) + + monkeypatch.setattr(grpc_mod.time, "sleep", tracking_sleep) + thread = threading.Thread(target=spinner) + thread.start() + try: + status = client.wait("job") + finally: + stop.set() + thread.join(timeout=2) + + assert status is JobStatus.COMPLETED + assert during_sleep, "wait() never slept between status polls" + assert during_sleep[0] > 0, ( + "spinner made no progress during wait sleep; GIL likely held " + f"(hits={during_sleep[0]})" + ) + + class TestTlsConfig: def test_mtls_requires_both_client_materials(self): pem = "-----BEGIN CERTIFICATE-----\nabc\n-----END CERTIFICATE-----" @@ -272,7 +422,7 @@ def get_solution( job_id = client.submit(problem, settings) client.start_incumbent_stream(job_id, settings=settings) - terminal = _poll_until_complete(client, job_id, _MIP_NAMES) + terminal = client.wait(job_id, timeout=120) assert terminal == JobStatus.COMPLETED client.join_incumbent_stream(job_id) @@ -284,6 +434,70 @@ def get_solution( assert solution is not None client.delete(job_id) + def test_mip_incumbent_stream_live_during_wait(self, grpc_server): + """Incumbent callbacks must fire during wait(), not in a burst after. + + The 2-variable MIP in test_mip_incumbent_stream finishes too fast to + tell. swath1 with a time limit stays PROCESSING long enough that a + GIL-holding wait() would delay every callback until join(). + """ + if not os.path.isfile(_SWATH1_MPS): + pytest.skip(f"dataset not found: {_SWATH1_MPS}") + + class TimedIncumbents(GetSolutionCallback): + def __init__(self): + super().__init__() + self.times = [] + self.costs = [] + + def get_solution( + self, solution, solution_cost, solution_bound, user_data + ): + self.times.append(time.monotonic()) + self.costs.append(float(solution_cost[0])) + + collector = TimedIncumbents() + settings = SolverSettings() + settings.set_mip_callback(collector, None) + settings.set_parameter(CUOPT_TIME_LIMIT, 8) + + client = Client("localhost", grpc_server) + job_id = client.submit(Read(_SWATH1_MPS), settings) + client.start_incumbent_stream( + job_id, settings=settings, poll_interval_ms=200 + ) + try: + terminal = client.wait(job_id, timeout=30) + wait_end = time.monotonic() + client.join_incumbent_stream(job_id) + finally: + client.delete(job_id) + + if terminal != JobStatus.COMPLETED: + pytest.skip(f"job did not complete ({terminal.name})") + if len(collector.times) < 2: + pytest.skip( + "need >=2 incumbents to test live delivery, got " + f"{len(collector.times)}" + ) + + n_before = sum(t < wait_end for t in collector.times) + spread = max(collector.times) - min(collector.times) + lag = wait_end - min(collector.times) + print( + f"incumbents={len(collector.times)} before_wait={n_before} " + f"spread={spread:.3f}s first_to_wait_end={lag:.3f}s" + ) + assert n_before >= 1, ( + f"all {len(collector.times)} incumbents arrived at/after " + f"wait() returned (spread={spread:.4f}s); GIL likely held" + ) + assert spread > 0.15, ( + f"incumbent timestamps clustered in {spread:.4f}s " + f"(n={len(collector.times)}, lag_to_wait_end={lag:.4f}s); " + "likely dumped as a burst at completion" + ) + @pytest.mark.xdist_group(name="grpc_server") @pytest.mark.filterwarnings("ignore::DeprecationWarning") From 8e3f264f97e6109a7930efe846f4be8d9042b63c Mon Sep 17 00:00:00 2001 From: Trevor McKay Date: Wed, 9 Sep 2026 11:15:50 -0400 Subject: [PATCH 051/113] add a smoke test to ensure server images start correctly (#1646) This adds a simple smoke test after image builds to make sure the server instances start. The goal is to detect defects in the Dockerfile used to build the image (like faulty configuration of shared library paths, etc). Authors: - Trevor McKay (https://github.com/tmckayus) - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) Approvers: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) URL: https://github.com/NVIDIA/cuopt/pull/1646 --- .github/workflows/test_images.yaml | 32 +++++++++ ci/docker/README.md | 14 ++++ ci/docker/smoke_image.sh | 104 +++++++++++++++++++++++++++++ 3 files changed, 150 insertions(+) create mode 100755 ci/docker/smoke_image.sh diff --git a/.github/workflows/test_images.yaml b/.github/workflows/test_images.yaml index afbbab4382..3767e64566 100644 --- a/.github/workflows/test_images.yaml +++ b/.github/workflows/test_images.yaml @@ -87,3 +87,35 @@ jobs: - name: Test cuopt run: | bash ./ci/docker/test_image.sh + + # Host-side startup smoke: exercises ENTRYPOINT/CMD (REST) and + # CUOPT_SERVER_TYPE=grpc. The jobs above run *inside* the image as a GHA + # container and never launch the servers, so they cannot catch packaging + # gaps such as UBI10's RHEL lib/lib64 NCCL path miss. + smoke: + name: smoke-images (${{ inputs.ARCH }}, cuda${{ needs.prepare.outputs.CUDA_SHORT }}) + runs-on: "linux-${{ inputs.ARCH }}-gpu-a100-latest-1" + needs: prepare + steps: + - name: Checkout code repo + uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 + with: + fetch-depth: 0 + ref: ${{ inputs.sha }} + persist-credentials: false + - name: Smoke Ubuntu image (REST + gRPC) + env: + IMAGE_TAG_PREFIX: ${{ inputs.IMAGE_TAG_PREFIX }} + CUDA_SHORT: ${{ needs.prepare.outputs.CUDA_SHORT }} + PYTHON_SHORT: ${{ needs.prepare.outputs.PYTHON_SHORT }} + run: | + bash ./ci/docker/smoke_image.sh \ + "nvidia/cuopt:${IMAGE_TAG_PREFIX}-cuda${CUDA_SHORT}-py${PYTHON_SHORT}" + - name: Smoke UBI10 image (REST + gRPC) + if: ${{ startsWith(inputs.CUDA_VER, '13.') }} + env: + IMAGE_TAG_PREFIX: ${{ inputs.IMAGE_TAG_PREFIX }} + CUDA_SHORT: ${{ needs.prepare.outputs.CUDA_SHORT }} + run: | + bash ./ci/docker/smoke_image.sh \ + "nvidia/cuopt:${IMAGE_TAG_PREFIX}-cuda${CUDA_SHORT}-ubi10" diff --git a/ci/docker/README.md b/ci/docker/README.md index 8886d4147d..9abf862ce6 100644 --- a/ci/docker/README.md +++ b/ci/docker/README.md @@ -26,3 +26,17 @@ docker run -it --rm --gpus all -u root --volume $PWD:/repo -w /repo --entrypoint # UBI10 image docker run -it --rm --gpus all -u root --volume $PWD:/repo -w /repo --entrypoint "/bin/bash" nvidia/cuopt:[TAG]-ubi10 ./ci/docker/test_image.sh ``` + +### Startup smoke (REST + gRPC) + +`test_image.sh` runs pytest inside the image and does not launch the servers. +To verify the published entrypoint starts both the default REST server and the +gRPC server (`CUOPT_SERVER_TYPE=grpc`): + +```bash +./ci/docker/smoke_image.sh nvidia/cuopt:[TAG] +./ci/docker/smoke_image.sh nvidia/cuopt:[TAG]-ubi10 +``` + +CI runs this for both variants after the multiarch manifests are published +(see `.github/workflows/test_images.yaml` job `smoke`). diff --git a/ci/docker/smoke_image.sh b/ci/docker/smoke_image.sh new file mode 100755 index 0000000000..c003c1f93e --- /dev/null +++ b/ci/docker/smoke_image.sh @@ -0,0 +1,104 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Smoke-test that a published cuOpt image starts its default REST server and +# the gRPC server via CUOPT_SERVER_TYPE=grpc. Runs on the host with docker so +# the real ENTRYPOINT/CMD path is exercised (unlike test_image.sh, which runs +# inside a GHA job container and never launches the servers). +# +# Usage (any published or locally built tag): +# ./ci/docker/smoke_image.sh nvidia/cuopt:[TAG] +# ./ci/docker/smoke_image.sh nvidia/cuopt:[TAG]-ubi10 +# +# Env: +# SMOKE_TIMEOUT_SECS Max seconds to wait for listen (default: 90) +# SMOKE_GPU_ARGS Docker GPU flags (default: --gpus all) + +set -euo pipefail + +IMAGE="${1:?usage: $0 }" +TIMEOUT_SECS="${SMOKE_TIMEOUT_SECS:-90}" +# shellcheck disable=SC2206 +GPU_ARGS=(${SMOKE_GPU_ARGS:---gpus all}) + +pass() { printf 'PASS %s\n' "$*"; } +fail() { printf 'FAIL %s\n' "$*" >&2; exit 1; } +info() { printf 'INFO %s\n' "$*"; } + +smoke_one() { + local label="$1" + local expect_re="$2" + shift 2 + # Remaining args are extra docker run flags (e.g. -e CUOPT_SERVER_TYPE=grpc). + + local name log cid i + name="cuopt-smoke-${label}-$$" + log="$(mktemp)" + cid="" + + smoke_fail() { + printf 'FAIL %s\n' "$*" >&2 + } + + cleanup() { + if [[ -n "${cid}" ]]; then + docker rm -f "${cid}" >/dev/null 2>&1 || true + fi + rm -f "${log}" + } + trap cleanup RETURN + + info "Starting ${label} server from ${IMAGE}" + # Do not use --rm: a fast crash (e.g. missing libnccl.so.2) would delete the + # container before we can collect logs. + if ! cid="$(docker run -d --name "${name}" "${GPU_ARGS[@]}" "$@" "${IMAGE}")"; then + smoke_fail "${label}: docker run failed" + return 1 + fi + + for ((i = 1; i <= TIMEOUT_SECS; i++)); do + docker logs "${cid}" >"${log}" 2>&1 || true + + if grep -qiE 'error while loading shared libraries|libnccl\.so|FATAL FIPS SELFTEST|OpenSSL internal error' "${log}"; then + echo "----- ${label} logs -----" + cat "${log}" + smoke_fail "${label}: loader/crypto failure while starting" + return 1 + fi + + if grep -qE "${expect_re}" "${log}"; then + pass "${label}: matched /${expect_re}/" + return 0 + fi + + # Container exited before listen — dump logs and fail. + if ! docker inspect -f '{{.State.Running}}' "${cid}" 2>/dev/null | grep -qx true; then + echo "----- ${label} logs -----" + cat "${log}" + smoke_fail "${label}: container exited before becoming ready" + return 1 + fi + + sleep 1 + done + + echo "----- ${label} logs -----" + cat "${log}" + smoke_fail "${label}: timed out after ${TIMEOUT_SECS}s waiting for /${expect_re}/" + return 1 +} + +info "Pulling ${IMAGE}" +if ! docker pull "${IMAGE}"; then + if docker image inspect "${IMAGE}" >/dev/null 2>&1; then + info "Pull failed; using local image ${IMAGE}" + else + fail "Pull failed and no local image named ${IMAGE}" + fi +fi + +smoke_one rest 'Uvicorn running on' +smoke_one grpc 'Listening on' -e CUOPT_SERVER_TYPE=grpc + +pass "Smoke OK for ${IMAGE}" From 52d6f08e1807a25edd16d00e0f4889fd0bd97cef Mon Sep 17 00:00:00 2001 From: Ramakrishna Prabhu <42624703+ramakrishnap-nv@users.noreply.github.com> Date: Wed, 9 Sep 2026 10:32:29 -0500 Subject: [PATCH 052/113] refactor: keep the GPU warm-start path out of host-only translation units (#1803) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `populate_from_data_model_view()` handled the GPU and CPU warm-start directions in one inlined `if/else`. Only the GPU direction needs a device, but the compiler instantiated both into every translation unit that includes the header, pulling `convert_to_gpu_warmstart` and friends into code that never touches a GPU. Split into three helpers, selected by a `kHostOnly` template parameter dispatched with `if constexpr` so a host-only caller never instantiates the GPU branch: - `apply_warmstart_gpu_target()` — real handle; defined in libcuopt - `apply_warmstart_cpu_target_with_device()` — null handle, caller has a device; defined in libcuopt - `apply_warmstart_cpu_target()` — host-only caller; inline in the header Two CPU-target variants because a `kHostOnly` caller cannot hold device-resident warm start, while a normal caller passing `handle == nullptr` can (`cython_solve.cu:181`) and needs the D2H copy. Also moves the trivial warm-start accessors into `solver_settings_accessors.cpp` so host-only consumers resolve them without the CUDA translation unit. Third of four steps toward a CUDA-free client library, after #1801 and #1802. ## Issue Follow-up for test coverage: #1867 Authors: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) Approvers: - Rajesh Gandham (https://github.com/rg20) - Trevor McKay (https://github.com/tmckayus) URL: https://github.com/NVIDIA/cuopt/pull/1803 --- .../optimization_problem_utils.hpp | 109 ++++++++++-------- cpp/src/grpc/client/cython_grpc_client.cpp | 3 +- cpp/src/pdlp/CMakeLists.txt | 1 + cpp/src/pdlp/optimization_problem.cu | 78 +++++++++++++ cpp/src/pdlp/solver_settings.cu | 21 ---- cpp/src/pdlp/solver_settings_accessors.cpp | 68 +++++++++++ 6 files changed, 208 insertions(+), 72 deletions(-) create mode 100644 cpp/src/pdlp/solver_settings_accessors.cpp diff --git a/cpp/include/cuopt/mathematical_optimization/optimization_problem_utils.hpp b/cpp/include/cuopt/mathematical_optimization/optimization_problem_utils.hpp index b1f81b8edb..dd251bd881 100644 --- a/cpp/include/cuopt/mathematical_optimization/optimization_problem_utils.hpp +++ b/cpp/include/cuopt/mathematical_optimization/optimization_problem_utils.hpp @@ -137,6 +137,54 @@ void populate_from_mps_data_model(optimization_problem_interface_t* pr } } +/** + * @brief Copy warm-start data into the form a GPU solve needs (H2D / view->device_uvector). + * + * Declared here, defined in libcuopt (optimization_problem.cu): it touches device memory, + * so keeping it out-of-line is what lets CUDA-free consumers of this header link without + * a CUDA runtime. Only call it with a real handle. + */ +template +void copy_warmstart_data_to_device(solver_settings_t& solver_settings, + const raft::handle_t* handle); + +/** + * @brief Copy warm-start data into the form a CPU / remote solve needs, including a + * device-to-host copy when the warm start is device-resident. + * + * Declared here, defined in libcuopt (optimization_problem.cu). This is the null-handle + * path for a normal (non host-only) caller: it has a device, so its settings may hold + * device_uvector-backed warm start that must be brought to host before a remote solve. + */ +template +void copy_warmstart_data_to_host(solver_settings_t& solver_settings); + +/** + * @brief Copy a host-span warm-start view into the form a CPU / remote solve needs. + * + * Handles the two cases reachable without a device: warm start already in host form + * (nothing to do), and a warm-start view over host spans (copy it). + * + * Deliberately does NOT handle device-resident warm start -- that needs a D2H copy and + * therefore CUDA. Callers that might be holding device data must use + * copy_warmstart_data_to_host() instead; only a kHostOnly caller, which by construction + * has no device to have populated it, may use this one. + */ +template +void copy_warmstart_view_to_host(solver_settings_t& solver_settings) +{ + auto& pdlp = solver_settings.get_pdlp_settings(); + + if (pdlp.get_cpu_pdlp_warm_start_data().is_populated()) { return; } + + // Warmstart view (host spans from Cython) -> CPU backend: copy directly, no CUDA needed. + if (solver_settings.get_pdlp_warm_start_data_view() + .last_restart_duality_gap_dual_solution_.size() > 0) { + pdlp.get_cpu_pdlp_warm_start_data() = + cpu_pdlp_warm_start_data_t(solver_settings.get_pdlp_warm_start_data_view()); + } +} + /** * @brief Transfer parsed MPS/QPS storage into a CPU-backed problem without copying payload arrays. * @@ -176,7 +224,7 @@ void adopt_from_mps_data_model(optimization_problem_interface_t* probl * @param[in] solver_settings Optional solver settings (for warmstart data, GPU only) * @param[in] handle Optional RAFT handle (for warmstart data, GPU only) */ -template +template void populate_from_data_model_view( optimization_problem_interface_t* problem, cuopt::mathematical_optimization::io::data_model_view_t* data_model, @@ -209,57 +257,18 @@ void populate_from_data_model_view( problem->set_objective_scaling_factor(data_model->get_objective_scaling_factor()); problem->set_objective_offset(data_model->get_objective_offset()); - // Handle warmstart data with GPU↔CPU conversion if needed + // `if constexpr`, not a runtime branch: a kHostOnly caller never *instantiates* the GPU + // helper, so it emits no reference to it and needs no CUDA runtime to link. if (solver_settings != nullptr) { - bool target_is_gpu = (handle != nullptr); - - // Check which warmstart type is populated - // Note: Python sets the VIEW (spans), so check both view and data for GPU warmstart - // CPU warmstart is set directly in the data structure - bool has_gpu_warmstart_view = (solver_settings->get_pdlp_warm_start_data_view() - .last_restart_duality_gap_dual_solution_.size() > 0); - bool has_gpu_warmstart_data = - solver_settings->get_pdlp_settings().get_pdlp_warm_start_data().is_populated(); - bool has_cpu_warmstart = - solver_settings->get_pdlp_settings().get_cpu_pdlp_warm_start_data().is_populated(); - - bool has_gpu_warmstart = has_gpu_warmstart_view || has_gpu_warmstart_data; - - if (has_gpu_warmstart || has_cpu_warmstart) { - if (target_is_gpu) { - // Target is GPU backend - if (has_gpu_warmstart_view) { - // GPU warmstart from Python → GPU backend: copy view (spans) to data (device_uvectors) - // Python sets the view (spans over cuDF), but solver needs device_uvectors - pdlp_warm_start_data_t pdlp_warm_start_data( - solver_settings->get_pdlp_warm_start_data_view(), handle->get_stream()); - solver_settings->get_pdlp_settings().set_pdlp_warm_start_data(pdlp_warm_start_data); - } else if (has_gpu_warmstart_data) { - // GPU warmstart from C++ API → GPU backend: data already set, nothing to do - // The device_uvectors are already populated in the settings - } else { - // CPU warmstart → GPU backend: convert H2D - pdlp_warm_start_data_t gpu_warmstart = convert_to_gpu_warmstart( - solver_settings->get_pdlp_settings().get_cpu_pdlp_warm_start_data(), - handle->get_stream()); - solver_settings->get_pdlp_settings().set_pdlp_warm_start_data(gpu_warmstart); - } + if constexpr (kHostOnly) { + copy_warmstart_view_to_host(*solver_settings); + } else { + if (handle != nullptr) { + copy_warmstart_data_to_device(*solver_settings, handle); } else { - // Target is CPU backend (remote execution) - if (has_cpu_warmstart) { - // CPU warmstart → CPU backend: data already in correct form, nothing to do - } else if (has_gpu_warmstart_view) { - // Warmstart view (host spans from Cython) → CPU backend: copy directly, no CUDA needed - solver_settings->get_pdlp_settings().get_cpu_pdlp_warm_start_data() = - cpu_pdlp_warm_start_data_t(solver_settings->get_pdlp_warm_start_data_view()); - } else { - // GPU warmstart data (device_uvectors) → CPU backend: convert D2H - auto& gpu_ws = solver_settings->get_pdlp_settings().get_pdlp_warm_start_data(); - cpu_pdlp_warm_start_data_t cpu_warmstart = - convert_to_cpu_warmstart(gpu_ws, gpu_ws.current_primal_solution_.stream()); - solver_settings->get_pdlp_settings().get_cpu_pdlp_warm_start_data() = - std::move(cpu_warmstart); - } + // No handle, but this caller has a device: the warm start may be device-resident, + // so it needs the variant that can copy it back to host. + copy_warmstart_data_to_host(*solver_settings); } } } diff --git a/cpp/src/grpc/client/cython_grpc_client.cpp b/cpp/src/grpc/client/cython_grpc_client.cpp index 74409fc93e..cc6a6fae15 100644 --- a/cpp/src/grpc/client/cython_grpc_client.cpp +++ b/cpp/src/grpc/client/cython_grpc_client.cpp @@ -109,7 +109,8 @@ grpc_submit_result_t grpc_python_client_t::submit( } cuopt::mathematical_optimization::cpu_optimization_problem_t cpu_problem; - cuopt::mathematical_optimization::populate_from_data_model_view( + // kHostOnly=true: remote client, so the GPU warm-start path is unreachable here. + cuopt::mathematical_optimization::populate_from_data_model_view( &cpu_problem, data_model, settings, nullptr); const bool is_mip = diff --git a/cpp/src/pdlp/CMakeLists.txt b/cpp/src/pdlp/CMakeLists.txt index b6a1f8a46d..1b1439b350 100644 --- a/cpp/src/pdlp/CMakeLists.txt +++ b/cpp/src/pdlp/CMakeLists.txt @@ -6,6 +6,7 @@ # Core LP files always included set(LP_CORE_FILES ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings.cu + ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings_accessors.cpp ${CMAKE_CURRENT_SOURCE_DIR}/optimization_problem.cu ${CMAKE_CURRENT_SOURCE_DIR}/cpu_optimization_problem.cpp ${CMAKE_CURRENT_SOURCE_DIR}/cpu_optimization_problem_to_gpu.cpp diff --git a/cpp/src/pdlp/optimization_problem.cu b/cpp/src/pdlp/optimization_problem.cu index 87fe438ca4..a80e0cdbcc 100644 --- a/cpp/src/pdlp/optimization_problem.cu +++ b/cpp/src/pdlp/optimization_problem.cu @@ -1637,4 +1637,82 @@ template CUOPT_EXPORT optimization_problem_t rmm::cuda_stream_view) const; #endif +// GPU-target warm-start handling, declared in optimization_problem_utils.hpp. +// +// Defined here rather than inline in the header so that CUDA-free consumers of that +// header (the gRPC client in cuopt_client) never instantiate the device conversions. +template +void copy_warmstart_data_to_device(solver_settings_t& solver_settings, + const raft::handle_t* handle) +{ + auto& pdlp = solver_settings.get_pdlp_settings(); + + const bool has_view = (solver_settings.get_pdlp_warm_start_data_view() + .last_restart_duality_gap_dual_solution_.size() > 0); + const bool has_device_data = pdlp.get_pdlp_warm_start_data().is_populated(); + const bool has_host_data = pdlp.get_cpu_pdlp_warm_start_data().is_populated(); + + if (!has_view && !has_device_data && !has_host_data) { return; } + + if (has_view) { + // Warmstart from Python (spans over cuDF) -> solver needs device_uvectors. + pdlp_warm_start_data_t warm_start(solver_settings.get_pdlp_warm_start_data_view(), + handle->get_stream()); + pdlp.set_pdlp_warm_start_data(warm_start); + } else if (has_device_data) { + // Already device-resident from the C++ API: nothing to do. + } else { + // Host warmstart -> GPU backend: convert H2D. + pdlp_warm_start_data_t warm_start = + convert_to_gpu_warmstart(pdlp.get_cpu_pdlp_warm_start_data(), handle->get_stream()); + pdlp.set_pdlp_warm_start_data(warm_start); + } +} + +// Null-handle CPU-target warm-start handling for callers that do have a device. +// +// Mirrors copy_warmstart_view_to_host() but adds the case that one cannot handle: warm +// start already sitting in device_uvectors, which needs a D2H copy before a remote solve. +// Dropping this silently loses a user's warm start on the +// populate_from_data_model_view(..., handle=nullptr) path (see cython_solve.cu). +template +void copy_warmstart_data_to_host(solver_settings_t& solver_settings) +{ + auto& pdlp = solver_settings.get_pdlp_settings(); + + // Already in host form. + if (pdlp.get_cpu_pdlp_warm_start_data().is_populated()) { return; } + + // Warm-start view (host spans from Cython) -> CPU backend: copy directly, no CUDA needed. + if (solver_settings.get_pdlp_warm_start_data_view() + .last_restart_duality_gap_dual_solution_.size() > 0) { + pdlp.get_cpu_pdlp_warm_start_data() = + cpu_pdlp_warm_start_data_t(solver_settings.get_pdlp_warm_start_data_view()); + return; + } + + // Device-resident warm start -> CPU backend: convert D2H. + auto& gpu_ws = pdlp.get_pdlp_warm_start_data(); + if (gpu_ws.is_populated()) { + pdlp.get_cpu_pdlp_warm_start_data() = + convert_to_cpu_warmstart(gpu_ws, gpu_ws.current_primal_solution_.stream()); + } +} + +// MIP_INSTANTIATE_FLOAT, not the wider MIP_INSTANTIATE_FLOAT || PDLP_INSTANTIATE_FLOAT used +// elsewhere: both helpers call solver_settings_t::get_pdlp_settings() and +// ::get_pdlp_warm_start_data_view(), which math_optimization/solver_settings.cpp only +// instantiates under MIP_INSTANTIATE_FLOAT. Widening the guard here without widening it +// there leaves libcuopt.so with undefined references to those accessors. +#if MIP_INSTANTIATE_FLOAT +template CUOPT_EXPORT void copy_warmstart_data_to_device(solver_settings_t&, + const raft::handle_t*); +template CUOPT_EXPORT void copy_warmstart_data_to_host(solver_settings_t&); +#endif +#if MIP_INSTANTIATE_DOUBLE +template CUOPT_EXPORT void copy_warmstart_data_to_device(solver_settings_t&, + const raft::handle_t*); +template CUOPT_EXPORT void copy_warmstart_data_to_host(solver_settings_t&); +#endif + } // namespace cuopt::mathematical_optimization diff --git a/cpp/src/pdlp/solver_settings.cu b/cpp/src/pdlp/solver_settings.cu index 33d8f1a64b..6ae0a84828 100644 --- a/cpp/src/pdlp/solver_settings.cu +++ b/cpp/src/pdlp/solver_settings.cu @@ -394,27 +394,6 @@ pdlp_warm_start_data_t& pdlp_solver_settings_t::get_pdlp_war return pdlp_warm_start_data_; } -template -const cpu_pdlp_warm_start_data_t& -pdlp_solver_settings_t::get_cpu_pdlp_warm_start_data() const noexcept -{ - return cpu_pdlp_warm_start_data_; -} - -template -cpu_pdlp_warm_start_data_t& -pdlp_solver_settings_t::get_cpu_pdlp_warm_start_data() noexcept -{ - return cpu_pdlp_warm_start_data_; -} - -template -const pdlp_warm_start_data_view_t& -pdlp_solver_settings_t::get_pdlp_warm_start_data_view() const noexcept -{ - return pdlp_warm_start_data_view_; -} - #if MIP_INSTANTIATE_FLOAT || PDLP_INSTANTIATE_FLOAT template class CUOPT_EXPORT pdlp_solver_settings_t; #endif diff --git a/cpp/src/pdlp/solver_settings_accessors.cpp b/cpp/src/pdlp/solver_settings_accessors.cpp new file mode 100644 index 0000000000..9d5efc7f9a --- /dev/null +++ b/cpp/src/pdlp/solver_settings_accessors.cpp @@ -0,0 +1,68 @@ +/* clang-format off */ +/* + * SPDX-FileCopyrightText: Copyright (c) 2024-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +/* clang-format on */ + +// Warm-start accessors of pdlp_solver_settings_t, split out of solver_settings.cu. +// +// These are trivial `return member_;` getters -- they hand back a reference and emit no +// device code, even where the referent is a GPU type. The gRPC client needs them, so they +// build into the CUDA-free cuopt_client library while the rest of the class (which does +// real thrust/rmm work) stays in solver_settings.cu. +// +// Only these members are instantiated below, deliberately NOT `template class`: the class +// holds a pdlp_warm_start_data_t, so instantiating all of it here would pull in device +// ctor/dtor code that belongs in the CUDA TU. + +#include +#include + +// Required: the explicit instantiations below are guarded on MIP_INSTANTIATE_* / +// PDLP_INSTANTIATE_*. Without this header those macros are undefined, the guards +// evaluate false, and this TU silently compiles to zero symbols. +#include + +namespace cuopt::mathematical_optimization { + +template +const cpu_pdlp_warm_start_data_t& +pdlp_solver_settings_t::get_cpu_pdlp_warm_start_data() const noexcept +{ + return cpu_pdlp_warm_start_data_; +} + +template +cpu_pdlp_warm_start_data_t& +pdlp_solver_settings_t::get_cpu_pdlp_warm_start_data() noexcept +{ + return cpu_pdlp_warm_start_data_; +} + +template +const pdlp_warm_start_data_view_t& +pdlp_solver_settings_t::get_pdlp_warm_start_data_view() const noexcept +{ + return pdlp_warm_start_data_view_; +} + +#if MIP_INSTANTIATE_FLOAT || PDLP_INSTANTIATE_FLOAT +template CUOPT_EXPORT const cpu_pdlp_warm_start_data_t& +pdlp_solver_settings_t::get_cpu_pdlp_warm_start_data() const noexcept; +template CUOPT_EXPORT cpu_pdlp_warm_start_data_t& +pdlp_solver_settings_t::get_cpu_pdlp_warm_start_data() noexcept; +template CUOPT_EXPORT const pdlp_warm_start_data_view_t& +pdlp_solver_settings_t::get_pdlp_warm_start_data_view() const noexcept; +#endif + +#if MIP_INSTANTIATE_DOUBLE +template CUOPT_EXPORT const cpu_pdlp_warm_start_data_t& +pdlp_solver_settings_t::get_cpu_pdlp_warm_start_data() const noexcept; +template CUOPT_EXPORT cpu_pdlp_warm_start_data_t& +pdlp_solver_settings_t::get_cpu_pdlp_warm_start_data() noexcept; +template CUOPT_EXPORT const pdlp_warm_start_data_view_t& +pdlp_solver_settings_t::get_pdlp_warm_start_data_view() const noexcept; +#endif + +} // namespace cuopt::mathematical_optimization From 0a4415bdd262e350929f3969d17a9dd85f1d4061 Mon Sep 17 00:00:00 2001 From: Trevor McKay Date: Wed, 9 Sep 2026 12:49:47 -0400 Subject: [PATCH 053/113] extract result file support logic to be shared by proxy (#1869) This is part of the refactoring of the cuopt server to make logic available to a new proxy server. It moves the local file support logic into a common place. Authors: - Trevor McKay (https://github.com/tmckayus) Approvers: - Ishika Roy (https://github.com/Iroy30) URL: https://github.com/NVIDIA/cuopt/pull/1869 --- .../tests/test_file_path_validation.py | 97 -------- .../cuopt_server/tests/test_local_files.py | 216 ++++++++++++++++++ .../cuopt_server/utils/job_queue.py | 110 ++------- .../cuopt_server/utils/local_files.py | 183 +++++++++++++++ python/cuopt_server/cuopt_server/webserver.py | 69 +----- 5 files changed, 422 insertions(+), 253 deletions(-) delete mode 100644 python/cuopt_server/cuopt_server/tests/test_file_path_validation.py create mode 100644 python/cuopt_server/cuopt_server/tests/test_local_files.py create mode 100644 python/cuopt_server/cuopt_server/utils/local_files.py diff --git a/python/cuopt_server/cuopt_server/tests/test_file_path_validation.py b/python/cuopt_server/cuopt_server/tests/test_file_path_validation.py deleted file mode 100644 index 57b8e50f21..0000000000 --- a/python/cuopt_server/cuopt_server/tests/test_file_path_validation.py +++ /dev/null @@ -1,97 +0,0 @@ -# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 - -import os - -import pytest -from fastapi import HTTPException - -from cuopt_server import webserver -from cuopt_server.utils import settings - - -@pytest.fixture(autouse=True) -def restore_data_dir(): - original_data_dir = settings.get_data_dir() - try: - yield - finally: - settings.set_data_dir(original_data_dir) - - -def test_validate_file_path_returns_file_if_relative_path_stays_in_data_dir( - tmp_path, -): - data_dir = tmp_path / "data" - data_dir.mkdir() - data_file = data_dir / "input.json" - data_file.write_text("{}", encoding="utf-8") - settings.set_data_dir(str(data_dir)) - - assert webserver.validate_file_path("input.json") == str(data_file) - - -def test_validate_file_path_rejects_unset_data_dir(): - settings.set_data_dir("") - - with pytest.raises(HTTPException) as exc_info: - webserver.validate_file_path("input.json") - - assert exc_info.value.status_code == 400 - assert "cuopt data directory not set" in exc_info.value.detail - - -def test_validate_file_path_rejects_absolute_path(tmp_path): - data_dir = tmp_path / "data" - data_dir.mkdir() - outside_file = tmp_path / "input.json" - outside_file.write_text("{}", encoding="utf-8") - settings.set_data_dir(str(data_dir)) - - with pytest.raises(HTTPException) as exc_info: - webserver.validate_file_path(str(outside_file)) - - assert exc_info.value.status_code == 400 - assert "relative to CUOPT_DATA_DIR" in exc_info.value.detail - - -def test_validate_file_path_rejects_parent_directory_escape(tmp_path): - data_dir = tmp_path / "data" - data_dir.mkdir() - outside_file = tmp_path / "input.json" - outside_file.write_text("{}", encoding="utf-8") - settings.set_data_dir(str(data_dir)) - - with pytest.raises(HTTPException) as exc_info: - webserver.validate_file_path("../input.json") - - assert exc_info.value.status_code == 400 - assert "stay inside CUOPT_DATA_DIR" in exc_info.value.detail - - -def test_validate_file_path_rejects_non_regular_file(tmp_path): - data_dir = tmp_path / "data" - data_dir.mkdir() - (data_dir / "input").mkdir() - settings.set_data_dir(str(data_dir)) - - with pytest.raises(HTTPException) as exc_info: - webserver.validate_file_path("input") - - assert exc_info.value.status_code == 400 - assert "not a regular file" in exc_info.value.detail - - -def test_validate_file_path_rejects_symlink_escape(tmp_path): - data_dir = tmp_path / "data" - data_dir.mkdir() - outside_file = tmp_path / "input.json" - outside_file.write_text("{}", encoding="utf-8") - os.symlink(outside_file, data_dir / "linked-input.json") - settings.set_data_dir(str(data_dir)) - - with pytest.raises(HTTPException) as exc_info: - webserver.validate_file_path("linked-input.json") - - assert exc_info.value.status_code == 400 - assert "stay inside CUOPT_DATA_DIR" in exc_info.value.detail diff --git a/python/cuopt_server/cuopt_server/tests/test_local_files.py b/python/cuopt_server/cuopt_server/tests/test_local_files.py new file mode 100644 index 0000000000..262d7a5fba --- /dev/null +++ b/python/cuopt_server/cuopt_server/tests/test_local_files.py @@ -0,0 +1,216 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import json +import os +import pickle +import stat + +import pytest +from fastapi import HTTPException + +from cuopt_server.utils import settings +from cuopt_server.utils.local_files import ( + decode_file_bytes, + file_result_message, + get_output_name, + load_optimization_file, + result_meets_threshold, + validate_file_path, + write_result_file, +) + + +@pytest.fixture(autouse=True) +def restore_data_dir(): + original_data_dir = settings.get_data_dir() + try: + yield + finally: + settings.set_data_dir(original_data_dir) + + +def test_get_output_name_empty_without_result_dir(): + assert get_output_name("", "problem.json", "out.json") == "" + + +def test_get_output_name_uses_requested_file(tmp_path): + result_dir = tmp_path / "results" + result_dir.mkdir() + assert get_output_name(str(result_dir), "problem.json", "out.bin") == ( + "out.bin" + ) + + +def test_get_output_name_rejects_result_path_escape(tmp_path): + result_dir = tmp_path / "results" + result_dir.mkdir() + name = get_output_name(str(result_dir), "problem.json", "../escape.bin") + assert name == "problem.json.result" + + +def test_get_output_name_from_data_file(tmp_path): + result_dir = tmp_path / "results" + result_dir.mkdir() + assert get_output_name(str(result_dir), "nested/problem.json", "") == ( + "problem.json.result" + ) + + +def test_validate_file_path_returns_file_in_data_dir(tmp_path): + data_dir = tmp_path / "data" + data_dir.mkdir() + data_file = data_dir / "input.json" + data_file.write_text("{}", encoding="utf-8") + settings.set_data_dir(str(data_dir)) + + assert validate_file_path("input.json") == str(data_file) + + +def test_validate_file_path_rejects_unset_data_dir(): + settings.set_data_dir("") + + with pytest.raises(HTTPException) as exc_info: + validate_file_path("input.json") + + assert exc_info.value.status_code == 400 + assert "cuopt data directory not set" in exc_info.value.detail + + +def test_validate_file_path_rejects_absolute_path(tmp_path): + data_dir = tmp_path / "data" + data_dir.mkdir() + outside_file = tmp_path / "input.json" + outside_file.write_text("{}", encoding="utf-8") + settings.set_data_dir(str(data_dir)) + + with pytest.raises(HTTPException) as exc_info: + validate_file_path(str(outside_file)) + + assert exc_info.value.status_code == 400 + assert "relative to CUOPT_DATA_DIR" in exc_info.value.detail + + +def test_validate_file_path_rejects_parent_directory_escape(tmp_path): + data_dir = tmp_path / "data" + data_dir.mkdir() + outside_file = tmp_path / "input.json" + outside_file.write_text("{}", encoding="utf-8") + settings.set_data_dir(str(data_dir)) + + with pytest.raises(HTTPException) as exc_info: + validate_file_path("../input.json") + + assert exc_info.value.status_code == 400 + assert "stay inside CUOPT_DATA_DIR" in exc_info.value.detail + + +def test_validate_file_path_rejects_non_regular_file(tmp_path): + data_dir = tmp_path / "data" + data_dir.mkdir() + (data_dir / "input").mkdir() + settings.set_data_dir(str(data_dir)) + + with pytest.raises(HTTPException) as exc_info: + validate_file_path("input") + + assert exc_info.value.status_code == 400 + assert "not a regular file" in exc_info.value.detail + + +def test_validate_file_path_rejects_symlink_escape(tmp_path): + data_dir = tmp_path / "data" + data_dir.mkdir() + outside_file = tmp_path / "input.json" + outside_file.write_text("{}", encoding="utf-8") + os.symlink(outside_file, data_dir / "linked-input.json") + settings.set_data_dir(str(data_dir)) + + with pytest.raises(HTTPException) as exc_info: + validate_file_path("linked-input.json") + + assert exc_info.value.status_code == 400 + assert "stay inside CUOPT_DATA_DIR" in exc_info.value.detail + + +def test_result_meets_threshold(): + assert result_meets_threshold("out", "/tmp", 250000, 250) is True + assert result_meets_threshold("out", "/tmp", 249999, 250) is False + assert result_meets_threshold("out", "/tmp", 1, 0) is True + assert result_meets_threshold("", "/tmp", 10**9, 0) is False + assert result_meets_threshold("out", "", 10**9, 0) is False + + +def test_write_result_file_and_mode(tmp_path): + result_dir = tmp_path / "results" + result_dir.mkdir() + payload = b'{"ok": true}' + + path = write_result_file(str(result_dir), "sol.json", payload, mode=0o600) + + assert path == str(result_dir / "sol.json") + assert (result_dir / "sol.json").read_bytes() == payload + assert stat.S_IMODE((result_dir / "sol.json").stat().st_mode) == 0o600 + + +def test_file_result_message_includes_notes_and_warnings(): + assert file_result_message("sol.json") == {"result_file": "sol.json"} + assert file_result_message("sol.json", warnings=["w"], notes=["n"]) == { + "result_file": "sol.json", + "warnings": ["w"], + "notes": ["n"], + } + + +def test_load_optimization_file_json(tmp_path): + problem = {"csr_constraint_matrix": {"offsets": [0, 1]}} + path = tmp_path / "problem.json" + path.write_text(json.dumps(problem), encoding="utf-8") + + assert load_optimization_file(str(path)) == problem + + +def test_load_optimization_file_without_extension(tmp_path): + problem = {"task_data": {"task_locations": [1]}} + path = tmp_path / "problem" + path.write_text(json.dumps(problem), encoding="utf-8") + + assert load_optimization_file(str(path)) == problem + + +def test_load_pickle_appends_deprecation_warning(tmp_path): + problem = {"csr_constraint_matrix": {"offsets": [0, 1]}} + path = tmp_path / "problem.pickle" + path.write_bytes(pickle.dumps(problem)) + warnings = [] + + assert load_optimization_file(str(path), warnings) == problem + assert warnings == [ + "Pickle data format is deprecated. Use zlib, msgpack, or plain JSON" + ] + + +def test_load_pickle_forbidden_class(tmp_path): + path = tmp_path / "bad.pickle" + path.write_bytes(pickle.dumps({"obj": object()})) + + with pytest.raises(HTTPException) as exc_info: + load_optimization_file(str(path)) + + assert exc_info.value.status_code == 422 + + +def test_decode_unsupported_extension(): + with pytest.raises(ValueError, match="unsupported"): + decode_file_bytes("txt", b"{}") + + +def test_pickle_forbidden_without_extension_does_not_fall_through(tmp_path): + path = tmp_path / "noext" + path.write_bytes(pickle.dumps({"obj": object()})) + + with pytest.raises(HTTPException) as exc_info: + load_optimization_file(str(path)) + + assert exc_info.value.status_code == 422 + assert "forbidden" in exc_info.value.detail diff --git a/python/cuopt_server/cuopt_server/utils/job_queue.py b/python/cuopt_server/cuopt_server/utils/job_queue.py index 006bd87704..8505652e4d 100644 --- a/python/cuopt_server/cuopt_server/utils/job_queue.py +++ b/python/cuopt_server/cuopt_server/utils/job_queue.py @@ -1,18 +1,15 @@ # SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -import json import logging import multiprocessing import os import time import uuid -import zlib from multiprocessing import shared_memory from multiprocessing.resource_tracker import unregister from threading import Event, Lock -import msgpack import msgpack_numpy from fastapi import HTTPException from fastapi.responses import JSONResponse @@ -52,6 +49,12 @@ from cuopt_server.utils.linear_programming.data_transformation import ( transform_lp_data, ) +from cuopt_server.utils.local_files import ( + file_result_message, + load_optimization_file, + result_meets_threshold, + write_result_file, +) from cuopt_server.utils.logutil import message from cuopt_server.utils.routing.initial_solution import add_initial_sol @@ -571,14 +574,7 @@ def set_termination(self, status_code, error): self.set_done() def get_file_result(self, result): - r = {"result_file": self.resultfile} - if self.warnings: - logging.debug("adding warnings to file result") - r["warnings"] = self.warnings - if self.notes: - logging.debug("adding notes to file result") - r["notes"] = self.notes - return r + return file_result_message(self.resultfile, self.warnings, self.notes) def set_result(self, result): # We expect normal cuOpt responses for a binary job to be @@ -588,14 +584,13 @@ def set_result(self, result): file_result = None if isinstance(result, str) or isinstance(result, bytes): try: - if ( - self.resultfile - and self.resultdir - and self.data_size >= self.maxresult * 1000 + if result_meets_threshold( + self.resultfile, + self.resultdir, + self.data_size, + self.maxresult, ): # TODO: set a file extension based on rtype? - op = os.path.join(self.resultdir, self.resultfile) - logging.debug(f"Writing large result to disk {op}") file_result = self.get_file_result(result) s = None @@ -610,10 +605,12 @@ def set_result(self, result): # Make sure to eliminate shm slice ref in buf # and unlink in all cases try: - with open(op, "wb") as out: - out.write(buf) - if self.mode: - os.chmod(op, self.mode) + write_result_file( + self.resultdir, + self.resultfile, + buf, + self.mode, + ) finally: buf = None if s: @@ -1199,77 +1196,8 @@ def __init__( _ = cuoptDataInternal.parse_obj(data) self._read_wrapper_data(data) - def _try_extension(self, ext, raw_data): - if ext == "zlib": - data = json.loads(zlib.decompress(raw_data)) - logging.debug("zlib data") - elif ext == "msgpack": - data = msgpack.loads(raw_data, strict_map_key=False) - logging.debug("msgpack serialized data") - elif ext == "json": - data = json.loads(raw_data) - logging.debug("uncompressed data") - elif ext == "pickle": - data = cuopt_pickle_load(raw_data, kind="") - self.warnings.append( - "Pickle data format is deprecated. " - "Use zlib, msgpack, or plain JSON" - ) - logging.warning("pickle data is deprecated") - logging.debug("pickle data") - else: - raise ValueError( - f"File extension {ext} is unsupported. " - "Supported file extensions are " - ".json, .zlib, .msgpack, or .pickle" - ) - return data - def _resolve_job(self): - # read the data from the file - # if we have an extension, use it otherwise try everything - try: - ext = ( - self.file_path.split(".")[-1] if "." in self.file_path else "" - ) - read_begin = time.time() - with open(self.file_path, "rb") as f: - raw_data = f.read() - if ext: - data = self._try_extension(ext, raw_data) - else: - for e in ["msgpack", "json", "zlib", "pickle"]: - try: - data = self._try_extension(e, raw_data) - break - except PickleForbidden: - # In this case we know it loaded as pickle but - # it failed the class restrictions, no reason - # to try anything else - raise - - except Exception: - pass - else: - raise HTTPException( - status_code=422, - detail="unable to read " - "optimization data file, " - "no file extension present and failed to load " - "as any supported format", - ) - logging.debug( - f"Total file load time {time.time() - read_begin}" - ) - - except HTTPException: - raise - - except Exception as e: - raise HTTPException( - status_code=422, - detail="unable to read optimization data file, %s" % (str(e)), - ) + data = load_optimization_file(self.file_path, self.warnings) initial_solutions = [] for init_sol in self.init_sols: diff --git a/python/cuopt_server/cuopt_server/utils/local_files.py b/python/cuopt_server/cuopt_server/utils/local_files.py new file mode 100644 index 0000000000..e4fb7f9bde --- /dev/null +++ b/python/cuopt_server/cuopt_server/utils/local_files.py @@ -0,0 +1,183 @@ +# SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Host-filesystem helpers for cuOpt HTTP: stage a problem under +# CUOPT_DATA_DIR and optionally write a result under CUOPT_RESULT_DIR +# when the serialized response meets the maxresult threshold. + +import json +import logging +import os +import time +import uuid +import zlib + +import msgpack +from fastapi import HTTPException + +import cuopt_server.utils.settings as settings +from cuopt_server.utils.http_codec import ( + PickleForbidden, + cuopt_pickle_load, +) + + +def get_output_name(resultdir, CUOPT_DATA_FILE, CUOPT_RESULT_FILE): + # Reject paths that escape resultdir using canonicalized containment check. + if CUOPT_RESULT_FILE and resultdir: + root = os.path.realpath(resultdir) + candidate = os.path.realpath(os.path.join(root, CUOPT_RESULT_FILE)) + if ( + os.path.isabs(CUOPT_RESULT_FILE) + or os.path.commonpath([root, candidate]) != root + ): + CUOPT_RESULT_FILE = "" + if not resultdir: + res = "" + elif CUOPT_RESULT_FILE: + res = CUOPT_RESULT_FILE + elif CUOPT_DATA_FILE: + res = os.path.basename(CUOPT_DATA_FILE) + ".result" + else: + res = str(uuid.uuid4()) + return res + + +def validate_file_path(cuopt_data_file): + ddir = settings.get_data_dir() + if not ddir: + logging.error("cuopt data directory not set!") + raise HTTPException( + status_code=400, + detail="cuopt data directory not set", + ) + + if os.path.isabs(cuopt_data_file): + raise HTTPException( + status_code=400, + detail="cuopt-data-file must be relative to CUOPT_DATA_DIR", + ) + + root = os.path.realpath(ddir) + file_path = os.path.realpath(os.path.join(root, cuopt_data_file)) + if os.path.commonpath([root, file_path]) != root: + raise HTTPException( + status_code=400, + detail="cuopt-data-file must stay inside CUOPT_DATA_DIR", + ) + + if not os.path.exists(file_path): + logging.error("cuopt-data-file does not exist") + raise HTTPException( + status_code=400, + detail=f"specified data file does not exist: {cuopt_data_file}", + ) + + if not os.path.isfile(file_path): + logging.error("cuopt-data-file is not a regular file") + raise HTTPException( + status_code=400, + detail=( + f"specified data file is not a regular file: {cuopt_data_file}" + ), + ) + + return file_path + + +def result_meets_threshold(resultfile, resultdir, data_size, maxresult): + return bool(resultfile and resultdir and data_size >= maxresult * 1000) + + +def write_result_file(resultdir, resultfile, buf, mode=None): + op = os.path.join(resultdir, resultfile) + logging.debug(f"Writing large result to disk {op}") + with open(op, "wb") as out: + out.write(buf) + if mode: + os.chmod(op, mode) + return op + + +def file_result_message(resultfile, warnings=None, notes=None): + r = {"result_file": resultfile} + if warnings: + logging.debug("adding warnings to file result") + r["warnings"] = warnings + if notes: + logging.debug("adding notes to file result") + r["notes"] = notes + return r + + +def decode_file_bytes(ext, raw_data, warnings=None): + if ext == "zlib": + data = json.loads(zlib.decompress(raw_data)) + logging.debug("zlib data") + elif ext == "msgpack": + data = msgpack.loads(raw_data, strict_map_key=False) + logging.debug("msgpack serialized data") + elif ext == "json": + data = json.loads(raw_data) + logging.debug("uncompressed data") + elif ext == "pickle": + data = cuopt_pickle_load(raw_data, kind="") + if warnings is not None: + warnings.append( + "Pickle data format is deprecated. " + "Use zlib, msgpack, or plain JSON" + ) + logging.warning("pickle data is deprecated") + logging.debug("pickle data") + else: + raise ValueError( + f"File extension {ext} is unsupported. " + "Supported file extensions are " + ".json, .zlib, .msgpack, or .pickle" + ) + return data + + +def load_optimization_file(file_path, warnings=None): + # read the data from the file + # if we have an extension, use it otherwise try everything + try: + ext = file_path.split(".")[-1] if "." in file_path else "" + read_begin = time.time() + with open(file_path, "rb") as f: + raw_data = f.read() + if ext: + data = decode_file_bytes(ext, raw_data, warnings) + else: + for e in ["msgpack", "json", "zlib", "pickle"]: + try: + data = decode_file_bytes(e, raw_data, warnings) + break + except PickleForbidden: + # In this case we know it loaded as pickle but + # it failed the class restrictions, no reason + # to try anything else + raise + + except Exception: + pass + else: + raise HTTPException( + status_code=422, + detail="unable to read " + "optimization data file, " + "no file extension present and failed to load " + "as any supported format", + ) + logging.debug(f"Total file load time {time.time() - read_begin}") + + except HTTPException: + raise + + except Exception as e: + raise HTTPException( + status_code=422, + detail="unable to read optimization data file, %s" % (str(e)), + ) + + return data diff --git a/python/cuopt_server/cuopt_server/webserver.py b/python/cuopt_server/cuopt_server/webserver.py index a0874d3ad7..189606950a 100644 --- a/python/cuopt_server/cuopt_server/webserver.py +++ b/python/cuopt_server/cuopt_server/webserver.py @@ -78,6 +78,10 @@ mime_wild, mime_zlib, ) +from cuopt_server.utils.local_files import ( + get_output_name, + validate_file_path, +) from cuopt_server.utils.job_queue import ( BaseResult, BinaryJobResult, @@ -177,71 +181,6 @@ def health(): raise HTTPException(status_code=500, detail=f"{msg}") -# Get name for file that stores the result of Solve -def get_output_name(resultdir, CUOPT_DATA_FILE, CUOPT_RESULT_FILE): - # Reject paths that escape resultdir using canonicalized containment check. - if CUOPT_RESULT_FILE and resultdir: - root = os.path.realpath(resultdir) - candidate = os.path.realpath(os.path.join(root, CUOPT_RESULT_FILE)) - if ( - os.path.isabs(CUOPT_RESULT_FILE) - or os.path.commonpath([root, candidate]) != root - ): - CUOPT_RESULT_FILE = "" - if not resultdir: - res = "" - elif CUOPT_RESULT_FILE: - res = CUOPT_RESULT_FILE - elif CUOPT_DATA_FILE: - res = os.path.basename(CUOPT_DATA_FILE) + ".result" - else: - res = str(uuid.uuid4()) - return res - - -# Validate if given data file and file path exists -def validate_file_path(cuopt_data_file): - ddir = settings.get_data_dir() - if not ddir: - logging.error("cuopt data directory not set!") - raise HTTPException( - status_code=400, - detail="cuopt data directory not set", - ) - - if os.path.isabs(cuopt_data_file): - raise HTTPException( - status_code=400, - detail="cuopt-data-file must be relative to CUOPT_DATA_DIR", - ) - - root = os.path.realpath(ddir) - file_path = os.path.realpath(os.path.join(root, cuopt_data_file)) - if os.path.commonpath([root, file_path]) != root: - raise HTTPException( - status_code=400, - detail="cuopt-data-file must stay inside CUOPT_DATA_DIR", - ) - - if not os.path.exists(file_path): - logging.error("cuopt-data-file does not exist") - raise HTTPException( - status_code=400, - detail=f"specified data file does not exist: {cuopt_data_file}", - ) - - if not os.path.isfile(file_path): - logging.error("cuopt-data-file is not a regular file") - raise HTTPException( - status_code=400, - detail=( - f"specified data file is not a regular file: {cuopt_data_file}" - ), - ) - - return file_path - - app_exit = None job_queue = None abort_queue = None From e6196f540e659e444f3e7ee48dbb0380c2de45f2 Mon Sep 17 00:00:00 2001 From: Ishika Roy <41401566+Iroy30@users.noreply.github.com> Date: Wed, 9 Sep 2026 13:00:40 -0500 Subject: [PATCH 054/113] update variable type (#1783) Addresses #1736 ## Issue Authors: - Ishika Roy (https://github.com/Iroy30) - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) Approvers: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) - Trevor McKay (https://github.com/tmckayus) URL: https://github.com/NVIDIA/cuopt/pull/1783 --- .../cuopt/cuopt/linear_programming/problem.py | 58 ++++++++++++++++--- .../linear_programming/test_python_API.py | 46 +++++++++++++++ 2 files changed, 95 insertions(+), 9 deletions(-) diff --git a/python/cuopt/cuopt/linear_programming/problem.py b/python/cuopt/cuopt/linear_programming/problem.py index 2c495af53c..44f543896b 100644 --- a/python/cuopt/cuopt/linear_programming/problem.py +++ b/python/cuopt/cuopt/linear_programming/problem.py @@ -36,6 +36,25 @@ class VType(str, Enum): SEMI_CONTINUOUS = VType.SEMI_CONTINUOUS +def _to_vtype(value: VType | str | bytes) -> VType: + """ + Coerces a variable type to a :py:class:`VType` member. + Besides VType members, the single character codes are accepted as ``str`` + or ``bytes``. + """ + if isinstance(value, VType): + return value + try: + # UnicodeDecodeError and the enum lookup failure are both ValueError. + return VType(value.decode() if isinstance(value, bytes) else value) + except ValueError: + valid = ", ".join(repr(t.value) for t in VType) + raise ValueError( + f"Invalid variable type {value!r}. Expected a VType member or " + f"one of {valid}." + ) from None + + class CType(str, Enum): """ The sense of a constraint is either LE, GE or EQ. @@ -95,8 +114,9 @@ class Variable: ---------- VariableName : str Name of the Variable. - VariableType : CONTINUOUS, INTEGER, or SEMI_CONTINUOUS - Variable type. + VariableType : VType + CONTINUOUS, INTEGER, or SEMI_CONTINUOUS. Assigning a ``str`` or + ``bytes`` character code converts it; anything else raises ValueError. LB : float Lower Bound of the Variable. UB : float @@ -141,6 +161,20 @@ def __init__( self.VariableName = vname self.MIPStart = float("nan") + @property + def VariableType(self) -> VType: + """ + Type of the variable. + """ + return self._variable_type + + @VariableType.setter + def VariableType(self, value: VType | str | bytes) -> None: + """ + Sets the type of the variable, coercing character codes to VType. + """ + self._variable_type = _to_vtype(value) + def __setattr__(self, name, value): object.__setattr__(self, name, value) # Solution fields are written in hot loops; skip tracking lookups. @@ -202,16 +236,20 @@ def getUpperBound(self): """ return self.UB - def setVariableType(self, val): + def setVariableType(self, val: VType | str | bytes) -> None: """ - Sets the variable type of the variable. - Variable types can be CONTINUOUS, INTEGER, or SEMI_CONTINUOUS. + Sets the variable type of the variable, equivalent to assigning + :py:attr:`VariableType`. + Variable types can be CONTINUOUS, INTEGER, or SEMI_CONTINUOUS, or the + equivalent character code as ``str`` or ``bytes``. + Raises ValueError for any other value. """ self.VariableType = val - def getVariableType(self): + def getVariableType(self) -> VType: """ - Returns the type of the variable. + Returns the type of the variable, equivalent to reading + :py:attr:`VariableType`. """ return self.VariableType @@ -2392,9 +2430,11 @@ def NumNZs(self): @property def IsMIP(self): - # Returns if the problem is a MIP problem. + """ + Returns True if any variable is integer or semi-continuous. + """ for var in self.vars: - if var.VariableType in ("I", "S", b"I", b"S"): + if var.VariableType in (INTEGER, SEMI_CONTINUOUS): return True return False diff --git a/python/cuopt/cuopt/tests/linear_programming/test_python_API.py b/python/cuopt/cuopt/tests/linear_programming/test_python_API.py index 1984f9ab72..04b4b9cf85 100644 --- a/python/cuopt/cuopt/tests/linear_programming/test_python_API.py +++ b/python/cuopt/cuopt/tests/linear_programming/test_python_API.py @@ -167,6 +167,52 @@ def test_constraint_duplicate_terms_slack(): assert c.Slack == pytest.approx(6.0) +def test_variable_type_is_normalized(): + """Every entry path stores a VType member and rejects other values.""" + prob = Problem() + from_enum = prob.addVariable(vtype=INTEGER) + from_str = prob.addVariable(vtype="I") + from_bytes = prob.addVariable(vtype=b"I") + default = prob.addVariable() + + for var in (from_enum, from_str, from_bytes): + assert var.VariableType is VType.INTEGER + assert default.VariableType is VType.CONTINUOUS + assert prob.IsMIP + + # Both the setter and direct assignment normalize. + from_str.setVariableType(b"S") + assert from_str.VariableType is VType.SEMI_CONTINUOUS + from_bytes.VariableType = "C" + assert from_bytes.VariableType is VType.CONTINUOUS + + with pytest.raises(ValueError): + from_enum.setVariableType(7) + + +def test_variable_type_normalized_from_mps(tmp_path): + """MPS parsing yields VType members and keeps each column's own type.""" + prob = Problem("mip") + x = prob.addVariable(lb=0.0, ub=10.0, vtype=INTEGER, name="x") + y = prob.addVariable(lb=0.0, ub=10.0, name="y") + prob.addConstraint(x + y <= 5, name="c") + prob.setObjective(x + y, sense=MAXIMIZE) + + path = str(tmp_path / "mip.mps") + prob.writeMPS(path) + + loaded = Problem.read(path) + # writeMPS emits integer columns inside INTORG/INTEND markers, so the + # column order is not preserved; key by name. `is` rather than `==` + # because VType subclasses str: "I" == VType.INTEGER. + types = { + v.getVariableName(): v.VariableType for v in loaded.getVariables() + } + assert types["x"] is VType.INTEGER + assert types["y"] is VType.CONTINUOUS + assert loaded.IsMIP + + def test_semi_continuous_variable(): prob = Problem("Semi-continuous") x = prob.addVariable(lb=5.0, ub=10.0, vtype=SEMI_CONTINUOUS, name="x") From cc21c5c6fccbdbcb40818d379c00f91eb47c721b Mon Sep 17 00:00:00 2001 From: Trevor McKay Date: Wed, 9 Sep 2026 16:12:39 -0400 Subject: [PATCH 055/113] Move deprecated server modules into utils/deprecated. (#1874) These files will not be used by the proxy server; move them so the old HTTP server can be deleted as a unit at a future date. Authors: - Trevor McKay (https://github.com/tmckayus) Approvers: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) URL: https://github.com/NVIDIA/cuopt/pull/1874 --- .../cuopt_server/cuopt_server/utils/deprecated/__init__.py | 6 ++++++ .../cuopt_server/utils/{ => deprecated}/billing_data.py | 0 .../utils/deprecated/linear_programming/__init__.py | 2 ++ .../linear_programming/generate_metadata.py | 0 .../cuopt_server/utils/{ => deprecated}/mock_store.py | 4 ++-- .../cuopt_server/utils/{ => deprecated}/response_util.py | 6 ++++-- .../cuopt_server/utils/{ => deprecated}/result_store.py | 0 .../cuopt_server/utils/deprecated/routing/__init__.py | 2 ++ .../utils/{ => deprecated}/routing/generate_metadata.py | 0 .../cuopt_server/utils/{ => deprecated}/sema.py | 0 10 files changed, 16 insertions(+), 4 deletions(-) create mode 100644 python/cuopt_server/cuopt_server/utils/deprecated/__init__.py rename python/cuopt_server/cuopt_server/utils/{ => deprecated}/billing_data.py (100%) create mode 100644 python/cuopt_server/cuopt_server/utils/deprecated/linear_programming/__init__.py rename python/cuopt_server/cuopt_server/utils/{ => deprecated}/linear_programming/generate_metadata.py (100%) rename python/cuopt_server/cuopt_server/utils/{ => deprecated}/mock_store.py (93%) rename python/cuopt_server/cuopt_server/utils/{ => deprecated}/response_util.py (79%) rename python/cuopt_server/cuopt_server/utils/{ => deprecated}/result_store.py (100%) create mode 100644 python/cuopt_server/cuopt_server/utils/deprecated/routing/__init__.py rename python/cuopt_server/cuopt_server/utils/{ => deprecated}/routing/generate_metadata.py (100%) rename python/cuopt_server/cuopt_server/utils/{ => deprecated}/sema.py (100%) diff --git a/python/cuopt_server/cuopt_server/utils/deprecated/__init__.py b/python/cuopt_server/cuopt_server/utils/deprecated/__init__.py new file mode 100644 index 0000000000..46cc85356d --- /dev/null +++ b/python/cuopt_server/cuopt_server/utils/deprecated/__init__.py @@ -0,0 +1,6 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Legacy-only cuOpt HTTP server modules. Permanent utils must not import +# this package. Deleting the old server is: remove cuopt_service.py, +# webserver.py, this package, and tests/deprecated/. diff --git a/python/cuopt_server/cuopt_server/utils/billing_data.py b/python/cuopt_server/cuopt_server/utils/deprecated/billing_data.py similarity index 100% rename from python/cuopt_server/cuopt_server/utils/billing_data.py rename to python/cuopt_server/cuopt_server/utils/deprecated/billing_data.py diff --git a/python/cuopt_server/cuopt_server/utils/deprecated/linear_programming/__init__.py b/python/cuopt_server/cuopt_server/utils/deprecated/linear_programming/__init__.py new file mode 100644 index 0000000000..d51c4fe1e0 --- /dev/null +++ b/python/cuopt_server/cuopt_server/utils/deprecated/linear_programming/__init__.py @@ -0,0 +1,2 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 diff --git a/python/cuopt_server/cuopt_server/utils/linear_programming/generate_metadata.py b/python/cuopt_server/cuopt_server/utils/deprecated/linear_programming/generate_metadata.py similarity index 100% rename from python/cuopt_server/cuopt_server/utils/linear_programming/generate_metadata.py rename to python/cuopt_server/cuopt_server/utils/deprecated/linear_programming/generate_metadata.py diff --git a/python/cuopt_server/cuopt_server/utils/mock_store.py b/python/cuopt_server/cuopt_server/utils/deprecated/mock_store.py similarity index 93% rename from python/cuopt_server/cuopt_server/utils/mock_store.py rename to python/cuopt_server/cuopt_server/utils/deprecated/mock_store.py index d905f862f5..a8fd210310 100644 --- a/python/cuopt_server/cuopt_server/utils/mock_store.py +++ b/python/cuopt_server/cuopt_server/utils/deprecated/mock_store.py @@ -1,9 +1,9 @@ -# SPDX-FileCopyrightText: Copyright (c) 2023-2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-FileCopyrightText: Copyright (c) 2023-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 import logging -from cuopt_server.utils.result_store import ResultStore +from cuopt_server.utils.deprecated.result_store import ResultStore class MockStore(ResultStore): diff --git a/python/cuopt_server/cuopt_server/utils/response_util.py b/python/cuopt_server/cuopt_server/utils/deprecated/response_util.py similarity index 79% rename from python/cuopt_server/cuopt_server/utils/response_util.py rename to python/cuopt_server/cuopt_server/utils/deprecated/response_util.py index 100eaf8470..56b7e0658e 100644 --- a/python/cuopt_server/cuopt_server/utils/response_util.py +++ b/python/cuopt_server/cuopt_server/utils/deprecated/response_util.py @@ -1,9 +1,11 @@ -# SPDX-FileCopyrightText: Copyright (c) 2022-2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-FileCopyrightText: Copyright (c) 2022-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 from typing import Dict -from .routing.optimization_data_model import OptimizationDataModel +from cuopt_server.utils.routing.optimization_data_model import ( + OptimizationDataModel, +) def get_full_response( diff --git a/python/cuopt_server/cuopt_server/utils/result_store.py b/python/cuopt_server/cuopt_server/utils/deprecated/result_store.py similarity index 100% rename from python/cuopt_server/cuopt_server/utils/result_store.py rename to python/cuopt_server/cuopt_server/utils/deprecated/result_store.py diff --git a/python/cuopt_server/cuopt_server/utils/deprecated/routing/__init__.py b/python/cuopt_server/cuopt_server/utils/deprecated/routing/__init__.py new file mode 100644 index 0000000000..d51c4fe1e0 --- /dev/null +++ b/python/cuopt_server/cuopt_server/utils/deprecated/routing/__init__.py @@ -0,0 +1,2 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 diff --git a/python/cuopt_server/cuopt_server/utils/routing/generate_metadata.py b/python/cuopt_server/cuopt_server/utils/deprecated/routing/generate_metadata.py similarity index 100% rename from python/cuopt_server/cuopt_server/utils/routing/generate_metadata.py rename to python/cuopt_server/cuopt_server/utils/deprecated/routing/generate_metadata.py diff --git a/python/cuopt_server/cuopt_server/utils/sema.py b/python/cuopt_server/cuopt_server/utils/deprecated/sema.py similarity index 100% rename from python/cuopt_server/cuopt_server/utils/sema.py rename to python/cuopt_server/cuopt_server/utils/deprecated/sema.py From 51f134d8436e30aceffde916c89e780594aebdbc Mon Sep 17 00:00:00 2001 From: Ramakrishna Prabhu <42624703+ramakrishnap-nv@users.noreply.github.com> Date: Wed, 9 Sep 2026 16:47:04 -0500 Subject: [PATCH 056/113] build: add the CUDA-free cuopt_client library (#1804) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds `cuopt_client`, a CPU-only library holding the host-side problem representation (parsers, `data_model_view`, `mps_data_model`, writers), the gRPC wire protocol, and the LP/MIP gRPC client. `libcuopt` and `cuopt_grpc_server` both link it, so there is one implementation rather than a client-side fork. This is the library-level half of letting a remote client talk to `cuopt_grpc_server` without `cudf`, `cupy`, `rmm` or `pylibraft`. The packaging half is not here: `libcuopt_client.so` still ships inside the `libcuopt` package, which depends on CUDA, so a GPU-free install is not yet possible. Tracked in #1872. Two notes for reviewers: - The routing gRPC arm stays in `libcuopt`. Its mappers call routing accessors that live in CUDA translation units, so moving it down would create a `libcuopt -> cuopt_client -> libcuopt` cycle. - Some public API changes which library exports it — `solver_settings_t::get_mip_callbacks()` and siblings now come from `libcuopt_client.so`. `cuopt` links `cuopt::cuopt_client` as `PUBLIC` and both are in `cuopt-exports`, so CMake consumers resolve them transitively; a bare `-lcuopt` link would also need `-lcuopt_client`. Verified: `libcuopt_client.so` has no CUDA, rmm or raft in `NEEDED`, and no undefined `cuopt::` symbols. Last of four steps toward a CUDA-free client library, after #1801, #1802 and #1803. Authors: - Ramakrishna Prabhu (https://github.com/ramakrishnap-nv) Approvers: - Trevor McKay (https://github.com/tmckayus) - Rajesh Gandham (https://github.com/rg20) URL: https://github.com/NVIDIA/cuopt/pull/1804 --- ci/build_wheel_cuopt.sh | 1 + conda/recipes/libcuopt/recipe.yaml | 2 + cpp/CMakeLists.txt | 146 ++++++++- cpp/src/CMakeLists.txt | 4 + cpp/src/io/CMakeLists.txt | 4 +- cpp/src/math_optimization/CMakeLists.txt | 12 +- cpp/src/math_optimization/solver_settings.cpp | 231 ++++---------- cpp/src/math_optimization/solver_settings.cu | 296 ++++++++++++++++++ .../math_optimization/solver_settings_gpu.cu | 181 ----------- cpp/src/mip_heuristics/CMakeLists.txt | 9 +- cpp/src/pdlp/CMakeLists.txt | 11 +- .../unit_tests/solver_settings_test.cu | 4 +- python/libcuopt/CMakeLists.txt | 4 + 13 files changed, 539 insertions(+), 366 deletions(-) create mode 100644 cpp/src/math_optimization/solver_settings.cu delete mode 100644 cpp/src/math_optimization/solver_settings_gpu.cu diff --git a/ci/build_wheel_cuopt.sh b/ci/build_wheel_cuopt.sh index 2e38167c0a..2cc81e3f27 100755 --- a/ci/build_wheel_cuopt.sh +++ b/ci/build_wheel_cuopt.sh @@ -41,6 +41,7 @@ EXCLUDE_ARGS=( --exclude "libcusolver.so.*" --exclude "libcusparse.so.*" --exclude "libcuopt.so" + --exclude "libcuopt_client.so" --exclude "libnvJitLink.so*" --exclude "librapids_logger.so" --exclude "librmm.so" diff --git a/conda/recipes/libcuopt/recipe.yaml b/conda/recipes/libcuopt/recipe.yaml index 18a26c172f..fcd83f749c 100644 --- a/conda/recipes/libcuopt/recipe.yaml +++ b/conda/recipes/libcuopt/recipe.yaml @@ -114,6 +114,7 @@ outputs: ignore: # See https://github.com/rapidsai/build-planning/issues/160 - lib/libcuopt.so + - lib/libcuopt_client.so string: cuda${{ cuda_major }}_${{ datetime_string }}_${{ head_rev }} requirements: build: @@ -165,6 +166,7 @@ outputs: - package_contents: files: - lib/libcuopt.so + - lib/libcuopt_client.so - bin/cuopt_cli - bin/cuopt_grpc_server about: diff --git a/cpp/CMakeLists.txt b/cpp/CMakeLists.txt index 72e52902e9..773f1b5a18 100644 --- a/cpp/CMakeLists.txt +++ b/cpp/CMakeLists.txt @@ -547,11 +547,12 @@ if (BUILD_TESTS) endif () set(CUOPT_SRC_FILES) +set(CUOPT_CLIENT_SRC_FILES) set(MPS_FAST_SRC_FILES) add_subdirectory(src) if (HOST_LINEINFO) - set_source_files_properties(${CUOPT_SRC_FILES} DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR} PROPERTIES COMPILE_OPTIONS "-g1") + set_source_files_properties(${CUOPT_SRC_FILES} ${CUOPT_CLIENT_SRC_FILES} DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR} PROPERTIES COMPILE_OPTIONS "-g1") endif () # Needed for the fast MPS parser, available on all x86-64-v3 compliant x86 CPUs (essentially since Haswell ~2013) @@ -564,11 +565,11 @@ endif () # TODO: figure out a set of flags for ARM that fits the range of CPUs we wish to support (neoverse?) # NEON should be universal on aarch64 and enough for our purposes (parsing) though -# Apply -UNDEBUG only to solver source files (not gRPC infrastructure). -# Must happen before gRPC files are appended to CUOPT_SRC_FILES. +# Apply -UNDEBUG only to solver and parser source files (not gRPC infrastructure). +# Must happen before gRPC files are appended to CUOPT_CLIENT_SRC_FILES. # Uses APPEND to preserve any existing per-file options (e.g. -g1 from HOST_LINEINFO). if (DEFINE_ASSERT) - set_property(SOURCE ${CUOPT_SRC_FILES} DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR} + set_property(SOURCE ${CUOPT_SRC_FILES} ${CUOPT_CLIENT_SRC_FILES} DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR} APPEND PROPERTY COMPILE_OPTIONS "-UNDEBUG") endif () @@ -595,9 +596,7 @@ if (NOT SKIP_GRPC_BUILD) src/grpc/client/grpc_client.cpp src/grpc/client/grpc_client_env.cpp src/grpc/client/cython_grpc_client.cpp - src/grpc/client/solve_remote.cpp ) - # Routing (VRP) arm: everything that depends on the routing engine. Kept as # its own list so a routing-only gRPC client can be split out of the # cuopt_grpc component without moving code around again. @@ -613,7 +612,19 @@ if (NOT SKIP_GRPC_BUILD) if (CUOPT_ENABLE_GRPC_ROUTING) list(APPEND GRPC_INFRA_FILES ${GRPC_ROUTING_FILES}) endif () - list(APPEND CUOPT_SRC_FILES ${GRPC_INFRA_FILES}) + + # The routing arm stays in libcuopt: its mappers call routing accessors that live in CUDA + # translation units, so moving it down would create a libcuopt -> cuopt_client -> libcuopt + # cycle. Those accessors need splitting out first, as was done for the LP/MIP settings. + list(APPEND CUOPT_CLIENT_SRC_FILES ${GRPC_PROTO_GENERATED_FILES} ${GRPC_MATHOPT_FILES}) + if (CUOPT_ENABLE_GRPC_ROUTING) + list(APPEND CUOPT_SRC_FILES ${GRPC_ROUTING_FILES}) + endif () + + # solve_remote.cpp is the local-vs-remote dispatcher: it calls into the GPU + # solver, so it stays in cuopt_objs rather than moving down to cuopt_client. + list(APPEND CUOPT_SRC_FILES src/grpc/client/solve_remote.cpp) + list(APPEND GRPC_INFRA_FILES src/grpc/client/solve_remote.cpp) # Always keep NDEBUG defined for gRPC infrastructure files so that abseil # headers inline Mutex::Dtor() instead of emitting an external call. @@ -628,6 +639,112 @@ if (NOT SKIP_GRPC_BUILD) APPEND PROPERTY COMPILE_OPTIONS "$<$:-fvisibility=default>") endif (NOT SKIP_GRPC_BUILD) +# ################################################################################################## +# - cuopt_client - CPU-only support library ---------------------------------------------------------- +# +# Host-side problem representation (parsers, data_model_view, mps_data_model, writers), the +# gRPC wire protocol, and the gRPC client. None of it touches CUDA, so Python extension +# modules that never call the GPU can link this instead of libcuopt.so. +# +# LANGUAGES is deliberately not CUDA: adding a .cu file here should fail loudly rather than +# quietly reintroduce a CUDA dependency. Built as an OBJECT library first, mirroring +# cuopt_objs/cuopt, so cuopt_static can link the objects directly for tests that reach +# parser internals the shared library does not export. +add_library(cuopt_client_objs OBJECT ${CUOPT_CLIENT_SRC_FILES}) +# Default visibility, deliberately unlike cuopt_objs: this library was carved out of the +# internals rather than designed as an export surface, so libcuopt resolves ~239 symbols +# from it. Hiding them would mean annotating essentially every host-side method with +# CUOPT_EXPORT. +set_target_properties(cuopt_client_objs + PROPERTIES POSITION_INDEPENDENT_CODE ON + CXX_SCAN_FOR_MODULES OFF +) + +add_library(cuopt_client SHARED $) +add_library(cuopt::cuopt_client ALIAS cuopt_client) + +target_include_directories(cuopt_client + PUBLIC + "$" + "$" + INTERFACE + "$" +) + +target_compile_definitions(cuopt_client + PUBLIC "CUOPT_LOG_ACTIVE_LEVEL=RAPIDS_LOGGER_LOG_LEVEL_${LIBCUOPT_LOGGING_LEVEL}" +) + +set_target_properties(cuopt_client + PROPERTIES POSITION_INDEPENDENT_CODE ON + CXX_SCAN_FOR_MODULES OFF + BUILD_RPATH "\$ORIGIN" + INSTALL_RPATH "\$ORIGIN" + LINKER_LANGUAGE CXX +) + +target_compile_definitions(cuopt_client_objs + PUBLIC "CUOPT_LOG_ACTIVE_LEVEL=RAPIDS_LOGGER_LOG_LEVEL_${LIBCUOPT_LOGGING_LEVEL}" +) + +target_compile_options(cuopt_client_objs + PRIVATE "$<$:${CUOPT_CXX_FLAGS}>" +) + +target_include_directories(cuopt_client_objs + PRIVATE + "${CMAKE_CURRENT_SOURCE_DIR}/../thirdparty" + "${CMAKE_CURRENT_SOURCE_DIR}/src" + "${CMAKE_CURRENT_SOURCE_DIR}/src/io" + "${CMAKE_CURRENT_SOURCE_DIR}/src/grpc" + "${CMAKE_CURRENT_SOURCE_DIR}/src/grpc/client" + "${CMAKE_CURRENT_SOURCE_DIR}/src/grpc/codegen/generated" + "${CMAKE_CURRENT_BINARY_DIR}" + "${CMAKE_CURRENT_BINARY_DIR}/include" + $<$:${BZIP2_INCLUDE_DIRS}> + $<$:${ZLIB_INCLUDE_DIRS}> + PUBLIC + "$" + "$" + INTERFACE + "$" +) + +# CCCL is a compile-time (header-only) dependency here: the fast MPS parser uses +# host helpers from (ceil_div, round_up). It pulls in no CUDA runtime. +# bzip2 / zlib / lz4 are dlopen'd at runtime by file_to_string.cpp, so they are +# header-only here too and deliberately absent from the link line. +# The OBJECT library needs these for their INTERFACE include dirs / defines at compile time. +target_link_libraries(cuopt_client_objs + PUBLIC + rapids_logger::rapids_logger + CCCL::CCCL + # Header-only here, exactly like CCCL: the client sources transitively include + # and via pdlp/solver_settings.hpp. + # No CUDA runtime is linked -- the device getters those headers declare only ever + # throw -- but without these the build relies on conda happening to put the headers on + # the default include path, and a CPM/fetched-rmm build fails to find them. + rmm::rmm + raft::raft + PRIVATE + simde::simde + OpenMP::OpenMP_CXX + $<$:protobuf::libprotobuf> + $<$:gRPC::grpc++> +) + +target_link_libraries(cuopt_client + PUBLIC + rapids_logger::rapids_logger + CCCL::CCCL + PRIVATE + simde::simde + OpenMP::OpenMP_CXX + ${CMAKE_DL_LIBS} + $<$:protobuf::libprotobuf> + $<$:gRPC::grpc++> +) + add_library(cuopt_objs OBJECT ${CUOPT_SRC_FILES} ) @@ -752,6 +869,7 @@ target_compile_definitions(cuopt_objs PUBLIC target_link_libraries(cuopt_objs PUBLIC + cuopt::cuopt_client CUDA::cublas CUDA::cusparse rmm::rmm @@ -772,7 +890,10 @@ target_link_libraries(cuopt_objs # - generate tests -------------------------------------------------------------------------------- if (BUILD_TESTS) include(CTest) - add_library(cuopt_static STATIC $) + # Embeds cuopt_client_objs directly rather than linking libcuopt_client.so: the internal + # test binaries reach parser internals that the shared library deliberately does not + # export. Do not also link cuopt::cuopt_client here -- that would duplicate every symbol. + add_library(cuopt_static STATIC $ $) target_link_libraries(cuopt_static PUBLIC CUDA::cublas @@ -834,6 +955,7 @@ target_include_directories(cuopt ) target_link_libraries(cuopt PUBLIC + cuopt::cuopt_client CUDA::cublas CUDA::cusparse rmm::rmm @@ -899,14 +1021,14 @@ else () endif () # adds the .so files to the runtime deb package -install(TARGETS cuopt +install(TARGETS cuopt cuopt_client DESTINATION ${_LIB_DEST} COMPONENT runtime EXPORT cuopt-exports ) # adds the .so files to the development deb package -install(TARGETS cuopt +install(TARGETS cuopt cuopt_client DESTINATION ${_LIB_DEST} COMPONENT dev ) @@ -934,7 +1056,7 @@ cuOpt library is a collection of GPU accelerated combinatorial optimization algo rapids_export(INSTALL cuopt EXPORT_SET cuopt-exports - GLOBAL_TARGETS cuopt + GLOBAL_TARGETS cuopt cuopt_client NAMESPACE cuopt:: DOCUMENTATION doc_string ) @@ -943,7 +1065,7 @@ rapids_export(INSTALL cuopt # - build export ------------------------------------------------------------------------------- rapids_export(BUILD cuopt EXPORT_SET cuopt-exports - GLOBAL_TARGETS cuopt + GLOBAL_TARGETS cuopt cuopt_client NAMESPACE cuopt:: DOCUMENTATION doc_string ) diff --git a/cpp/src/CMakeLists.txt b/cpp/src/CMakeLists.txt index db71f8fa4d..75b04b5c42 100644 --- a/cpp/src/CMakeLists.txt +++ b/cpp/src/CMakeLists.txt @@ -8,6 +8,9 @@ set(UTIL_SRC_FILES ${CMAKE_CURRENT_SOURCE_DIR}/utilities/seed_generator.cu ${CMAKE_CURRENT_SOURCE_DIR}/utilities/timestamp_utils.cpp ${CMAKE_CURRENT_SOURCE_DIR}/utilities/work_unit_scheduler.cpp) +# No sources: became header-only in #1778. +set(UTIL_CLIENT_SRC_FILES) + add_subdirectory(linear_algebra) add_subdirectory(pdlp) add_subdirectory(math_optimization) @@ -25,4 +28,5 @@ add_subdirectory(branch_and_bound) add_subdirectory(cuts) set(CUOPT_SRC_FILES ${CUOPT_SRC_FILES} ${UTIL_SRC_FILES} PARENT_SCOPE) +set(CUOPT_CLIENT_SRC_FILES ${CUOPT_CLIENT_SRC_FILES} ${UTIL_CLIENT_SRC_FILES} PARENT_SCOPE) set(MPS_FAST_SRC_FILES ${MPS_FAST_SRC_FILES} PARENT_SCOPE) diff --git a/cpp/src/io/CMakeLists.txt b/cpp/src/io/CMakeLists.txt index cafcffb23f..f7851aa4fd 100644 --- a/cpp/src/io/CMakeLists.txt +++ b/cpp/src/io/CMakeLists.txt @@ -23,5 +23,7 @@ set(PARSERS_SRC_FILES ${MPS_FAST_SRC_FILES} ) -set(CUOPT_SRC_FILES ${CUOPT_SRC_FILES} ${PARSERS_SRC_FILES} PARENT_SCOPE) +# Parsers and the host-side problem representation are CUDA-free (the only `cuda::` uses +# are host integer helpers from header-only libcu++), so they build into cuopt_client. +set(CUOPT_CLIENT_SRC_FILES ${CUOPT_CLIENT_SRC_FILES} ${PARSERS_SRC_FILES} PARENT_SCOPE) set(MPS_FAST_SRC_FILES ${MPS_FAST_SRC_FILES} PARENT_SCOPE) diff --git a/cpp/src/math_optimization/CMakeLists.txt b/cpp/src/math_optimization/CMakeLists.txt index 3aed4e8fb3..8e0f7203de 100644 --- a/cpp/src/math_optimization/CMakeLists.txt +++ b/cpp/src/math_optimization/CMakeLists.txt @@ -5,13 +5,21 @@ list(PREPEND MATH_OPT_SRC_FILES - ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings.cpp - ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings_gpu.cu + ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings.cu ${CMAKE_CURRENT_SOURCE_DIR}/solution_reader.cu ${CMAKE_CURRENT_SOURCE_DIR}/solution_writer.cu ${CMAKE_CURRENT_SOURCE_DIR}/tic_toc.cpp ${CMAKE_CURRENT_SOURCE_DIR}/logger_entry.cpp ) +# solver_settings_t is host-only apart from the device members in solver_settings.cu, +# so the bulk of it builds into cuopt_client. The gRPC client takes a solver_settings_t +# in its public API, so this is required for the client library to resolve standalone. +set(MATH_OPT_CLIENT_SRC_FILES + ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings.cpp + ) + set(CUOPT_SRC_FILES ${CUOPT_SRC_FILES} ${MATH_OPT_SRC_FILES} PARENT_SCOPE) +set(CUOPT_CLIENT_SRC_FILES ${CUOPT_CLIENT_SRC_FILES} + ${MATH_OPT_CLIENT_SRC_FILES} PARENT_SCOPE) diff --git a/cpp/src/math_optimization/solver_settings.cpp b/cpp/src/math_optimization/solver_settings.cpp index 6646e09bbf..349dbd8abd 100644 --- a/cpp/src/math_optimization/solver_settings.cpp +++ b/cpp/src/math_optimization/solver_settings.cpp @@ -79,169 +79,6 @@ bool string_to_bool(const std::string& value, bool& result) } // namespace -template -solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings() -{ - // clang-format off - // Float parameters - float_parameters = { - {CUOPT_TIME_LIMIT, &mip_settings.time_limit, f_t(0.0), std::numeric_limits::infinity(), std::numeric_limits::infinity()}, - {CUOPT_TIME_LIMIT, &pdlp_settings.time_limit, f_t(0.0), std::numeric_limits::infinity(), std::numeric_limits::infinity()}, - {CUOPT_WORK_LIMIT, &mip_settings.work_limit, f_t(0.0), std::numeric_limits::infinity(), std::numeric_limits::infinity()}, - {CUOPT_ABSOLUTE_DUAL_TOLERANCE, &pdlp_settings.tolerances.absolute_dual_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, - {CUOPT_RELATIVE_DUAL_TOLERANCE, &pdlp_settings.tolerances.relative_dual_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, - {CUOPT_ABSOLUTE_PRIMAL_TOLERANCE, &pdlp_settings.tolerances.absolute_primal_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, - {CUOPT_RELATIVE_PRIMAL_TOLERANCE, &pdlp_settings.tolerances.relative_primal_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, - {CUOPT_ABSOLUTE_GAP_TOLERANCE, &pdlp_settings.tolerances.absolute_gap_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, - {CUOPT_RELATIVE_GAP_TOLERANCE, &pdlp_settings.tolerances.relative_gap_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, - {CUOPT_MIP_ABSOLUTE_TOLERANCE, &mip_settings.tolerances.absolute_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-6)}, - {CUOPT_MIP_RELATIVE_TOLERANCE, &mip_settings.tolerances.relative_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-12)}, - {CUOPT_MIP_INTEGRALITY_TOLERANCE, &mip_settings.tolerances.integrality_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-5)}, - {CUOPT_MIP_ABSOLUTE_GAP, &mip_settings.tolerances.absolute_mip_gap, f_t(0.0), std::numeric_limits::infinity(), std::max(f_t(1e-10), std::numeric_limits::epsilon())}, - {CUOPT_MIP_RELATIVE_GAP, &mip_settings.tolerances.relative_mip_gap, f_t(0.0), f_t(1e-1), f_t(1e-4)}, - {CUOPT_PRIMAL_INFEASIBLE_TOLERANCE, &pdlp_settings.tolerances.primal_infeasible_tolerance, f_t(0.0), f_t(1e-1), std::max(f_t(1e-10), std::numeric_limits::epsilon())}, - {CUOPT_DUAL_INFEASIBLE_TOLERANCE, &pdlp_settings.tolerances.dual_infeasible_tolerance, f_t(0.0), f_t(1e-1), std::max(f_t(1e-10), std::numeric_limits::epsilon())}, - {CUOPT_MIP_CUT_CHANGE_THRESHOLD, &mip_settings.cut_change_threshold, f_t(-1.0), std::numeric_limits::infinity(), f_t(-1.0)}, - {CUOPT_MIP_CUT_MIN_ORTHOGONALITY, &mip_settings.cut_min_orthogonality, f_t(0.0), f_t(1.0), f_t(0.5)}, - {CUOPT_BARRIER_PRIMAL_REGULARIZATION, &pdlp_settings.barrier_primal_regularization, f_t(-1.0), std::numeric_limits::infinity(), f_t(-1.0), "initial primal regularization for the augmented system; -1 automatic"}, - {CUOPT_BARRIER_DUAL_REGULARIZATION, &pdlp_settings.barrier_dual_regularization, f_t(-1.0), std::numeric_limits::infinity(), f_t(-1.0), "initial dual regularization for the augmented system; -1 automatic"}, - {CUOPT_BARRIER_STEP_SCALE, &pdlp_settings.barrier_step_scale, f_t(0.5), f_t(0.9999), f_t(0.9)}, - {CUOPT_BARRIER_INITIAL_POINT_SAFEGUARD, &pdlp_settings.barrier_initial_point_safeguard, f_t(0.0), std::numeric_limits::infinity(), f_t(10.0), "margin pushing the barrier initial iterate into the interior of the nonnegative orthant / SOC"}, - // MIP heuristic hyper-parameters (hidden from default --help: name contains "hyper_") - {CUOPT_MIP_HYPER_HEURISTIC_ROOT_LP_TIME_RATIO, &mip_settings.heuristic_params.root_lp_time_ratio, f_t(0.0), f_t(1.0), f_t(0.1), "fraction of total time for root LP"}, - {CUOPT_MIP_HYPER_HEURISTIC_ROOT_LP_MAX_TIME, &mip_settings.heuristic_params.root_lp_max_time, f_t(0.0), std::numeric_limits::infinity(), f_t(15.0), "hard cap on root LP seconds"}, - {CUOPT_MIP_HYPER_HEURISTIC_RINS_TIME_LIMIT, &mip_settings.heuristic_params.rins_time_limit, f_t(0.0), std::numeric_limits::infinity(), f_t(3.0), "per-call RINS sub-MIP time"}, - {CUOPT_MIP_HYPER_HEURISTIC_RINS_MAX_TIME_LIMIT, &mip_settings.heuristic_params.rins_max_time_limit, f_t(0.0), std::numeric_limits::infinity(), f_t(20.0), "ceiling for RINS adaptive time budget"}, - {CUOPT_MIP_HYPER_HEURISTIC_RINS_FIX_RATE, &mip_settings.heuristic_params.rins_fix_rate, f_t(0.0), f_t(1.0), f_t(0.5), "RINS variable fix rate"}, - {CUOPT_MIP_HYPER_HEURISTIC_INITIAL_INFEASIBILITY_WEIGHT, &mip_settings.heuristic_params.initial_infeasibility_weight, f_t(1e-9), std::numeric_limits::infinity(), f_t(1000.0), "constraint violation penalty seed"}, - {CUOPT_MIP_HYPER_HEURISTIC_RELAXED_LP_TIME_LIMIT, &mip_settings.heuristic_params.relaxed_lp_time_limit, f_t(1e-9), std::numeric_limits::infinity(), f_t(1.0), "base relaxed LP time cap in heuristics"}, - {CUOPT_MIP_HYPER_HEURISTIC_RELATED_VARS_TIME_LIMIT, &mip_settings.heuristic_params.related_vars_time_limit, f_t(1e-9), std::numeric_limits::infinity(), f_t(30.0), "time for related-variable structure build"}, - {CUOPT_MIP_SEMICONTINUOUS_BIG_M, &mip_settings.semi_continuous_big_m, f_t(1.0), std::numeric_limits::infinity(), f_t(1e10), "big-M value for semi-continuous variables with no finite upper bound"}, - // Diving heuristic hyper-parameters (hidden from default --help: name contains "hyper_") - {CUOPT_MIP_HYPER_DIVING_ITERATION_LIMIT_FACTOR, &mip_settings.diving_params.iteration_limit_factor, f_t(0.0), f_t(1.0), f_t(0.05), "fraction of best-first iterations allowed per dive"}, - // Recursive sub-MIP (RINS) hyper-parameters (hidden from default --help: name contains "hyper_") - {CUOPT_MIP_HYPER_SUBMIP_BASE_TARGET_FIXRATE, &mip_settings.submip_params.base_target_fixrate, f_t(0.0), f_t(1.0), f_t(0.6), "base target fix rate for the RINS neighbourhood"}, - {CUOPT_MIP_HYPER_SUBMIP_MIN_FIXRATE, &mip_settings.submip_params.min_fixrate, f_t(0.0), f_t(1.0), f_t(0.25), "minimum fix rate for accepting the RINS neighbourhood"}, - {CUOPT_MIP_HYPER_SUBMIP_MIN_FIXRATE_CAP, &mip_settings.submip_params.min_fixrate_cap, f_t(0.0), f_t(1.0), f_t(0.1), "hard cap on the minimum fix rate for solving a sub-MIP"}, - {CUOPT_MIP_HYPER_SUBMIP_TARGET_MIP_GAP, &mip_settings.submip_params.target_mip_gap, f_t(0.0), f_t(1.0), f_t(0.01), "MIP gap target for the sub-MIP"}, - {CUOPT_MIP_HYPER_SUBMIP_ITERATION_LIMIT_RATIO, &mip_settings.submip_params.iteration_limit_ratio, f_t(0.0), f_t(1.0), f_t(0.8), "sub-MIP simplex-iteration limit as a factor of parent B&B iterations"}, - {CUOPT_MIP_HYPER_SUBMIP_ROUND_CLOSE_RATIO, &mip_settings.submip_params.round_close_ratio, f_t(0.0), f_t(1.0), f_t(0.8), "share of the still-unfixed integers left for later neighbourhood rounds (0 reaches the target fix rate in a single round)"}, - }; - - // Int parameters - // TODO should we have Stable2 and Methodolical1 here? - int_parameters = { - {CUOPT_ITERATION_LIMIT, &pdlp_settings.iteration_limit, 0, std::numeric_limits::max(), std::numeric_limits::max()}, - {CUOPT_NODE_LIMIT, &mip_settings.node_limit, 0, std::numeric_limits::max(), std::numeric_limits::max()}, - {CUOPT_PDLP_SOLVER_MODE, reinterpret_cast(&pdlp_settings.pdlp_solver_mode), CUOPT_PDLP_SOLVER_MODE_STABLE1, CUOPT_PDLP_SOLVER_MODE_STABLE3, CUOPT_PDLP_SOLVER_MODE_STABLE3}, - {CUOPT_METHOD, reinterpret_cast(&pdlp_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_PRIMAL, CUOPT_METHOD_CONCURRENT}, - {CUOPT_METHOD, reinterpret_cast(&mip_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_BARRIER, CUOPT_METHOD_CONCURRENT}, - {CUOPT_NUM_CPU_THREADS, &mip_settings.num_cpu_threads, -1, std::numeric_limits::max(), -1}, - {CUOPT_AUGMENTED, &pdlp_settings.augmented, -1, 1, -1}, - {CUOPT_FOLDING, &pdlp_settings.folding, -1, 1, -1}, - {CUOPT_DUALIZE, &pdlp_settings.dualize, -1, 1, -1}, - {CUOPT_ORDERING, &pdlp_settings.ordering, -1, 1, -1}, - {CUOPT_INITIAL_PERTURBATION, &pdlp_settings.initial_perturbation, -1, 1, -1}, - {CUOPT_REMOVE_PERTURBATION, &pdlp_settings.remove_perturbation, -1, 1, -1}, - {CUOPT_PRIMAL_PRICING, &pdlp_settings.primal_pricing, 0, 1, 0}, - {CUOPT_BARRIER_DUAL_INITIAL_POINT, reinterpret_cast(&pdlp_settings.barrier_dual_initial_point), -1, 2, -1}, - {CUOPT_POSTSOLVE_INFO, &pdlp_settings.postsolve_info, -1, 1, -1}, - {CUOPT_MIP_CUT_PASSES, &mip_settings.max_cut_passes, -1, std::numeric_limits::max(), 10}, - {CUOPT_MIP_MIXED_INTEGER_ROUNDING_CUTS, &mip_settings.mir_cuts, -1, 1, -1}, - {CUOPT_MIP_MIXED_INTEGER_GOMORY_CUTS, &mip_settings.mixed_integer_gomory_cuts, -1, 1, -1}, - {CUOPT_MIP_KNAPSACK_CUTS, &mip_settings.knapsack_cuts, -1, 1, -1}, - {CUOPT_MIP_FLOW_COVER_CUTS, &mip_settings.flow_cover_cuts, -1, 1, -1}, - {CUOPT_MIP_CLIQUE_CUTS, &mip_settings.clique_cuts, -1, 1, -1}, - {CUOPT_MIP_ZERO_HALF_CUTS, &mip_settings.zero_half_cuts, -1, 1, -1}, - {CUOPT_MIP_IMPLIED_BOUND_CUTS, &mip_settings.implied_bound_cuts, -1, 1, -1}, - {CUOPT_MIP_STRONG_CHVATAL_GOMORY_CUTS, &mip_settings.strong_chvatal_gomory_cuts, -1, 1, -1}, - {CUOPT_MIP_REDUCED_COST_STRENGTHENING, &mip_settings.reduced_cost_strengthening, -1, std::numeric_limits::max(), -1}, - {CUOPT_MIP_DUAL_DEGENERATE_FEASIBILITY_PUMP, &mip_settings.dual_degenerate_feasibility_pump, -1, 1, -1}, - {CUOPT_MIP_PRIMAL_DEGENERATE_PIVOTS, &mip_settings.primal_degenerate_pivots, -1, 1, -1}, - {CUOPT_MIP_DUAL_DEGENERATE_PIVOTS, &mip_settings.dual_degenerate_pivots, -1, 1, -1}, - {CUOPT_MIP_RINS, &mip_settings.submip_params.rins, -1, 1, -1}, - {CUOPT_MIP_RENS, &mip_settings.submip_params.rens, -1, 1, -1}, - {CUOPT_MIP_OBJECTIVE_STEP, &mip_settings.objective_step, 0, 1, 1}, - {CUOPT_NUM_GPUS, &pdlp_settings.num_gpus, -1, 72, 1}, - {CUOPT_NUM_GPUS, &mip_settings.num_gpus, -1, 72, 1}, - {CUOPT_MIP_BATCH_PDLP_STRONG_BRANCHING, &mip_settings.mip_batch_pdlp_strong_branching, 0, 2, 0}, - {CUOPT_MIP_BATCH_PDLP_RELIABILITY_BRANCHING, &mip_settings.mip_batch_pdlp_reliability_branching, 0, 2, 0}, - {CUOPT_MIP_STRONG_BRANCHING_SIMPLEX_ITERATION_LIMIT, &mip_settings.strong_branching_simplex_iteration_limit, -1,std::numeric_limits::max(), -1}, - {CUOPT_PRESOLVE, reinterpret_cast(&pdlp_settings.presolver), CUOPT_PRESOLVE_DEFAULT, CUOPT_PRESOLVE_PSLP, CUOPT_PRESOLVE_DEFAULT}, - {CUOPT_PRESOLVE, reinterpret_cast(&mip_settings.presolver), CUOPT_PRESOLVE_DEFAULT, CUOPT_PRESOLVE_PSLP, CUOPT_PRESOLVE_DEFAULT}, - {CUOPT_DISTRIBUTED_PDLP_PARTITIONER, reinterpret_cast(&pdlp_settings.distributed_pdlp_partitioner), CUOPT_DISTRIBUTED_PDLP_PARTITIONER_AUTO, CUOPT_DISTRIBUTED_PDLP_PARTITIONER_ROUND_ROBIN, CUOPT_DISTRIBUTED_PDLP_PARTITIONER_AUTO}, - {CUOPT_MIP_DETERMINISM_MODE, &mip_settings.determinism_mode, CUOPT_MODE_OPPORTUNISTIC, CUOPT_MODE_DETERMINISTIC, CUOPT_MODE_OPPORTUNISTIC}, - {CUOPT_RANDOM_SEED, &mip_settings.seed, -1, std::numeric_limits::max(), -1}, - {CUOPT_MIP_RELIABILITY_BRANCHING, &mip_settings.reliability_branching, -1, std::numeric_limits::max(), -1}, - {CUOPT_PDLP_PRECISION, reinterpret_cast(&pdlp_settings.pdlp_precision), CUOPT_PDLP_DEFAULT_PRECISION, CUOPT_PDLP_MIXED_PRECISION, CUOPT_PDLP_DEFAULT_PRECISION}, - {CUOPT_MIP_SYMMETRY, &mip_settings.symmetry, -1, 2, -1}, - {CUOPT_MIP_SCALING, &mip_settings.mip_scaling, CUOPT_MIP_SCALING_OFF, CUOPT_MIP_SCALING_NO_OBJECTIVE, CUOPT_MIP_SCALING_NO_OBJECTIVE}, - // MIP heuristic hyper-parameters (hidden from default --help: name contains "hyper_") - {CUOPT_MIP_HYPER_HEURISTIC_POPULATION_SIZE, &mip_settings.heuristic_params.population_size, 1, std::numeric_limits::max(), 32, "max solutions in pool"}, - {CUOPT_MIP_HYPER_HEURISTIC_NUM_CPUFJ_THREADS, &mip_settings.heuristic_params.num_cpufj_threads, 0, std::numeric_limits::max(), 8, "parallel CPU FJ climbers"}, - {CUOPT_MIP_HYPER_HEURISTIC_PRESOLVE_MAX_ROUNDS, &mip_settings.heuristic_params.presolve_max_rounds, -1, std::numeric_limits::max(), -1, "Papilo presolve rounds cap (<0 derives it from the problem, 0 keeps Papilo default)"}, - {CUOPT_MIP_HYPER_HEURISTIC_PAPILO_PROBING_MAX_BADGESIZE, &mip_settings.heuristic_params.papilo_probing_max_badgesize, -1, std::numeric_limits::max(), -1, "ceiling on Papilo probing.minbadgesize (<0 derives it from the problem, 0 leaves it uncapped)"}, - {CUOPT_MIP_HYPER_HEURISTIC_STAGNATION_TRIGGER, &mip_settings.heuristic_params.stagnation_trigger, 1, std::numeric_limits::max(), 3, "FP loops w/o improvement before recombination"}, - {CUOPT_MIP_HYPER_HEURISTIC_MAX_ITERS_WITHOUT_IMPROVEMENT, &mip_settings.heuristic_params.max_iterations_without_improvement, 1, std::numeric_limits::max(), 8, "diversity step depth after stagnation"}, - {CUOPT_MIP_HYPER_HEURISTIC_N_OF_MINIMUMS_FOR_EXIT, &mip_settings.heuristic_params.n_of_minimums_for_exit, 1, std::numeric_limits::max(), 7000, "FJ baseline local-minima exit threshold"}, - {CUOPT_MIP_HYPER_HEURISTIC_ENABLED_RECOMBINERS, &mip_settings.heuristic_params.enabled_recombiners, 0, 15, 15, "bitmask: 1=BP 2=FP 4=LS 8=SubMIP"}, - {CUOPT_MIP_HYPER_HEURISTIC_CYCLE_DETECTION_LENGTH, &mip_settings.heuristic_params.cycle_detection_length, 1, std::numeric_limits::max(), 30, "FP assignment cycle ring buffer length"}, - // Diving heuristic hyper-parameters (hidden from default --help: name contains "hyper_") - {CUOPT_MIP_HYPER_DIVING_LINE_SEARCH, &mip_settings.diving_params.line_search_diving, -1, 1, -1, "line-search diving toggle: -1 automatic, 0 disabled, 1 enabled"}, - {CUOPT_MIP_HYPER_DIVING_PSEUDOCOST, &mip_settings.diving_params.pseudocost_diving, -1, 1, -1, "pseudocost diving toggle: -1 automatic, 0 disabled, 1 enabled"}, - {CUOPT_MIP_HYPER_DIVING_GUIDED, &mip_settings.diving_params.guided_diving, -1, 1, -1, "guided diving toggle: -1 automatic, 0 disabled, 1 enabled"}, - {CUOPT_MIP_HYPER_DIVING_COEFFICIENT, &mip_settings.diving_params.coefficient_diving, -1, 1, -1, "coefficient diving toggle: -1 automatic, 0 disabled, 1 enabled"}, - {CUOPT_MIP_HYPER_DIVING_FARKAS, &mip_settings.diving_params.farkas_diving, -1, 1, -1, "Farkas diving toggle: -1 automatic, 0 disabled, 1 enabled"}, - {CUOPT_MIP_HYPER_DIVING_VECTOR_LENGTH, &mip_settings.diving_params.vector_length_diving, -1, 1, -1, "vector-length diving toggle: -1 automatic, 0 disabled, 1 enabled"}, - {CUOPT_MIP_HYPER_DIVING_NODE_LIMIT, &mip_settings.diving_params.node_limit, 0, std::numeric_limits::max(), 500, "maximum nodes explored per dive"}, - {CUOPT_MIP_HYPER_DIVING_BACKTRACK_LIMIT, &mip_settings.diving_params.backtrack_limit, 0, std::numeric_limits::max(), 5, "maximum backtracking allowed per dive"}, - // Recursive sub-MIP (RINS) hyper-parameters (hidden from default --help: name contains "hyper_") - {CUOPT_MIP_HYPER_SUBMIP_NODE_LIMIT_OFFSET, &mip_settings.submip_params.node_limit_offset, 0, std::numeric_limits::max(), 200, "base node limit for the sub-MIP"}, - {CUOPT_MIP_HYPER_SUBMIP_ITERATION_LIMIT_OFFSET, &mip_settings.submip_params.iteration_limit_offset, 0, std::numeric_limits::max(), 10000, "base sub-MIP simplex-iteration limit for root heuristics"}, - {CUOPT_MIP_HYPER_SUBMIP_MAX_LEVEL, &mip_settings.submip_params.max_level, 0, std::numeric_limits::max(), 10, "maximum sub-MIP recursion level"}, - {CUOPT_BARRIER_PRESOLVE_BOUND_FREE_VARIABLES, &pdlp_settings.barrier_presolve_bound_free_variables, -1, 1, -1, "Bound free variables during barrier presolve: -1 automatic (default behavior), 0 disabled, 1 enabled"}, - {CUOPT_BARRIER_ADAPTIVE_REGULARIZATION, &pdlp_settings.barrier_adaptive_regularization, -1, 1, -1, "Adaptive regularization for barrier method: -1 automatic (default behavior), 0 disabled, 1 enabled"}, - // QCQP (barrier) scaling hyper-parameter - {CUOPT_QCQP_HYPER_RUIZ_EQUILIBRATION, &pdlp_settings.qcqp_ruiz_equilibration, -1, 1, -1, "Ruiz equilibration for QCQP barrier scaling: -1 automatic (row/column imbalance heuristic), 0 disabled, 1 enabled"}, - }; - - // Bool parameters - bool_parameters = { - {CUOPT_INFEASIBILITY_DETECTION, &pdlp_settings.detect_infeasibility, false}, - {CUOPT_STRICT_INFEASIBILITY, &pdlp_settings.strict_infeasibility, false}, - {CUOPT_PER_CONSTRAINT_RESIDUAL, &pdlp_settings.per_constraint_residual, false}, - {CUOPT_SAVE_BEST_PRIMAL_SO_FAR, &pdlp_settings.save_best_primal_so_far, false}, - {CUOPT_FIRST_PRIMAL_FEASIBLE, &pdlp_settings.first_primal_feasible, false}, - {CUOPT_MIP_HEURISTICS_ONLY, &mip_settings.heuristics_only, false}, - {CUOPT_LOG_TO_CONSOLE, &pdlp_settings.log_to_console, true}, - {CUOPT_LOG_TO_CONSOLE, &mip_settings.log_to_console, true}, - {CUOPT_CROSSOVER, &pdlp_settings.crossover, false}, - {CUOPT_ELIMINATE_DENSE_COLUMNS, &pdlp_settings.eliminate_dense_columns, true}, - {CUOPT_CUDSS_DETERMINISTIC, &pdlp_settings.cudss_deterministic, false}, - {CUOPT_DUAL_POSTSOLVE, &pdlp_settings.dual_postsolve, true}, - {CUOPT_BARRIER_ITERATIVE_REFINEMENT, &pdlp_settings.barrier_iterative_refinement, true}, - {CUOPT_MIP_PROBING, &mip_settings.probing, true}, - {CUOPT_USE_DISTRIBUTED_PDLP, &pdlp_settings.use_distributed_pdlp, false}, - // Diving heuristic hyper-parameters (hidden from default --help: name contains "hyper_") - {CUOPT_MIP_HYPER_DIVING_SHOW_TYPE, &mip_settings.diving_params.show_type, false, "log diving heuristic type when it finds a new incumbent"}, - // Recursive sub-MIP (RINS) hyper-parameters (hidden from default --help: name contains "hyper_") - {CUOPT_MIP_HYPER_SUBMIP_ENABLE_CPUFJ, &mip_settings.submip_params.enable_cpufj, true, "run CPU FJ over the sub-MIP"}, - {CUOPT_MIP_HYPER_BLOCK_BVE, &mip_settings.block_bve, true, "eliminate blocks of binaries in cuOpt's MIP presolve (needs " CUOPT_MIP_PROBING ")"}, - }; - // String parameters - string_parameters = { - {CUOPT_LOG_FILE, &mip_settings.log_file, ""}, - {CUOPT_LOG_FILE, &pdlp_settings.log_file, ""}, - {CUOPT_SOLUTION_FILE, &mip_settings.sol_file, ""}, - {CUOPT_SOLUTION_FILE, &pdlp_settings.sol_file, ""}, - {CUOPT_USER_PROBLEM_FILE, &mip_settings.user_problem_file, ""}, - {CUOPT_USER_PROBLEM_FILE, &pdlp_settings.user_problem_file, ""}, - {CUOPT_PRESOLVE_FILE, &mip_settings.presolve_file, ""}, - {CUOPT_PRESOLVE_FILE, &pdlp_settings.presolve_file, ""}, - }; - // clang-format on -} - template void solver_settings_t::set_parameter_from_string(const std::string& name, const std::string& value) @@ -604,8 +441,45 @@ bool solver_settings_t::dump_parameters_to_file(const std::string& pat return true; } +// NOTE: deliberately no `template class solver_settings_t<...>` here. +// +// That would instantiate every member, including the implicitly-defined constructor and +// copy constructor. Those construct a pdlp_solver_settings_t, which holds a +// pdlp_warm_start_data_t by value, whose default ctor lives in a CUDA translation unit -- +// so the whole class instantiation drags a CUDA dependency into this CUDA-free library and +// leaves libcuopt_client.so with an undefined symbol. Members are therefore instantiated +// individually below; libcuopt emits the constructors via its own `template class` +// in solver_settings.cu. + #if MIP_INSTANTIATE_FLOAT -template class CUOPT_EXPORT solver_settings_t; +template CUOPT_EXPORT void solver_settings_t::set_parameter_from_string( + const std::string&, const std::string&); +template CUOPT_EXPORT std::string solver_settings_t::get_parameter_as_string( + const std::string&) const; +template CUOPT_EXPORT void solver_settings_t::set_mip_callback( + internals::base_solution_callback_t*, void*); +template CUOPT_EXPORT const std::vector +solver_settings_t::get_mip_callbacks() const; +template CUOPT_EXPORT pdlp_solver_settings_t& +solver_settings_t::get_pdlp_settings(); +template CUOPT_EXPORT mip_solver_settings_t& +solver_settings_t::get_mip_settings(); +template CUOPT_EXPORT const std::vector>& +solver_settings_t::get_float_parameters() const; +template CUOPT_EXPORT const std::vector>& +solver_settings_t::get_int_parameters() const; +template CUOPT_EXPORT const std::vector>& +solver_settings_t::get_bool_parameters() const; +template CUOPT_EXPORT const std::vector +solver_settings_t::get_parameter_names() const; +template CUOPT_EXPORT const std::vector>& +solver_settings_t::get_string_parameters() const; +template CUOPT_EXPORT const pdlp_warm_start_data_view_t& +solver_settings_t::get_pdlp_warm_start_data_view() const noexcept; +template CUOPT_EXPORT void solver_settings_t::load_parameters_from_file( + const std::string&); +template CUOPT_EXPORT bool solver_settings_t::dump_parameters_to_file( + const std::string&, bool) const; template CUOPT_EXPORT void solver_settings_t::set_parameter(const std::string& name, int value); template CUOPT_EXPORT void solver_settings_t::set_parameter(const std::string& name, @@ -623,7 +497,34 @@ template CUOPT_EXPORT std::string solver_settings_t::get_parameter( #endif #if MIP_INSTANTIATE_DOUBLE -template class CUOPT_EXPORT solver_settings_t; +template CUOPT_EXPORT void solver_settings_t::set_parameter_from_string( + const std::string&, const std::string&); +template CUOPT_EXPORT std::string solver_settings_t::get_parameter_as_string( + const std::string&) const; +template CUOPT_EXPORT void solver_settings_t::set_mip_callback( + internals::base_solution_callback_t*, void*); +template CUOPT_EXPORT const std::vector +solver_settings_t::get_mip_callbacks() const; +template CUOPT_EXPORT pdlp_solver_settings_t& +solver_settings_t::get_pdlp_settings(); +template CUOPT_EXPORT mip_solver_settings_t& +solver_settings_t::get_mip_settings(); +template CUOPT_EXPORT const std::vector>& +solver_settings_t::get_float_parameters() const; +template CUOPT_EXPORT const std::vector>& +solver_settings_t::get_int_parameters() const; +template CUOPT_EXPORT const std::vector>& +solver_settings_t::get_bool_parameters() const; +template CUOPT_EXPORT const std::vector +solver_settings_t::get_parameter_names() const; +template CUOPT_EXPORT const std::vector>& +solver_settings_t::get_string_parameters() const; +template CUOPT_EXPORT const pdlp_warm_start_data_view_t& +solver_settings_t::get_pdlp_warm_start_data_view() const noexcept; +template CUOPT_EXPORT void solver_settings_t::load_parameters_from_file( + const std::string&); +template CUOPT_EXPORT bool solver_settings_t::dump_parameters_to_file( + const std::string&, bool) const; template CUOPT_EXPORT void solver_settings_t::set_parameter(const std::string& name, int value); template CUOPT_EXPORT void solver_settings_t::set_parameter(const std::string& name, diff --git a/cpp/src/math_optimization/solver_settings.cu b/cpp/src/math_optimization/solver_settings.cu new file mode 100644 index 0000000000..16c8dacf5d --- /dev/null +++ b/cpp/src/math_optimization/solver_settings.cu @@ -0,0 +1,296 @@ +/* clang-format off */ +/* + * SPDX-FileCopyrightText: Copyright (c) 2024-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +/* clang-format on */ + +// Device-facing members of solver_settings_t, split out of solver_settings.cu. +// +// Everything else in that class is host-only parameter handling, so the remainder now +// builds as solver_settings.cpp into the CUDA-free cuopt_client library. Only these +// members take an rmm::cuda_stream_view or hand back a device_uvector, so they are the +// only ones that must stay in a CUDA TU inside libcuopt. +// +// solver_settings.cpp deliberately has no `template class` at all -- that would instantiate +// the constructor, which needs CUDA (see the note there) -- so it cannot emit these members +// either. The `template class` below covers the class as a whole for libcuopt; members +// defined in the CUDA-free TU are instantiated individually there. + +#include + +#include +#include + +#include + +namespace cuopt { +namespace CUOPT_EXPORT mathematical_optimization { + +template +void solver_settings_t::set_initial_pdlp_primal_solution(const f_t* solution, + i_t size, + rmm::cuda_stream_view stream) +{ + pdlp_settings.set_initial_primal_solution(solution, size, stream); +} + +template +void solver_settings_t::set_initial_pdlp_dual_solution(const f_t* solution, + i_t size, + rmm::cuda_stream_view stream) +{ + pdlp_settings.set_initial_dual_solution(solution, size, stream); +} + +template +void solver_settings_t::set_pdlp_warm_start_data( + const f_t* current_primal_solution, + const f_t* current_dual_solution, + const f_t* initial_primal_average, + const f_t* initial_dual_average, + const f_t* current_ATY, + const f_t* sum_primal_solutions, + const f_t* sum_dual_solutions, + const f_t* last_restart_duality_gap_primal_solution, + const f_t* last_restart_duality_gap_dual_solution, + i_t primal_size, + i_t dual_size, + f_t initial_primal_weight, + f_t initial_step_size, + i_t total_pdlp_iterations, + i_t total_pdhg_iterations, + f_t last_candidate_kkt_score, + f_t last_restart_kkt_score, + f_t sum_solution_weight, + i_t iterations_since_last_restart) +{ + pdlp_settings.set_pdlp_warm_start_data(current_primal_solution, + current_dual_solution, + initial_primal_average, + initial_dual_average, + current_ATY, + sum_primal_solutions, + sum_dual_solutions, + last_restart_duality_gap_primal_solution, + last_restart_duality_gap_dual_solution, + primal_size, + dual_size, + initial_primal_weight, + initial_step_size, + total_pdlp_iterations, + total_pdhg_iterations, + last_candidate_kkt_score, + last_restart_kkt_score, + sum_solution_weight, + iterations_since_last_restart); +} + +template +const rmm::device_uvector& solver_settings_t::get_initial_pdlp_primal_solution() + const +{ + return pdlp_settings.get_initial_primal_solution(); +} + +template +const rmm::device_uvector& solver_settings_t::get_initial_pdlp_dual_solution() const +{ + return pdlp_settings.get_initial_dual_solution(); +} + +template +void solver_settings_t::add_initial_mip_solution(const f_t* solution, + i_t size, + rmm::cuda_stream_view stream) +{ + mip_settings.add_initial_solution(solution, size, stream); +} + +// The constructor is here, not in solver_settings.cpp, and its body is a red herring: it +// only builds parameter tables. What forces the placement is the member it default- +// constructs. pdlp_solver_settings_t holds a pdlp_warm_start_data_t by value, and that +// type's default ctor -- defined in pdlp/pdlp_warm_start_data.cu -- constructs nine +// rmm::device_uvectors on cudaStreamDefault. Compiling this constructor into the CUDA-free +// cuopt_client would therefore leave libcuopt_client.so with an undefined reference that +// only surfaces at call time. +// +// Giving pdlp_warm_start_data_t a default ctor that does not touch device memory would let +// this move to the host translation unit. +template +solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings() +{ + // clang-format off + // Float parameters + float_parameters = { + {CUOPT_TIME_LIMIT, &mip_settings.time_limit, f_t(0.0), std::numeric_limits::infinity(), std::numeric_limits::infinity()}, + {CUOPT_TIME_LIMIT, &pdlp_settings.time_limit, f_t(0.0), std::numeric_limits::infinity(), std::numeric_limits::infinity()}, + {CUOPT_WORK_LIMIT, &mip_settings.work_limit, f_t(0.0), std::numeric_limits::infinity(), std::numeric_limits::infinity()}, + {CUOPT_ABSOLUTE_DUAL_TOLERANCE, &pdlp_settings.tolerances.absolute_dual_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, + {CUOPT_RELATIVE_DUAL_TOLERANCE, &pdlp_settings.tolerances.relative_dual_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, + {CUOPT_ABSOLUTE_PRIMAL_TOLERANCE, &pdlp_settings.tolerances.absolute_primal_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, + {CUOPT_RELATIVE_PRIMAL_TOLERANCE, &pdlp_settings.tolerances.relative_primal_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, + {CUOPT_ABSOLUTE_GAP_TOLERANCE, &pdlp_settings.tolerances.absolute_gap_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, + {CUOPT_RELATIVE_GAP_TOLERANCE, &pdlp_settings.tolerances.relative_gap_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-4)}, + {CUOPT_MIP_ABSOLUTE_TOLERANCE, &mip_settings.tolerances.absolute_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-6)}, + {CUOPT_MIP_RELATIVE_TOLERANCE, &mip_settings.tolerances.relative_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-12)}, + {CUOPT_MIP_INTEGRALITY_TOLERANCE, &mip_settings.tolerances.integrality_tolerance, f_t(0.0), f_t(1e-1), f_t(1e-5)}, + {CUOPT_MIP_ABSOLUTE_GAP, &mip_settings.tolerances.absolute_mip_gap, f_t(0.0), std::numeric_limits::infinity(), std::max(f_t(1e-10), std::numeric_limits::epsilon())}, + {CUOPT_MIP_RELATIVE_GAP, &mip_settings.tolerances.relative_mip_gap, f_t(0.0), f_t(1e-1), f_t(1e-4)}, + {CUOPT_PRIMAL_INFEASIBLE_TOLERANCE, &pdlp_settings.tolerances.primal_infeasible_tolerance, f_t(0.0), f_t(1e-1), std::max(f_t(1e-10), std::numeric_limits::epsilon())}, + {CUOPT_DUAL_INFEASIBLE_TOLERANCE, &pdlp_settings.tolerances.dual_infeasible_tolerance, f_t(0.0), f_t(1e-1), std::max(f_t(1e-10), std::numeric_limits::epsilon())}, + {CUOPT_MIP_CUT_CHANGE_THRESHOLD, &mip_settings.cut_change_threshold, f_t(-1.0), std::numeric_limits::infinity(), f_t(-1.0)}, + {CUOPT_MIP_CUT_MIN_ORTHOGONALITY, &mip_settings.cut_min_orthogonality, f_t(0.0), f_t(1.0), f_t(0.5)}, + {CUOPT_BARRIER_PRIMAL_REGULARIZATION, &pdlp_settings.barrier_primal_regularization, f_t(-1.0), std::numeric_limits::infinity(), f_t(-1.0), "initial primal regularization for the augmented system; -1 automatic"}, + {CUOPT_BARRIER_DUAL_REGULARIZATION, &pdlp_settings.barrier_dual_regularization, f_t(-1.0), std::numeric_limits::infinity(), f_t(-1.0), "initial dual regularization for the augmented system; -1 automatic"}, + {CUOPT_BARRIER_STEP_SCALE, &pdlp_settings.barrier_step_scale, f_t(0.5), f_t(0.9999), f_t(0.9)}, + {CUOPT_BARRIER_INITIAL_POINT_SAFEGUARD, &pdlp_settings.barrier_initial_point_safeguard, f_t(0.0), std::numeric_limits::infinity(), f_t(10.0), "margin pushing the barrier initial iterate into the interior of the nonnegative orthant / SOC"}, + // MIP heuristic hyper-parameters (hidden from default --help: name contains "hyper_") + {CUOPT_MIP_HYPER_HEURISTIC_ROOT_LP_TIME_RATIO, &mip_settings.heuristic_params.root_lp_time_ratio, f_t(0.0), f_t(1.0), f_t(0.1), "fraction of total time for root LP"}, + {CUOPT_MIP_HYPER_HEURISTIC_ROOT_LP_MAX_TIME, &mip_settings.heuristic_params.root_lp_max_time, f_t(0.0), std::numeric_limits::infinity(), f_t(15.0), "hard cap on root LP seconds"}, + {CUOPT_MIP_HYPER_HEURISTIC_RINS_TIME_LIMIT, &mip_settings.heuristic_params.rins_time_limit, f_t(0.0), std::numeric_limits::infinity(), f_t(3.0), "per-call RINS sub-MIP time"}, + {CUOPT_MIP_HYPER_HEURISTIC_RINS_MAX_TIME_LIMIT, &mip_settings.heuristic_params.rins_max_time_limit, f_t(0.0), std::numeric_limits::infinity(), f_t(20.0), "ceiling for RINS adaptive time budget"}, + {CUOPT_MIP_HYPER_HEURISTIC_RINS_FIX_RATE, &mip_settings.heuristic_params.rins_fix_rate, f_t(0.0), f_t(1.0), f_t(0.5), "RINS variable fix rate"}, + {CUOPT_MIP_HYPER_HEURISTIC_INITIAL_INFEASIBILITY_WEIGHT, &mip_settings.heuristic_params.initial_infeasibility_weight, f_t(1e-9), std::numeric_limits::infinity(), f_t(1000.0), "constraint violation penalty seed"}, + {CUOPT_MIP_HYPER_HEURISTIC_RELAXED_LP_TIME_LIMIT, &mip_settings.heuristic_params.relaxed_lp_time_limit, f_t(1e-9), std::numeric_limits::infinity(), f_t(1.0), "base relaxed LP time cap in heuristics"}, + {CUOPT_MIP_HYPER_HEURISTIC_RELATED_VARS_TIME_LIMIT, &mip_settings.heuristic_params.related_vars_time_limit, f_t(1e-9), std::numeric_limits::infinity(), f_t(30.0), "time for related-variable structure build"}, + {CUOPT_MIP_SEMICONTINUOUS_BIG_M, &mip_settings.semi_continuous_big_m, f_t(1.0), std::numeric_limits::infinity(), f_t(1e10), "big-M value for semi-continuous variables with no finite upper bound"}, + // Diving heuristic hyper-parameters (hidden from default --help: name contains "hyper_") + {CUOPT_MIP_HYPER_DIVING_ITERATION_LIMIT_FACTOR, &mip_settings.diving_params.iteration_limit_factor, f_t(0.0), f_t(1.0), f_t(0.05), "fraction of best-first iterations allowed per dive"}, + // Recursive sub-MIP (RINS) hyper-parameters (hidden from default --help: name contains "hyper_") + {CUOPT_MIP_HYPER_SUBMIP_BASE_TARGET_FIXRATE, &mip_settings.submip_params.base_target_fixrate, f_t(0.0), f_t(1.0), f_t(0.6), "base target fix rate for the RINS neighbourhood"}, + {CUOPT_MIP_HYPER_SUBMIP_MIN_FIXRATE, &mip_settings.submip_params.min_fixrate, f_t(0.0), f_t(1.0), f_t(0.25), "minimum fix rate for accepting the RINS neighbourhood"}, + {CUOPT_MIP_HYPER_SUBMIP_MIN_FIXRATE_CAP, &mip_settings.submip_params.min_fixrate_cap, f_t(0.0), f_t(1.0), f_t(0.1), "hard cap on the minimum fix rate for solving a sub-MIP"}, + {CUOPT_MIP_HYPER_SUBMIP_TARGET_MIP_GAP, &mip_settings.submip_params.target_mip_gap, f_t(0.0), f_t(1.0), f_t(0.01), "MIP gap target for the sub-MIP"}, + {CUOPT_MIP_HYPER_SUBMIP_ITERATION_LIMIT_RATIO, &mip_settings.submip_params.iteration_limit_ratio, f_t(0.0), f_t(1.0), f_t(0.8), "sub-MIP simplex-iteration limit as a factor of parent B&B iterations"}, + {CUOPT_MIP_HYPER_SUBMIP_ROUND_CLOSE_RATIO, &mip_settings.submip_params.round_close_ratio, f_t(0.0), f_t(1.0), f_t(0.8), "share of the still-unfixed integers left for later neighbourhood rounds (0 reaches the target fix rate in a single round)"}, + }; + + // Int parameters + // TODO should we have Stable2 and Methodolical1 here? + int_parameters = { + {CUOPT_ITERATION_LIMIT, &pdlp_settings.iteration_limit, 0, std::numeric_limits::max(), std::numeric_limits::max()}, + {CUOPT_NODE_LIMIT, &mip_settings.node_limit, 0, std::numeric_limits::max(), std::numeric_limits::max()}, + {CUOPT_PDLP_SOLVER_MODE, reinterpret_cast(&pdlp_settings.pdlp_solver_mode), CUOPT_PDLP_SOLVER_MODE_STABLE1, CUOPT_PDLP_SOLVER_MODE_STABLE3, CUOPT_PDLP_SOLVER_MODE_STABLE3}, + {CUOPT_METHOD, reinterpret_cast(&pdlp_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_PRIMAL, CUOPT_METHOD_CONCURRENT}, + {CUOPT_METHOD, reinterpret_cast(&mip_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_BARRIER, CUOPT_METHOD_CONCURRENT}, + {CUOPT_NUM_CPU_THREADS, &mip_settings.num_cpu_threads, -1, std::numeric_limits::max(), -1}, + {CUOPT_AUGMENTED, &pdlp_settings.augmented, -1, 1, -1}, + {CUOPT_FOLDING, &pdlp_settings.folding, -1, 1, -1}, + {CUOPT_DUALIZE, &pdlp_settings.dualize, -1, 1, -1}, + {CUOPT_ORDERING, &pdlp_settings.ordering, -1, 1, -1}, + {CUOPT_INITIAL_PERTURBATION, &pdlp_settings.initial_perturbation, -1, 1, -1}, + {CUOPT_REMOVE_PERTURBATION, &pdlp_settings.remove_perturbation, -1, 1, -1}, + {CUOPT_PRIMAL_PRICING, &pdlp_settings.primal_pricing, 0, 1, 0}, + {CUOPT_BARRIER_DUAL_INITIAL_POINT, reinterpret_cast(&pdlp_settings.barrier_dual_initial_point), -1, 2, -1}, + {CUOPT_POSTSOLVE_INFO, &pdlp_settings.postsolve_info, -1, 1, -1}, + {CUOPT_MIP_CUT_PASSES, &mip_settings.max_cut_passes, -1, std::numeric_limits::max(), 10}, + {CUOPT_MIP_MIXED_INTEGER_ROUNDING_CUTS, &mip_settings.mir_cuts, -1, 1, -1}, + {CUOPT_MIP_MIXED_INTEGER_GOMORY_CUTS, &mip_settings.mixed_integer_gomory_cuts, -1, 1, -1}, + {CUOPT_MIP_KNAPSACK_CUTS, &mip_settings.knapsack_cuts, -1, 1, -1}, + {CUOPT_MIP_FLOW_COVER_CUTS, &mip_settings.flow_cover_cuts, -1, 1, -1}, + {CUOPT_MIP_CLIQUE_CUTS, &mip_settings.clique_cuts, -1, 1, -1}, + {CUOPT_MIP_ZERO_HALF_CUTS, &mip_settings.zero_half_cuts, -1, 1, -1}, + {CUOPT_MIP_IMPLIED_BOUND_CUTS, &mip_settings.implied_bound_cuts, -1, 1, -1}, + {CUOPT_MIP_STRONG_CHVATAL_GOMORY_CUTS, &mip_settings.strong_chvatal_gomory_cuts, -1, 1, -1}, + {CUOPT_MIP_REDUCED_COST_STRENGTHENING, &mip_settings.reduced_cost_strengthening, -1, std::numeric_limits::max(), -1}, + {CUOPT_MIP_DUAL_DEGENERATE_FEASIBILITY_PUMP, &mip_settings.dual_degenerate_feasibility_pump, -1, 1, -1}, + {CUOPT_MIP_PRIMAL_DEGENERATE_PIVOTS, &mip_settings.primal_degenerate_pivots, -1, 1, -1}, + {CUOPT_MIP_DUAL_DEGENERATE_PIVOTS, &mip_settings.dual_degenerate_pivots, -1, 1, -1}, + {CUOPT_MIP_RINS, &mip_settings.submip_params.rins, -1, 1, -1}, + {CUOPT_MIP_RENS, &mip_settings.submip_params.rens, -1, 1, -1}, + {CUOPT_MIP_OBJECTIVE_STEP, &mip_settings.objective_step, 0, 1, 1}, + {CUOPT_NUM_GPUS, &pdlp_settings.num_gpus, -1, 72, 1}, + {CUOPT_NUM_GPUS, &mip_settings.num_gpus, -1, 72, 1}, + {CUOPT_MIP_BATCH_PDLP_STRONG_BRANCHING, &mip_settings.mip_batch_pdlp_strong_branching, 0, 2, 0}, + {CUOPT_MIP_BATCH_PDLP_RELIABILITY_BRANCHING, &mip_settings.mip_batch_pdlp_reliability_branching, 0, 2, 0}, + {CUOPT_MIP_STRONG_BRANCHING_SIMPLEX_ITERATION_LIMIT, &mip_settings.strong_branching_simplex_iteration_limit, -1,std::numeric_limits::max(), -1}, + {CUOPT_PRESOLVE, reinterpret_cast(&pdlp_settings.presolver), CUOPT_PRESOLVE_DEFAULT, CUOPT_PRESOLVE_PSLP, CUOPT_PRESOLVE_DEFAULT}, + {CUOPT_PRESOLVE, reinterpret_cast(&mip_settings.presolver), CUOPT_PRESOLVE_DEFAULT, CUOPT_PRESOLVE_PSLP, CUOPT_PRESOLVE_DEFAULT}, + {CUOPT_DISTRIBUTED_PDLP_PARTITIONER, reinterpret_cast(&pdlp_settings.distributed_pdlp_partitioner), CUOPT_DISTRIBUTED_PDLP_PARTITIONER_AUTO, CUOPT_DISTRIBUTED_PDLP_PARTITIONER_ROUND_ROBIN, CUOPT_DISTRIBUTED_PDLP_PARTITIONER_AUTO}, + {CUOPT_MIP_DETERMINISM_MODE, &mip_settings.determinism_mode, CUOPT_MODE_OPPORTUNISTIC, CUOPT_MODE_DETERMINISTIC, CUOPT_MODE_OPPORTUNISTIC}, + {CUOPT_RANDOM_SEED, &mip_settings.seed, -1, std::numeric_limits::max(), -1}, + {CUOPT_MIP_RELIABILITY_BRANCHING, &mip_settings.reliability_branching, -1, std::numeric_limits::max(), -1}, + {CUOPT_PDLP_PRECISION, reinterpret_cast(&pdlp_settings.pdlp_precision), CUOPT_PDLP_DEFAULT_PRECISION, CUOPT_PDLP_MIXED_PRECISION, CUOPT_PDLP_DEFAULT_PRECISION}, + {CUOPT_MIP_SYMMETRY, &mip_settings.symmetry, -1, 2, -1}, + {CUOPT_MIP_SCALING, &mip_settings.mip_scaling, CUOPT_MIP_SCALING_OFF, CUOPT_MIP_SCALING_NO_OBJECTIVE, CUOPT_MIP_SCALING_NO_OBJECTIVE}, + // MIP heuristic hyper-parameters (hidden from default --help: name contains "hyper_") + {CUOPT_MIP_HYPER_HEURISTIC_POPULATION_SIZE, &mip_settings.heuristic_params.population_size, 1, std::numeric_limits::max(), 32, "max solutions in pool"}, + {CUOPT_MIP_HYPER_HEURISTIC_NUM_CPUFJ_THREADS, &mip_settings.heuristic_params.num_cpufj_threads, 0, std::numeric_limits::max(), 8, "parallel CPU FJ climbers"}, + {CUOPT_MIP_HYPER_HEURISTIC_PRESOLVE_MAX_ROUNDS, &mip_settings.heuristic_params.presolve_max_rounds, -1, std::numeric_limits::max(), -1, "Papilo presolve rounds cap (<0 derives it from the problem, 0 keeps Papilo default)"}, + {CUOPT_MIP_HYPER_HEURISTIC_PAPILO_PROBING_MAX_BADGESIZE, &mip_settings.heuristic_params.papilo_probing_max_badgesize, -1, std::numeric_limits::max(), -1, "ceiling on Papilo probing.minbadgesize (<0 derives it from the problem, 0 leaves it uncapped)"}, + {CUOPT_MIP_HYPER_HEURISTIC_STAGNATION_TRIGGER, &mip_settings.heuristic_params.stagnation_trigger, 1, std::numeric_limits::max(), 3, "FP loops w/o improvement before recombination"}, + {CUOPT_MIP_HYPER_HEURISTIC_MAX_ITERS_WITHOUT_IMPROVEMENT, &mip_settings.heuristic_params.max_iterations_without_improvement, 1, std::numeric_limits::max(), 8, "diversity step depth after stagnation"}, + {CUOPT_MIP_HYPER_HEURISTIC_N_OF_MINIMUMS_FOR_EXIT, &mip_settings.heuristic_params.n_of_minimums_for_exit, 1, std::numeric_limits::max(), 7000, "FJ baseline local-minima exit threshold"}, + {CUOPT_MIP_HYPER_HEURISTIC_ENABLED_RECOMBINERS, &mip_settings.heuristic_params.enabled_recombiners, 0, 15, 15, "bitmask: 1=BP 2=FP 4=LS 8=SubMIP"}, + {CUOPT_MIP_HYPER_HEURISTIC_CYCLE_DETECTION_LENGTH, &mip_settings.heuristic_params.cycle_detection_length, 1, std::numeric_limits::max(), 30, "FP assignment cycle ring buffer length"}, + // Diving heuristic hyper-parameters (hidden from default --help: name contains "hyper_") + {CUOPT_MIP_HYPER_DIVING_LINE_SEARCH, &mip_settings.diving_params.line_search_diving, -1, 1, -1, "line-search diving toggle: -1 automatic, 0 disabled, 1 enabled"}, + {CUOPT_MIP_HYPER_DIVING_PSEUDOCOST, &mip_settings.diving_params.pseudocost_diving, -1, 1, -1, "pseudocost diving toggle: -1 automatic, 0 disabled, 1 enabled"}, + {CUOPT_MIP_HYPER_DIVING_GUIDED, &mip_settings.diving_params.guided_diving, -1, 1, -1, "guided diving toggle: -1 automatic, 0 disabled, 1 enabled"}, + {CUOPT_MIP_HYPER_DIVING_COEFFICIENT, &mip_settings.diving_params.coefficient_diving, -1, 1, -1, "coefficient diving toggle: -1 automatic, 0 disabled, 1 enabled"}, + {CUOPT_MIP_HYPER_DIVING_FARKAS, &mip_settings.diving_params.farkas_diving, -1, 1, -1, "Farkas diving toggle: -1 automatic, 0 disabled, 1 enabled"}, + {CUOPT_MIP_HYPER_DIVING_VECTOR_LENGTH, &mip_settings.diving_params.vector_length_diving, -1, 1, -1, "vector-length diving toggle: -1 automatic, 0 disabled, 1 enabled"}, + {CUOPT_MIP_HYPER_DIVING_NODE_LIMIT, &mip_settings.diving_params.node_limit, 0, std::numeric_limits::max(), 500, "maximum nodes explored per dive"}, + {CUOPT_MIP_HYPER_DIVING_BACKTRACK_LIMIT, &mip_settings.diving_params.backtrack_limit, 0, std::numeric_limits::max(), 5, "maximum backtracking allowed per dive"}, + // Recursive sub-MIP (RINS) hyper-parameters (hidden from default --help: name contains "hyper_") + {CUOPT_MIP_HYPER_SUBMIP_NODE_LIMIT_OFFSET, &mip_settings.submip_params.node_limit_offset, 0, std::numeric_limits::max(), 200, "base node limit for the sub-MIP"}, + {CUOPT_MIP_HYPER_SUBMIP_ITERATION_LIMIT_OFFSET, &mip_settings.submip_params.iteration_limit_offset, 0, std::numeric_limits::max(), 10000, "base sub-MIP simplex-iteration limit for root heuristics"}, + {CUOPT_MIP_HYPER_SUBMIP_MAX_LEVEL, &mip_settings.submip_params.max_level, 0, std::numeric_limits::max(), 10, "maximum sub-MIP recursion level"}, + {CUOPT_BARRIER_PRESOLVE_BOUND_FREE_VARIABLES, &pdlp_settings.barrier_presolve_bound_free_variables, -1, 1, -1, "Bound free variables during barrier presolve: -1 automatic (default behavior), 0 disabled, 1 enabled"}, + {CUOPT_BARRIER_ADAPTIVE_REGULARIZATION, &pdlp_settings.barrier_adaptive_regularization, -1, 1, -1, "Adaptive regularization for barrier method: -1 automatic (default behavior), 0 disabled, 1 enabled"}, + // QCQP (barrier) scaling hyper-parameter + {CUOPT_QCQP_HYPER_RUIZ_EQUILIBRATION, &pdlp_settings.qcqp_ruiz_equilibration, -1, 1, -1, "Ruiz equilibration for QCQP barrier scaling: -1 automatic (row/column imbalance heuristic), 0 disabled, 1 enabled"}, + }; + + // Bool parameters + bool_parameters = { + {CUOPT_INFEASIBILITY_DETECTION, &pdlp_settings.detect_infeasibility, false}, + {CUOPT_STRICT_INFEASIBILITY, &pdlp_settings.strict_infeasibility, false}, + {CUOPT_PER_CONSTRAINT_RESIDUAL, &pdlp_settings.per_constraint_residual, false}, + {CUOPT_SAVE_BEST_PRIMAL_SO_FAR, &pdlp_settings.save_best_primal_so_far, false}, + {CUOPT_FIRST_PRIMAL_FEASIBLE, &pdlp_settings.first_primal_feasible, false}, + {CUOPT_MIP_HEURISTICS_ONLY, &mip_settings.heuristics_only, false}, + {CUOPT_LOG_TO_CONSOLE, &pdlp_settings.log_to_console, true}, + {CUOPT_LOG_TO_CONSOLE, &mip_settings.log_to_console, true}, + {CUOPT_CROSSOVER, &pdlp_settings.crossover, false}, + {CUOPT_ELIMINATE_DENSE_COLUMNS, &pdlp_settings.eliminate_dense_columns, true}, + {CUOPT_CUDSS_DETERMINISTIC, &pdlp_settings.cudss_deterministic, false}, + {CUOPT_DUAL_POSTSOLVE, &pdlp_settings.dual_postsolve, true}, + {CUOPT_BARRIER_ITERATIVE_REFINEMENT, &pdlp_settings.barrier_iterative_refinement, true}, + {CUOPT_MIP_PROBING, &mip_settings.probing, true}, + {CUOPT_USE_DISTRIBUTED_PDLP, &pdlp_settings.use_distributed_pdlp, false}, + // Diving heuristic hyper-parameters (hidden from default --help: name contains "hyper_") + {CUOPT_MIP_HYPER_DIVING_SHOW_TYPE, &mip_settings.diving_params.show_type, false, "log diving heuristic type when it finds a new incumbent"}, + // Recursive sub-MIP (RINS) hyper-parameters (hidden from default --help: name contains "hyper_") + {CUOPT_MIP_HYPER_SUBMIP_ENABLE_CPUFJ, &mip_settings.submip_params.enable_cpufj, true, "run CPU FJ over the sub-MIP"}, + {CUOPT_MIP_HYPER_BLOCK_BVE, &mip_settings.block_bve, true, "eliminate blocks of binaries in cuOpt's MIP presolve (needs " CUOPT_MIP_PROBING ")"}, + }; + // String parameters + string_parameters = { + {CUOPT_LOG_FILE, &mip_settings.log_file, ""}, + {CUOPT_LOG_FILE, &pdlp_settings.log_file, ""}, + {CUOPT_SOLUTION_FILE, &mip_settings.sol_file, ""}, + {CUOPT_SOLUTION_FILE, &pdlp_settings.sol_file, ""}, + {CUOPT_USER_PROBLEM_FILE, &mip_settings.user_problem_file, ""}, + {CUOPT_USER_PROBLEM_FILE, &pdlp_settings.user_problem_file, ""}, + {CUOPT_PRESOLVE_FILE, &mip_settings.presolve_file, ""}, + {CUOPT_PRESOLVE_FILE, &pdlp_settings.presolve_file, ""}, + }; + // clang-format on +} + +#if MIP_INSTANTIATE_FLOAT +// Emits the ctor/dtor/copy for the whole class; solver_settings.cpp deliberately does not, +// because those need CUDA (see the note there). +template class CUOPT_EXPORT solver_settings_t; +#endif + +#if MIP_INSTANTIATE_DOUBLE +// Emits the ctor/dtor/copy for the whole class; solver_settings.cpp deliberately does not, +// because those need CUDA (see the note there). +template class CUOPT_EXPORT solver_settings_t; +#endif + +} // namespace CUOPT_EXPORT mathematical_optimization +} // namespace cuopt diff --git a/cpp/src/math_optimization/solver_settings_gpu.cu b/cpp/src/math_optimization/solver_settings_gpu.cu deleted file mode 100644 index a23fbf104a..0000000000 --- a/cpp/src/math_optimization/solver_settings_gpu.cu +++ /dev/null @@ -1,181 +0,0 @@ -/* clang-format off */ -/* - * SPDX-FileCopyrightText: Copyright (c) 2024-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. - * SPDX-License-Identifier: Apache-2.0 - */ -/* clang-format on */ - -// Device-facing members of solver_settings_t, split out of solver_settings.cu. -// -// Everything else in that class is host-only parameter handling, so the remainder now -// builds as solver_settings.cpp into the CUDA-free cuopt_client library. Only these -// members take an rmm::cuda_stream_view or hand back a device_uvector, so they are the -// only ones that must stay in a CUDA TU inside libcuopt. -// -// The `template class` instantiation in solver_settings.cpp cannot emit these members -// (their definitions are not visible there), so they are instantiated explicitly below. - -#include - -#include -#include - -#include - -namespace cuopt { -namespace CUOPT_EXPORT mathematical_optimization { - -template -void solver_settings_t::set_initial_pdlp_primal_solution(const f_t* solution, - i_t size, - rmm::cuda_stream_view stream) -{ - pdlp_settings.set_initial_primal_solution(solution, size, stream); -} - -template -void solver_settings_t::set_initial_pdlp_dual_solution(const f_t* solution, - i_t size, - rmm::cuda_stream_view stream) -{ - pdlp_settings.set_initial_dual_solution(solution, size, stream); -} - -template -void solver_settings_t::set_pdlp_warm_start_data( - const f_t* current_primal_solution, - const f_t* current_dual_solution, - const f_t* initial_primal_average, - const f_t* initial_dual_average, - const f_t* current_ATY, - const f_t* sum_primal_solutions, - const f_t* sum_dual_solutions, - const f_t* last_restart_duality_gap_primal_solution, - const f_t* last_restart_duality_gap_dual_solution, - i_t primal_size, - i_t dual_size, - f_t initial_primal_weight, - f_t initial_step_size, - i_t total_pdlp_iterations, - i_t total_pdhg_iterations, - f_t last_candidate_kkt_score, - f_t last_restart_kkt_score, - f_t sum_solution_weight, - i_t iterations_since_last_restart) -{ - pdlp_settings.set_pdlp_warm_start_data(current_primal_solution, - current_dual_solution, - initial_primal_average, - initial_dual_average, - current_ATY, - sum_primal_solutions, - sum_dual_solutions, - last_restart_duality_gap_primal_solution, - last_restart_duality_gap_dual_solution, - primal_size, - dual_size, - initial_primal_weight, - initial_step_size, - total_pdlp_iterations, - total_pdhg_iterations, - last_candidate_kkt_score, - last_restart_kkt_score, - sum_solution_weight, - iterations_since_last_restart); -} - -template -const rmm::device_uvector& solver_settings_t::get_initial_pdlp_primal_solution() - const -{ - return pdlp_settings.get_initial_primal_solution(); -} - -template -const rmm::device_uvector& solver_settings_t::get_initial_pdlp_dual_solution() const -{ - return pdlp_settings.get_initial_dual_solution(); -} - -template -void solver_settings_t::add_initial_mip_solution(const f_t* solution, - i_t size, - rmm::cuda_stream_view stream) -{ - mip_settings.add_initial_solution(solution, size, stream); -} - -#if MIP_INSTANTIATE_FLOAT -template CUOPT_EXPORT void solver_settings_t::set_initial_pdlp_primal_solution( - const float*, int, rmm::cuda_stream_view); -template CUOPT_EXPORT void solver_settings_t::set_initial_pdlp_dual_solution( - const float*, int, rmm::cuda_stream_view); -template CUOPT_EXPORT const rmm::device_uvector& -solver_settings_t::get_initial_pdlp_primal_solution() const; -template CUOPT_EXPORT const rmm::device_uvector& -solver_settings_t::get_initial_pdlp_dual_solution() const; -template CUOPT_EXPORT void solver_settings_t::add_initial_mip_solution( - const float*, int, rmm::cuda_stream_view); -// The 19-argument host overload. It was moved into this TU with the rest of the block, but -// `template class` in solver_settings.cpp cannot emit it (definition not visible there), so -// without this line the symbol disappears -- and it is the one the Cython layer binds to, -// which takes down every Python test, docs-build and wheel-test job. -template CUOPT_EXPORT void solver_settings_t::set_pdlp_warm_start_data(const float*, - const float*, - const float*, - const float*, - const float*, - const float*, - const float*, - const float*, - const float*, - int, - int, - float, - float, - int, - int, - float, - float, - float, - int); -#endif - -#if MIP_INSTANTIATE_DOUBLE -template CUOPT_EXPORT void solver_settings_t::set_initial_pdlp_primal_solution( - const double*, int, rmm::cuda_stream_view); -template CUOPT_EXPORT void solver_settings_t::set_initial_pdlp_dual_solution( - const double*, int, rmm::cuda_stream_view); -template CUOPT_EXPORT const rmm::device_uvector& -solver_settings_t::get_initial_pdlp_primal_solution() const; -template CUOPT_EXPORT const rmm::device_uvector& -solver_settings_t::get_initial_pdlp_dual_solution() const; -template CUOPT_EXPORT void solver_settings_t::add_initial_mip_solution( - const double*, int, rmm::cuda_stream_view); -// The 19-argument host overload. It was moved into this TU with the rest of the block, but -// `template class` in solver_settings.cpp cannot emit it (definition not visible there), so -// without this line the symbol disappears -- and it is the one the Cython layer binds to, -// which takes down every Python test, docs-build and wheel-test job. -template CUOPT_EXPORT void solver_settings_t::set_pdlp_warm_start_data(const double*, - const double*, - const double*, - const double*, - const double*, - const double*, - const double*, - const double*, - const double*, - int, - int, - double, - double, - int, - int, - double, - double, - double, - int); -#endif - -} // namespace CUOPT_EXPORT mathematical_optimization -} // namespace cuopt diff --git a/cpp/src/mip_heuristics/CMakeLists.txt b/cpp/src/mip_heuristics/CMakeLists.txt index a7c341f555..6beadaac8d 100644 --- a/cpp/src/mip_heuristics/CMakeLists.txt +++ b/cpp/src/mip_heuristics/CMakeLists.txt @@ -9,7 +9,6 @@ set(MIP_LP_NECESSARY_FILES ${CMAKE_CURRENT_SOURCE_DIR}/problem/problem.cu ${CMAKE_CURRENT_SOURCE_DIR}/problem/presolve_data.cu ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings.cu - ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings.cpp ${CMAKE_CURRENT_SOURCE_DIR}/solver_solution.cu ${CMAKE_CURRENT_SOURCE_DIR}/local_search/rounding/simple_rounding.cu ${CMAKE_CURRENT_SOURCE_DIR}/presolve/third_party_presolve.cpp @@ -59,5 +58,13 @@ else() set(MIP_SRC_FILES ${MIP_LP_NECESSARY_FILES} ${MIP_NON_LP_FILES}) endif() +# Host-only members of mip_solver_settings_t (callbacks, tolerances). The gRPC client +# needs them, so they build into cuopt_client; add_initial_solution stays in the .cu. +set(MIP_CLIENT_SRC_FILES + ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings.cpp +) + set(CUOPT_SRC_FILES ${CUOPT_SRC_FILES} ${MIP_SRC_FILES} PARENT_SCOPE) +set(CUOPT_CLIENT_SRC_FILES ${CUOPT_CLIENT_SRC_FILES} + ${MIP_CLIENT_SRC_FILES} PARENT_SCOPE) diff --git a/cpp/src/pdlp/CMakeLists.txt b/cpp/src/pdlp/CMakeLists.txt index 1b1439b350..f53ccf2ea0 100644 --- a/cpp/src/pdlp/CMakeLists.txt +++ b/cpp/src/pdlp/CMakeLists.txt @@ -8,7 +8,6 @@ set(LP_CORE_FILES ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings.cu ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings_accessors.cpp ${CMAKE_CURRENT_SOURCE_DIR}/optimization_problem.cu - ${CMAKE_CURRENT_SOURCE_DIR}/cpu_optimization_problem.cpp ${CMAKE_CURRENT_SOURCE_DIR}/cpu_optimization_problem_to_gpu.cpp ${CMAKE_CURRENT_SOURCE_DIR}/backend_selection.cpp ${CMAKE_CURRENT_SOURCE_DIR}/utilities/problem_checking.cu @@ -17,7 +16,6 @@ set(LP_CORE_FILES ${CMAKE_CURRENT_SOURCE_DIR}/pdhg.cu ${CMAKE_CURRENT_SOURCE_DIR}/solver_solution.cu ${CMAKE_CURRENT_SOURCE_DIR}/solution_conversion.cu - ${CMAKE_CURRENT_SOURCE_DIR}/solution_conversion_cpu.cpp ${CMAKE_CURRENT_SOURCE_DIR}/saddle_point.cu ${CMAKE_CURRENT_SOURCE_DIR}/cusparse_view.cu ${CMAKE_CURRENT_SOURCE_DIR}/pdlp_warm_start_data.cu @@ -52,4 +50,13 @@ else() set(LP_SRC_FILES ${LP_CORE_FILES} ${LP_ADAPTER_FILES}) endif() +# Host-only LP sources the gRPC client needs, so cuopt_client resolves standalone. Their +# GPU-facing members were split into the CUDA translation units above. +set(LP_CLIENT_FILES + ${CMAKE_CURRENT_SOURCE_DIR}/cpu_optimization_problem.cpp + ${CMAKE_CURRENT_SOURCE_DIR}/solution_conversion_cpu.cpp + ${CMAKE_CURRENT_SOURCE_DIR}/solver_settings_accessors.cpp +) + set(CUOPT_SRC_FILES ${CUOPT_SRC_FILES} ${LP_SRC_FILES} PARENT_SCOPE) +set(CUOPT_CLIENT_SRC_FILES ${CUOPT_CLIENT_SRC_FILES} ${LP_CLIENT_FILES} PARENT_SCOPE) diff --git a/cpp/tests/linear_programming/unit_tests/solver_settings_test.cu b/cpp/tests/linear_programming/unit_tests/solver_settings_test.cu index 3f1edcf06d..a2b1a69515 100644 --- a/cpp/tests/linear_programming/unit_tests/solver_settings_test.cu +++ b/cpp/tests/linear_programming/unit_tests/solver_settings_test.cu @@ -285,10 +285,10 @@ TEST(SolverSettingsTest, warm_start_bigger_vector) // ============================================================================= // solver_settings_t (the CUDA-free wrapper split across -// math_optimization/solver_settings.cpp and solver_settings_gpu.cu) +// math_optimization/solver_settings.cpp and solver_settings.cu) // ============================================================================= // -// These exercise every member that solver_settings_gpu.cu explicitly instantiates. +// These exercise every member that solver_settings.cu explicitly instantiates. // A member with a missing explicit instantiation compiles and links this test binary // fine (cuopt_static resolves it internally), but disappears from libcuopt.so's // exported symbols -- the failure mode described in the PR that introduced this split. diff --git a/python/libcuopt/CMakeLists.txt b/python/libcuopt/CMakeLists.txt index 4d24169645..adc1b8e171 100644 --- a/python/libcuopt/CMakeLists.txt +++ b/python/libcuopt/CMakeLists.txt @@ -96,5 +96,9 @@ endif() message(STATUS "libcuopt: Final RPATH = ${rpaths}") set_property(TARGET cuopt PROPERTY INSTALL_RPATH ${rpaths} APPEND) +# cuopt_client needs these too: it PUBLIC-links rapids_logger, which +# build_wheel_libcuopt.sh excludes from vendoring, so the only way to find it is +# $ORIGIN/../../rapids_logger/lib64 from this list. +set_property(TARGET cuopt_client PROPERTY INSTALL_RPATH ${rpaths} APPEND) set_property(TARGET cuopt_cli PROPERTY INSTALL_RPATH ${rpaths} APPEND) set_property(TARGET cuopt_grpc_server PROPERTY INSTALL_RPATH ${rpaths} APPEND) From deade2c601fada136ca847d9184afff73ac68f53 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 11:18:23 -0700 Subject: [PATCH 057/113] Remove BFRT debug code --- .../bound_flipping_ratio_test.cpp | 50 --------------- .../bound_flipping_ratio_test.hpp | 20 ------ cpp/src/dual_simplex/phase2.cpp | 64 +------------------ 3 files changed, 2 insertions(+), 132 deletions(-) diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index d18ed95e90..b73e2b8957 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -138,17 +138,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, std::vector ratios(nz); std::vector harris_ratios(nz); work_estimate_ += 3 * nz; - double t0 = tic(); i_t num_breakpoints = compute_breakpoints(indicies, ratios, harris_ratios); - time_compute_breakpoints_ += toc(t0); - num_breakpoints_ = num_breakpoints; - // Count zero ratios - num_harris_zero_ = 0; - num_exact_zero_ = 0; - for (i_t k = 0; k < num_breakpoints; k++) { - if (harris_ratios[k] == 0.0) num_harris_zero_++; - if (ratios[k] == 0.0) num_exact_zero_++; - } work_estimate_ += 2 * num_breakpoints; if constexpr (verbose) { settings_.log.printf("Initial breakpoints %d\n", num_breakpoints); } if (num_breakpoints == 0) { @@ -161,7 +151,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, i_t entering_index = RATIO_TEST_NO_ENTERING_VARIABLE; f_t max_step_length; - t0 = tic(); i_t k_idx = single_pass(0, num_breakpoints, indicies, @@ -170,7 +159,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, nonbasic_entering, entering_index, max_step_length); - time_single_pass_ += toc(t0); if (k_idx == RATIO_TEST_NUMERICAL_ISSUES) { return RATIO_TEST_NUMERICAL_ISSUES; } // The variable selected by single_pass is guaranteed to be in the first bucket: it // defines the minimum Harris ratio, and its exact ratio is no greater than its Harris @@ -185,8 +173,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, entering_index, std::abs(delta_z_[entering_index])); } - num_buckets_used_ = 0; - step_length_result_ = step_length; determine_flips(step_length, entering_index, flip_indices); return entering_index; } @@ -296,7 +282,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, i_t num_candidates = 0; // This is O( log10(max_step_length/min_step_length) * num_breakpoints) - t0 = tic(); while (total_slope >= 0.0 && coarse_threshold <= max_step_length && scan_start < num_breakpoints && !found_unbounded) { for (i_t h = scan_start; h < num_breakpoints; ++h) { @@ -317,7 +302,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, scan_start = num_candidates; coarse_threshold *= 10.0; } - time_coarse_filter_ += toc(t0); candidates.resize(num_candidates); @@ -357,8 +341,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, work_estimate_ += num_candidates + 1; // This is O(num_buckets * num_candidates) - i_t slope_breaker_k = -1; // the candidate k that made slope go negative - t0 = tic(); while (cumulative_slope >= 0.0 && scan_start < num_candidates && threshold <= max_step_length) { f_t next_threshold = inf; i_t write = scan_start; @@ -371,7 +353,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, const i_t j = nonbasic_list_[indicies[k]]; if (bounded_variables_[j]) { cumulative_slope -= std::abs(delta_z_[j]) * (upper_[j] - lower_[j]); - if (cumulative_slope < 0.0 && slope_breaker_k < 0) { slope_breaker_k = k; } } std::swap(candidates[h], candidates[write]); write++; @@ -390,8 +371,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, if (cumulative_slope < 0.0) break; } - time_bucket_sort_ += toc(t0); - bucket0_size_ = (num_buckets > 0) ? bucket_start[1] : 0; // Compute the maximum pivot // This is O(num_candidates) @@ -433,13 +412,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, } // Step = entering variable's breakpoint ratio - num_buckets_used_ = num_buckets; if (entering_k < 0) { // Fallback to single_pass result - used_fallback_ = true; - bucket_selected_ = -1; - step_length_result_ = step_length; - selected_is_slope_breaker_ = false; determine_flips(step_length, entering_index, flip_indices); return entering_index; } @@ -447,30 +421,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, nonbasic_entering = indicies[entering_k]; entering_index = nonbasic_list_[nonbasic_entering]; - // Record whether we selected the slope breaker - selected_is_slope_breaker_ = (entering_k == slope_breaker_k); - - // Record which bucket was selected - used_fallback_ = false; - i_t pos = -1; - for (i_t b = 0; b < num_buckets; b++) { - if (entering_k >= 0) { - // Find which bucket entering_k is in based on its position in candidates - pos = -1; - for (i_t h = 0; h < num_candidates; h++) { - if (candidates[h] == entering_k) { - pos = h; - break; - } - } - if (pos >= bucket_start[b] && pos < bucket_start[b + 1]) { - bucket_selected_ = b; - break; - } - } - } - work_estimate_ += (bucket_selected_ + 1) * (pos + 3); - step_length_result_ = step_length; determine_flips(step_length, entering_index, flip_indices); return entering_index; diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp index 4587037889..8076be5927 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp @@ -56,26 +56,6 @@ class bound_flipping_ratio_test_t { i_t compute_step_length(f_t& step_length, i_t& nonbasic_entering, std::vector& flip_indices); f_t work_estimate() const { return work_estimate_; } - // Timing fields (filled by compute_step_length) - f_t time_compute_breakpoints_{0.0}; - f_t time_single_pass_{0.0}; - f_t time_coarse_filter_{0.0}; - f_t time_bucket_sort_{0.0}; - f_t time_pivot_selection_{0.0}; - - // Diagnostic fields - i_t num_buckets_used_{0}; // number of buckets in bucket sort - i_t bucket_selected_{ - -1}; // which bucket the entering variable came from (-1 = single_pass/fallback) - f_t step_length_result_{0.0}; // the step length chosen - bool used_fallback_{false}; // true if we fell back to single_pass result - i_t bucket0_size_{0}; // size of first bucket (candidates with ratio <= min_harris) - i_t num_breakpoints_{0}; // total breakpoints computed - bool selected_is_slope_breaker_{ - false}; // true if we selected the variable that made slope go negative - i_t num_harris_zero_{0}; // number of harris_ratios that are exactly 0 - i_t num_exact_zero_{0}; // number of exact ratios that are exactly 0 - private: i_t compute_breakpoints(std::vector& indices, std::vector& ratios, diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 7213861fe4..ba16747b2a 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2806,28 +2806,9 @@ class phase2_timers_t { update_infeasibility_time.work; // clang-format off print_one(settings, "BFRT time", bfrt_time, total_time, total_work); - if (bfrt_time.time > 0.1) { - settings.log.printf(" BFRT breakpoints: %.2fs\n", bfrt_breakpoints_time); - settings.log.printf(" BFRT single_pass: %.2fs\n", bfrt_single_pass_time); - settings.log.printf(" BFRT coarse: %.2fs\n", bfrt_coarse_time); - settings.log.printf(" BFRT bucket: %.2fs\n", bfrt_bucket_time); - settings.log.printf(" BFRT select: %.2fs\n", bfrt_select_time); - } if (bfrt_calls > 0) { - settings.log.printf(" BFRT calls: %d, zero_steps: %d (%.1f%%), single_pass_only: %d, bucket_used: %d, not_last_bucket: %d, fallback: %d\n", - bfrt_calls, bfrt_zero_steps, 100.0 * bfrt_zero_steps / bfrt_calls, - bfrt_single_pass_only, bfrt_bucket_used, bfrt_not_last_bucket, bfrt_fallback); - settings.log.printf(" BFRT slope_breaker: %d, not_slope_breaker: %d (%.1f%%)\n", - bfrt_selected_slope_breaker, bfrt_not_slope_breaker, - bfrt_bucket_used > 0 ? 100.0 * bfrt_not_slope_breaker / bfrt_bucket_used : 0.0); - if (bfrt_zero_steps > 0) { - settings.log.printf(" BFRT zero-step avg: num_buckets=%.1f, bucket0_size=%.1f, num_breakpoints=%.1f, harris_zero=%.1f, exact_zero=%.1f\n", - 1.0 * bfrt_zero_step_num_buckets_sum / bfrt_zero_steps, - 1.0 * bfrt_zero_step_bucket0_sum / bfrt_zero_steps, - 1.0 * bfrt_zero_step_num_breakpoints_sum / bfrt_zero_steps, - 1.0 * bfrt_zero_step_harris_zero_sum / bfrt_zero_steps, - 1.0 * bfrt_zero_step_exact_zero_sum / bfrt_zero_steps); - } + settings.log.printf(" BFRT calls: %d, zero_steps: %d (%.1f%%)\n", + bfrt_calls, bfrt_zero_steps, 100.0 * bfrt_zero_steps / bfrt_calls); } print_one(settings, "Pricing time", pricing_time, total_time, total_work); print_one(settings, "BTran time", btran_time, total_time, total_work); @@ -2849,25 +2830,9 @@ class phase2_timers_t { // clang-format on } work_timer_t bfrt_time; - f_t bfrt_breakpoints_time{0.0}; - f_t bfrt_single_pass_time{0.0}; - f_t bfrt_coarse_time{0.0}; - f_t bfrt_bucket_time{0.0}; - f_t bfrt_select_time{0.0}; // BFRT diagnostic counters i_t bfrt_calls{0}; i_t bfrt_zero_steps{0}; // step_length == 0 - i_t bfrt_single_pass_only{0}; // no bound flips (single_pass decided) - i_t bfrt_bucket_used{0}; // bucket sort was used - i_t bfrt_not_last_bucket{0}; // selected from a bucket other than the last - i_t bfrt_fallback{0}; // fell back to single_pass result after bucket sort - i_t bfrt_zero_step_num_buckets_sum{0}; // sum of num_buckets on zero-step iters - i_t bfrt_zero_step_bucket0_sum{0}; // sum of bucket0 size on zero-step iters - i_t bfrt_zero_step_num_breakpoints_sum{0}; // sum of num_breakpoints on zero-step iters - i_t bfrt_zero_step_harris_zero_sum{0}; // sum of harris_ratios==0 on zero-step iters - i_t bfrt_zero_step_exact_zero_sum{0}; // sum of exact ratios==0 on zero-step iters - i_t bfrt_selected_slope_breaker{0}; // times we selected the slope breaker - i_t bfrt_not_slope_breaker{0}; // times we selected something else work_timer_t pricing_time; work_timer_t btran_time; work_timer_t ftran_time; @@ -3716,35 +3681,10 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, return dual_status_t::NUMERICAL; } timers.bfrt_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); - timers.bfrt_breakpoints_time += bfrt.time_compute_breakpoints_; - timers.bfrt_single_pass_time += bfrt.time_single_pass_; - timers.bfrt_coarse_time += bfrt.time_coarse_filter_; - timers.bfrt_bucket_time += bfrt.time_bucket_sort_; - timers.bfrt_select_time += bfrt.time_pivot_selection_; // BFRT diagnostics timers.bfrt_calls++; if (step_length == 0.0) { timers.bfrt_zero_steps++; - timers.bfrt_zero_step_num_buckets_sum += bfrt.num_buckets_used_; - timers.bfrt_zero_step_bucket0_sum += bfrt.bucket0_size_; - timers.bfrt_zero_step_num_breakpoints_sum += bfrt.num_breakpoints_; - timers.bfrt_zero_step_harris_zero_sum += bfrt.num_harris_zero_; - timers.bfrt_zero_step_exact_zero_sum += bfrt.num_exact_zero_; - } - if (bfrt.num_buckets_used_ == 0) { - timers.bfrt_single_pass_only++; - } else { - timers.bfrt_bucket_used++; - if (bfrt.used_fallback_) { - timers.bfrt_fallback++; - } else if (bfrt.bucket_selected_ < bfrt.num_buckets_used_ - 1) { - timers.bfrt_not_last_bucket++; - } - if (bfrt.selected_is_slope_breaker_) { - timers.bfrt_selected_slope_breaker++; - } else { - timers.bfrt_not_slope_breaker++; - } } } else { entering_index = phase2::phase2_ratio_test( From 60efd8ba8fe317e2ccc3d5bf88d0ea3b795eb3a7 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 10 Sep 2026 13:26:10 -0700 Subject: [PATCH 058/113] Remove tiny perturbations before retrying dual simplex --- cpp/src/dual_simplex/phase2.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 9eb3224817..7213861fe4 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2492,7 +2492,7 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, // Check if there's any perturbation const f_t perturbation = amount_of_perturbation(lp, objective); - if (perturbation <= 1e-6) return 0; // OPTIMAL + if (perturbation == 0.0) return 0; // OPTIMAL // Count perturbations on basic vs nonbasic variables i_t num_basic_perturbed = 0; From 4d9309cbf98a8dbcbb5d2fe03453925036401299 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 11:18:23 -0700 Subject: [PATCH 059/113] Remove BFRT debug code --- .../bound_flipping_ratio_test.cpp | 50 --------------- .../bound_flipping_ratio_test.hpp | 20 ------ cpp/src/dual_simplex/phase2.cpp | 64 +------------------ 3 files changed, 2 insertions(+), 132 deletions(-) diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index d18ed95e90..b73e2b8957 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -138,17 +138,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, std::vector ratios(nz); std::vector harris_ratios(nz); work_estimate_ += 3 * nz; - double t0 = tic(); i_t num_breakpoints = compute_breakpoints(indicies, ratios, harris_ratios); - time_compute_breakpoints_ += toc(t0); - num_breakpoints_ = num_breakpoints; - // Count zero ratios - num_harris_zero_ = 0; - num_exact_zero_ = 0; - for (i_t k = 0; k < num_breakpoints; k++) { - if (harris_ratios[k] == 0.0) num_harris_zero_++; - if (ratios[k] == 0.0) num_exact_zero_++; - } work_estimate_ += 2 * num_breakpoints; if constexpr (verbose) { settings_.log.printf("Initial breakpoints %d\n", num_breakpoints); } if (num_breakpoints == 0) { @@ -161,7 +151,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, i_t entering_index = RATIO_TEST_NO_ENTERING_VARIABLE; f_t max_step_length; - t0 = tic(); i_t k_idx = single_pass(0, num_breakpoints, indicies, @@ -170,7 +159,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, nonbasic_entering, entering_index, max_step_length); - time_single_pass_ += toc(t0); if (k_idx == RATIO_TEST_NUMERICAL_ISSUES) { return RATIO_TEST_NUMERICAL_ISSUES; } // The variable selected by single_pass is guaranteed to be in the first bucket: it // defines the minimum Harris ratio, and its exact ratio is no greater than its Harris @@ -185,8 +173,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, entering_index, std::abs(delta_z_[entering_index])); } - num_buckets_used_ = 0; - step_length_result_ = step_length; determine_flips(step_length, entering_index, flip_indices); return entering_index; } @@ -296,7 +282,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, i_t num_candidates = 0; // This is O( log10(max_step_length/min_step_length) * num_breakpoints) - t0 = tic(); while (total_slope >= 0.0 && coarse_threshold <= max_step_length && scan_start < num_breakpoints && !found_unbounded) { for (i_t h = scan_start; h < num_breakpoints; ++h) { @@ -317,7 +302,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, scan_start = num_candidates; coarse_threshold *= 10.0; } - time_coarse_filter_ += toc(t0); candidates.resize(num_candidates); @@ -357,8 +341,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, work_estimate_ += num_candidates + 1; // This is O(num_buckets * num_candidates) - i_t slope_breaker_k = -1; // the candidate k that made slope go negative - t0 = tic(); while (cumulative_slope >= 0.0 && scan_start < num_candidates && threshold <= max_step_length) { f_t next_threshold = inf; i_t write = scan_start; @@ -371,7 +353,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, const i_t j = nonbasic_list_[indicies[k]]; if (bounded_variables_[j]) { cumulative_slope -= std::abs(delta_z_[j]) * (upper_[j] - lower_[j]); - if (cumulative_slope < 0.0 && slope_breaker_k < 0) { slope_breaker_k = k; } } std::swap(candidates[h], candidates[write]); write++; @@ -390,8 +371,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, if (cumulative_slope < 0.0) break; } - time_bucket_sort_ += toc(t0); - bucket0_size_ = (num_buckets > 0) ? bucket_start[1] : 0; // Compute the maximum pivot // This is O(num_candidates) @@ -433,13 +412,8 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, } // Step = entering variable's breakpoint ratio - num_buckets_used_ = num_buckets; if (entering_k < 0) { // Fallback to single_pass result - used_fallback_ = true; - bucket_selected_ = -1; - step_length_result_ = step_length; - selected_is_slope_breaker_ = false; determine_flips(step_length, entering_index, flip_indices); return entering_index; } @@ -447,30 +421,6 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, nonbasic_entering = indicies[entering_k]; entering_index = nonbasic_list_[nonbasic_entering]; - // Record whether we selected the slope breaker - selected_is_slope_breaker_ = (entering_k == slope_breaker_k); - - // Record which bucket was selected - used_fallback_ = false; - i_t pos = -1; - for (i_t b = 0; b < num_buckets; b++) { - if (entering_k >= 0) { - // Find which bucket entering_k is in based on its position in candidates - pos = -1; - for (i_t h = 0; h < num_candidates; h++) { - if (candidates[h] == entering_k) { - pos = h; - break; - } - } - if (pos >= bucket_start[b] && pos < bucket_start[b + 1]) { - bucket_selected_ = b; - break; - } - } - } - work_estimate_ += (bucket_selected_ + 1) * (pos + 3); - step_length_result_ = step_length; determine_flips(step_length, entering_index, flip_indices); return entering_index; diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp index 4587037889..8076be5927 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp @@ -56,26 +56,6 @@ class bound_flipping_ratio_test_t { i_t compute_step_length(f_t& step_length, i_t& nonbasic_entering, std::vector& flip_indices); f_t work_estimate() const { return work_estimate_; } - // Timing fields (filled by compute_step_length) - f_t time_compute_breakpoints_{0.0}; - f_t time_single_pass_{0.0}; - f_t time_coarse_filter_{0.0}; - f_t time_bucket_sort_{0.0}; - f_t time_pivot_selection_{0.0}; - - // Diagnostic fields - i_t num_buckets_used_{0}; // number of buckets in bucket sort - i_t bucket_selected_{ - -1}; // which bucket the entering variable came from (-1 = single_pass/fallback) - f_t step_length_result_{0.0}; // the step length chosen - bool used_fallback_{false}; // true if we fell back to single_pass result - i_t bucket0_size_{0}; // size of first bucket (candidates with ratio <= min_harris) - i_t num_breakpoints_{0}; // total breakpoints computed - bool selected_is_slope_breaker_{ - false}; // true if we selected the variable that made slope go negative - i_t num_harris_zero_{0}; // number of harris_ratios that are exactly 0 - i_t num_exact_zero_{0}; // number of exact ratios that are exactly 0 - private: i_t compute_breakpoints(std::vector& indices, std::vector& ratios, diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 7213861fe4..ba16747b2a 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2806,28 +2806,9 @@ class phase2_timers_t { update_infeasibility_time.work; // clang-format off print_one(settings, "BFRT time", bfrt_time, total_time, total_work); - if (bfrt_time.time > 0.1) { - settings.log.printf(" BFRT breakpoints: %.2fs\n", bfrt_breakpoints_time); - settings.log.printf(" BFRT single_pass: %.2fs\n", bfrt_single_pass_time); - settings.log.printf(" BFRT coarse: %.2fs\n", bfrt_coarse_time); - settings.log.printf(" BFRT bucket: %.2fs\n", bfrt_bucket_time); - settings.log.printf(" BFRT select: %.2fs\n", bfrt_select_time); - } if (bfrt_calls > 0) { - settings.log.printf(" BFRT calls: %d, zero_steps: %d (%.1f%%), single_pass_only: %d, bucket_used: %d, not_last_bucket: %d, fallback: %d\n", - bfrt_calls, bfrt_zero_steps, 100.0 * bfrt_zero_steps / bfrt_calls, - bfrt_single_pass_only, bfrt_bucket_used, bfrt_not_last_bucket, bfrt_fallback); - settings.log.printf(" BFRT slope_breaker: %d, not_slope_breaker: %d (%.1f%%)\n", - bfrt_selected_slope_breaker, bfrt_not_slope_breaker, - bfrt_bucket_used > 0 ? 100.0 * bfrt_not_slope_breaker / bfrt_bucket_used : 0.0); - if (bfrt_zero_steps > 0) { - settings.log.printf(" BFRT zero-step avg: num_buckets=%.1f, bucket0_size=%.1f, num_breakpoints=%.1f, harris_zero=%.1f, exact_zero=%.1f\n", - 1.0 * bfrt_zero_step_num_buckets_sum / bfrt_zero_steps, - 1.0 * bfrt_zero_step_bucket0_sum / bfrt_zero_steps, - 1.0 * bfrt_zero_step_num_breakpoints_sum / bfrt_zero_steps, - 1.0 * bfrt_zero_step_harris_zero_sum / bfrt_zero_steps, - 1.0 * bfrt_zero_step_exact_zero_sum / bfrt_zero_steps); - } + settings.log.printf(" BFRT calls: %d, zero_steps: %d (%.1f%%)\n", + bfrt_calls, bfrt_zero_steps, 100.0 * bfrt_zero_steps / bfrt_calls); } print_one(settings, "Pricing time", pricing_time, total_time, total_work); print_one(settings, "BTran time", btran_time, total_time, total_work); @@ -2849,25 +2830,9 @@ class phase2_timers_t { // clang-format on } work_timer_t bfrt_time; - f_t bfrt_breakpoints_time{0.0}; - f_t bfrt_single_pass_time{0.0}; - f_t bfrt_coarse_time{0.0}; - f_t bfrt_bucket_time{0.0}; - f_t bfrt_select_time{0.0}; // BFRT diagnostic counters i_t bfrt_calls{0}; i_t bfrt_zero_steps{0}; // step_length == 0 - i_t bfrt_single_pass_only{0}; // no bound flips (single_pass decided) - i_t bfrt_bucket_used{0}; // bucket sort was used - i_t bfrt_not_last_bucket{0}; // selected from a bucket other than the last - i_t bfrt_fallback{0}; // fell back to single_pass result after bucket sort - i_t bfrt_zero_step_num_buckets_sum{0}; // sum of num_buckets on zero-step iters - i_t bfrt_zero_step_bucket0_sum{0}; // sum of bucket0 size on zero-step iters - i_t bfrt_zero_step_num_breakpoints_sum{0}; // sum of num_breakpoints on zero-step iters - i_t bfrt_zero_step_harris_zero_sum{0}; // sum of harris_ratios==0 on zero-step iters - i_t bfrt_zero_step_exact_zero_sum{0}; // sum of exact ratios==0 on zero-step iters - i_t bfrt_selected_slope_breaker{0}; // times we selected the slope breaker - i_t bfrt_not_slope_breaker{0}; // times we selected something else work_timer_t pricing_time; work_timer_t btran_time; work_timer_t ftran_time; @@ -3716,35 +3681,10 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, return dual_status_t::NUMERICAL; } timers.bfrt_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); - timers.bfrt_breakpoints_time += bfrt.time_compute_breakpoints_; - timers.bfrt_single_pass_time += bfrt.time_single_pass_; - timers.bfrt_coarse_time += bfrt.time_coarse_filter_; - timers.bfrt_bucket_time += bfrt.time_bucket_sort_; - timers.bfrt_select_time += bfrt.time_pivot_selection_; // BFRT diagnostics timers.bfrt_calls++; if (step_length == 0.0) { timers.bfrt_zero_steps++; - timers.bfrt_zero_step_num_buckets_sum += bfrt.num_buckets_used_; - timers.bfrt_zero_step_bucket0_sum += bfrt.bucket0_size_; - timers.bfrt_zero_step_num_breakpoints_sum += bfrt.num_breakpoints_; - timers.bfrt_zero_step_harris_zero_sum += bfrt.num_harris_zero_; - timers.bfrt_zero_step_exact_zero_sum += bfrt.num_exact_zero_; - } - if (bfrt.num_buckets_used_ == 0) { - timers.bfrt_single_pass_only++; - } else { - timers.bfrt_bucket_used++; - if (bfrt.used_fallback_) { - timers.bfrt_fallback++; - } else if (bfrt.bucket_selected_ < bfrt.num_buckets_used_ - 1) { - timers.bfrt_not_last_bucket++; - } - if (bfrt.selected_is_slope_breaker_) { - timers.bfrt_selected_slope_breaker++; - } else { - timers.bfrt_not_slope_breaker++; - } } } else { entering_index = phase2::phase2_ratio_test( From f6f82f43af05a1d16a5c6793788864c156f6d79d Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 12:07:56 -0700 Subject: [PATCH 060/113] Remove more debugging code --- cpp/src/dual_simplex/phase2.cpp | 33 +++++++++++---------------------- 1 file changed, 11 insertions(+), 22 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index ba16747b2a..7cf7dc395f 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -484,7 +484,7 @@ void initial_perturbation(const lp_problem_t& lp, // that case. The original costs are restored before declaring optimality. const f_t perturbation_base = (strongly_degenerate ? 1e-5 : 5e-7) * max_abs_obj_coeff; - settings.log.printf( + settings.log.debug( "Perturbation debug: max_abs_obj_coeff=%e (dampened), perturbation_base=%e, n=%d, " "num_boxed=%d\n", max_abs_obj_coeff, @@ -2424,7 +2424,7 @@ i_t set_primal_variables_on_bounds(const lp_problem_t& lp, i_t total_changes = num_fixed_to_lower + num_fixed_to_upper + num_lower_to_upper + num_upper_to_lower + num_set_fixed; if (total_changes > 0) { - settings.log.printf( + settings.log.debug( "set_primal_variables_on_bounds: %d changes (fixed->lower=%d, fixed->upper=%d, " "lower->upper=%d, upper->lower=%d, ->fixed=%d)\n", total_changes, @@ -2555,7 +2555,7 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, work_estimate += 4 * m + 2 * n; if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL - settings.log.printf("Removed perturbation. Continuing dual simplex (primal_inf=%.2e)\n", + settings.log.debug("Removed perturbation. Continuing dual simplex (primal_inf=%.2e)\n", primal_infeasibility); return 1; // CONTINUE_DUAL } @@ -2611,7 +2611,7 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, if (residual_dual_infeas > settings.dual_tol) { // One-sided infeasibility remains — can't continue with dual simplex. // new_vstatus is discarded; vstatus unchanged. - settings.log.printf( + settings.log.debug( "Perturbation removal: %d flips, residual_dual_infeas=%.2e (PRIMAL_CLEANUP)\n", num_flipped, residual_dual_infeas); @@ -2638,10 +2638,10 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, primal_infeasibility); work_estimate += 4 * m + 2 * n; - settings.log.printf( + settings.log.debug( "Perturbation removal: %d flips, primal_inf=%.2e\n", num_flipped, primal_infeasibility); if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL - settings.log.printf("Continuing dual after flip (primal_inf=%.2e)\n", primal_infeasibility); + settings.log.debug("Continuing dual after flip (primal_inf=%.2e)\n", primal_infeasibility); return 1; // CONTINUE_DUAL } @@ -3014,7 +3014,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } } } - settings.log.printf( + settings.log.debug( "NONBASIC_FIXED boxed: %d, degenerate (|z_j| < dual_tol): %d\n", num_fixed, num_degen); } @@ -3082,7 +3082,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } vstatus = best_vstatus; x = best_x; - settings.log.printf( + settings.log.debug( "Bound assignment: default(%d/%.2e) colsum(%d/%.2e) abs-bound(%d/%.2e) -> %s\n", all_num_infeas[0], all_sum_infeas[0], @@ -3107,7 +3107,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } bool near_optimal = (num_primal_infeas < 1000 && max_primal_infeas < 1e-3); bool apply_perturbation = (settings.initial_perturbation == 1) || !near_optimal; - settings.log.printf( + settings.log.debug( "Near-optimal check: num_primal_infeas=%d, max_primal_infeas=%.2e, near_optimal=%d, " "apply_perturbation=%d\n", num_primal_infeas, @@ -3281,10 +3281,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t sparse_delta_z = 0; i_t dense_delta_z = 0; i_t num_refactors = 0; - i_t total_bound_flips = 0; - i_t max_bound_flips = 0; f_t delta_y_nz_percentage = 0.0; - phase2::phase2_timers_t timers(true); + phase2::phase2_timers_t timers(false); // Sparse vectors for main loop (declared outside loop for instrumentation) sparse_vector_t delta_y_sparse(m, 0); @@ -3489,7 +3487,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } if (removal_status == 2) { // PRIMAL_CLEANUP const f_t perturbation = phase2::amount_of_perturbation(lp, objective); - settings.log.printf("Failed to remove perturbation of %.2e.\n", perturbation); + settings.log.printf("Failed to remove perturbation of %.2e. Using primal simplex for cleanup.\n", perturbation); settings.log.printf("Num updates: %d\n", ft.num_updates()); settings.log.printf("Iterations: %d\n", iter); i_t dual_iter = iter; @@ -3847,8 +3845,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, : 0; timers.flip_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); - total_bound_flips += num_flipped; - if (num_flipped > max_bound_flips) max_bound_flips = num_flipped; delta_xB_0_sparse.clear(); if (num_flipped > 0) { @@ -4237,13 +4233,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (phase == 2) { timers.print_timers(settings); - i_t num_iters = iter - start_iter; - if (num_iters > 0) { - settings.log.printf("Bound flips: total=%d, avg=%.1f, max=%d\n", - total_bound_flips, - 1.0 * total_bound_flips / num_iters, - max_bound_flips); - } constexpr bool print_stats = false; if constexpr (print_stats) { settings.log.printf("Sparse delta_z %8d %8.2f%\n", From 3e041794679e1507cb2eb42271e1a427d3b3b80c Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 12:26:43 -0700 Subject: [PATCH 061/113] Remove unused unscaled max obj coeff --- cpp/src/dual_simplex/phase2.cpp | 4 ++-- cpp/src/dual_simplex/simplex_solver_settings.hpp | 2 -- cpp/src/dual_simplex/solve.cpp | 8 -------- 3 files changed, 2 insertions(+), 12 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 7cf7dc395f..2e5fd918f8 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -1633,7 +1633,7 @@ i_t compute_perturbation(const lp_problem_t& lp, sum_perturb += violation; } } - // On degenerate steps, shift the entering variable's cost (like HiGHS) + // On degenerate steps, shift the entering variable's cost // This accumulates shifts that break degeneracy at the next refactorization if (entering_index >= 0 && step_length == 0.0) { assert(vstatus[entering_index] != variable_status_t::BASIC); @@ -2325,7 +2325,7 @@ i_t set_primal_variables_on_bounds(const lp_problem_t& lp, vstatus[j] = variable_status_t::NONBASIC_LOWER; } } else { - // degen_type == 3: abs_bound (prefer bound closer to zero, like HiGHS) + // degen_type == 3: abs_bound (prefer bound closer to zero) if (std::abs(lp.upper[j]) < std::abs(lp.lower[j])) { x[j] = lp.upper[j]; vstatus[j] = variable_status_t::NONBASIC_UPPER; diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index bed2c4eee5..577a82e1b1 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -91,7 +91,6 @@ struct simplex_solver_settings_t { barrier_initial_point_safeguard(10.0), check_Q(false), crossover(false), - unscaled_max_abs_obj_coeff(-1.0), refactor_frequency(100), iteration_log_frequency(1000), first_iteration_log(2), @@ -208,7 +207,6 @@ struct simplex_solver_settings_t { // the interior of the nonnegative orthant / SOC bool check_Q; // true to check if Q is positive semidefinite bool crossover; // true to do crossover, false to not - f_t unscaled_max_abs_obj_coeff; // max |c_j| before scaling (-1 = not set, compute from lp) i_t refactor_frequency; // number of basis updates before refactorization i_t iteration_log_frequency; // number of iterations between log updates i_t first_iteration_log; // number of iterations to log at beginning of solve diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index 867225d79e..ba69699af0 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -227,14 +227,6 @@ lp_status_t solve_linear_program_with_advanced_basis( presolved_lp.A.col_start[presolved_lp.num_cols]); std::vector column_scales; std::vector row_scales_simplex; - // Compute max |c_j| before scaling for perturbation calibration - if (settings.unscaled_max_abs_obj_coeff < 0.0) { - f_t max_obj = 0.0; - for (i_t j = 0; j < presolved_lp.num_cols; ++j) { - max_obj = std::max(max_obj, std::abs(presolved_lp.objective[j])); - } - const_cast&>(settings).unscaled_max_abs_obj_coeff = max_obj; - } { raft::common::nvtx::range scope_scaling("DualSimplex::scaling"); scaling(presolved_lp, settings, lp, column_scales, row_scales_simplex); From 2d948ecc24f52b2cb85e5c260f416540e026fd66 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 12:26:43 -0700 Subject: [PATCH 062/113] Remove unused unscaled max obj coeff --- cpp/src/dual_simplex/phase2.cpp | 4 ++-- cpp/src/dual_simplex/simplex_solver_settings.hpp | 2 -- cpp/src/dual_simplex/solve.cpp | 8 -------- 3 files changed, 2 insertions(+), 12 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index ba16747b2a..d73e656694 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -1633,7 +1633,7 @@ i_t compute_perturbation(const lp_problem_t& lp, sum_perturb += violation; } } - // On degenerate steps, shift the entering variable's cost (like HiGHS) + // On degenerate steps, shift the entering variable's cost // This accumulates shifts that break degeneracy at the next refactorization if (entering_index >= 0 && step_length == 0.0) { assert(vstatus[entering_index] != variable_status_t::BASIC); @@ -2325,7 +2325,7 @@ i_t set_primal_variables_on_bounds(const lp_problem_t& lp, vstatus[j] = variable_status_t::NONBASIC_LOWER; } } else { - // degen_type == 3: abs_bound (prefer bound closer to zero, like HiGHS) + // degen_type == 3: abs_bound (prefer bound closer to zero) if (std::abs(lp.upper[j]) < std::abs(lp.lower[j])) { x[j] = lp.upper[j]; vstatus[j] = variable_status_t::NONBASIC_UPPER; diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 54af1b7d83..9669a17eed 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -91,7 +91,6 @@ struct simplex_solver_settings_t { barrier_initial_point_safeguard(10.0), check_Q(false), crossover(false), - unscaled_max_abs_obj_coeff(-1.0), refactor_frequency(100), iteration_log_frequency(1000), first_iteration_log(2), @@ -205,7 +204,6 @@ struct simplex_solver_settings_t { // the interior of the nonnegative orthant / SOC bool check_Q; // true to check if Q is positive semidefinite bool crossover; // true to do crossover, false to not - f_t unscaled_max_abs_obj_coeff; // max |c_j| before scaling (-1 = not set, compute from lp) i_t refactor_frequency; // number of basis updates before refactorization i_t iteration_log_frequency; // number of iterations between log updates i_t first_iteration_log; // number of iterations to log at beginning of solve diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index 867225d79e..ba69699af0 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -227,14 +227,6 @@ lp_status_t solve_linear_program_with_advanced_basis( presolved_lp.A.col_start[presolved_lp.num_cols]); std::vector column_scales; std::vector row_scales_simplex; - // Compute max |c_j| before scaling for perturbation calibration - if (settings.unscaled_max_abs_obj_coeff < 0.0) { - f_t max_obj = 0.0; - for (i_t j = 0; j < presolved_lp.num_cols; ++j) { - max_obj = std::max(max_obj, std::abs(presolved_lp.objective[j])); - } - const_cast&>(settings).unscaled_max_abs_obj_coeff = max_obj; - } { raft::common::nvtx::range scope_scaling("DualSimplex::scaling"); scaling(presolved_lp, settings, lp, column_scales, row_scales_simplex); From ae8f9030d773fffe3e9d9c89f5650284e9fb87b2 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 12:07:56 -0700 Subject: [PATCH 063/113] Remove more debugging code --- cpp/src/dual_simplex/phase2.cpp | 33 +++++++++++---------------------- 1 file changed, 11 insertions(+), 22 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index d73e656694..2e5fd918f8 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -484,7 +484,7 @@ void initial_perturbation(const lp_problem_t& lp, // that case. The original costs are restored before declaring optimality. const f_t perturbation_base = (strongly_degenerate ? 1e-5 : 5e-7) * max_abs_obj_coeff; - settings.log.printf( + settings.log.debug( "Perturbation debug: max_abs_obj_coeff=%e (dampened), perturbation_base=%e, n=%d, " "num_boxed=%d\n", max_abs_obj_coeff, @@ -2424,7 +2424,7 @@ i_t set_primal_variables_on_bounds(const lp_problem_t& lp, i_t total_changes = num_fixed_to_lower + num_fixed_to_upper + num_lower_to_upper + num_upper_to_lower + num_set_fixed; if (total_changes > 0) { - settings.log.printf( + settings.log.debug( "set_primal_variables_on_bounds: %d changes (fixed->lower=%d, fixed->upper=%d, " "lower->upper=%d, upper->lower=%d, ->fixed=%d)\n", total_changes, @@ -2555,7 +2555,7 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, work_estimate += 4 * m + 2 * n; if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL - settings.log.printf("Removed perturbation. Continuing dual simplex (primal_inf=%.2e)\n", + settings.log.debug("Removed perturbation. Continuing dual simplex (primal_inf=%.2e)\n", primal_infeasibility); return 1; // CONTINUE_DUAL } @@ -2611,7 +2611,7 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, if (residual_dual_infeas > settings.dual_tol) { // One-sided infeasibility remains — can't continue with dual simplex. // new_vstatus is discarded; vstatus unchanged. - settings.log.printf( + settings.log.debug( "Perturbation removal: %d flips, residual_dual_infeas=%.2e (PRIMAL_CLEANUP)\n", num_flipped, residual_dual_infeas); @@ -2638,10 +2638,10 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, primal_infeasibility); work_estimate += 4 * m + 2 * n; - settings.log.printf( + settings.log.debug( "Perturbation removal: %d flips, primal_inf=%.2e\n", num_flipped, primal_infeasibility); if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL - settings.log.printf("Continuing dual after flip (primal_inf=%.2e)\n", primal_infeasibility); + settings.log.debug("Continuing dual after flip (primal_inf=%.2e)\n", primal_infeasibility); return 1; // CONTINUE_DUAL } @@ -3014,7 +3014,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } } } - settings.log.printf( + settings.log.debug( "NONBASIC_FIXED boxed: %d, degenerate (|z_j| < dual_tol): %d\n", num_fixed, num_degen); } @@ -3082,7 +3082,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } vstatus = best_vstatus; x = best_x; - settings.log.printf( + settings.log.debug( "Bound assignment: default(%d/%.2e) colsum(%d/%.2e) abs-bound(%d/%.2e) -> %s\n", all_num_infeas[0], all_sum_infeas[0], @@ -3107,7 +3107,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } bool near_optimal = (num_primal_infeas < 1000 && max_primal_infeas < 1e-3); bool apply_perturbation = (settings.initial_perturbation == 1) || !near_optimal; - settings.log.printf( + settings.log.debug( "Near-optimal check: num_primal_infeas=%d, max_primal_infeas=%.2e, near_optimal=%d, " "apply_perturbation=%d\n", num_primal_infeas, @@ -3281,10 +3281,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t sparse_delta_z = 0; i_t dense_delta_z = 0; i_t num_refactors = 0; - i_t total_bound_flips = 0; - i_t max_bound_flips = 0; f_t delta_y_nz_percentage = 0.0; - phase2::phase2_timers_t timers(true); + phase2::phase2_timers_t timers(false); // Sparse vectors for main loop (declared outside loop for instrumentation) sparse_vector_t delta_y_sparse(m, 0); @@ -3489,7 +3487,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } if (removal_status == 2) { // PRIMAL_CLEANUP const f_t perturbation = phase2::amount_of_perturbation(lp, objective); - settings.log.printf("Failed to remove perturbation of %.2e.\n", perturbation); + settings.log.printf("Failed to remove perturbation of %.2e. Using primal simplex for cleanup.\n", perturbation); settings.log.printf("Num updates: %d\n", ft.num_updates()); settings.log.printf("Iterations: %d\n", iter); i_t dual_iter = iter; @@ -3847,8 +3845,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, : 0; timers.flip_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); - total_bound_flips += num_flipped; - if (num_flipped > max_bound_flips) max_bound_flips = num_flipped; delta_xB_0_sparse.clear(); if (num_flipped > 0) { @@ -4237,13 +4233,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (phase == 2) { timers.print_timers(settings); - i_t num_iters = iter - start_iter; - if (num_iters > 0) { - settings.log.printf("Bound flips: total=%d, avg=%.1f, max=%d\n", - total_bound_flips, - 1.0 * total_bound_flips / num_iters, - max_bound_flips); - } constexpr bool print_stats = false; if constexpr (print_stats) { settings.log.printf("Sparse delta_z %8d %8.2f%\n", From 6deb6bbffb733c68a9d88fc9ffd397f5a9e2894a Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 12:47:49 -0700 Subject: [PATCH 064/113] Check time and concurrent halt in BFRT --- cpp/src/dual_simplex/bound_flipping_ratio_test.cpp | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index b73e2b8957..8e628a9dd2 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -342,6 +342,10 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // This is O(num_buckets * num_candidates) while (cumulative_slope >= 0.0 && scan_start < num_candidates && threshold <= max_step_length) { + if (toc(start_time_) > settings_.time_limit) { return RATIO_TEST_TIME_LIMIT; } + if (settings_.concurrent_halt != nullptr && *settings_.concurrent_halt == 1) { + return CONCURRENT_HALT_RETURN; + } f_t next_threshold = inf; i_t write = scan_start; From 3311cdb96c88f96d7eb9e744ab201a03d81fb2ef Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 13:17:53 -0700 Subject: [PATCH 065/113] Improve handling of primal simplex return status --- cpp/src/dual_simplex/phase2.cpp | 23 ++++++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 2e5fd918f8..6deb8c8aa0 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -3506,11 +3506,32 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (primal_status == primal_status_t::OPTIMAL) { settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); objective = lp.objective; + } else if (primal_status == primal_status_t::TIME_LIMIT) { + return dual_status_t::TIME_LIMIT; + } else if (primal_status == primal_status_t::CONCURRENT_LIMIT) { + return dual_status_t::CONCURRENT_LIMIT; + } else if (primal_status == primal_status_t::WORK_LIMIT) { + return dual_status_t::WORK_LIMIT; + } else if (primal_status == primal_status_t::ITERATION_LIMIT) { + return dual_status_t::ITERATION_LIMIT; } else { settings.log.printf("Primal cleanup failed.\n"); + const f_t primal_infeas = phase2::primal_infeasibility(lp, settings, vstatus, sol.x); const f_t dual_infeas = phase2::dual_infeasibility( lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); - if (dual_infeas > 10.0 * settings.dual_tol) { return dual_status_t::NUMERICAL; } + // Failed cleanup may leave duals from phase I or from the previous basis. + const f_t primal_residual = phase2::l2_primal_residual(lp, sol); + const f_t dual_residual = phase2::l2_dual_residual(lp, sol); + phase2_work_estimate += 4.0 * lp.A.nnz() + 3 * m + 4 * n; + bool is_optimal = primal_infeas <= 10.0 * settings.primal_tol && + dual_infeas <= 10.0 * settings.dual_tol && + primal_residual <= settings.primal_tol && + dual_residual <= settings.dual_tol; + for (const i_t j : basic_list) { + is_optimal = is_optimal && std::abs(sol.z[j]) <= settings.dual_tol; + } + phase2_work_estimate += m; + if (!is_optimal) { return dual_status_t::NUMERICAL; } } } // removal_status == 0 (OPTIMAL) or primal cleanup done: fall through to prepare_optimality From 2920576dd5e7d12b26bd6d52670daa178b9f435b Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 13:19:18 -0700 Subject: [PATCH 066/113] Remove primal clean up. We now do this inside dual simplex --- cpp/src/dual_simplex/solve.cpp | 6 ------ 1 file changed, 6 deletions(-) diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index ba69699af0..9541f94304 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -348,12 +348,6 @@ lp_status_t solve_linear_program_with_advanced_basis( edge_norms, work_unit_context); } - constexpr bool primal_cleanup = false; - if (status == dual_status_t::OPTIMAL && primal_cleanup) { - settings.log.printf("Running primal cleanup\n"); - primal_phase2(2, start_time, lp, settings, vstatus, solution, iter); - // TODO: We need to update ft if the basis changed - } if (settings.inside_mip == 1 && settings.concurrent_halt != nullptr) { settings.log.debug("Setting concurrent halt to 1 inside_mip\n"); *settings.concurrent_halt = 1; From 52d8ccb960f8a34525e0197df9e5de2b9cf2f33d Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 12:47:49 -0700 Subject: [PATCH 067/113] Check time and concurrent halt in BFRT --- cpp/src/dual_simplex/bound_flipping_ratio_test.cpp | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index b73e2b8957..8e628a9dd2 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -342,6 +342,10 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, // This is O(num_buckets * num_candidates) while (cumulative_slope >= 0.0 && scan_start < num_candidates && threshold <= max_step_length) { + if (toc(start_time_) > settings_.time_limit) { return RATIO_TEST_TIME_LIMIT; } + if (settings_.concurrent_halt != nullptr && *settings_.concurrent_halt == 1) { + return CONCURRENT_HALT_RETURN; + } f_t next_threshold = inf; i_t write = scan_start; From ffaec1137f331173e2db34b78e1a24d1dcacb5f4 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 13:17:53 -0700 Subject: [PATCH 068/113] Improve handling of primal simplex return status --- cpp/src/dual_simplex/phase2.cpp | 23 ++++++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 2e5fd918f8..6deb8c8aa0 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -3506,11 +3506,32 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (primal_status == primal_status_t::OPTIMAL) { settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); objective = lp.objective; + } else if (primal_status == primal_status_t::TIME_LIMIT) { + return dual_status_t::TIME_LIMIT; + } else if (primal_status == primal_status_t::CONCURRENT_LIMIT) { + return dual_status_t::CONCURRENT_LIMIT; + } else if (primal_status == primal_status_t::WORK_LIMIT) { + return dual_status_t::WORK_LIMIT; + } else if (primal_status == primal_status_t::ITERATION_LIMIT) { + return dual_status_t::ITERATION_LIMIT; } else { settings.log.printf("Primal cleanup failed.\n"); + const f_t primal_infeas = phase2::primal_infeasibility(lp, settings, vstatus, sol.x); const f_t dual_infeas = phase2::dual_infeasibility( lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); - if (dual_infeas > 10.0 * settings.dual_tol) { return dual_status_t::NUMERICAL; } + // Failed cleanup may leave duals from phase I or from the previous basis. + const f_t primal_residual = phase2::l2_primal_residual(lp, sol); + const f_t dual_residual = phase2::l2_dual_residual(lp, sol); + phase2_work_estimate += 4.0 * lp.A.nnz() + 3 * m + 4 * n; + bool is_optimal = primal_infeas <= 10.0 * settings.primal_tol && + dual_infeas <= 10.0 * settings.dual_tol && + primal_residual <= settings.primal_tol && + dual_residual <= settings.dual_tol; + for (const i_t j : basic_list) { + is_optimal = is_optimal && std::abs(sol.z[j]) <= settings.dual_tol; + } + phase2_work_estimate += m; + if (!is_optimal) { return dual_status_t::NUMERICAL; } } } // removal_status == 0 (OPTIMAL) or primal cleanup done: fall through to prepare_optimality From 2ede6e125f72c3a199e540a447b76bd4ac7b5656 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 13:19:18 -0700 Subject: [PATCH 069/113] Remove primal clean up. We now do this inside dual simplex --- cpp/src/dual_simplex/solve.cpp | 6 ------ 1 file changed, 6 deletions(-) diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index ba69699af0..9541f94304 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -348,12 +348,6 @@ lp_status_t solve_linear_program_with_advanced_basis( edge_norms, work_unit_context); } - constexpr bool primal_cleanup = false; - if (status == dual_status_t::OPTIMAL && primal_cleanup) { - settings.log.printf("Running primal cleanup\n"); - primal_phase2(2, start_time, lp, settings, vstatus, solution, iter); - // TODO: We need to update ft if the basis changed - } if (settings.inside_mip == 1 && settings.concurrent_halt != nullptr) { settings.log.debug("Setting concurrent halt to 1 inside_mip\n"); *settings.concurrent_halt = 1; From 74146b9ee34463f6476bdd08dcae0848653a3cd7 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 13:53:51 -0700 Subject: [PATCH 070/113] Fix comment --- cpp/src/dual_simplex/primal.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 1a67956e47..8826c7bdca 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -822,8 +822,8 @@ primal_status_t primal_phase2(i_t phase, iter, work_estimate); } -// Note this implementation of primal simplex is experimental -// It is meant only to serve as a method to remove the perturbation to the objective +// Note this implementation of primal simplex is not well optimized +// It is really meant as a method to remove the perturbation to the objective // after dual simplex has found a primal feasible solution template primal_status_t primal_phase2_with_advanced_basis( From 53be7d42dc7e446d35319c7f8948bb2142fa0db5 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 13:53:51 -0700 Subject: [PATCH 071/113] Fix comment --- cpp/src/dual_simplex/primal.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 1a67956e47..8826c7bdca 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -822,8 +822,8 @@ primal_status_t primal_phase2(i_t phase, iter, work_estimate); } -// Note this implementation of primal simplex is experimental -// It is meant only to serve as a method to remove the perturbation to the objective +// Note this implementation of primal simplex is not well optimized +// It is really meant as a method to remove the perturbation to the objective // after dual simplex has found a primal feasible solution template primal_status_t primal_phase2_with_advanced_basis( From 88e37c5bd8f60dd5b01ea476aef36d5f01654f62 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 14:04:53 -0700 Subject: [PATCH 072/113] Fix messages --- cpp/src/pdlp/solve.cu | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index 63b2f9681e..bbeecde2d5 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -1884,15 +1884,15 @@ optimization_problem_solution_t solve_lp_with_method( } else { cuopt_expects(false, error_type_t::ValidationError, - "Invalid LP method. Valid values: Concurrent(0), PDLP(1), DualSimplex(2), " - "Barrier(3), Primal(4)."); + "Invalid LP method. Valid values: Concurrent(0), PDLP(1), Dual Simplex(2), " + "Barrier(3), Primal Simplex(4)."); return run_pdlp(problem, settings, timer, is_batch_mode); } } else { // Float precision only supports PDLP without presolve/crossover cuopt_expects(settings.method == method_t::PDLP, error_type_t::ValidationError, - "Float precision only supports PDLP method. DualSimplex, Primal, Barrier, and " + "Float precision only supports PDLP method. Dual Simplex, Primal Simplex, Barrier, and " "Concurrent require double precision."); return run_pdlp(problem, settings, timer, is_batch_mode); } From 8e0d79cae778c1355f4a90925f535fd7900ba536 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 14:04:53 -0700 Subject: [PATCH 073/113] Fix messages --- cpp/src/pdlp/solve.cu | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index 63b2f9681e..bbeecde2d5 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -1884,15 +1884,15 @@ optimization_problem_solution_t solve_lp_with_method( } else { cuopt_expects(false, error_type_t::ValidationError, - "Invalid LP method. Valid values: Concurrent(0), PDLP(1), DualSimplex(2), " - "Barrier(3), Primal(4)."); + "Invalid LP method. Valid values: Concurrent(0), PDLP(1), Dual Simplex(2), " + "Barrier(3), Primal Simplex(4)."); return run_pdlp(problem, settings, timer, is_batch_mode); } } else { // Float precision only supports PDLP without presolve/crossover cuopt_expects(settings.method == method_t::PDLP, error_type_t::ValidationError, - "Float precision only supports PDLP method. DualSimplex, Primal, Barrier, and " + "Float precision only supports PDLP method. Dual Simplex, Primal Simplex, Barrier, and " "Concurrent require double precision."); return run_pdlp(problem, settings, timer, is_batch_mode); } From 7e9545a190de9b4d623d9283cf7f3955a01c22f5 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 14:16:38 -0700 Subject: [PATCH 074/113] Remove unused variable --- cpp/src/dual_simplex/phase2.cpp | 2 -- 1 file changed, 2 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 6deb8c8aa0..88011a8c43 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -3294,8 +3294,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, sparse_vector_t v_sparse(m, 0); // For steepest edge norms sparse_vector_t atilde_sparse(m, 0); // For flip adjustments - // Track iteration interval start time for runtime measurement - [[maybe_unused]] f_t interval_start_time = toc(start_time); i_t last_feature_log_iter = iter; phase2_work_estimate += ft.work_estimate(); From b6a93183ebbb595aee7a1b69e4c0337c6627f6c9 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 14:49:24 -0700 Subject: [PATCH 075/113] Fix method. Add limit check at top of loop to avoid infinite loop --- cpp/src/dual_simplex/primal.cpp | 7 +++++++ cpp/src/math_optimization/solver_settings.cu | 2 +- 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 8826c7bdca..03ad284428 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -968,6 +968,13 @@ primal_status_t primal_phase2_with_advanced_basis( primal_timers_t timers(false); while (iter < iter_limit) { + if (toc(start_time) > settings.time_limit) { return primal_status_t::TIME_LIMIT; } + if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { + return primal_status_t::CONCURRENT_LIMIT; + } + if (work_estimate + basis_update.work_estimate() > settings.work_limit) { + return primal_status_t::WORK_LIMIT; + } timers.start_timer(work_estimate + basis_update.work_estimate()); i_t nonbasic_entering = -1; i_t direction; diff --git a/cpp/src/math_optimization/solver_settings.cu b/cpp/src/math_optimization/solver_settings.cu index 16c8dacf5d..12d3c67ed5 100644 --- a/cpp/src/math_optimization/solver_settings.cu +++ b/cpp/src/math_optimization/solver_settings.cu @@ -173,7 +173,7 @@ solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings {CUOPT_NODE_LIMIT, &mip_settings.node_limit, 0, std::numeric_limits::max(), std::numeric_limits::max()}, {CUOPT_PDLP_SOLVER_MODE, reinterpret_cast(&pdlp_settings.pdlp_solver_mode), CUOPT_PDLP_SOLVER_MODE_STABLE1, CUOPT_PDLP_SOLVER_MODE_STABLE3, CUOPT_PDLP_SOLVER_MODE_STABLE3}, {CUOPT_METHOD, reinterpret_cast(&pdlp_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_PRIMAL, CUOPT_METHOD_CONCURRENT}, - {CUOPT_METHOD, reinterpret_cast(&mip_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_BARRIER, CUOPT_METHOD_CONCURRENT}, + {CUOPT_METHOD, reinterpret_cast(&mip_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_PRIMAL, CUOPT_METHOD_CONCURRENT}, {CUOPT_NUM_CPU_THREADS, &mip_settings.num_cpu_threads, -1, std::numeric_limits::max(), -1}, {CUOPT_AUGMENTED, &pdlp_settings.augmented, -1, 1, -1}, {CUOPT_FOLDING, &pdlp_settings.folding, -1, 1, -1}, From 61ddbf1f19d8197865201379f3ae2c399dc5714e Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 14:49:24 -0700 Subject: [PATCH 076/113] Fix method. Add limit check at top of loop to avoid infinite loop --- cpp/src/dual_simplex/primal.cpp | 7 +++++++ cpp/src/math_optimization/solver_settings.cu | 4 ++-- 2 files changed, 9 insertions(+), 2 deletions(-) diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 8826c7bdca..03ad284428 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -968,6 +968,13 @@ primal_status_t primal_phase2_with_advanced_basis( primal_timers_t timers(false); while (iter < iter_limit) { + if (toc(start_time) > settings.time_limit) { return primal_status_t::TIME_LIMIT; } + if (settings.concurrent_halt != nullptr && *settings.concurrent_halt == 1) { + return primal_status_t::CONCURRENT_LIMIT; + } + if (work_estimate + basis_update.work_estimate() > settings.work_limit) { + return primal_status_t::WORK_LIMIT; + } timers.start_timer(work_estimate + basis_update.work_estimate()); i_t nonbasic_entering = -1; i_t direction; diff --git a/cpp/src/math_optimization/solver_settings.cu b/cpp/src/math_optimization/solver_settings.cu index e6ec6e0927..4a76b5813b 100644 --- a/cpp/src/math_optimization/solver_settings.cu +++ b/cpp/src/math_optimization/solver_settings.cu @@ -172,8 +172,8 @@ solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings {CUOPT_ITERATION_LIMIT, &pdlp_settings.iteration_limit, 0, std::numeric_limits::max(), std::numeric_limits::max()}, {CUOPT_NODE_LIMIT, &mip_settings.node_limit, 0, std::numeric_limits::max(), std::numeric_limits::max()}, {CUOPT_PDLP_SOLVER_MODE, reinterpret_cast(&pdlp_settings.pdlp_solver_mode), CUOPT_PDLP_SOLVER_MODE_STABLE1, CUOPT_PDLP_SOLVER_MODE_STABLE3, CUOPT_PDLP_SOLVER_MODE_STABLE3}, - {CUOPT_METHOD, reinterpret_cast(&pdlp_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_BARRIER, CUOPT_METHOD_CONCURRENT}, - {CUOPT_METHOD, reinterpret_cast(&mip_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_BARRIER, CUOPT_METHOD_CONCURRENT}, + {CUOPT_METHOD, reinterpret_cast(&pdlp_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_PRIMAL, CUOPT_METHOD_CONCURRENT}, + {CUOPT_METHOD, reinterpret_cast(&mip_settings.method), CUOPT_METHOD_CONCURRENT, CUOPT_METHOD_PRIMAL, CUOPT_METHOD_CONCURRENT}, {CUOPT_NUM_CPU_THREADS, &mip_settings.num_cpu_threads, -1, std::numeric_limits::max(), -1}, {CUOPT_AUGMENTED, &pdlp_settings.augmented, -1, 1, -1}, {CUOPT_FOLDING, &pdlp_settings.folding, -1, 1, -1}, From fa84590515e2f0a66c45aa18663c4b1c66693fa8 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 15:47:15 -0700 Subject: [PATCH 077/113] Refactor primal cleanup --- cpp/src/dual_simplex/phase2.cpp | 143 +++++++++++++++++++++----------- 1 file changed, 93 insertions(+), 50 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 6deb8c8aa0..a15737b7a8 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2645,6 +2645,70 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, return 1; // CONTINUE_DUAL } +template +dual_status_t run_primal_cleanup(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + f_t start_time, + basis_update_mpf_t& ft, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + std::vector& objective, + lp_solution_t& sol, + i_t& iter, + f_t& work_estimate) +{ + const f_t perturbation = amount_of_perturbation(lp, objective); + settings.log.printf( + "Failed to remove perturbation of %.2e. Using primal simplex for cleanup.\n", perturbation); + settings.log.printf("Num updates: %d\n", ft.num_updates()); + settings.log.printf("Iterations: %d\n", iter); + const i_t dual_iter = iter; + const primal_status_t primal_status = primal_phase2_with_advanced_basis(2, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + work_estimate, + false); + if (primal_status == primal_status_t::OPTIMAL) { + settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); + objective = lp.objective; + } else if (primal_status == primal_status_t::TIME_LIMIT) { + return dual_status_t::TIME_LIMIT; + } else if (primal_status == primal_status_t::CONCURRENT_LIMIT) { + return dual_status_t::CONCURRENT_LIMIT; + } else if (primal_status == primal_status_t::WORK_LIMIT) { + return dual_status_t::WORK_LIMIT; + } else if (primal_status == primal_status_t::ITERATION_LIMIT) { + return dual_status_t::ITERATION_LIMIT; + } else { + settings.log.printf("Primal cleanup failed.\n"); + const f_t primal_infeas = primal_infeasibility(lp, settings, vstatus, sol.x); + const f_t dual_infeas = dual_infeasibility( + lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); + // Failed cleanup may leave duals from phase I or from the previous basis. + const f_t primal_residual = l2_primal_residual(lp, sol); + const f_t dual_residual = l2_dual_residual(lp, sol); + work_estimate += 4.0 * lp.A.nnz() + 3 * lp.num_rows + 4 * lp.num_cols; + bool is_optimal = primal_infeas <= 10.0 * settings.primal_tol && + dual_infeas <= 10.0 * settings.dual_tol && + primal_residual <= settings.primal_tol && + dual_residual <= settings.dual_tol; + for (const i_t j : basic_list) { + is_optimal = is_optimal && std::abs(sol.z[j]) <= settings.dual_tol; + } + work_estimate += lp.num_rows; + if (!is_optimal) { return dual_status_t::NUMERICAL; } + } + return dual_status_t::OPTIMAL; +} + template void prepare_optimality(i_t info, f_t orig_primal_infeas, @@ -3486,53 +3550,18 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, continue; } if (removal_status == 2) { // PRIMAL_CLEANUP - const f_t perturbation = phase2::amount_of_perturbation(lp, objective); - settings.log.printf("Failed to remove perturbation of %.2e. Using primal simplex for cleanup.\n", perturbation); - settings.log.printf("Num updates: %d\n", ft.num_updates()); - settings.log.printf("Iterations: %d\n", iter); - i_t dual_iter = iter; - primal_status_t primal_status = primal_phase2_with_advanced_basis(2, - start_time, - lp, - settings, - vstatus, - ft, - basic_list, - nonbasic_list, - sol, - iter, - phase2_work_estimate, - false); - if (primal_status == primal_status_t::OPTIMAL) { - settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); - objective = lp.objective; - } else if (primal_status == primal_status_t::TIME_LIMIT) { - return dual_status_t::TIME_LIMIT; - } else if (primal_status == primal_status_t::CONCURRENT_LIMIT) { - return dual_status_t::CONCURRENT_LIMIT; - } else if (primal_status == primal_status_t::WORK_LIMIT) { - return dual_status_t::WORK_LIMIT; - } else if (primal_status == primal_status_t::ITERATION_LIMIT) { - return dual_status_t::ITERATION_LIMIT; - } else { - settings.log.printf("Primal cleanup failed.\n"); - const f_t primal_infeas = phase2::primal_infeasibility(lp, settings, vstatus, sol.x); - const f_t dual_infeas = phase2::dual_infeasibility( - lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); - // Failed cleanup may leave duals from phase I or from the previous basis. - const f_t primal_residual = phase2::l2_primal_residual(lp, sol); - const f_t dual_residual = phase2::l2_dual_residual(lp, sol); - phase2_work_estimate += 4.0 * lp.A.nnz() + 3 * m + 4 * n; - bool is_optimal = primal_infeas <= 10.0 * settings.primal_tol && - dual_infeas <= 10.0 * settings.dual_tol && - primal_residual <= settings.primal_tol && - dual_residual <= settings.dual_tol; - for (const i_t j : basic_list) { - is_optimal = is_optimal && std::abs(sol.z[j]) <= settings.dual_tol; - } - phase2_work_estimate += m; - if (!is_optimal) { return dual_status_t::NUMERICAL; } - } + const dual_status_t cleanup_status = phase2::run_primal_cleanup(lp, + settings, + start_time, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + sol, + iter, + phase2_work_estimate); + if (cleanup_status != dual_status_t::OPTIMAL) { return cleanup_status; } } // removal_status == 0 (OPTIMAL) or primal cleanup done: fall through to prepare_optimality } @@ -3735,10 +3764,25 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, primal_infeasibility, primal_infeasibility_squared, phase2_work_estimate); - if (removal_status == 0) { // OPTIMAL + if (removal_status == 2) { // PRIMAL_CLEANUP + const dual_status_t cleanup_status = phase2::run_primal_cleanup(lp, + settings, + start_time, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + sol, + iter, + phase2_work_estimate); + if (cleanup_status != dual_status_t::OPTIMAL) { return cleanup_status; } + objective = lp.objective; + } + if (removal_status == 0 || removal_status == 2) { // OPTIMAL or successful cleanup obj = phase2::compute_perturbed_objective(objective, x); phase2_work_estimate += 2 * n; - if (primal_infeasibility <= settings.primal_tol) { + if (removal_status == 2 || primal_infeasibility <= settings.primal_tol) { phase2_work_estimate += ft.work_estimate(); ft.clear_work_estimate(); phase2::prepare_optimality(1, @@ -3775,7 +3819,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); continue; } - // removal_status == 2 (PRIMAL_CLEANUP): fall through to existing logic below } if (perturbation == 0.0 && phase == 2) { From f084344ac9ae95e806f45204dbf85d43154cc790 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 15:47:15 -0700 Subject: [PATCH 078/113] Refactor primal cleanup --- cpp/src/dual_simplex/phase2.cpp | 143 +++++++++++++++++++++----------- 1 file changed, 93 insertions(+), 50 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 88011a8c43..c52fb60723 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2645,6 +2645,70 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, return 1; // CONTINUE_DUAL } +template +dual_status_t run_primal_cleanup(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + f_t start_time, + basis_update_mpf_t& ft, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + std::vector& objective, + lp_solution_t& sol, + i_t& iter, + f_t& work_estimate) +{ + const f_t perturbation = amount_of_perturbation(lp, objective); + settings.log.printf( + "Failed to remove perturbation of %.2e. Using primal simplex for cleanup.\n", perturbation); + settings.log.printf("Num updates: %d\n", ft.num_updates()); + settings.log.printf("Iterations: %d\n", iter); + const i_t dual_iter = iter; + const primal_status_t primal_status = primal_phase2_with_advanced_basis(2, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + work_estimate, + false); + if (primal_status == primal_status_t::OPTIMAL) { + settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); + objective = lp.objective; + } else if (primal_status == primal_status_t::TIME_LIMIT) { + return dual_status_t::TIME_LIMIT; + } else if (primal_status == primal_status_t::CONCURRENT_LIMIT) { + return dual_status_t::CONCURRENT_LIMIT; + } else if (primal_status == primal_status_t::WORK_LIMIT) { + return dual_status_t::WORK_LIMIT; + } else if (primal_status == primal_status_t::ITERATION_LIMIT) { + return dual_status_t::ITERATION_LIMIT; + } else { + settings.log.printf("Primal cleanup failed.\n"); + const f_t primal_infeas = primal_infeasibility(lp, settings, vstatus, sol.x); + const f_t dual_infeas = dual_infeasibility( + lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); + // Failed cleanup may leave duals from phase I or from the previous basis. + const f_t primal_residual = l2_primal_residual(lp, sol); + const f_t dual_residual = l2_dual_residual(lp, sol); + work_estimate += 4.0 * lp.A.nnz() + 3 * lp.num_rows + 4 * lp.num_cols; + bool is_optimal = primal_infeas <= 10.0 * settings.primal_tol && + dual_infeas <= 10.0 * settings.dual_tol && + primal_residual <= settings.primal_tol && + dual_residual <= settings.dual_tol; + for (const i_t j : basic_list) { + is_optimal = is_optimal && std::abs(sol.z[j]) <= settings.dual_tol; + } + work_estimate += lp.num_rows; + if (!is_optimal) { return dual_status_t::NUMERICAL; } + } + return dual_status_t::OPTIMAL; +} + template void prepare_optimality(i_t info, f_t orig_primal_infeas, @@ -3484,53 +3548,18 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, continue; } if (removal_status == 2) { // PRIMAL_CLEANUP - const f_t perturbation = phase2::amount_of_perturbation(lp, objective); - settings.log.printf("Failed to remove perturbation of %.2e. Using primal simplex for cleanup.\n", perturbation); - settings.log.printf("Num updates: %d\n", ft.num_updates()); - settings.log.printf("Iterations: %d\n", iter); - i_t dual_iter = iter; - primal_status_t primal_status = primal_phase2_with_advanced_basis(2, - start_time, - lp, - settings, - vstatus, - ft, - basic_list, - nonbasic_list, - sol, - iter, - phase2_work_estimate, - false); - if (primal_status == primal_status_t::OPTIMAL) { - settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); - objective = lp.objective; - } else if (primal_status == primal_status_t::TIME_LIMIT) { - return dual_status_t::TIME_LIMIT; - } else if (primal_status == primal_status_t::CONCURRENT_LIMIT) { - return dual_status_t::CONCURRENT_LIMIT; - } else if (primal_status == primal_status_t::WORK_LIMIT) { - return dual_status_t::WORK_LIMIT; - } else if (primal_status == primal_status_t::ITERATION_LIMIT) { - return dual_status_t::ITERATION_LIMIT; - } else { - settings.log.printf("Primal cleanup failed.\n"); - const f_t primal_infeas = phase2::primal_infeasibility(lp, settings, vstatus, sol.x); - const f_t dual_infeas = phase2::dual_infeasibility( - lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); - // Failed cleanup may leave duals from phase I or from the previous basis. - const f_t primal_residual = phase2::l2_primal_residual(lp, sol); - const f_t dual_residual = phase2::l2_dual_residual(lp, sol); - phase2_work_estimate += 4.0 * lp.A.nnz() + 3 * m + 4 * n; - bool is_optimal = primal_infeas <= 10.0 * settings.primal_tol && - dual_infeas <= 10.0 * settings.dual_tol && - primal_residual <= settings.primal_tol && - dual_residual <= settings.dual_tol; - for (const i_t j : basic_list) { - is_optimal = is_optimal && std::abs(sol.z[j]) <= settings.dual_tol; - } - phase2_work_estimate += m; - if (!is_optimal) { return dual_status_t::NUMERICAL; } - } + const dual_status_t cleanup_status = phase2::run_primal_cleanup(lp, + settings, + start_time, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + sol, + iter, + phase2_work_estimate); + if (cleanup_status != dual_status_t::OPTIMAL) { return cleanup_status; } } // removal_status == 0 (OPTIMAL) or primal cleanup done: fall through to prepare_optimality } @@ -3733,10 +3762,25 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, primal_infeasibility, primal_infeasibility_squared, phase2_work_estimate); - if (removal_status == 0) { // OPTIMAL + if (removal_status == 2) { // PRIMAL_CLEANUP + const dual_status_t cleanup_status = phase2::run_primal_cleanup(lp, + settings, + start_time, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + sol, + iter, + phase2_work_estimate); + if (cleanup_status != dual_status_t::OPTIMAL) { return cleanup_status; } + objective = lp.objective; + } + if (removal_status == 0 || removal_status == 2) { // OPTIMAL or successful cleanup obj = phase2::compute_perturbed_objective(objective, x); phase2_work_estimate += 2 * n; - if (primal_infeasibility <= settings.primal_tol) { + if (removal_status == 2 || primal_infeasibility <= settings.primal_tol) { phase2_work_estimate += ft.work_estimate(); ft.clear_work_estimate(); phase2::prepare_optimality(1, @@ -3773,7 +3817,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); continue; } - // removal_status == 2 (PRIMAL_CLEANUP): fall through to existing logic below } if (perturbation == 0.0 && phase == 2) { From cfcc5a444cd62ba55099759f5365d798c41ebe6a Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 16:26:30 -0700 Subject: [PATCH 079/113] Fix work estimate initialization --- cpp/src/dual_simplex/crossover.cpp | 10 +-- cpp/src/dual_simplex/phase2.cpp | 134 +++++++++++++++++++---------- cpp/src/dual_simplex/phase2.hpp | 10 +-- cpp/src/dual_simplex/solve.cpp | 8 +- 4 files changed, 104 insertions(+), 58 deletions(-) diff --git a/cpp/src/dual_simplex/crossover.cpp b/cpp/src/dual_simplex/crossover.cpp index 977f5e5511..bd51675363 100644 --- a/cpp/src/dual_simplex/crossover.cpp +++ b/cpp/src/dual_simplex/crossover.cpp @@ -1428,7 +1428,7 @@ crossover_status_t crossover(const lp_problem_t& lp, simplex_solver_settings_t dual_settings = settings; dual_settings.iteration_limit = std::numeric_limits::max(); dual_status_t status = dual_phase2( - 2, 0, start_time, lp, dual_settings, vstatus, solution, dual_iter, work_estimate, edge_norms); + 2, 0, start_time, lp, dual_settings, vstatus, solution, dual_iter, edge_norms, work_estimate); if (toc(start_time) > settings.time_limit) { settings.log.printf("Time limit exceeded\n"); return crossover_status_t::TIME_LIMIT; @@ -1508,8 +1508,8 @@ crossover_status_t crossover(const lp_problem_t& lp, phase1_vstatus, phase1_solution, iter, - phase1_work_estimate, - junk); + junk, + phase1_work_estimate); if (phase1_status == dual_status_t::NUMERICAL || phase1_status == dual_status_t::DUAL_UNBOUNDED) { settings.log.printf("Failed in Phase 1\n"); @@ -1633,8 +1633,8 @@ crossover_status_t crossover(const lp_problem_t& lp, vstatus, solution, iter, - phase2_work_estimate, - edge_norms); + edge_norms, + phase2_work_estimate); if (toc(start_time) > settings.time_limit) { settings.log.printf("Time limit exceeded\n"); return crossover_status_t::TIME_LIMIT; diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index a15737b7a8..b8c5093dd1 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2920,44 +2920,7 @@ class phase2_timers_t { } // namespace phase2 template -dual_status_t dual_phase2(i_t phase, - i_t slack_basis, - f_t start_time, - const lp_problem_t& lp, - const simplex_solver_settings_t& settings, - std::vector& vstatus, - lp_solution_t& sol, - i_t& iter, - f_t& work_estimate, - std::vector& delta_y_steepest_edge, - work_limit_context_t* work_unit_context) -{ - PHASE2_NVTX_RANGE("DualSimplex::phase2"); - const i_t m = lp.num_rows; - const i_t n = lp.num_cols; - std::vector basic_list(m); - std::vector nonbasic_list; - basis_update_mpf_t ft(m, settings.refactor_frequency); - const bool initialize_basis = true; - return dual_phase2_with_advanced_basis(phase, - slack_basis, - initialize_basis, - start_time, - lp, - settings, - vstatus, - ft, - basic_list, - nonbasic_list, - sol, - iter, - work_estimate, - delta_y_steepest_edge, - work_unit_context); -} - -template -dual_status_t dual_phase2_with_advanced_basis(i_t phase, +static dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t slack_basis, bool initialize_basis, f_t start_time, @@ -2969,8 +2932,9 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, std::vector& nonbasic_list, lp_solution_t& sol, i_t& iter, - f_t& phase2_work_estimate, std::vector& delta_y_steepest_edge, + f_t& phase2_work_estimate, + f_t& last_work_reported, work_limit_context_t* work_unit_context) { PHASE2_NVTX_RANGE("DualSimplex::phase2_advanced"); @@ -3364,11 +3328,11 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += ft.work_estimate(); ft.clear_work_estimate(); - f_t last_work_reported = 0.0; if (work_unit_context) { - work_unit_context->record_work_sync_on_horizon((phase2_work_estimate) / 1e8); - last_work_reported = phase2_work_estimate; + work_unit_context->record_work_sync_on_horizon((phase2_work_estimate - last_work_reported) / + 1e8); } + last_work_reported = phase2_work_estimate; if (phase == 2) { settings.log.printf("%5d %+.16e %7d %.8e %.2e %.2f\n", @@ -4311,6 +4275,88 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, return status; } +template +dual_status_t dual_phase2_with_advanced_basis( + i_t phase, + i_t slack_basis, + bool initialize_basis, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& ft, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + i_t& iter, + std::vector& delta_y_steepest_edge, + f_t& phase2_work_estimate, + work_limit_context_t* work_unit_context) +{ + f_t last_work_reported = phase2_work_estimate; + const dual_status_t status = dual_phase2_with_advanced_basis(phase, + slack_basis, + initialize_basis, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + delta_y_steepest_edge, + phase2_work_estimate, + last_work_reported, + work_unit_context); + + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); + if (work_unit_context && phase2_work_estimate > last_work_reported) { + work_unit_context->record_work_sync_on_horizon((phase2_work_estimate - last_work_reported) / + 1e8); + } + return status; +} + +template +dual_status_t dual_phase2(i_t phase, + i_t slack_basis, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + lp_solution_t& sol, + i_t& iter, + std::vector& delta_y_steepest_edge, + f_t& work_estimate, + work_limit_context_t* work_unit_context) +{ + PHASE2_NVTX_RANGE("DualSimplex::phase2"); + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + std::vector basic_list(m); + std::vector nonbasic_list; + basis_update_mpf_t ft(m, settings.refactor_frequency); + const bool initialize_basis = true; + return dual_phase2_with_advanced_basis(phase, + slack_basis, + initialize_basis, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + delta_y_steepest_edge, + work_estimate, + work_unit_context); +} + #ifdef DUAL_SIMPLEX_INSTANTIATE_DOUBLE template dual_status_t dual_phase2( @@ -4322,8 +4368,8 @@ template dual_status_t dual_phase2( std::vector& vstatus, lp_solution_t& sol, int& iter, - double& work_estimate, std::vector& steepest_edge_norms, + double& work_estimate, work_limit_context_t* work_unit_context); template dual_status_t dual_phase2_with_advanced_basis( @@ -4339,8 +4385,8 @@ template dual_status_t dual_phase2_with_advanced_basis( std::vector& nonbasic_list, lp_solution_t& sol, int& iter, - double& work_estimate, std::vector& steepest_edge_norms, + double& work_estimate, work_limit_context_t* work_unit_context); template void compute_reduced_cost_update(const lp_problem_t& lp, diff --git a/cpp/src/dual_simplex/phase2.hpp b/cpp/src/dual_simplex/phase2.hpp index cfa46d8152..4c038c299b 100644 --- a/cpp/src/dual_simplex/phase2.hpp +++ b/cpp/src/dual_simplex/phase2.hpp @@ -60,8 +60,8 @@ dual_status_t dual_phase2(i_t phase, std::vector& vstatus, lp_solution_t& sol, i_t& iter, - f_t& work_estimate, std::vector& steepest_edge_norms, + f_t& work_estimate, work_limit_context_t* work_unit_context = nullptr); template @@ -85,8 +85,8 @@ dual_status_t dual_phase2(i_t phase, vstatus, sol, iter, - work_estimate, steepest_edge_norms, + work_estimate, work_unit_context); } @@ -103,9 +103,9 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, std::vector& nonbasic_list, lp_solution_t& sol, i_t& iter, + std::vector& delta_y_steepest_edge, f_t& work_estimate, - std::vector& delta_y_steepest_edge, - work_limit_context_t* work_unit_context = nullptr); + work_limit_context_t* work_unit_context = nullptr); template dual_status_t dual_phase2_with_advanced_basis(i_t phase, @@ -136,8 +136,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, nonbasic_list, sol, iter, - work_estimate, delta_y_steepest_edge, + work_estimate, work_unit_context); } diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index 9541f94304..4bb80e8471 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -267,8 +267,8 @@ lp_status_t solve_linear_program_with_advanced_basis( phase1_vstatus, phase1_solution, iter, - work_estimate, edge_norms, + work_estimate, work_unit_context); } if (phase1_status == dual_status_t::NUMERICAL) { @@ -306,8 +306,8 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, solution, iter, - work_estimate, edge_norms, + work_estimate, work_unit_context); if (status == dual_status_t::NUMERICAL) { // Became dual infeasible. Try phase 1 again @@ -327,8 +327,8 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, phase1_solution, iter, - work_estimate, edge_norms, + work_estimate, work_unit_context); vstatus = phase1_vstatus; edge_norms.clear(); @@ -344,8 +344,8 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, solution, iter, - work_estimate, edge_norms, + work_estimate, work_unit_context); } if (settings.inside_mip == 1 && settings.concurrent_halt != nullptr) { From dba71c347c7d1e87bc43b777b889ebf3d35fa860 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 16:26:30 -0700 Subject: [PATCH 080/113] Fix work estimate initialization --- cpp/src/branch_and_bound/branch_and_bound.cpp | 16 +-- cpp/src/branch_and_bound/pseudo_costs.cpp | 8 +- cpp/src/dual_simplex/crossover.cpp | 10 +- cpp/src/dual_simplex/phase2.cpp | 134 ++++++++++++------ cpp/src/dual_simplex/phase2.hpp | 4 +- cpp/src/dual_simplex/solve.cpp | 8 +- 6 files changed, 113 insertions(+), 67 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index cf0008d80d..01302fe8d7 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -812,8 +812,8 @@ bool branch_and_bound_t::repair_solution(const std::vector& edge_ vstatus, lp_solution, iter, - repair_work_estimate, - leaf_edge_norms); + leaf_edge_norms, + repair_work_estimate); repaired_solution = lp_solution.x; if (lp_status == dual_status_t::OPTIMAL) { @@ -1758,8 +1758,8 @@ dual_status_t branch_and_bound_t::solve_node_lp( worker->nonbasic_list, worker->leaf_solution, node_iter, - node_work_estimate, - worker->leaf_edge_norms); + worker->leaf_edge_norms, + node_work_estimate); if (lp_status == dual_status_t::NUMERICAL) { log.debug_format("Numerical issue node {}. Resolving from scratch.\n", node_ptr->node_id); @@ -3651,8 +3651,8 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t 1.0) { @@ -6436,8 +6436,8 @@ node_status_t branch_and_bound_t::solve_node_deterministic( worker.nonbasic_list, worker.leaf_solution, node_iter, - dual_work_estimate, leaf_edge_norms, + dual_work_estimate, &worker.work_context); if (lp_status == dual_status_t::NUMERICAL) { @@ -7055,8 +7055,8 @@ void branch_and_bound_t::deterministic_dive( worker.nonbasic_list, worker.leaf_solution, node_iter, - dual_work_estimate, leaf_edge_norms, + dual_work_estimate, &worker.work_context); if (lp_status == dual_status_t::NUMERICAL) { diff --git a/cpp/src/branch_and_bound/pseudo_costs.cpp b/cpp/src/branch_and_bound/pseudo_costs.cpp index 8f86594e79..28cf93129e 100644 --- a/cpp/src/branch_and_bound/pseudo_costs.cpp +++ b/cpp/src/branch_and_bound/pseudo_costs.cpp @@ -379,8 +379,8 @@ void strong_branch_helper(i_t start, vstatus, solution, iter, - child_work_estimate, - child_edge_norms); + child_edge_norms, + child_work_estimate); f_t obj = std::numeric_limits::quiet_NaN(); if (status == dual_status_t::DUAL_UNBOUNDED) { @@ -521,8 +521,8 @@ std::pair trial_branching(const lp_problem_t& orig child_nonbasic_list, solution, iter, - child_work_estimate, - child_edge_norms); + child_edge_norms, + child_work_estimate); settings.log.debug("Trial branching on variable %d. Lo: %e Up: %e. Iter %d. Status %s. Obj %e\n", branch_var, diff --git a/cpp/src/dual_simplex/crossover.cpp b/cpp/src/dual_simplex/crossover.cpp index 977f5e5511..bd51675363 100644 --- a/cpp/src/dual_simplex/crossover.cpp +++ b/cpp/src/dual_simplex/crossover.cpp @@ -1428,7 +1428,7 @@ crossover_status_t crossover(const lp_problem_t& lp, simplex_solver_settings_t dual_settings = settings; dual_settings.iteration_limit = std::numeric_limits::max(); dual_status_t status = dual_phase2( - 2, 0, start_time, lp, dual_settings, vstatus, solution, dual_iter, work_estimate, edge_norms); + 2, 0, start_time, lp, dual_settings, vstatus, solution, dual_iter, edge_norms, work_estimate); if (toc(start_time) > settings.time_limit) { settings.log.printf("Time limit exceeded\n"); return crossover_status_t::TIME_LIMIT; @@ -1508,8 +1508,8 @@ crossover_status_t crossover(const lp_problem_t& lp, phase1_vstatus, phase1_solution, iter, - phase1_work_estimate, - junk); + junk, + phase1_work_estimate); if (phase1_status == dual_status_t::NUMERICAL || phase1_status == dual_status_t::DUAL_UNBOUNDED) { settings.log.printf("Failed in Phase 1\n"); @@ -1633,8 +1633,8 @@ crossover_status_t crossover(const lp_problem_t& lp, vstatus, solution, iter, - phase2_work_estimate, - edge_norms); + edge_norms, + phase2_work_estimate); if (toc(start_time) > settings.time_limit) { settings.log.printf("Time limit exceeded\n"); return crossover_status_t::TIME_LIMIT; diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index c52fb60723..d8c023f022 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2920,44 +2920,7 @@ class phase2_timers_t { } // namespace phase2 template -dual_status_t dual_phase2(i_t phase, - i_t slack_basis, - f_t start_time, - const lp_problem_t& lp, - const simplex_solver_settings_t& settings, - std::vector& vstatus, - lp_solution_t& sol, - i_t& iter, - f_t& work_estimate, - std::vector& delta_y_steepest_edge, - work_limit_context_t* work_unit_context) -{ - PHASE2_NVTX_RANGE("DualSimplex::phase2"); - const i_t m = lp.num_rows; - const i_t n = lp.num_cols; - std::vector basic_list(m); - std::vector nonbasic_list; - basis_update_mpf_t ft(m, settings.refactor_frequency); - const bool initialize_basis = true; - return dual_phase2_with_advanced_basis(phase, - slack_basis, - initialize_basis, - start_time, - lp, - settings, - vstatus, - ft, - basic_list, - nonbasic_list, - sol, - iter, - work_estimate, - delta_y_steepest_edge, - work_unit_context); -} - -template -dual_status_t dual_phase2_with_advanced_basis(i_t phase, +static dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t slack_basis, bool initialize_basis, f_t start_time, @@ -2969,8 +2932,9 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, std::vector& nonbasic_list, lp_solution_t& sol, i_t& iter, - f_t& phase2_work_estimate, std::vector& delta_y_steepest_edge, + f_t& phase2_work_estimate, + f_t& last_work_reported, work_limit_context_t* work_unit_context) { PHASE2_NVTX_RANGE("DualSimplex::phase2_advanced"); @@ -3362,11 +3326,11 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += ft.work_estimate(); ft.clear_work_estimate(); - f_t last_work_reported = 0.0; if (work_unit_context) { - work_unit_context->record_work_sync_on_horizon((phase2_work_estimate) / 1e8); - last_work_reported = phase2_work_estimate; + work_unit_context->record_work_sync_on_horizon((phase2_work_estimate - last_work_reported) / + 1e8); } + last_work_reported = phase2_work_estimate; if (phase == 2) { settings.log.printf("%5d %+.16e %7d %.8e %.2e %.2f\n", @@ -4309,6 +4273,88 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, return status; } +template +dual_status_t dual_phase2_with_advanced_basis( + i_t phase, + i_t slack_basis, + bool initialize_basis, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& ft, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + i_t& iter, + std::vector& delta_y_steepest_edge, + f_t& phase2_work_estimate, + work_limit_context_t* work_unit_context) +{ + f_t last_work_reported = phase2_work_estimate; + const dual_status_t status = dual_phase2_with_advanced_basis(phase, + slack_basis, + initialize_basis, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + delta_y_steepest_edge, + phase2_work_estimate, + last_work_reported, + work_unit_context); + + phase2_work_estimate += ft.work_estimate(); + ft.clear_work_estimate(); + if (work_unit_context && phase2_work_estimate > last_work_reported) { + work_unit_context->record_work_sync_on_horizon((phase2_work_estimate - last_work_reported) / + 1e8); + } + return status; +} + +template +dual_status_t dual_phase2(i_t phase, + i_t slack_basis, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + lp_solution_t& sol, + i_t& iter, + std::vector& delta_y_steepest_edge, + f_t& work_estimate, + work_limit_context_t* work_unit_context) +{ + PHASE2_NVTX_RANGE("DualSimplex::phase2"); + const i_t m = lp.num_rows; + const i_t n = lp.num_cols; + std::vector basic_list(m); + std::vector nonbasic_list; + basis_update_mpf_t ft(m, settings.refactor_frequency); + const bool initialize_basis = true; + return dual_phase2_with_advanced_basis(phase, + slack_basis, + initialize_basis, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + delta_y_steepest_edge, + work_estimate, + work_unit_context); +} + #ifdef DUAL_SIMPLEX_INSTANTIATE_DOUBLE template dual_status_t dual_phase2( @@ -4320,8 +4366,8 @@ template dual_status_t dual_phase2( std::vector& vstatus, lp_solution_t& sol, int& iter, - double& work_estimate, std::vector& steepest_edge_norms, + double& work_estimate, work_limit_context_t* work_unit_context); template dual_status_t dual_phase2_with_advanced_basis( @@ -4337,8 +4383,8 @@ template dual_status_t dual_phase2_with_advanced_basis( std::vector& nonbasic_list, lp_solution_t& sol, int& iter, - double& work_estimate, std::vector& steepest_edge_norms, + double& work_estimate, work_limit_context_t* work_unit_context); template void compute_reduced_cost_update(const lp_problem_t& lp, diff --git a/cpp/src/dual_simplex/phase2.hpp b/cpp/src/dual_simplex/phase2.hpp index e5a4bacf62..391cca6e11 100644 --- a/cpp/src/dual_simplex/phase2.hpp +++ b/cpp/src/dual_simplex/phase2.hpp @@ -60,8 +60,8 @@ dual_status_t dual_phase2(i_t phase, std::vector& vstatus, lp_solution_t& sol, i_t& iter, - f_t& work_estimate, std::vector& steepest_edge_norms, + f_t& work_estimate, work_limit_context_t* work_unit_context = nullptr); template @@ -77,8 +77,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, std::vector& nonbasic_list, lp_solution_t& sol, i_t& iter, - f_t& work_estimate, std::vector& delta_y_steepest_edge, + f_t& work_estimate, work_limit_context_t* work_unit_context = nullptr); template diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index 9541f94304..4bb80e8471 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -267,8 +267,8 @@ lp_status_t solve_linear_program_with_advanced_basis( phase1_vstatus, phase1_solution, iter, - work_estimate, edge_norms, + work_estimate, work_unit_context); } if (phase1_status == dual_status_t::NUMERICAL) { @@ -306,8 +306,8 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, solution, iter, - work_estimate, edge_norms, + work_estimate, work_unit_context); if (status == dual_status_t::NUMERICAL) { // Became dual infeasible. Try phase 1 again @@ -327,8 +327,8 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, phase1_solution, iter, - work_estimate, edge_norms, + work_estimate, work_unit_context); vstatus = phase1_vstatus; edge_norms.clear(); @@ -344,8 +344,8 @@ lp_status_t solve_linear_program_with_advanced_basis( nonbasic_list, solution, iter, - work_estimate, edge_norms, + work_estimate, work_unit_context); } if (settings.inside_mip == 1 && settings.concurrent_halt != nullptr) { From 952cf05788a1ffdfc161233cd02c055eee387141 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 17:02:29 -0700 Subject: [PATCH 081/113] Remove unused variable --- cpp/src/dual_simplex/phase2.cpp | 2 -- 1 file changed, 2 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index b8c5093dd1..d8c023f022 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -3322,8 +3322,6 @@ static dual_status_t dual_phase2_with_advanced_basis(i_t phase, sparse_vector_t v_sparse(m, 0); // For steepest edge norms sparse_vector_t atilde_sparse(m, 0); // For flip adjustments - // Track iteration interval start time for runtime measurement - [[maybe_unused]] f_t interval_start_time = toc(start_time); i_t last_feature_log_iter = iter; phase2_work_estimate += ft.work_estimate(); From a7659d75c945def2e8597ac506a44363ceeb06f3 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 17:09:28 -0700 Subject: [PATCH 082/113] Format simplex improvements --- cpp/src/dual_simplex/phase2.cpp | 201 +++++++++--------- cpp/src/dual_simplex/phase2.hpp | 8 +- .../dual_simplex/simplex_solver_settings.hpp | 8 +- cpp/src/dual_simplex/solve.hpp | 4 +- cpp/src/pdlp/solve.cu | 9 +- 5 files changed, 114 insertions(+), 116 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index d8c023f022..a93acb4efd 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2556,7 +2556,7 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, if (primal_infeasibility <= settings.primal_tol) return 0; // OPTIMAL settings.log.debug("Removed perturbation. Continuing dual simplex (primal_inf=%.2e)\n", - primal_infeasibility); + primal_infeasibility); return 1; // CONTINUE_DUAL } @@ -2647,35 +2647,35 @@ i_t attempt_to_remove_perturbations(const lp_problem_t& lp, template dual_status_t run_primal_cleanup(const lp_problem_t& lp, - const simplex_solver_settings_t& settings, - f_t start_time, - basis_update_mpf_t& ft, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& vstatus, - std::vector& objective, - lp_solution_t& sol, - i_t& iter, - f_t& work_estimate) + const simplex_solver_settings_t& settings, + f_t start_time, + basis_update_mpf_t& ft, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + std::vector& objective, + lp_solution_t& sol, + i_t& iter, + f_t& work_estimate) { const f_t perturbation = amount_of_perturbation(lp, objective); - settings.log.printf( - "Failed to remove perturbation of %.2e. Using primal simplex for cleanup.\n", perturbation); + settings.log.printf("Failed to remove perturbation of %.2e. Using primal simplex for cleanup.\n", + perturbation); settings.log.printf("Num updates: %d\n", ft.num_updates()); settings.log.printf("Iterations: %d\n", iter); - const i_t dual_iter = iter; + const i_t dual_iter = iter; const primal_status_t primal_status = primal_phase2_with_advanced_basis(2, - start_time, - lp, - settings, - vstatus, - ft, - basic_list, - nonbasic_list, - sol, - iter, - work_estimate, - false); + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + work_estimate, + false); if (primal_status == primal_status_t::OPTIMAL) { settings.log.printf("Primal cleanup successful. Iterations %d\n", iter - dual_iter); objective = lp.objective; @@ -2690,16 +2690,15 @@ dual_status_t run_primal_cleanup(const lp_problem_t& lp, } else { settings.log.printf("Primal cleanup failed.\n"); const f_t primal_infeas = primal_infeasibility(lp, settings, vstatus, sol.x); - const f_t dual_infeas = dual_infeasibility( - lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); + const f_t dual_infeas = + dual_infeasibility(lp, settings, vstatus, sol.z, settings.tight_tol, settings.dual_tol); // Failed cleanup may leave duals from phase I or from the previous basis. const f_t primal_residual = l2_primal_residual(lp, sol); const f_t dual_residual = l2_dual_residual(lp, sol); work_estimate += 4.0 * lp.A.nnz() + 3 * lp.num_rows + 4 * lp.num_cols; bool is_optimal = primal_infeas <= 10.0 * settings.primal_tol && dual_infeas <= 10.0 * settings.dual_tol && - primal_residual <= settings.primal_tol && - dual_residual <= settings.dual_tol; + primal_residual <= settings.primal_tol && dual_residual <= settings.dual_tol; for (const i_t j : basic_list) { is_optimal = is_optimal && std::abs(sol.z[j]) <= settings.dual_tol; } @@ -2896,7 +2895,7 @@ class phase2_timers_t { work_timer_t bfrt_time; // BFRT diagnostic counters i_t bfrt_calls{0}; - i_t bfrt_zero_steps{0}; // step_length == 0 + i_t bfrt_zero_steps{0}; // step_length == 0 work_timer_t pricing_time; work_timer_t btran_time; work_timer_t ftran_time; @@ -2920,22 +2919,23 @@ class phase2_timers_t { } // namespace phase2 template -static dual_status_t dual_phase2_with_advanced_basis(i_t phase, - i_t slack_basis, - bool initialize_basis, - f_t start_time, - const lp_problem_t& lp, - const simplex_solver_settings_t& settings, - std::vector& vstatus, - basis_update_mpf_t& ft, - std::vector& basic_list, - std::vector& nonbasic_list, - lp_solution_t& sol, - i_t& iter, - std::vector& delta_y_steepest_edge, - f_t& phase2_work_estimate, - f_t& last_work_reported, - work_limit_context_t* work_unit_context) +static dual_status_t dual_phase2_with_advanced_basis( + i_t phase, + i_t slack_basis, + bool initialize_basis, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& ft, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + i_t& iter, + std::vector& delta_y_steepest_edge, + f_t& phase2_work_estimate, + f_t& last_work_reported, + work_limit_context_t* work_unit_context) { PHASE2_NVTX_RANGE("DualSimplex::phase2_advanced"); const i_t m = lp.num_rows; @@ -3322,7 +3322,7 @@ static dual_status_t dual_phase2_with_advanced_basis(i_t phase, sparse_vector_t v_sparse(m, 0); // For steepest edge norms sparse_vector_t atilde_sparse(m, 0); // For flip adjustments - i_t last_feature_log_iter = iter; + i_t last_feature_log_iter = iter; phase2_work_estimate += ft.work_estimate(); ft.clear_work_estimate(); @@ -3513,16 +3513,16 @@ static dual_status_t dual_phase2_with_advanced_basis(i_t phase, } if (removal_status == 2) { // PRIMAL_CLEANUP const dual_status_t cleanup_status = phase2::run_primal_cleanup(lp, - settings, - start_time, - ft, - basic_list, - nonbasic_list, - vstatus, - objective, - sol, - iter, - phase2_work_estimate); + settings, + start_time, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + sol, + iter, + phase2_work_estimate); if (cleanup_status != dual_status_t::OPTIMAL) { return cleanup_status; } } // removal_status == 0 (OPTIMAL) or primal cleanup done: fall through to prepare_optimality @@ -3693,9 +3693,7 @@ static dual_status_t dual_phase2_with_advanced_basis(i_t phase, timers.bfrt_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); // BFRT diagnostics timers.bfrt_calls++; - if (step_length == 0.0) { - timers.bfrt_zero_steps++; - } + if (step_length == 0.0) { timers.bfrt_zero_steps++; } } else { entering_index = phase2::phase2_ratio_test( lp, settings, vstatus, nonbasic_list, z, delta_z, step_length, nonbasic_entering_index); @@ -3728,16 +3726,16 @@ static dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate); if (removal_status == 2) { // PRIMAL_CLEANUP const dual_status_t cleanup_status = phase2::run_primal_cleanup(lp, - settings, - start_time, - ft, - basic_list, - nonbasic_list, - vstatus, - objective, - sol, - iter, - phase2_work_estimate); + settings, + start_time, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + sol, + iter, + phase2_work_estimate); if (cleanup_status != dual_status_t::OPTIMAL) { return cleanup_status; } objective = lp.objective; } @@ -4274,40 +4272,39 @@ static dual_status_t dual_phase2_with_advanced_basis(i_t phase, } template -dual_status_t dual_phase2_with_advanced_basis( - i_t phase, - i_t slack_basis, - bool initialize_basis, - f_t start_time, - const lp_problem_t& lp, - const simplex_solver_settings_t& settings, - std::vector& vstatus, - basis_update_mpf_t& ft, - std::vector& basic_list, - std::vector& nonbasic_list, - lp_solution_t& sol, - i_t& iter, - std::vector& delta_y_steepest_edge, - f_t& phase2_work_estimate, - work_limit_context_t* work_unit_context) +dual_status_t dual_phase2_with_advanced_basis(i_t phase, + i_t slack_basis, + bool initialize_basis, + f_t start_time, + const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + std::vector& vstatus, + basis_update_mpf_t& ft, + std::vector& basic_list, + std::vector& nonbasic_list, + lp_solution_t& sol, + i_t& iter, + std::vector& delta_y_steepest_edge, + f_t& phase2_work_estimate, + work_limit_context_t* work_unit_context) { - f_t last_work_reported = phase2_work_estimate; + f_t last_work_reported = phase2_work_estimate; const dual_status_t status = dual_phase2_with_advanced_basis(phase, - slack_basis, - initialize_basis, - start_time, - lp, - settings, - vstatus, - ft, - basic_list, - nonbasic_list, - sol, - iter, - delta_y_steepest_edge, - phase2_work_estimate, - last_work_reported, - work_unit_context); + slack_basis, + initialize_basis, + start_time, + lp, + settings, + vstatus, + ft, + basic_list, + nonbasic_list, + sol, + iter, + delta_y_steepest_edge, + phase2_work_estimate, + last_work_reported, + work_unit_context); phase2_work_estimate += ft.work_estimate(); ft.clear_work_estimate(); diff --git a/cpp/src/dual_simplex/phase2.hpp b/cpp/src/dual_simplex/phase2.hpp index 4c038c299b..5f8b0da1f8 100644 --- a/cpp/src/dual_simplex/phase2.hpp +++ b/cpp/src/dual_simplex/phase2.hpp @@ -53,16 +53,16 @@ static std::string dual_status_to_string(dual_status_t status) template dual_status_t dual_phase2(i_t phase, - i_t slack_basis, - f_t start_time, + i_t slack_basis, + f_t start_time, const lp_problem_t& lp, const simplex_solver_settings_t& settings, std::vector& vstatus, lp_solution_t& sol, i_t& iter, - std::vector& steepest_edge_norms, + std::vector& steepest_edge_norms, f_t& work_estimate, - work_limit_context_t* work_unit_context = nullptr); + work_limit_context_t* work_unit_context = nullptr); template dual_status_t dual_phase2(i_t phase, diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 9669a17eed..477da4403b 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -221,10 +221,10 @@ struct simplex_solver_settings_t { i_t strong_chvatal_gomory_cuts; // -1 automatic, 0 to disable, >0 to enable strong Chvatal Gomory // cuts i_t symmetry; // -1 automatic, 0 to disable, >0 to enable different symmetry methods - i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost - // strengthening - f_t cut_change_threshold; // threshold for cut change - f_t cut_min_orthogonality; // minimum orthogonality for cuts + i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost + // strengthening + f_t cut_change_threshold; // threshold for cut change + f_t cut_min_orthogonality; // minimum orthogonality for cuts i_t mip_batch_pdlp_strong_branching; // 0 = DS only, 1 = cooperative DS + PDLP, 2 = batch PDLP only i_t mip_batch_pdlp_reliability_branching; // 0 = DS only, 1 = cooperative DS + PDLP, 2 = batch diff --git a/cpp/src/dual_simplex/solve.hpp b/cpp/src/dual_simplex/solve.hpp index 4bbc908976..7dd22e749d 100644 --- a/cpp/src/dual_simplex/solve.hpp +++ b/cpp/src/dual_simplex/solve.hpp @@ -73,8 +73,8 @@ lp_status_t solve_linear_program_advanced(const lp_problem_t& original lp_solution_t& original_solution, std::vector& vstatus, std::vector& edge_norms, - f_t& work_estimate, - work_limit_context_t* work_unit_context = nullptr); + f_t& work_estimate, + work_limit_context_t* work_unit_context = nullptr); template lp_status_t solve_linear_program_advanced(const lp_problem_t& original_lp, diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index bbeecde2d5..87f97980f7 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -1890,10 +1890,11 @@ optimization_problem_solution_t solve_lp_with_method( } } else { // Float precision only supports PDLP without presolve/crossover - cuopt_expects(settings.method == method_t::PDLP, - error_type_t::ValidationError, - "Float precision only supports PDLP method. Dual Simplex, Primal Simplex, Barrier, and " - "Concurrent require double precision."); + cuopt_expects( + settings.method == method_t::PDLP, + error_type_t::ValidationError, + "Float precision only supports PDLP method. Dual Simplex, Primal Simplex, Barrier, and " + "Concurrent require double precision."); return run_pdlp(problem, settings, timer, is_batch_mode); } } From c56bdaefe5a4834cb42c1d6320b3c352c1b666d3 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 17:14:06 -0700 Subject: [PATCH 083/113] Format MIP extensions --- cpp/src/branch_and_bound/branch_and_bound.cpp | 26 ++++---- cpp/src/branch_and_bound/branch_and_bound.hpp | 13 ++-- .../deterministic_workers.hpp | 59 +++++++++---------- cpp/src/branch_and_bound/worker_pool.hpp | 31 +++++----- .../dual_simplex/simplex_solver_settings.hpp | 8 +-- cpp/src/mip_heuristics/root_heuristics.hpp | 19 +++--- 6 files changed, 76 insertions(+), 80 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 01302fe8d7..d43776b2f4 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -5795,21 +5795,21 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut var_types_, symmetry_, settings_, - pc_, - root_relax_soln_.x, - edge_norms_, - new_slacks_); + pc_, + root_relax_soln_.x, + edge_norms_, + new_slacks_); submip_worker_pool_.init(num_submip_workers, original_lp_, Arow_, var_types_, symmetry_, settings_, - pc_, - root_relax_soln_.x, - edge_norms_, - new_slacks_, - num_bfs_workers); + pc_, + root_relax_soln_.x, + edge_norms_, + new_slacks_, + num_bfs_workers); if (num_diving_workers > 0) { diving_worker_pool_.init(num_diving_workers, @@ -6021,10 +6021,10 @@ void branch_and_bound_t::run_deterministic_coordinator(const csr_matri Arow, var_types_, settings_, - pc_, - root_relax_soln_.x, - edge_norms_, - new_slacks_); + pc_, + root_relax_soln_.x, + edge_norms_, + new_slacks_); if (num_diving_workers > 0) { // Extract diving types from search_strategies (skip BEST_FIRST at index 0) diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 9d5df5a842..633803ab13 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -611,13 +611,12 @@ class branch_and_bound_t { root_heuristics_t& root_heuristics); // Solve the LP relaxation of a leaf node - simplex::dual_status_t solve_node_lp( - mip_node_t* node_ptr, - const simplex::simplex_solver_settings_t& settings, - branch_and_bound_worker_t* worker, - branch_and_bound_stats_t& stats, - simplex::logger_t& log, - int64_t iter_limit = std::numeric_limits::max()); + simplex::dual_status_t solve_node_lp(mip_node_t* node_ptr, + const simplex::simplex_solver_settings_t& settings, + branch_and_bound_worker_t* worker, + branch_and_bound_stats_t& stats, + simplex::logger_t& log, + int64_t iter_limit = std::numeric_limits::max()); // Apply symmetry-based bound reductions (orbital fixing and, when // settings_.symmetry == 2, lexical reduction) to the current node. diff --git a/cpp/src/branch_and_bound/deterministic_workers.hpp b/cpp/src/branch_and_bound/deterministic_workers.hpp index 7c31d023cc..200e64b472 100644 --- a/cpp/src/branch_and_bound/deterministic_workers.hpp +++ b/cpp/src/branch_and_bound/deterministic_workers.hpp @@ -89,11 +89,11 @@ class deterministic_worker_base_t : public branch_and_bound_worker_t { const csr_matrix_t& Arow, const std::vector& var_types, const simplex::simplex_solver_settings_t& settings, - pseudo_costs_t& pc, - const std::vector& root_solution, - const std::vector& root_edge_norm, - const std::vector& new_slacks, - const std::string& context_name) + pseudo_costs_t& pc, + const std::vector& root_solution, + const std::vector& root_edge_norm, + const std::vector& new_slacks, + const std::string& context_name) : base_t( id, original_lp, Arow, var_types, settings, pc, root_solution, root_edge_norm, new_slacks), work_context(context_name), @@ -146,20 +146,20 @@ class deterministic_bfs_worker_t const csr_matrix_t& Arow, const std::vector& var_types, const simplex::simplex_solver_settings_t& settings, - pseudo_costs_t& pc, - const std::vector& root_solution, - const std::vector& root_edge_norm, - const std::vector& new_slacks) + pseudo_costs_t& pc, + const std::vector& root_solution, + const std::vector& root_edge_norm, + const std::vector& new_slacks) : base_t(id, original_lp, Arow, var_types, settings, - pc, - root_solution, - root_edge_norm, - new_slacks, - "BB_Worker_" + std::to_string(id)) + pc, + root_solution, + root_edge_norm, + new_slacks, + "BB_Worker_" + std::to_string(id)) { } @@ -324,11 +324,11 @@ class deterministic_diving_worker_t Arow, var_types, settings, - pc, - root_solution, - root_edge_norm, - new_slacks, - "Diving_Worker_" + std::to_string(id)), + pc, + root_solution, + root_edge_norm, + new_slacks, + "Diving_Worker_" + std::to_string(id)), diving_type(type) { dive_lower = original_lp.lower; @@ -482,17 +482,16 @@ class deterministic_diving_worker_pool_t this->workers_.reserve(num_workers); for (int i = 0; i < num_workers; ++i) { search_strategy_t type = diving_types[i % diving_types.size()]; - this->workers_.emplace_back( - i, - type, - original_lp, - Arow, - var_types, - settings, - pc, - root_solution, - root_edge_norm, - new_slacks); + this->workers_.emplace_back(i, + type, + original_lp, + Arow, + var_types, + settings, + pc, + root_solution, + root_edge_norm, + new_slacks); } } diff --git a/cpp/src/branch_and_bound/worker_pool.hpp b/cpp/src/branch_and_bound/worker_pool.hpp index bdb420a405..2360e8920d 100644 --- a/cpp/src/branch_and_bound/worker_pool.hpp +++ b/cpp/src/branch_and_bound/worker_pool.hpp @@ -24,11 +24,11 @@ class worker_pool_t { const std::vector& var_type, mip_symmetry_t* symmetry, const simplex::simplex_solver_settings_t& settings, - pseudo_costs_t& pc, - const std::vector& root_solution, - const std::vector& root_edge_norm, - const std::vector& new_slacks, - const uint64_t rng_offset = 0) + pseudo_costs_t& pc, + const std::vector& root_solution, + const std::vector& root_edge_norm, + const std::vector& new_slacks, + const uint64_t rng_offset = 0) { assert(!is_initialized_); assert(num_workers > 0); @@ -37,17 +37,16 @@ class worker_pool_t { num_idle_workers_ = num_workers; idle_workers_.clear_resize(num_workers); for (i_t i = 0; i < num_workers; ++i) { - workers_[i] = std::make_unique( - i, - original_lp, - Arow, - var_type, - settings, - pc, - root_solution, - root_edge_norm, - new_slacks, - rng_offset); + workers_[i] = std::make_unique(i, + original_lp, + Arow, + var_type, + settings, + pc, + root_solution, + root_edge_norm, + new_slacks, + rng_offset); idle_workers_.push_back(i); // Propagate the (possibly null) symmetry pointer; workers lazily build // their orbital_fixing/lexical_reduction state via ensure_orbital_fixing(). diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 08cc3b55a1..577a82e1b1 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -224,13 +224,13 @@ struct simplex_solver_settings_t { i_t strong_chvatal_gomory_cuts; // -1 automatic, 0 to disable, >0 to enable strong Chvatal Gomory // cuts i_t symmetry; // -1 automatic, 0 to disable, >0 to enable different symmetry methods - i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost - // strengthening + i_t reduced_cost_strengthening; // -1 automatic, 0 to disable, >0 to enable reduced cost + // strengthening i_t dual_degenerate_feasibility_pump; // 0 to disable, 1 to enable i_t primal_degenerate_pivots; // 0 to disable, 1 to enable i_t dual_degenerate_pivots; // 0 to disable, 1 to enable - f_t cut_change_threshold; // threshold for cut change - f_t cut_min_orthogonality; // minimum orthogonality for cuts + f_t cut_change_threshold; // threshold for cut change + f_t cut_min_orthogonality; // minimum orthogonality for cuts i_t mip_batch_pdlp_strong_branching; // 0 = DS only, 1 = cooperative DS + PDLP, 2 = batch PDLP only i_t mip_batch_pdlp_reliability_branching; // 0 = DS only, 1 = cooperative DS + PDLP, 2 = batch diff --git a/cpp/src/mip_heuristics/root_heuristics.hpp b/cpp/src/mip_heuristics/root_heuristics.hpp index 2a893bcc19..38ddaf9921 100644 --- a/cpp/src/mip_heuristics/root_heuristics.hpp +++ b/cpp/src/mip_heuristics/root_heuristics.hpp @@ -84,16 +84,15 @@ struct cut_pass_heuristics_t { const std::vector& sol, search_strategy_t type) { - submip_worker_ = std::make_unique>( - id, - lp, - Arow_, - var_types_, - settings, - pseudo_costs_, - root_solution_, - root_edge_norm_, - new_slacks_); + submip_worker_ = std::make_unique>(id, + lp, + Arow_, + var_types_, + settings, + pseudo_costs_, + root_solution_, + root_edge_norm_, + new_slacks_); submip_worker_->start_node = mip_node_t(root_obj, root_vstatus); submip_worker_->leaf_vstatus = root_vstatus; submip_worker_->leaf_solution.x = sol; From 8523f2b2fbb85762fff82beec7837e95fef6bb98 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 17:35:55 -0700 Subject: [PATCH 084/113] Rename simplex parameters and default primal pricing to Devex --- cpp/include/cuopt/mathematical_optimization/constants.h | 6 +++--- .../mathematical_optimization/pdlp/solver_settings.hpp | 2 +- cpp/src/dual_simplex/simplex_solver_settings.hpp | 2 +- cpp/src/math_optimization/solver_settings.cu | 6 +++--- 4 files changed, 8 insertions(+), 8 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/constants.h b/cpp/include/cuopt/mathematical_optimization/constants.h index aae5c1ce6f..8e1d485153 100644 --- a/cpp/include/cuopt/mathematical_optimization/constants.h +++ b/cpp/include/cuopt/mathematical_optimization/constants.h @@ -57,9 +57,9 @@ #define CUOPT_ELIMINATE_DENSE_COLUMNS "eliminate_dense_columns" #define CUOPT_CUDSS_DETERMINISTIC "cudss_deterministic" #define CUOPT_PRESOLVE "presolve" -#define CUOPT_INITIAL_PERTURBATION "initial_perturbation" -#define CUOPT_REMOVE_PERTURBATION "remove_perturbation" -#define CUOPT_PRIMAL_PRICING "primal_pricing" +#define CUOPT_DUAL_SIMPLEX_INITIAL_PERTURBATION "dual_simplex_initial_perturbation" +#define CUOPT_DUAL_SIMPLEX_REMOVE_PERTURBATION "dual_simplex_remove_perturbation" +#define CUOPT_PRIMAL_SIMPLEX_PRICING "primal_simplex_pricing" #define CUOPT_MIP_PROBING "mip_probing" #define CUOPT_DUAL_POSTSOLVE "dual_postsolve" #define CUOPT_MIP_DETERMINISM_MODE "mip_determinism_mode" diff --git a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp index 1c4d0ce71b..5e16b5428a 100644 --- a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp @@ -304,7 +304,7 @@ class pdlp_solver_settings_t { i_t ordering{-1}; i_t initial_perturbation{-1}; i_t remove_perturbation{-1}; - i_t primal_pricing{0}; + i_t primal_pricing{1}; barrier_dual_initial_point_t barrier_dual_initial_point{barrier_dual_initial_point_t::Automatic}; i_t postsolve_info{-1}; i_t barrier_presolve_bound_free_variables{-1}; // -1 automatic, 0 disabled, 1 enabled diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 577a82e1b1..f4b0e52f21 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -83,7 +83,7 @@ struct simplex_solver_settings_t { ordering(-1), initial_perturbation(-1), remove_perturbation(-1), - primal_pricing(0), + primal_pricing(1), barrier_dual_initial_point(barrier_dual_initial_point_t::Automatic), postsolve_info(-1), barrier_presolve_bound_free_variables(-1), diff --git a/cpp/src/math_optimization/solver_settings.cu b/cpp/src/math_optimization/solver_settings.cu index 12d3c67ed5..2c5846aeee 100644 --- a/cpp/src/math_optimization/solver_settings.cu +++ b/cpp/src/math_optimization/solver_settings.cu @@ -179,9 +179,9 @@ solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings {CUOPT_FOLDING, &pdlp_settings.folding, -1, 1, -1}, {CUOPT_DUALIZE, &pdlp_settings.dualize, -1, 1, -1}, {CUOPT_ORDERING, &pdlp_settings.ordering, -1, 1, -1}, - {CUOPT_INITIAL_PERTURBATION, &pdlp_settings.initial_perturbation, -1, 1, -1}, - {CUOPT_REMOVE_PERTURBATION, &pdlp_settings.remove_perturbation, -1, 1, -1}, - {CUOPT_PRIMAL_PRICING, &pdlp_settings.primal_pricing, 0, 1, 0}, + {CUOPT_DUAL_SIMPLEX_INITIAL_PERTURBATION, &pdlp_settings.initial_perturbation, -1, 1, -1}, + {CUOPT_DUAL_SIMPLEX_REMOVE_PERTURBATION, &pdlp_settings.remove_perturbation, -1, 1, -1}, + {CUOPT_PRIMAL_SIMPLEX_PRICING, &pdlp_settings.primal_pricing, 0, 1, 1}, {CUOPT_BARRIER_DUAL_INITIAL_POINT, reinterpret_cast(&pdlp_settings.barrier_dual_initial_point), -1, 2, -1}, {CUOPT_POSTSOLVE_INFO, &pdlp_settings.postsolve_info, -1, 1, -1}, {CUOPT_MIP_CUT_PASSES, &mip_settings.max_cut_passes, -1, std::numeric_limits::max(), 10}, From 9d1043014bb2d23c06ce0ae766105c4e6455c882 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 17:35:55 -0700 Subject: [PATCH 085/113] Rename simplex parameters and default primal pricing to Devex --- cpp/include/cuopt/mathematical_optimization/constants.h | 6 +++--- .../mathematical_optimization/pdlp/solver_settings.hpp | 2 +- cpp/src/dual_simplex/simplex_solver_settings.hpp | 2 +- cpp/src/math_optimization/solver_settings.cu | 6 +++--- 4 files changed, 8 insertions(+), 8 deletions(-) diff --git a/cpp/include/cuopt/mathematical_optimization/constants.h b/cpp/include/cuopt/mathematical_optimization/constants.h index c76867d0c0..ee91b017b0 100644 --- a/cpp/include/cuopt/mathematical_optimization/constants.h +++ b/cpp/include/cuopt/mathematical_optimization/constants.h @@ -57,9 +57,9 @@ #define CUOPT_ELIMINATE_DENSE_COLUMNS "eliminate_dense_columns" #define CUOPT_CUDSS_DETERMINISTIC "cudss_deterministic" #define CUOPT_PRESOLVE "presolve" -#define CUOPT_INITIAL_PERTURBATION "initial_perturbation" -#define CUOPT_REMOVE_PERTURBATION "remove_perturbation" -#define CUOPT_PRIMAL_PRICING "primal_pricing" +#define CUOPT_DUAL_SIMPLEX_INITIAL_PERTURBATION "dual_simplex_initial_perturbation" +#define CUOPT_DUAL_SIMPLEX_REMOVE_PERTURBATION "dual_simplex_remove_perturbation" +#define CUOPT_PRIMAL_SIMPLEX_PRICING "primal_simplex_pricing" #define CUOPT_MIP_PROBING "mip_probing" #define CUOPT_DUAL_POSTSOLVE "dual_postsolve" #define CUOPT_MIP_DETERMINISM_MODE "mip_determinism_mode" diff --git a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp index 1c4d0ce71b..5e16b5428a 100644 --- a/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp +++ b/cpp/include/cuopt/mathematical_optimization/pdlp/solver_settings.hpp @@ -304,7 +304,7 @@ class pdlp_solver_settings_t { i_t ordering{-1}; i_t initial_perturbation{-1}; i_t remove_perturbation{-1}; - i_t primal_pricing{0}; + i_t primal_pricing{1}; barrier_dual_initial_point_t barrier_dual_initial_point{barrier_dual_initial_point_t::Automatic}; i_t postsolve_info{-1}; i_t barrier_presolve_bound_free_variables{-1}; // -1 automatic, 0 disabled, 1 enabled diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 477da4403b..647b11f48e 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -83,7 +83,7 @@ struct simplex_solver_settings_t { ordering(-1), initial_perturbation(-1), remove_perturbation(-1), - primal_pricing(0), + primal_pricing(1), barrier_dual_initial_point(barrier_dual_initial_point_t::Automatic), postsolve_info(-1), barrier_presolve_bound_free_variables(-1), diff --git a/cpp/src/math_optimization/solver_settings.cu b/cpp/src/math_optimization/solver_settings.cu index 4a76b5813b..97a6eaba02 100644 --- a/cpp/src/math_optimization/solver_settings.cu +++ b/cpp/src/math_optimization/solver_settings.cu @@ -179,9 +179,9 @@ solver_settings_t::solver_settings_t() : pdlp_settings(), mip_settings {CUOPT_FOLDING, &pdlp_settings.folding, -1, 1, -1}, {CUOPT_DUALIZE, &pdlp_settings.dualize, -1, 1, -1}, {CUOPT_ORDERING, &pdlp_settings.ordering, -1, 1, -1}, - {CUOPT_INITIAL_PERTURBATION, &pdlp_settings.initial_perturbation, -1, 1, -1}, - {CUOPT_REMOVE_PERTURBATION, &pdlp_settings.remove_perturbation, -1, 1, -1}, - {CUOPT_PRIMAL_PRICING, &pdlp_settings.primal_pricing, 0, 1, 0}, + {CUOPT_DUAL_SIMPLEX_INITIAL_PERTURBATION, &pdlp_settings.initial_perturbation, -1, 1, -1}, + {CUOPT_DUAL_SIMPLEX_REMOVE_PERTURBATION, &pdlp_settings.remove_perturbation, -1, 1, -1}, + {CUOPT_PRIMAL_SIMPLEX_PRICING, &pdlp_settings.primal_pricing, 0, 1, 1}, {CUOPT_BARRIER_DUAL_INITIAL_POINT, reinterpret_cast(&pdlp_settings.barrier_dual_initial_point), -1, 2, -1}, {CUOPT_POSTSOLVE_INFO, &pdlp_settings.postsolve_info, -1, 1, -1}, {CUOPT_MIP_CUT_PASSES, &mip_settings.max_cut_passes, -1, std::numeric_limits::max(), 10}, From bc99b1edef5195d20fc0a3f694c7eeb8f59a159d Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 11 Sep 2026 21:37:54 -0700 Subject: [PATCH 086/113] Stop recursive sub-MIPs when branch-and-bound terminates --- cpp/src/branch_and_bound/branch_and_bound.cpp | 53 +++++++++++++++++-- cpp/src/branch_and_bound/branch_and_bound.hpp | 15 ++++++ 2 files changed, 64 insertions(+), 4 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index d43776b2f4..d1b26f21d4 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -891,6 +891,7 @@ template void branch_and_bound_t::set_solution_at_root(mip_solution_t& solution, const cut_info_t& cut_info) { + request_stop(); mutex_upper_.lock(); incumbent_.set_incumbent_solution(root_objective_, root_relax_soln_.x); upper_bound_ = root_objective_; @@ -920,6 +921,7 @@ template void branch_and_bound_t::set_final_solution(mip_solution_t& solution, f_t lower_bound) { + request_stop(); if (solver_status_ == mip_status_t::HALT) { settings_.log.debug("Stopping the solver...\n"); } if (solver_status_ == mip_status_t::NUMERICAL) { @@ -2068,6 +2070,40 @@ void branch_and_bound_t::work_stealing(bfs_worker_t* worker) } } +template +void branch_and_bound_t::request_stop() +{ + // Protect child lifetimes and acquire locks only from parent to child. + std::lock_guard lock(children_mutex_); + node_concurrent_halt_.store(1, std::memory_order_release); + for (branch_and_bound_t* child : active_children_) { + child->request_stop(); + } +} + +template +branch_and_bound_t::submip_registration_t::submip_registration_t( + branch_and_bound_t& parent, branch_and_bound_t& child) + : parent(parent), child(child) +{ + std::lock_guard lock(parent.children_mutex_); + parent.active_children_.push_back(&child); + if (parent.node_concurrent_halt_.load(std::memory_order_acquire) || + parent.received_halt_signal()) { + child.request_stop(); + } +} + +template +branch_and_bound_t::submip_registration_t::~submip_registration_t() +{ + std::lock_guard lock(parent.children_mutex_); + const typename std::vector::iterator position = + std::find(parent.active_children_.begin(), parent.active_children_.end(), &child); + assert(position != parent.active_children_.end()); + parent.active_children_.erase(position); +} + template void branch_and_bound_t::best_first_search_with(bfs_worker_t* worker) { @@ -2183,9 +2219,7 @@ void branch_and_bound_t::best_first_search_with(bfs_worker_t } } - if (solver_status_ == mip_status_t::TIME_LIMIT || solver_status_ == mip_status_t::OPTIMAL) { - node_concurrent_halt_ = 1; - } + if (solver_status_ != mip_status_t::UNSET) { request_stop(); } // If the worker has still nodes in the queue (this can happen if it was stopped due to // time limit, small gap or other reason), then do not add back to the pool to avoid @@ -2199,6 +2233,7 @@ void branch_and_bound_t::best_first_search_with(bfs_worker_t if (exploration_stats_.nodes_unexplored == 0 && bfs_worker_pool_.num_idle() == bfs_worker_pool_.size()) { is_running_ = false; + request_stop(); } } @@ -2372,6 +2407,10 @@ bool branch_and_bound_t::launch_diving_worker(bfs_worker_t* template bool branch_and_bound_t::launch_submip_worker(const std::vector& sol) { + if (solver_status_ != mip_status_t::UNSET || node_concurrent_halt_.load() || + received_halt_signal()) { + return false; + } if (settings_.submip_settings.rins == 0 && settings_.submip_settings.rens == 0) return false; if (settings_.submip_settings.rens == 0 && !incumbent_.has_incumbent) return false; if (submip_worker_pool_.num_idle() == 0) return false; @@ -2528,6 +2567,7 @@ void branch_and_bound_t::solve_submip(diving_worker_t* worke probing_implied_bound_t empty_probing(submip_problem.num_cols); branch_and_bound_t submip_bnb(submip_problem, submip_settings, tic(), empty_probing); + submip_registration_t registration(*this, submip_bnb); mip_solution_t submip_solution(submip_problem.num_cols); std::vector presolved_incumbent; @@ -5246,6 +5286,10 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut root_lp_current_lower_bound_ = -inf; exploration_stats_.nodes_unexplored = 0; exploration_stats_.nodes_explored = 0; + if (node_concurrent_halt_.load() || received_halt_signal()) { + solver_status_ = mip_status_t::HALT; + return solver_status_; + } original_lp_.A.to_compressed_row(Arow_); settings_.log.debug("Reduced cost strengthening enabled: %d\n", @@ -5615,6 +5659,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut mutex_upper_.unlock(); if (cut_pass_action == cut_pass_action_t::RETURN) { + request_stop(); if (settings_.benchmark_info_ptr != nullptr) { settings_.benchmark_info_ptr->cut_generation_time_sec = toc(cut_generation_start_time); } @@ -5770,7 +5815,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut } settings_.log.printf("Exploring the B&B tree using %d threads\n\n", settings_.num_threads); - node_concurrent_halt_ = 0; + // Do not clear a stop request delivered by an ancestor before tree startup. exploration_stats_.nodes_explored = 0; exploration_stats_.nodes_unexplored = 2; diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 633803ab13..b70a337bfb 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -44,6 +44,7 @@ #include #include #include +#include #include namespace cuopt::mathematical_optimization::mip { @@ -398,6 +399,20 @@ class branch_and_bound_t { bool enable_concurrent_lp_root_solve_{false}; std::atomic root_concurrent_halt_{0}; std::atomic node_concurrent_halt_{0}; + std::mutex children_mutex_; + std::vector active_children_; + + void request_stop(); + + // Declared after the child solver, so unregistration precedes child destruction. + struct submip_registration_t { + branch_and_bound_t& parent; + branch_and_bound_t& child; + submip_registration_t(branch_and_bound_t& parent, branch_and_bound_t& child); + ~submip_registration_t(); + submip_registration_t(const submip_registration_t&) = delete; + submip_registration_t& operator=(const submip_registration_t&) = delete; + }; bool is_root_solution_set{false}; bool has_initial_pseudocost_{false}; From 8dd900be07a0aee5e8b0d7d024a2d4581aa1f9e6 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Sat, 12 Sep 2026 11:17:05 -0700 Subject: [PATCH 087/113] Use sparse reduced-cost updates for MIP strengthening --- cpp/src/branch_and_bound/branch_and_bound.cpp | 64 +++++++++++++++---- 1 file changed, 53 insertions(+), 11 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index d1b26f21d4..a3d74758e0 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -5032,6 +5032,9 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( const f_t relaxation_objective, reduced_cost_bounds_t& reduced_cost_bounds) { + const double strengthening_start = tic(); + double btran_time = 0.0; + double reduced_cost_update_time = 0.0; // Count primal degenerate basic variables i_t num_degenerate = 0; i_t num_degenerate_continuous = 0; @@ -5055,10 +5058,17 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( if (num_degenerate_integer == 0) return; + settings_.log.printf("RCS timing start: candidates=%d elapsed=%.6f\n", + num_degenerate_integer, + toc(exploration_stats_.start_time)); std::vector variable_to_basic_position(lp.num_cols, -1); for (i_t k = 0; k < lp.num_rows; k++) { variable_to_basic_position[basic_list[k]] = k; } + // The basis stays fixed across candidates; preserve Arow_'s ordering for cut generation. + csr_matrix_t local_Arow = Arow_; + std::vector nonbasic_end(lp.num_rows); + simplex::compute_initial_nonbasic_end(variable_to_basic_position, local_Arow, nonbasic_end); std::vector delta_y(lp.num_rows, 0); std::vector delta_z(lp.num_cols, 0); std::vector delta_z_mark(lp.num_cols, 0); @@ -5090,20 +5100,43 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( ep.x[0] = 1.0; sparse_vector_t delta_y_sparse; sparse_vector_t UTsol_sparse; + const double btran_start = tic(); basis_update.b_transpose_solve(ep, delta_y_sparse, UTsol_sparse); + btran_time += toc(btran_start); // delta_zN = -N^T * delta_y, delta_z[leaving] = -1 - delta_y_sparse.to_dense(delta_y); - simplex::compute_reduced_cost_update(lp, - basic_list, - nonbasic_list, - delta_y, - leaving_index, - /*direction=*/-1, - delta_z_mark, - delta_z_indices, - delta_z, - work_estimate); + const double reduced_cost_update_start = tic(); + i_t delta_y_nz0 = 0; + for (const f_t value : delta_y_sparse.x) { + if (std::abs(value) > 1e-12) { delta_y_nz0++; } + } + work_estimate += delta_y_sparse.i.size(); + const f_t delta_y_nz_percentage = delta_y_nz0 / static_cast(lp.num_rows) * 100.0; + if (delta_y_nz_percentage <= 30.0) { + simplex::compute_delta_z(local_Arow, + delta_y_sparse, + leaving_index, + /*direction=*/-1, + nonbasic_end, + delta_z_mark, + delta_z_indices, + delta_z, + work_estimate); + } else { + delta_y_sparse.to_dense(delta_y); + work_estimate += delta_y.size(); + simplex::compute_reduced_cost_update(lp, + basic_list, + nonbasic_list, + delta_y, + leaving_index, + /*direction=*/-1, + delta_z_mark, + delta_z_indices, + delta_z, + work_estimate); + } + reduced_cost_update_time += toc(reduced_cost_update_start); const f_t lower_j = lp.lower[j]; const f_t upper_j = lp.upper[j]; @@ -5271,6 +5304,15 @@ void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( } } settings_.log.printf("Added %d bounds for reduced cost strengthening\n", num_bounds_added); + settings_.log.printf( + "RCS timing end: candidates=%d bounds=%d total=%.6f btran=%.6f reduced_cost_update=%.6f " + "elapsed=%.6f\n", + num_degenerate_integer, + num_bounds_added, + toc(strengthening_start), + btran_time, + reduced_cost_update_time, + toc(exploration_stats_.start_time)); } template From 30cfc1513f3aa598a396f4d85481e8cc18031b99 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Sat, 12 Sep 2026 11:49:41 -0700 Subject: [PATCH 088/113] Fix primal simplex pricing parameter test Signed-off-by: Christopher Maes --- python/cuopt/cuopt/tests/linear_programming/test_lp_solver.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/python/cuopt/cuopt/tests/linear_programming/test_lp_solver.py b/python/cuopt/cuopt/tests/linear_programming/test_lp_solver.py index ae58752ffb..ab517eee0b 100644 --- a/python/cuopt/cuopt/tests/linear_programming/test_lp_solver.py +++ b/python/cuopt/cuopt/tests/linear_programming/test_lp_solver.py @@ -289,6 +289,8 @@ def _non_default_solver_param_value(name, current): return 0 if int(current) == 1 else 1 if name == "pdlp_precision": return 1 if int(current) == 0 else 0 + if name == "primal_simplex_pricing": + return 0 if int(current) == 1 else 1 if name == "mip_objective_step": return 0 if int(current) == 1 else 1 if isinstance(current, bool): From c6429624e125190a59e07f37e79bbcfb13271f12 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Mon, 14 Sep 2026 15:14:04 -0700 Subject: [PATCH 089/113] Validate dual simplex cutoffs with the original dual objective --- cpp/src/dual_simplex/phase2.cpp | 44 +++++++++++++++++-- .../dual_simplex/simplex_solver_settings.hpp | 2 + 2 files changed, 43 insertions(+), 3 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index a5f10c3229..b78bfc62d2 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2826,6 +2826,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, toc(start_time)); } i_t iterations_since_refactor = 0; + i_t last_cutoff_check = -1; while (iter < iter_limit) { PHASE2_NVTX_RANGE("DualSimplex::phase2_main_loop"); @@ -3708,9 +3709,46 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } } - if (obj >= settings.cut_off) { - settings.log.printf("Solve cutoff. Current objecive %e. Cutoff %e\n", obj, settings.cut_off); - return dual_status_t::CUTOFF; + if (obj >= settings.cut_off && + (last_cutoff_check == -1 || iter - last_cutoff_check >= settings.cutoff_check_frequency)) { + last_cutoff_check = iter; + const f_t unperturb_obj = compute_objective(lp, x); + if (unperturb_obj >= settings.cut_off) { + // Validate the cutoff using the original objective, not the perturbed costs. + std::vector trial_y = y; + std::vector trial_z = z; + if (phase2::amount_of_perturbation(lp, objective) != 0.0) { + phase2::compute_dual_solution_from_basis( + lp, ft, basic_list, nonbasic_list, trial_y, trial_z, phase2_work_estimate); + } + const f_t dual_infeas = phase2::dual_infeasibility( + lp, settings, vstatus, trial_z, settings.tight_tol, settings.dual_tol); + const bool dual_feasible = dual_infeas <= settings.dual_tol; + + // Include residual reduced costs for basic variables in the dual bound. + std::vector reduced_cost = lp.objective; + matrix_transpose_vector_multiply(lp.A, -1.0, trial_y, 1.0, reduced_cost); + f_t dual_objective = dot(lp.rhs, trial_y); + for (i_t j = 0; j < n; j++) { + const bool missing_bound = (reduced_cost[j] > 0.0 && lp.lower[j] == -inf) || + (reduced_cost[j] < 0.0 && lp.upper[j] == inf); + // Tolerate roundoff at infinite bounds only; this is an approximate certificate. + if (missing_bound && std::abs(reduced_cost[j]) <= settings.zero_tol) { continue; } + if (reduced_cost[j] > 0.0) { + dual_objective += reduced_cost[j] * lp.lower[j]; + } else if (reduced_cost[j] < 0.0) { + dual_objective += reduced_cost[j] * lp.upper[j]; + } + } + phase2_work_estimate += 3 * lp.A.col_start[n] + 12 * n + 2 * m; + + if (dual_feasible && std::isfinite(dual_objective) && dual_objective >= settings.cut_off) { + z = trial_z; + y = trial_y; + return dual_status_t::CUTOFF; + } + } + phase2_work_estimate += 2 * n; } if (work_unit_context && work_unit_context->global_work_units_elapsed >= settings.work_limit) { diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index 8b3eba56d3..f75f89ebfc 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -91,6 +91,7 @@ struct simplex_solver_settings_t { refactor_frequency(100), iteration_log_frequency(1000), first_iteration_log(2), + cutoff_check_frequency(1000), num_threads(omp_get_max_threads() - 1), max_cut_passes(0), mir_cuts(-1), @@ -201,6 +202,7 @@ struct simplex_solver_settings_t { i_t refactor_frequency; // number of basis updates before refactorization i_t iteration_log_frequency; // number of iterations between log updates i_t first_iteration_log; // number of iterations to log at beginning of solve + i_t cutoff_check_frequency; // number of iterations between cutoff checks i_t num_threads; // number of threads to use i_t random_seed; // random seed i_t max_cut_passes; // number of cut passes to make From d11f578a036e3880c1ae3a5fc7b9ff3bbdacb984 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Mon, 14 Sep 2026 18:42:40 -0700 Subject: [PATCH 090/113] Fix for cycling in bnat400 and mas74; limit last bucket so that slope is always nonnegative --- .../bound_flipping_ratio_test.cpp | 62 ++++++++++++++++++- .../bound_flipping_ratio_test.hpp | 6 ++ 2 files changed, 65 insertions(+), 3 deletions(-) diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp index 8e628a9dd2..2dec031fab 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.cpp @@ -122,6 +122,52 @@ void bound_flipping_ratio_test_t::determine_flips(f_t step_length, work_estimate_ += 5 * delta_z_indices_.size() + flip_indices.size(); } +template +i_t bound_flipping_ratio_test_t::limit_last_bucket(std::vector& candidates, + i_t first, + i_t end, + const std::vector& indices, + const std::vector& ratios, + f_t slope) +{ + // Three-way weighted selection. Discarded lower partitions remain eligible; + // only the partition containing the slope crossing needs another scan. + while (first < end) { + if (toc(start_time_) > settings_.time_limit) return RATIO_TEST_TIME_LIMIT; + if (settings_.concurrent_halt != nullptr && *settings_.concurrent_halt == 1) { + return CONCURRENT_HALT_RETURN; + } + const f_t split = ratios[candidates[first + (end - first) / 2]]; + i_t lower = first, scan = first, upper = end; + f_t lower_weight = 0.0, equal_weight = 0.0; + while (scan < upper) { + const i_t k = candidates[scan]; + const i_t j = nonbasic_list_[indices[k]]; + const f_t weight = + bounded_variables_[j] ? std::abs(delta_z_[j]) * (upper_[j] - lower_[j]) : inf; + if (ratios[k] < split) { + lower_weight += weight; + std::swap(candidates[lower++], candidates[scan++]); + } else if (ratios[k] > split) { + std::swap(candidates[scan], candidates[--upper]); + } else { + equal_weight += weight; + ++scan; + } + } + work_estimate_ += 12 * (end - first); + if (lower > first && lower_weight >= slope) { + end = lower; + } else if (lower_weight + equal_weight >= slope) { + return upper; // Keep every candidate at the crossing breakpoint. + } else { + slope -= lower_weight + equal_weight; + first = upper; + } + } + return end; +} + template i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, i_t& nonbasic_entering, @@ -336,8 +382,9 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, f_t threshold = minimum_harris_ratio; i_t num_buckets = 0; std::vector bucket_start(num_candidates + 1, 0); - f_t cumulative_slope = slope; - scan_start = 0; + f_t cumulative_slope = slope; + f_t last_bucket_slope = slope; + scan_start = 0; work_estimate_ += num_candidates + 1; // This is O(num_buckets * num_candidates) @@ -348,6 +395,7 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, } f_t next_threshold = inf; i_t write = scan_start; + last_bucket_slope = cumulative_slope; for (i_t h = scan_start; h < num_candidates; h++) { const i_t k = candidates[h]; @@ -373,7 +421,15 @@ i_t bound_flipping_ratio_test_t::compute_step_length(f_t& step_length, scan_start = write; threshold = next_threshold; - if (cumulative_slope < 0.0) break; + if (cumulative_slope <= 0.0) break; + } + + if (num_buckets > 0) { + const i_t end = bucket_start[num_buckets]; + const i_t retained_end = limit_last_bucket( + candidates, bucket_start[num_buckets - 1], end, indicies, ratios, last_bucket_slope); + if (retained_end < 0) return retained_end; + bucket_start[num_buckets] = retained_end; } // Compute the maximum pivot diff --git a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp index 8076be5927..4a4d1221fa 100644 --- a/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp +++ b/cpp/src/dual_simplex/bound_flipping_ratio_test.hpp @@ -69,6 +69,12 @@ class bound_flipping_ratio_test_t { i_t& entering_index, f_t& max_val); void determine_flips(f_t step_length, i_t entering_index, std::vector& flip_indices); + i_t limit_last_bucket(std::vector& candidates, + i_t first, + i_t end, + const std::vector& indices, + const std::vector& ratios, + f_t slope); const std::vector& lower_; const std::vector& upper_; const std::vector& bounded_variables_; From 2e5e02e72ec21e87d1f16311c914f08bb1f2d76d Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 15 Sep 2026 11:07:00 -0700 Subject: [PATCH 091/113] Adapt cutoff checks to BTRAN density and remove redundant status check --- cpp/src/dual_simplex/phase2.cpp | 12 ++++++------ cpp/src/dual_simplex/simplex_solver_settings.hpp | 2 -- 2 files changed, 6 insertions(+), 8 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index b78bfc62d2..03e5e0b924 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -3709,8 +3709,12 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } } + // Use the pivotal BTRAN density already measured in this iteration. Always + // attempt the first cutoff; subsequent checks are spaced 1 to 100 iterations apart. + const i_t cutoff_check_frequency = + static_cast(100.0 / std::clamp(delta_y_nz_percentage, f_t{1}, f_t{100})); if (obj >= settings.cut_off && - (last_cutoff_check == -1 || iter - last_cutoff_check >= settings.cutoff_check_frequency)) { + (last_cutoff_check == -1 || iter - last_cutoff_check >= cutoff_check_frequency)) { last_cutoff_check = iter; const f_t unperturb_obj = compute_objective(lp, x); if (unperturb_obj >= settings.cut_off) { @@ -3721,10 +3725,6 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::compute_dual_solution_from_basis( lp, ft, basic_list, nonbasic_list, trial_y, trial_z, phase2_work_estimate); } - const f_t dual_infeas = phase2::dual_infeasibility( - lp, settings, vstatus, trial_z, settings.tight_tol, settings.dual_tol); - const bool dual_feasible = dual_infeas <= settings.dual_tol; - // Include residual reduced costs for basic variables in the dual bound. std::vector reduced_cost = lp.objective; matrix_transpose_vector_multiply(lp.A, -1.0, trial_y, 1.0, reduced_cost); @@ -3742,7 +3742,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, } phase2_work_estimate += 3 * lp.A.col_start[n] + 12 * n + 2 * m; - if (dual_feasible && std::isfinite(dual_objective) && dual_objective >= settings.cut_off) { + if (std::isfinite(dual_objective) && dual_objective >= settings.cut_off) { z = trial_z; y = trial_y; return dual_status_t::CUTOFF; diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index f75f89ebfc..8b3eba56d3 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -91,7 +91,6 @@ struct simplex_solver_settings_t { refactor_frequency(100), iteration_log_frequency(1000), first_iteration_log(2), - cutoff_check_frequency(1000), num_threads(omp_get_max_threads() - 1), max_cut_passes(0), mir_cuts(-1), @@ -202,7 +201,6 @@ struct simplex_solver_settings_t { i_t refactor_frequency; // number of basis updates before refactorization i_t iteration_log_frequency; // number of iterations between log updates i_t first_iteration_log; // number of iterations to log at beginning of solve - i_t cutoff_check_frequency; // number of iterations between cutoff checks i_t num_threads; // number of threads to use i_t random_seed; // random seed i_t max_cut_passes; // number of cut passes to make From 9206716ed041420e144cc64ba337ca6a16174b2d Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 15 Sep 2026 20:41:17 -0700 Subject: [PATCH 092/113] Avoid redundant reduced-cost product in cutoff validation --- cpp/src/dual_simplex/phase2.cpp | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 03e5e0b924..0e6589a32b 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -3720,10 +3720,13 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, if (unperturb_obj >= settings.cut_off) { // Validate the cutoff using the original objective, not the perturbed costs. std::vector trial_y = y; - std::vector trial_z = z; if (phase2::amount_of_perturbation(lp, objective) != 0.0) { - phase2::compute_dual_solution_from_basis( - lp, ft, basic_list, nonbasic_list, trial_y, trial_z, phase2_work_estimate); + std::vector original_basic_cost(m); + for (i_t k = 0; k < m; ++k) { + original_basic_cost[k] = lp.objective[basic_list[k]]; + } + phase2_work_estimate += 5 * m; + ft.b_transpose_solve(original_basic_cost, trial_y); } // Include residual reduced costs for basic variables in the dual bound. std::vector reduced_cost = lp.objective; @@ -3743,7 +3746,11 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += 3 * lp.A.col_start[n] + 12 * n + 2 * m; if (std::isfinite(dual_objective) && dual_objective >= settings.cut_off) { - z = trial_z; + // Preserve the basic reduced-cost convention only after evaluating the bound. + for (const i_t j : basic_list) + reduced_cost[j] = 0.0; + phase2_work_estimate += m; + z = std::move(reduced_cost); y = trial_y; return dual_status_t::CUTOFF; } From b966d5384ef8e37a7274ed8d688a375ece97e26a Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 17 Sep 2026 17:38:23 -0700 Subject: [PATCH 093/113] Use sparse input for U multiplication --- cpp/src/dual_simplex/basis_updates.cpp | 45 ++++++++++++++++++++------ cpp/src/dual_simplex/basis_updates.hpp | 3 +- 2 files changed, 38 insertions(+), 10 deletions(-) diff --git a/cpp/src/dual_simplex/basis_updates.cpp b/cpp/src/dual_simplex/basis_updates.cpp index c2a7027548..6db50f0f69 100644 --- a/cpp/src/dual_simplex/basis_updates.cpp +++ b/cpp/src/dual_simplex/basis_updates.cpp @@ -2021,20 +2021,47 @@ void basis_update_mpf_t::u_multiply(const std::vector& x, std::ve work_estimate_ += 2 * U0_.col_start[U0_.n]; } -// Sparse-in/sparse-out overload of u_multiply. Same semantics as the dense version. +// Sparse-in/sparse-out overload of u_multiply. Input and output must not alias. template void basis_update_mpf_t::u_multiply(const sparse_vector_t& x, sparse_vector_t& y) const { const i_t m = L0_.m; - // Scatter x into a dense workspace, compute U0 * x, gather back to sparse. - std::vector x_dense; - x.to_dense(x_dense); - std::vector y_dense(m, 0.0); - matrix_vector_multiply(U0_, f_t(1.0), x_dense, f_t(0.0), y_dense); - work_estimate_ += 2 * U0_.col_start[U0_.n]; - y.from_dense(y_dense); - work_estimate_ += m; + i_t nz = 0; + const i_t x_nz = x.i.size(); + // The first half of xi_workspace_ holds marks, the second the touched rows. + for (i_t k = 0; k < x_nz; ++k) { + const i_t j = x.i[k]; + const f_t x_j = x.x[k]; + if (x_j == 0) { continue; } + const i_t col_start = U0_.col_start[j]; + const i_t col_end = U0_.col_start[j + 1]; + for (i_t p = col_start; p < col_end; ++p) { + const i_t i = U0_.i[p]; + if (!xi_workspace_[i]) { + xi_workspace_[i] = 1; + xi_workspace_[m + nz++] = i; + } + x_workspace_[i] += U0_.x[p] * x_j; + } + work_estimate_ += 2 * (col_end - col_start); + } + y.n = m; + y.i.clear(); + y.x.clear(); + y.i.reserve(nz); + y.x.reserve(nz); + for (i_t k = 0; k < nz; ++k) { + const i_t i = xi_workspace_[m + k]; + if (x_workspace_[i] != 0) { + y.i.push_back(i); + y.x.push_back(x_workspace_[i]); + } + x_workspace_[i] = 0.0; + xi_workspace_[i] = 0; + xi_workspace_[m + k] = 0; + } + work_estimate_ += x_nz + 5 * nz + 3 * y.i.size(); } // Solve for x such that L*x = y diff --git a/cpp/src/dual_simplex/basis_updates.hpp b/cpp/src/dual_simplex/basis_updates.hpp index bdedcc4a18..d7a92cd618 100644 --- a/cpp/src/dual_simplex/basis_updates.hpp +++ b/cpp/src/dual_simplex/basis_updates.hpp @@ -358,7 +358,8 @@ class basis_update_mpf_t { // against U0. void u_multiply(const std::vector& x, std::vector& y) const; - // Sparse-in/sparse-out overload of u_multiply. + // Sparse-in/sparse-out overload of u_multiply. Input and output must not alias. + // Output indices are unsorted; only exact zeros are omitted. void u_multiply(const sparse_vector_t& x, sparse_vector_t& y) const; // Replace the column B(:, leaving_index) with the vector abar. Pass in utilde such that L*utilde From 4d8e232055170bee724955ecea5d0749ad54b468 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Thu, 17 Sep 2026 17:52:42 -0700 Subject: [PATCH 094/113] Defer sparse U multiplication until integer pivots are accepted --- cpp/src/branch_and_bound/branch_and_bound.cpp | 114 ++++++++++-------- cpp/src/branch_and_bound/branch_and_bound.hpp | 6 +- 2 files changed, 69 insertions(+), 51 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index a3d74758e0..ee2279156a 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -4304,16 +4304,20 @@ i_t branch_and_bound_t::apply_delta_x_for_integer_pivot( std::vector& basic_list, std::vector& nonbasic_list, std::vector& nonbasic_index, + std::vector& variable_to_basic, std::vector& vstatus, i_t entering_index, i_t nonbasic_entering, i_t direction, - std::vector& delta_x, - const sparse_vector_t& utilde_sparse, + const sparse_vector_t& delta_x, + sparse_vector_t& utilde_sparse, simplex::lp_solution_t& solution, simplex::basis_update_mpf_t& basis_update, f_t& work_estimate) { + // Keep the existing dense ratio test and full integrality scan. + std::vector delta_x_dense; + delta_x.to_dense(delta_x_dense); f_t step_length; i_t basic_leaving; const i_t leaving_index = simplex::primal_ratio_test(lp, @@ -4321,7 +4325,7 @@ i_t branch_and_bound_t::apply_delta_x_for_integer_pivot( vstatus, basic_list, solution.x, - delta_x, + delta_x_dense, step_length, basic_leaving, entering_index, @@ -4343,7 +4347,7 @@ i_t branch_and_bound_t::apply_delta_x_for_integer_pivot( std::vector test_x = solution.x; i_t integer_destroyed = 0; for (i_t h = 0; h < lp.num_cols; ++h) { - test_x[h] += step_length * delta_x[h]; + test_x[h] += step_length * delta_x_dense[h]; if (var_types_[h] != variable_type_t::INTEGER) { continue; } const bool was_fractional = is_fractional(solution.x[h], var_types_[h], settings_.integer_tol); const bool now_fractional = is_fractional(test_x[h], var_types_[h], settings_.integer_tol); @@ -4356,13 +4360,33 @@ i_t branch_and_bound_t::apply_delta_x_for_integer_pivot( // Require a strict net decrease in fractional integers. if (integer_destroyed >= 0) { return -2; } - solution.x = test_x; - basic_list[basic_leaving] = entering_index; - nonbasic_list[nonbasic_entering] = leaving_index; - vstatus[entering_index] = variable_status_t::BASIC; + if (utilde_sparse.i.empty()) { + // Recover B^{-1} abar from the direction before changing the basis: + // delta_x[basic_list[h]] = -direction * (B^{-1} abar)[h]. + // In MPF, utilde = U0 * (B^{-1} abar), since all updates are absorbed into L. + sparse_vector_t b_inv_abar(lp.num_rows, 0); + b_inv_abar.i.reserve(delta_x.i.size()); + b_inv_abar.x.reserve(delta_x.x.size()); + const i_t nz = delta_x.i.size(); + for (i_t k = 0; k < nz; ++k) { + const i_t h = variable_to_basic[delta_x.i[k]]; + if (h >= 0 && delta_x.x[k] != 0) { + b_inv_abar.i.push_back(h); + b_inv_abar.x.push_back(-direction * delta_x.x[k]); + } + } + basis_update.u_multiply(b_inv_abar, utilde_sparse); + } + + solution.x = test_x; + basic_list[basic_leaving] = entering_index; + variable_to_basic[entering_index] = basic_leaving; + variable_to_basic[leaving_index] = -1; + nonbasic_list[nonbasic_entering] = leaving_index; + vstatus[entering_index] = variable_status_t::BASIC; if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { vstatus[leaving_index] = variable_status_t::NONBASIC_FIXED; - } else if (delta_x[leaving_index] < 0) { + } else if (delta_x_dense[leaving_index] < 0) { vstatus[leaving_index] = variable_status_t::NONBASIC_LOWER; } else { vstatus[leaving_index] = variable_status_t::NONBASIC_UPPER; @@ -4405,6 +4429,9 @@ i_t branch_and_bound_t::apply_delta_x_for_integer_pivot( if (rank == CONCURRENT_HALT_RETURN || rank == TIME_LIMIT_RETURN) { return -3; } if (rank < 0 || rank != lp.num_rows) { return -3; } simplex::reorder_basic_list(q, basic_list); + for (i_t k = 0; k < m; ++k) { + variable_to_basic[basic_list[k]] = k; + } basis_update.reset(L, U, p); } @@ -4421,6 +4448,7 @@ void branch_and_bound_t::fast_slack_integer_pivots( std::vector& basic_list, std::vector& nonbasic_list, std::vector& nonbasic_index, + std::vector& variable_to_basic, std::vector& vstatus, simplex::lp_solution_t& soln, simplex::basis_update_mpf_t& basis_update, @@ -4505,7 +4533,7 @@ void branch_and_bound_t::fast_slack_integer_pivots( // delta_x[nonbasic_slack] = -delta_xj * a_ij > 0, i.e. the entering slack moves up from // its lower bound 0. We build the sparse version to feed the feasibility scan, then // normalize so that delta_x[nonbasic_slack] == 1 (the convention primal_ratio_test expects - // for entering variables) and scatter into a dense vector. + // for entering variables). sparse_vector_t delta_x_sparse; delta_x_sparse.n = lp.num_cols; delta_x_sparse.i.reserve(col_end - col_start + 1); @@ -4542,38 +4570,25 @@ void branch_and_bound_t::fast_slack_integer_pivots( val /= scale; } - std::vector delta_x(lp.num_cols, 0.0); - delta_x_sparse.to_dense(delta_x); - // Entering variable is the nonbasic slack, moving up from its lower bound 0. const i_t entering_index = nonbasic_slack; const i_t nonbasic_entering = nonbasic_index[nonbasic_slack]; if (nonbasic_entering < 0) { continue; } const i_t direction = 1; - // Recover B^{-1} * abar from the full-vector delta_x. In our sign convention, - // delta_x[basic_list[h]] = -direction * (B^{-1} abar)[h], so - // (B^{-1} abar)[h] = -direction * delta_x[basic_list[h]]. - // Then utilde = L^{-1} P abar = U * (B^{-1} abar). In MPF, U == U0 (rank-1 updates all - // live in L), so u_multiply is a single sparse matvec against U0. - std::vector b_inv_abar(lp.num_rows); - for (i_t h = 0; h < lp.num_rows; ++h) { - b_inv_abar[h] = -direction * delta_x[basic_list[h]]; - } - std::vector utilde_dense; - basis_update.u_multiply(b_inv_abar, utilde_dense); + // The common helper computes utilde only if the pivot is accepted. sparse_vector_t utilde_sparse; - utilde_sparse.from_dense(utilde_dense); i_t error = apply_delta_x_for_integer_pivot(lp, basic_list, nonbasic_list, nonbasic_index, + variable_to_basic, vstatus, entering_index, nonbasic_entering, direction, - delta_x, + delta_x_sparse, utilde_sparse, soln, basis_update, @@ -4681,6 +4696,10 @@ i_t branch_and_bound_t::pivot_out_integer_variables( } std::vector nonbasic_index; + std::vector variable_to_basic(lp.num_cols, -1); + for (i_t k = 0; k < lp.num_rows; k++) { + variable_to_basic[basic_list_copy[k]] = k; + } fast_slack_integer_pivots(lp, settings, fractional, @@ -4689,17 +4708,13 @@ i_t branch_and_bound_t::pivot_out_integer_variables( basic_list_copy, nonbasic_list_copy, nonbasic_index, + variable_to_basic, vstatus_copy, soln_copy, basis_update_copy, work_estimate); std::vector work_list = fractional; - std::vector to_basic_position(lp.num_cols, -1); - - for (i_t k = 0; k < lp.num_rows; k++) { - to_basic_position[basic_list_copy[k]] = k; - } sparse_vector_t ep; ep.n = lp.num_rows; @@ -4733,7 +4748,7 @@ i_t branch_and_bound_t::pivot_out_integer_variables( while (!work_list.empty()) { const i_t j = work_list.back(); - const i_t p = to_basic_position[j]; + const i_t p = variable_to_basic[j]; work_list.pop_back(); worklist_total_processed++; @@ -4862,18 +4877,22 @@ i_t branch_and_bound_t::pivot_out_integer_variables( worklist_ftran_time += toc(ftran_start); worklist_ftran_done++; - std::vector delta_xB_dense; - delta_xB.to_dense(delta_xB_dense); - std::vector delta_x(lp.num_cols, 0.0); - for (i_t i = 0; i < lp.num_rows; i++) { - delta_x[basic_list_copy[i]] = -direction * delta_xB_dense[i]; + sparse_vector_t delta_x(lp.num_cols, 0); + delta_x.i.reserve(delta_xB.i.size() + 1); + delta_x.x.reserve(delta_xB.x.size() + 1); + const i_t nz = delta_xB.i.size(); + for (i_t k = 0; k < nz; ++k) { + delta_x.i.push_back(basic_list_copy[delta_xB.i[k]]); + delta_x.x.push_back(-direction * delta_xB.x[k]); } - delta_x[q] = direction; + delta_x.i.push_back(q); + delta_x.x.push_back(direction); i_t error = apply_delta_x_for_integer_pivot(lp, basic_list_copy, nonbasic_list_copy, nonbasic_index, + variable_to_basic, vstatus_copy, entering_index, nonbasic_entering, @@ -4900,21 +4919,18 @@ i_t branch_and_bound_t::pivot_out_integer_variables( if (!error) { worklist_pivots_succeeded++; - // Update to_basic_position for the variables that changed status - // entering_index is now basic, leaving_index is now nonbasic - // Find the leaving variable: it's the one that took entering_index's slot in nonbasic_list - const i_t leaving_index = nonbasic_list_copy[nonbasic_entering]; - to_basic_position[entering_index] = to_basic_position[leaving_index]; - to_basic_position[leaving_index] = -1; - - // We did a successful pivot; add fractional variables whose values changed to work list +#ifdef READD_TO_WORKLIST + // We did a successful pivot; add fractional variables whose values changed to work list. + std::vector delta_x_dense; + delta_x.to_dense(delta_x_dense); for (i_t k : fractional) { if (vstatus_copy[k] != variable_status_t::BASIC) { continue; } - if (std::abs(delta_x[k]) > settings_.zero_tol) { - // work_list.push_back(k); - // worklist_readded++; + if (std::abs(delta_x_dense[k]) > settings_.zero_tol) { + work_list.push_back(k); + worklist_readded++; } } +#endif break; } } diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index b70a337bfb..4a5badf3a4 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -516,6 +516,7 @@ class branch_and_bound_t { std::vector& basic_list, std::vector& nonbasic_list, std::vector& nonbasic_index, + std::vector& variable_to_basic, std::vector& vstatus, simplex::lp_solution_t& soln, simplex::basis_update_mpf_t& basis_update, @@ -536,12 +537,13 @@ class branch_and_bound_t { std::vector& basic_list, std::vector& nonbasic_list, std::vector& nonbasic_index, + std::vector& variable_to_basic, std::vector& vstatus, i_t entering_index, i_t nonbasic_entering, i_t direction, - std::vector& delta_x, - const sparse_vector_t& utilde_sparse, + const sparse_vector_t& delta_x, + sparse_vector_t& utilde_sparse, simplex::lp_solution_t& solution, simplex::basis_update_mpf_t& basis_update, f_t& work_estimate); From 5a17774f488f8199869f2c86747b96c182091ee8 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Mon, 14 Sep 2026 15:14:04 -0700 Subject: [PATCH 095/113] Validate dual simplex cutoffs with the original dual objective --- cpp/src/dual_simplex/phase2.cpp | 44 +++++++++++++++++-- .../dual_simplex/simplex_solver_settings.hpp | 2 + 2 files changed, 43 insertions(+), 3 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index a93acb4efd..917e0d3a43 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -3342,6 +3342,7 @@ static dual_status_t dual_phase2_with_advanced_basis( toc(start_time)); } i_t iterations_since_refactor = 0; + i_t last_cutoff_check = -1; while (iter < iter_limit) { PHASE2_NVTX_RANGE("DualSimplex::phase2_main_loop"); @@ -4229,9 +4230,46 @@ static dual_status_t dual_phase2_with_advanced_basis( } } - if (obj >= settings.cut_off) { - settings.log.printf("Solve cutoff. Current objecive %e. Cutoff %e\n", obj, settings.cut_off); - return dual_status_t::CUTOFF; + if (obj >= settings.cut_off && + (last_cutoff_check == -1 || iter - last_cutoff_check >= settings.cutoff_check_frequency)) { + last_cutoff_check = iter; + const f_t unperturb_obj = compute_objective(lp, x); + if (unperturb_obj >= settings.cut_off) { + // Validate the cutoff using the original objective, not the perturbed costs. + std::vector trial_y = y; + std::vector trial_z = z; + if (phase2::amount_of_perturbation(lp, objective) != 0.0) { + phase2::compute_dual_solution_from_basis( + lp, ft, basic_list, nonbasic_list, trial_y, trial_z, phase2_work_estimate); + } + const f_t dual_infeas = phase2::dual_infeasibility( + lp, settings, vstatus, trial_z, settings.tight_tol, settings.dual_tol); + const bool dual_feasible = dual_infeas <= settings.dual_tol; + + // Include residual reduced costs for basic variables in the dual bound. + std::vector reduced_cost = lp.objective; + matrix_transpose_vector_multiply(lp.A, -1.0, trial_y, 1.0, reduced_cost); + f_t dual_objective = dot(lp.rhs, trial_y); + for (i_t j = 0; j < n; j++) { + const bool missing_bound = (reduced_cost[j] > 0.0 && lp.lower[j] == -inf) || + (reduced_cost[j] < 0.0 && lp.upper[j] == inf); + // Tolerate roundoff at infinite bounds only; this is an approximate certificate. + if (missing_bound && std::abs(reduced_cost[j]) <= settings.zero_tol) { continue; } + if (reduced_cost[j] > 0.0) { + dual_objective += reduced_cost[j] * lp.lower[j]; + } else if (reduced_cost[j] < 0.0) { + dual_objective += reduced_cost[j] * lp.upper[j]; + } + } + phase2_work_estimate += 3 * lp.A.col_start[n] + 12 * n + 2 * m; + + if (dual_feasible && std::isfinite(dual_objective) && dual_objective >= settings.cut_off) { + z = trial_z; + y = trial_y; + return dual_status_t::CUTOFF; + } + } + phase2_work_estimate += 2 * n; } if (work_unit_context && work_unit_context->global_work_units_elapsed >= settings.work_limit) { diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index f4b0e52f21..da686c095a 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -94,6 +94,7 @@ struct simplex_solver_settings_t { refactor_frequency(100), iteration_log_frequency(1000), first_iteration_log(2), + cutoff_check_frequency(1000), num_threads(omp_get_max_threads() - 1), max_cut_passes(0), mir_cuts(-1), @@ -210,6 +211,7 @@ struct simplex_solver_settings_t { i_t refactor_frequency; // number of basis updates before refactorization i_t iteration_log_frequency; // number of iterations between log updates i_t first_iteration_log; // number of iterations to log at beginning of solve + i_t cutoff_check_frequency; // number of iterations between cutoff checks i_t num_threads; // number of threads to use i_t random_seed; // random seed i_t max_cut_passes; // number of cut passes to make From 80d6bf284d9e485a738de65e805ea027364f85c5 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 15 Sep 2026 11:07:00 -0700 Subject: [PATCH 096/113] Adapt cutoff checks to BTRAN density and remove redundant status check --- cpp/src/dual_simplex/phase2.cpp | 12 ++++++------ cpp/src/dual_simplex/simplex_solver_settings.hpp | 2 -- 2 files changed, 6 insertions(+), 8 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 917e0d3a43..d7decd8783 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -4230,8 +4230,12 @@ static dual_status_t dual_phase2_with_advanced_basis( } } + // Use the pivotal BTRAN density already measured in this iteration. Always + // attempt the first cutoff; subsequent checks are spaced 1 to 100 iterations apart. + const i_t cutoff_check_frequency = + static_cast(100.0 / std::clamp(delta_y_nz_percentage, f_t{1}, f_t{100})); if (obj >= settings.cut_off && - (last_cutoff_check == -1 || iter - last_cutoff_check >= settings.cutoff_check_frequency)) { + (last_cutoff_check == -1 || iter - last_cutoff_check >= cutoff_check_frequency)) { last_cutoff_check = iter; const f_t unperturb_obj = compute_objective(lp, x); if (unperturb_obj >= settings.cut_off) { @@ -4242,10 +4246,6 @@ static dual_status_t dual_phase2_with_advanced_basis( phase2::compute_dual_solution_from_basis( lp, ft, basic_list, nonbasic_list, trial_y, trial_z, phase2_work_estimate); } - const f_t dual_infeas = phase2::dual_infeasibility( - lp, settings, vstatus, trial_z, settings.tight_tol, settings.dual_tol); - const bool dual_feasible = dual_infeas <= settings.dual_tol; - // Include residual reduced costs for basic variables in the dual bound. std::vector reduced_cost = lp.objective; matrix_transpose_vector_multiply(lp.A, -1.0, trial_y, 1.0, reduced_cost); @@ -4263,7 +4263,7 @@ static dual_status_t dual_phase2_with_advanced_basis( } phase2_work_estimate += 3 * lp.A.col_start[n] + 12 * n + 2 * m; - if (dual_feasible && std::isfinite(dual_objective) && dual_objective >= settings.cut_off) { + if (std::isfinite(dual_objective) && dual_objective >= settings.cut_off) { z = trial_z; y = trial_y; return dual_status_t::CUTOFF; diff --git a/cpp/src/dual_simplex/simplex_solver_settings.hpp b/cpp/src/dual_simplex/simplex_solver_settings.hpp index da686c095a..f4b0e52f21 100644 --- a/cpp/src/dual_simplex/simplex_solver_settings.hpp +++ b/cpp/src/dual_simplex/simplex_solver_settings.hpp @@ -94,7 +94,6 @@ struct simplex_solver_settings_t { refactor_frequency(100), iteration_log_frequency(1000), first_iteration_log(2), - cutoff_check_frequency(1000), num_threads(omp_get_max_threads() - 1), max_cut_passes(0), mir_cuts(-1), @@ -211,7 +210,6 @@ struct simplex_solver_settings_t { i_t refactor_frequency; // number of basis updates before refactorization i_t iteration_log_frequency; // number of iterations between log updates i_t first_iteration_log; // number of iterations to log at beginning of solve - i_t cutoff_check_frequency; // number of iterations between cutoff checks i_t num_threads; // number of threads to use i_t random_seed; // random seed i_t max_cut_passes; // number of cut passes to make From 4db0c724cd923d3c9b501a102e3969f75f990a36 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 15 Sep 2026 20:41:17 -0700 Subject: [PATCH 097/113] Avoid redundant reduced-cost product in cutoff validation --- cpp/src/dual_simplex/phase2.cpp | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index d7decd8783..f1b335755c 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -4241,10 +4241,13 @@ static dual_status_t dual_phase2_with_advanced_basis( if (unperturb_obj >= settings.cut_off) { // Validate the cutoff using the original objective, not the perturbed costs. std::vector trial_y = y; - std::vector trial_z = z; if (phase2::amount_of_perturbation(lp, objective) != 0.0) { - phase2::compute_dual_solution_from_basis( - lp, ft, basic_list, nonbasic_list, trial_y, trial_z, phase2_work_estimate); + std::vector original_basic_cost(m); + for (i_t k = 0; k < m; ++k) { + original_basic_cost[k] = lp.objective[basic_list[k]]; + } + phase2_work_estimate += 5 * m; + ft.b_transpose_solve(original_basic_cost, trial_y); } // Include residual reduced costs for basic variables in the dual bound. std::vector reduced_cost = lp.objective; @@ -4264,7 +4267,11 @@ static dual_status_t dual_phase2_with_advanced_basis( phase2_work_estimate += 3 * lp.A.col_start[n] + 12 * n + 2 * m; if (std::isfinite(dual_objective) && dual_objective >= settings.cut_off) { - z = trial_z; + // Preserve the basic reduced-cost convention only after evaluating the bound. + for (const i_t j : basic_list) + reduced_cost[j] = 0.0; + phase2_work_estimate += m; + z = std::move(reduced_cost); y = trial_y; return dual_status_t::CUTOFF; } From adeadd1daa32dcd1e5faedf096aad28c6d58ff60 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 18 Sep 2026 15:30:07 -0700 Subject: [PATCH 098/113] Validate infeasibility and repair objectives below variable bounds --- cpp/src/dual_simplex/phase2.cpp | 189 ++++++++++++++++---------------- cpp/src/dual_simplex/primal.cpp | 1 - 2 files changed, 92 insertions(+), 98 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index a93acb4efd..524f0c37a7 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -3342,6 +3342,17 @@ static dual_status_t dual_phase2_with_advanced_basis( toc(start_time)); } i_t iterations_since_refactor = 0; + f_t box_objective_bound = 0.0; + if (phase == 2) { + for (i_t j = 0; j < n; ++j) { + const f_t cost = lp.objective[j]; + if (cost > 0.0) + box_objective_bound += cost * lp.lower[j]; + else if (cost < 0.0) + box_objective_bound += cost * lp.upper[j]; + } + phase2_work_estimate += 3 * n; + } while (iter < iter_limit) { PHASE2_NVTX_RANGE("DualSimplex::phase2_main_loop"); @@ -3528,26 +3539,72 @@ static dual_status_t dual_phase2_with_advanced_basis( // removal_status == 0 (OPTIMAL) or primal cleanup done: fall through to prepare_optimality } - phase2::prepare_optimality(0, - primal_infeasibility, - lp, - settings, - ft, - objective, - basic_list, - nonbasic_list, - vstatus, - phase, - start_time, - max_val, - phase2_work_estimate, - iter, - x, - y, - z, - sol); - status = dual_status_t::OPTIMAL; - break; + if (phase == 2 && std::isfinite(box_objective_bound)) { + // This helps prevent small negative bound O(-1e-8) on cbs-cta. + const f_t original_objective = compute_objective(lp, x); + phase2_work_estimate += 2 * n; + if (box_objective_bound - original_objective >= settings.tight_tol) { + // Normal pricing ignores sub-primal_tol violations, but their objective + // contribution can still put the solution below the variable-box bound. + f_t largest_objective_violation = -1.0; + for (i_t k = 0; k < m; ++k) { + const i_t j = basic_list[k]; + const f_t lower_violation = lp.lower[j] - x[j]; + const f_t upper_violation = x[j] - lp.upper[j]; + const f_t violation = std::max(lower_violation, upper_violation); + const f_t objective_violation = std::abs(lp.objective[j]) * violation; + if (violation > 0.0 && objective_violation > largest_objective_violation) { + largest_objective_violation = objective_violation; + leaving_index = j; + basic_leaving_index = k; + direction = lower_violation >= upper_violation ? 1 : -1; + } + } + phase2_work_estimate += 9 * m; + if (leaving_index < 0) { + settings.log.printf( + "Objective below box bound without a repairable basic violation.\n"); + return dual_status_t::NUMERICAL; + } + // Primal cleanup may have changed the basis since pricing. + phase2::reset_basis_mark( + basic_list, nonbasic_list, basic_mark, nonbasic_mark, phase2_work_estimate); + compute_initial_nonbasic_end(basic_mark, Arow, nonbasic_end); + const f_t violation = direction == 1 ? lp.lower[leaving_index] - x[leaving_index] + : x[leaving_index] - lp.upper[leaving_index]; + max_val = violation * violation; + settings.log.printf( + "Objective %.17g below box bound %.17g; forcing basic variable %d out (violation " + "%.17g).\n", + original_objective, + box_objective_bound, + leaving_index, + violation); + } + } + + if (leaving_index == -1) { + phase2::prepare_optimality(0, + primal_infeasibility, + lp, + settings, + ft, + objective, + basic_list, + nonbasic_list, + vstatus, + phase, + start_time, + max_val, + phase2_work_estimate, + iter, + x, + y, + z, + sol); + status = dual_status_t::OPTIMAL; + break; + } } if (toc(start_time) > settings.time_limit) { return dual_status_t::TIME_LIMIT; } @@ -3702,85 +3759,16 @@ static dual_status_t dual_phase2_with_advanced_basis( if (entering_index == RATIO_TEST_TIME_LIMIT) { return dual_status_t::TIME_LIMIT; } if (entering_index == CONCURRENT_HALT_RETURN) { return dual_status_t::CONCURRENT_LIMIT; } if (entering_index == RATIO_TEST_NO_ENTERING_VARIABLE) { + // While solving the (possibly) perturbed problem, we were unable to find an entering + // variable. We need to check if this implies the original problem is primal infeasible. Note + // that perturbing the primal objective does not affect whether the original problem is primal + // infeasible. + settings.log.printf("No entering variable found. Iter %d\n", iter); settings.log.printf("Scaled infeasibility %e\n", max_val); f_t perturbation = phase2::amount_of_perturbation(lp, objective); phase2_work_estimate += 2 * n; - if (perturbation > 0.0 && phase == 2) { - i_t removal_status = phase2::attempt_to_remove_perturbations(lp, - settings, - ft, - basic_list, - nonbasic_list, - vstatus, - objective, - z, - y, - x, - xB_workspace, - squared_infeasibilities, - infeasibility_indices, - primal_infeasibility, - primal_infeasibility_squared, - phase2_work_estimate); - if (removal_status == 2) { // PRIMAL_CLEANUP - const dual_status_t cleanup_status = phase2::run_primal_cleanup(lp, - settings, - start_time, - ft, - basic_list, - nonbasic_list, - vstatus, - objective, - sol, - iter, - phase2_work_estimate); - if (cleanup_status != dual_status_t::OPTIMAL) { return cleanup_status; } - objective = lp.objective; - } - if (removal_status == 0 || removal_status == 2) { // OPTIMAL or successful cleanup - obj = phase2::compute_perturbed_objective(objective, x); - phase2_work_estimate += 2 * n; - if (removal_status == 2 || primal_infeasibility <= settings.primal_tol) { - phase2_work_estimate += ft.work_estimate(); - ft.clear_work_estimate(); - phase2::prepare_optimality(1, - primal_infeasibility, - lp, - settings, - ft, - objective, - basic_list, - nonbasic_list, - vstatus, - phase, - start_time, - max_val, - phase2_work_estimate, - iter, - x, - y, - z, - sol); - status = dual_status_t::OPTIMAL; - break; - } - settings.log.printf("Continuing with perturbation removed\n"); - phase2_work_estimate += 3 * delta_z_indices.size(); - phase2::clear_delta_z( - entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); - continue; - } else if (removal_status == 1) { // CONTINUE_DUAL - obj = phase2::compute_perturbed_objective(objective, x); - phase2_work_estimate += 2 * n; - phase2_work_estimate += 3 * delta_z_indices.size(); - phase2::clear_delta_z( - entering_index, leaving_index, delta_z_mark, delta_z_indices, delta_z); - continue; - } - } - if (perturbation == 0.0 && phase == 2) { constexpr bool use_farkas = false; if constexpr (use_farkas) { @@ -3815,17 +3803,24 @@ static dual_status_t dual_phase2_with_advanced_basis( phase2::dual_infeasibility(lp, settings, vstatus, z, settings.tight_tol, settings.dual_tol); phase2_work_estimate += 3 * n; settings.log.printf("Dual infeasibility %e\n", dual_infeas); + std::vector dual_res1; + phase2::compute_dual_residual(lp.A, objective, y, z, dual_res1); + f_t dual_res_norm = vector_norm_inf(dual_res1); + phase2_work_estimate += 2.0 * lp.A.nnz() + 6.0 * n; + settings.log.printf("Dual residual %e\n", dual_res_norm); const f_t primal_inf = simplex::phase2::primal_infeasibility(lp, settings, vstatus, x); phase2_work_estimate += 3 * n; settings.log.printf("Primal infeasibility %e\n", primal_inf); settings.log.printf("Updates %d\n", ft.num_updates()); settings.log.printf("Steepest edge %e\n", max_val); - if (dual_infeas > settings.dual_tol) { + if (dual_infeas <= settings.dual_tol && dual_res_norm <= settings.dual_tol) { + return dual_status_t::DUAL_UNBOUNDED; + } else { settings.log.printf( - "Numerical issues encountered. No entering variable found with large infeasibility.\n"); + "Numerical issues encountered. No entering variable found with large infeasibility or " + "residual.\n"); return dual_status_t::NUMERICAL; } - return dual_status_t::DUAL_UNBOUNDED; } timers.start_timer(phase2_work_estimate + ft.work_estimate()); diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 03ad284428..04536d0369 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -1136,7 +1136,6 @@ primal_status_t primal_phase2_with_advanced_basis( f_t retry_dual_tol = pricing_dual_tol; while (entering_index == -1 && retry_dual_tol > f_t(1e-10)) { retry_dual_tol *= f_t(0.1); - settings.log.printf("Retrying phase-I pricing with dual_tol %e\n", retry_dual_tol); entering_index = phase2_pricing(lp, z, nonbasic_list, From ebe0c3114159d87077b20e72a870f313e3a233c0 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 18 Sep 2026 15:30:28 -0700 Subject: [PATCH 099/113] Restrict objective cutoffs to dual simplex phase two --- cpp/src/dual_simplex/phase2.cpp | 4 +++- cpp/src/dual_simplex/solve.cpp | 2 +- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 0e6589a32b..84656953df 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -3713,7 +3713,7 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, // attempt the first cutoff; subsequent checks are spaced 1 to 100 iterations apart. const i_t cutoff_check_frequency = static_cast(100.0 / std::clamp(delta_y_nz_percentage, f_t{1}, f_t{100})); - if (obj >= settings.cut_off && + if (phase == 2 && obj >= settings.cut_off && (last_cutoff_check == -1 || iter - last_cutoff_check >= cutoff_check_frequency)) { last_cutoff_check = iter; const f_t unperturb_obj = compute_objective(lp, x); @@ -3752,6 +3752,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate += m; z = std::move(reduced_cost); y = trial_y; + settings.log.printf( + "Solve cutoff. Current objective %e. Cutoff %e\n", dual_objective, settings.cut_off); return dual_status_t::CUTOFF; } } diff --git a/cpp/src/dual_simplex/solve.cpp b/cpp/src/dual_simplex/solve.cpp index 388bb43b35..56d9906e73 100644 --- a/cpp/src/dual_simplex/solve.cpp +++ b/cpp/src/dual_simplex/solve.cpp @@ -220,7 +220,7 @@ lp_status_t solve_linear_program_with_advanced_basis( edge_norms, work_unit_context); } - if (phase1_status == dual_status_t::NUMERICAL) { + if (phase1_status == dual_status_t::NUMERICAL || phase1_status == dual_status_t::CUTOFF) { settings.log.printf("Failed in Phase 1\n"); return lp_status_t::NUMERICAL_ISSUES; } From 8a71d1543dca2329c5cc346aa12af73a3c5bc29a Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Fri, 18 Sep 2026 17:16:18 -0700 Subject: [PATCH 100/113] Refactor simplex setup, add primal tests, and propagate crossover settings --- cpp/src/dual_simplex/phase2.cpp | 22 +-- cpp/src/dual_simplex/primal.cpp | 28 +--- cpp/src/dual_simplex/scaling.cpp | 11 +- cpp/src/pdlp/solve.cu | 12 +- cpp/tests/dual_simplex/unit_tests/primal.cpp | 159 +++++++++++++++++++ cpp/tests/internal/CMakeLists.txt | 1 + 6 files changed, 183 insertions(+), 50 deletions(-) create mode 100644 cpp/tests/dual_simplex/unit_tests/primal.cpp diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 524f0c37a7..54b33930af 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -2715,19 +2715,14 @@ void prepare_optimality(i_t info, const simplex_solver_settings_t& settings, basis_update_mpf_t& ft, const std::vector& objective, - // Primal cleanup below pivots, so the basis, the statuses - // and the iteration count are updated in place. - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& vstatus, + const std::vector& vstatus, int phase, f_t start_time, - f_t max_val, - f_t& work_estimate, - i_t& iter, - const std::vector& x, - std::vector& y, - std::vector& z, + f_t work_estimate, + i_t iter, + const std::vector x, + const std::vector y, + const std::vector z, lp_solution_t& sol) { const i_t m = lp.num_rows; @@ -3590,12 +3585,9 @@ static dual_status_t dual_phase2_with_advanced_basis( settings, ft, objective, - basic_list, - nonbasic_list, vstatus, phase, start_time, - max_val, phase2_work_estimate, iter, x, @@ -3750,7 +3742,7 @@ static dual_status_t dual_phase2_with_advanced_basis( timers.bfrt_time += timers.stop_timer(phase2_work_estimate + ft.work_estimate()); // BFRT diagnostics timers.bfrt_calls++; - if (step_length == 0.0) { timers.bfrt_zero_steps++; } + if (entering_index >= 0 && step_length == 0.0) { timers.bfrt_zero_steps++; } } else { entering_index = phase2::phase2_ratio_test( lp, settings, vstatus, nonbasic_list, z, delta_z, step_length, nonbasic_entering_index); diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 04536d0369..9f1f5d591f 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -855,8 +855,6 @@ primal_status_t primal_phase2_with_advanced_basis( std::vector& y = sol.y; std::vector& z = sol.z; - std::vector incoming_x = x; - std::vector incoming_vstatus = vstatus; work_estimate += 2.0 * n; settings.log.printf("Primal Simplex\n"); settings.log.printf("Pricing: %s\n", settings.primal_pricing == 1 ? "Devex" : "Dantzig"); @@ -864,32 +862,8 @@ primal_status_t primal_phase2_with_advanced_basis( // Setting them after the solve leaves ||A*x - b|| large whenever x_N != 0. set_primal_variables_on_bounds(lp, settings, vstatus, x, work_estimate); - std::vector rhs = lp.rhs; - work_estimate += m; - // rhs = b - sum_{j : x_j = l_j} A(:, j) l(j) - sum_{j : x_j = u_j} A(:, j) * - // u(j) - for (i_t k = 0; k < n - m; ++k) { - const i_t j = nonbasic_list[k]; - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - const f_t xj = x[j]; - for (i_t p = col_start; p < col_end; ++p) { - rhs[lp.A.i[p]] -= xj * lp.A.x[p]; - } - work_estimate += 3.0 * (col_end - col_start); - } - work_estimate += 4 * (n - m); - - std::vector xB(m); work_estimate += m; - - basis_update.b_solve(rhs, xB); - - for (i_t k = 0; k < m; ++k) { - const i_t j = basic_list[k]; - x[j] = xB[k]; - } - work_estimate += 3 * m; + compute_basic_primal_variables(lp, basis_update, basic_list, nonbasic_list, x, work_estimate); constexpr bool print_norms = false; if constexpr (print_norms) { settings.log.printf("|| x || %e\n", vector_norm2(x)); } diff --git a/cpp/src/dual_simplex/scaling.cpp b/cpp/src/dual_simplex/scaling.cpp index 102036e635..b814cb17ff 100644 --- a/cpp/src/dual_simplex/scaling.cpp +++ b/cpp/src/dual_simplex/scaling.cpp @@ -258,15 +258,16 @@ i_t scaling(const lp_problem_t& unscaled, const bool use_lp_row_scaling = !settings.inside_mip && unscaled.second_order_cone_dims.empty() && unscaled.Q.n == 0; if (use_lp_row_scaling) { - csr_matrix_t Arow(0, 0, 0); - scaled.A.to_compressed_row(Arow); std::vector row_norm(m, 1.0); + for (i_t j = 0; j < n; ++j) { + for (i_t p = scaled.A.col_start[j]; p < scaled.A.col_start[j + 1]; ++p) { + const i_t i = scaled.A.i[p]; + row_norm[i] = std::max(row_norm[i], std::abs(scaled.A.x[p])); + } + } f_t max_row_norm = 0.0; f_t min_row_norm = inf; for (i_t i = 0; i < m; ++i) { - for (i_t p = Arow.row_start[i]; p < Arow.row_start[i + 1]; ++p) { - row_norm[i] = std::max(row_norm[i], std::abs(Arow.x[p])); - } max_row_norm = std::max(max_row_norm, row_norm[i]); min_row_norm = std::min(min_row_norm, row_norm[i]); } diff --git a/cpp/src/pdlp/solve.cu b/cpp/src/pdlp/solve.cu index 08135ebf67..14beb1e108 100644 --- a/cpp/src/pdlp/solve.cu +++ b/cpp/src/pdlp/solve.cu @@ -504,6 +504,9 @@ std::tuple, simplex::lp_status_t, f_t, f_t, f_t barrier_settings.time_limit = settings.time_limit; barrier_settings.iteration_limit = settings.iteration_limit; barrier_settings.concurrent_halt = settings.concurrent_halt; + barrier_settings.initial_perturbation = settings.initial_perturbation; + barrier_settings.remove_perturbation = settings.remove_perturbation; + barrier_settings.primal_pricing = settings.primal_pricing; barrier_settings.folding = settings.folding; barrier_settings.augmented = settings.augmented; barrier_settings.dualize = settings.dualize; @@ -914,9 +917,12 @@ optimization_problem_solution_t run_pdlp(mip::problem_t& pro simplex::lp_solution_t initial_solution(1, 1); translate_to_crossover_problem(problem, sol, lp, initial_solution); simplex::simplex_solver_settings_t dual_simplex_settings; - dual_simplex_settings.time_limit = settings.time_limit; - dual_simplex_settings.iteration_limit = settings.iteration_limit; - dual_simplex_settings.concurrent_halt = settings.concurrent_halt; + dual_simplex_settings.time_limit = settings.time_limit; + dual_simplex_settings.iteration_limit = settings.iteration_limit; + dual_simplex_settings.concurrent_halt = settings.concurrent_halt; + dual_simplex_settings.initial_perturbation = settings.initial_perturbation; + dual_simplex_settings.remove_perturbation = settings.remove_perturbation; + dual_simplex_settings.primal_pricing = settings.primal_pricing; simplex::lp_solution_t vertex_solution(lp.num_rows, lp.num_cols); std::vector vstatus(lp.num_cols); simplex::crossover_status_t crossover_status = simplex::crossover(lp, diff --git a/cpp/tests/dual_simplex/unit_tests/primal.cpp b/cpp/tests/dual_simplex/unit_tests/primal.cpp new file mode 100644 index 0000000000..bf9118f40e --- /dev/null +++ b/cpp/tests/dual_simplex/unit_tests/primal.cpp @@ -0,0 +1,159 @@ +/* clang-format off */ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +/* clang-format on */ + +#include +#include +#include +#include + +#include + +#include + +namespace cuopt::mathematical_optimization::simplex::test { + +class primal_simplex : public ::testing::Test { + protected: + void SetUp() override + { + // min -3*x - 2*y, x + y <= 4, 2*x + y <= 5, 0 <= x,y <= 10. + // The last two columns are slacks; every solve starts from a fresh slack basis. + lp.A.col_start = {0, 2, 4, 5, 6}; + lp.A.i = {0, 1, 0, 1, 0, 1}; + lp.A.x = {1.0, 2.0, 1.0, 1.0, 1.0, 1.0}; + lp.objective = {-3.0, -2.0, 0.0, 0.0}; + lp.rhs = {4.0, 5.0}; + lp.lower = {0.0, 0.0, 0.0, 0.0}; + lp.upper = {10.0, 10.0, inf, inf}; + lp.obj_scale = 1.0; + settings.iteration_limit = 100; + } + + void expect_optimum(const std::vector& expected_x, double expected_objective) + { + ASSERT_EQ(solution.x.size(), expected_x.size()); + std::vector residual = lp.rhs; + double objective = 0.0; + for (int j = 0; j < lp.num_cols; ++j) { + EXPECT_NEAR(solution.x[j], expected_x[j], 1e-6) << "column " << j; + EXPECT_GE(solution.x[j], lp.lower[j] - 1e-6) << "column " << j; + EXPECT_LE(solution.x[j], lp.upper[j] + 1e-6) << "column " << j; + objective += lp.objective[j] * solution.x[j]; + for (int k = lp.A.col_start[j]; k < lp.A.col_start[j + 1]; ++k) { + residual[lp.A.i[k]] -= lp.A.x[k] * solution.x[j]; + } + } + EXPECT_NEAR(objective, expected_objective, 1e-6); + EXPECT_NEAR(solution.objective, expected_objective, 1e-6); + EXPECT_NEAR(solution.user_objective, expected_objective, 1e-6); + for (int i = 0; i < lp.num_rows; ++i) { + EXPECT_NEAR(residual[i], 0.0, 1e-6) << "row " << i; + } + } + + cuopt::init_logger_t log{"", true}; + raft::handle_t handle{}; + lp_problem_t lp{&handle, 2, 4, 6}; + simplex_solver_settings_t settings; + lp_solution_t solution{2, 4}; + std::vector vstatus{variable_status_t::NONBASIC_LOWER, + variable_status_t::NONBASIC_LOWER, + variable_status_t::BASIC, + variable_status_t::BASIC}; + int iter = 0; +}; + +TEST_F(primal_simplex, bounded_phase2) +{ + ASSERT_EQ(primal_phase2(2, tic(), lp, settings, vstatus, solution, iter), + primal_status_t::OPTIMAL); + EXPECT_GT(iter, 0); + expect_optimum({1.0, 3.0, 0.0, 0.0}, -9.0); + + // Exercise conversion, presolve, scaling and postsolve through the user entry point too. + user_problem_t user_problem(&handle); + user_problem.num_rows = 2; + user_problem.num_cols = 2; + user_problem.A.resize(2, 2, 4); + user_problem.A.col_start = {0, 2, 4}; + user_problem.A.i = {0, 1, 0, 1}; + user_problem.A.x = {1.0, 2.0, 1.0, 1.0}; + user_problem.objective = {-3.0, -2.0}; + user_problem.rhs = {4.0, 5.0}; + user_problem.row_sense = {'L', 'L'}; + user_problem.lower = {0.0, 0.0}; + user_problem.upper = {10.0, 10.0}; + user_problem.num_range_rows = 0; + user_problem.problem_name = "primal_bounded_phase2"; + user_problem.row_names = {"capacity", "resource"}; + user_problem.col_names = {"x", "y"}; + user_problem.var_types = {variable_type_t::CONTINUOUS, variable_type_t::CONTINUOUS}; + settings.dualize = 0; + lp_solution_t user_solution(2, 2); + ASSERT_EQ(solve_linear_program_with_primal(user_problem, settings, tic(), user_solution), + lp_status_t::OPTIMAL); + ASSERT_EQ(user_solution.x.size(), 2); + EXPECT_NEAR(user_solution.x[0], 1.0, 1e-6); + EXPECT_NEAR(user_solution.x[1], 3.0, 1e-6); + EXPECT_NEAR(user_solution.objective, -9.0, 1e-6); + EXPECT_NEAR(user_solution.user_objective, -9.0, 1e-6); + EXPECT_NEAR(-3.0 * user_solution.x[0] - 2.0 * user_solution.x[1], -9.0, 1e-6); + EXPECT_NEAR(user_solution.x[0] + user_solution.x[1] - 4.0, 0.0, 1e-6); + EXPECT_NEAR(2.0 * user_solution.x[0] + user_solution.x[1] - 5.0, 0.0, 1e-6); +} + +TEST_F(primal_simplex, phase1_recovery) +{ + // x + y >= 2 gives an infeasible initial slack s = -2. + // Phase I must recover feasibility, then Phase II minimizes the original objective. + lp.A.x[0] = -1.0; + lp.A.x[2] = -1.0; + lp.rhs[0] = -2.0; + ASSERT_EQ(primal_phase2(2, tic(), lp, settings, vstatus, solution, iter), + primal_status_t::OPTIMAL); + EXPECT_GT(iter, 0); + expect_optimum({0.0, 5.0, 3.0, 0.0}, -10.0); +} + +TEST_F(primal_simplex, infeasible) +{ + // x + y >= 6 contradicts 2*x + y <= 5 for nonnegative x,y. + lp.A.x[0] = -1.0; + lp.A.x[2] = -1.0; + lp.rhs[0] = -6.0; + ASSERT_EQ(primal_phase2(2, tic(), lp, settings, vstatus, solution, iter), + primal_status_t::PRIMAL_INFEASIBLE); +} + +TEST_F(primal_simplex, unbounded) +{ + // -x - y <= 4 and -2*x - y <= 5 permit x to increase without limit. + lp.A.x = {-1.0, -2.0, -1.0, -1.0, 1.0, 1.0}; + lp.upper = {inf, inf, inf, inf}; + ASSERT_EQ(primal_phase2(2, tic(), lp, settings, vstatus, solution, iter), + primal_status_t::PRIMAL_UNBOUNDED); +} + +TEST_F(primal_simplex, termination_limits) +{ + // Each limit gets its own fresh solution and slack basis, with no sleeps needed. + for (bool expired_time : {false, true}) { + SCOPED_TRACE(expired_time ? "expired time limit" : "zero iteration limit"); + lp_solution_t limited_solution(lp.num_rows, lp.num_cols); + auto limited_vstatus = vstatus; + int limited_iter = 0; + settings.iteration_limit = expired_time ? 100 : 0; + settings.time_limit = expired_time ? 1.0 : inf; + const double start_time = tic() - (expired_time ? 2.0 : 0.0); + EXPECT_EQ( + primal_phase2(2, start_time, lp, settings, limited_vstatus, limited_solution, limited_iter), + expired_time ? primal_status_t::TIME_LIMIT : primal_status_t::ITERATION_LIMIT); + EXPECT_EQ(limited_iter, 0); + } +} + +} // namespace cuopt::mathematical_optimization::simplex::test diff --git a/cpp/tests/internal/CMakeLists.txt b/cpp/tests/internal/CMakeLists.txt index 9beef78df3..8c1f05f9c0 100644 --- a/cpp/tests/internal/CMakeLists.txt +++ b/cpp/tests/internal/CMakeLists.txt @@ -9,6 +9,7 @@ ConfigureTest(NUMOPT_INTERNAL_TEST ${CUOPT_TEST_DIR}/internal/main.cu # dual_simplex ${CUOPT_TEST_DIR}/dual_simplex/unit_tests/solve.cpp + ${CUOPT_TEST_DIR}/dual_simplex/unit_tests/primal.cpp ${CUOPT_TEST_DIR}/dual_simplex/unit_tests/solve_barrier.cu ${CUOPT_TEST_DIR}/dual_simplex/unit_tests/right_looking_ldlt.cpp # linear_programming From c65b3c35eb276c82cdb447cd2a64640450db9984 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Mon, 21 Sep 2026 15:29:12 -0700 Subject: [PATCH 101/113] Recompute dual variables and check dual feasibility after basis repair After basis repair, we recompute dual variables using the working objective. We check the new dual variables for feasibility. If infeasible, we try to restore feasibility by flipping boxed nonbasic variables to their opposite bounds. If recovery fails, we return numerical failure. Callers that cannot recover from basis repairs also propagate numerical failure. We report repaired columns separately from remaining factorization deficiencies. On 34 fast to solve MIPLIB problems, basis repair occurred on 13 problems. In a subsequent run of those 13 problems on this branch, 12,367 basis repairs occurred. Of the 12,327 evaluated by dual-feasibility recovery, 1,972 produced dual infeasibility, 1,509 were recovered by flipping bounds, and 463 remained infeasible. --- cpp/src/branch_and_bound/branch_and_bound.cpp | 14 +- cpp/src/cuts/cuts.cpp | 13 +- cpp/src/cuts/cuts.hpp | 1 + cpp/src/dual_simplex/basis_updates.cpp | 5 +- cpp/src/dual_simplex/basis_updates.hpp | 5 +- cpp/src/dual_simplex/phase2.cpp | 124 ++++++++++++++++-- cpp/src/linear_algebra/vector_math.hpp | 9 ++ 7 files changed, 153 insertions(+), 18 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index a320cc0602..de5cabc2c7 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3259,16 +3259,18 @@ lp_status_t branch_and_bound_t::solve_root_relaxation( assert(nonbasic_list.size() == original_lp_.num_cols - original_lp_.num_rows); } // Populate the basis_update from the crossover vstatus - i_t refactor_status = basis_update.refactor_basis(original_lp_.A, + i_t deficient_repaired = 0; + i_t refactor_status = basis_update.refactor_basis(original_lp_.A, root_crossover_settings, original_lp_.lower, original_lp_.upper, exploration_stats_.start_time, basic_list, nonbasic_list, - crossover_vstatus_); - if (refactor_status != 0) { - settings_.log.printf("Failed to refactor basis. %d deficient columns.\n", refactor_status); + crossover_vstatus_, + deficient_repaired); + if (refactor_status != 0 || deficient_repaired > 0) { + settings_.log.printf("Failed to refactor basis. %d deficient columns.\n", deficient_repaired); assert(refactor_status == 0); root_status = lp_status_t::NUMERICAL_ISSUES; } @@ -3578,6 +3580,10 @@ auto branch_and_bound_t::do_cut_pass( set_final_solution(solution, root_objective_); return cut_pass_action_t::RETURN; } + if (remove_cuts_status != 0) { + solver_status_ = mip_status_t::NUMERICAL; + return cut_pass_action_t::RETURN; + } f_t remove_cuts_time = toc(remove_cuts_start_time); if (remove_cuts_time > 1.0) { diff --git a/cpp/src/cuts/cuts.cpp b/cpp/src/cuts/cuts.cpp index e9f51666dc..ce01ed92c1 100644 --- a/cpp/src/cuts/cuts.cpp +++ b/cpp/src/cuts/cuts.cpp @@ -6559,10 +6559,19 @@ i_t remove_cuts(lp_problem_t& lp, lp.A.col_start[lp.A.n]); basis_update.resize(lp.num_rows); - i_t refactor_status = basis_update.refactor_basis( - lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); + i_t deficient_repaired = 0; + i_t refactor_status = basis_update.refactor_basis(lp.A, + settings, + lp.lower, + lp.upper, + start_time, + basic_list, + nonbasic_list, + vstatus, + deficient_repaired); if (refactor_status == CONCURRENT_HALT_RETURN) { return CONCURRENT_HALT_RETURN; } if (refactor_status == TIME_LIMIT_RETURN) { return TIME_LIMIT_RETURN; } + if (refactor_status != 0 || deficient_repaired > 0) { return -1; } } return 0; diff --git a/cpp/src/cuts/cuts.hpp b/cpp/src/cuts/cuts.hpp index ca87e26c39..12e8fb1ce9 100644 --- a/cpp/src/cuts/cuts.hpp +++ b/cpp/src/cuts/cuts.hpp @@ -1263,6 +1263,7 @@ i_t add_cuts(const simplex::simplex_solver_settings_t& settings, std::vector& vstatus, std::vector& edge_norms); +// Returns -1 on numerical failure, or the halt/time-limit return code. template i_t remove_cuts(simplex::lp_problem_t& lp, const simplex::simplex_solver_settings_t& settings, diff --git a/cpp/src/dual_simplex/basis_updates.cpp b/cpp/src/dual_simplex/basis_updates.cpp index f81962d054..6efde71413 100644 --- a/cpp/src/dual_simplex/basis_updates.cpp +++ b/cpp/src/dual_simplex/basis_updates.cpp @@ -2346,8 +2346,10 @@ int basis_update_mpf_t::refactor_basis( f_t start_time, std::vector& basic_list, std::vector& nonbasic_list, - std::vector& vstatus) + std::vector& vstatus, + i_t& deficient_repaired) { + deficient_repaired = 0; raft::common::nvtx::range scope("LU::refactor_basis"); std::vector deficient; std::vector slacks_needed; @@ -2374,6 +2376,7 @@ int basis_update_mpf_t::refactor_basis( if (status == TIME_LIMIT_RETURN) { return TIME_LIMIT_RETURN; } if (status == -1) { settings.log.debug("Initial factorization failed\n"); + deficient_repaired = static_cast(deficient.size()); basis_repair(A, settings, lower, diff --git a/cpp/src/dual_simplex/basis_updates.hpp b/cpp/src/dual_simplex/basis_updates.hpp index d1c623db55..f5e35b978a 100644 --- a/cpp/src/dual_simplex/basis_updates.hpp +++ b/cpp/src/dual_simplex/basis_updates.hpp @@ -377,7 +377,7 @@ class basis_update_mpf_t { void multiply_lu(csc_matrix_t& out) const; - // Compute L*U = A(p, basic_list) + // Compute L*U = A(p, basic_list). Report the number of deficient columns repaired. int refactor_basis(const csc_matrix_t& A, const simplex_solver_settings_t& settings, const std::vector& lower, @@ -385,7 +385,8 @@ class basis_update_mpf_t { f_t start_time, std::vector& basic_list, std::vector& nonbasic_list, - std::vector& vstatus); + std::vector& vstatus, + i_t& deficient_repaired); void set_refactor_frequency(i_t new_frequency) { refactor_frequency_ = new_frequency; } diff --git a/cpp/src/dual_simplex/phase2.cpp b/cpp/src/dual_simplex/phase2.cpp index 84656953df..468361d4cb 100644 --- a/cpp/src/dual_simplex/phase2.cpp +++ b/cpp/src/dual_simplex/phase2.cpp @@ -1957,6 +1957,48 @@ f_t dual_infeasibility(const lp_problem_t& lp, return sum_infeasible; } +template +bool recover_dual_feasibility_after_repair(const lp_problem_t& lp, + const simplex_solver_settings_t& settings, + basis_update_mpf_t& ft, + const std::vector& basic_list, + const std::vector& nonbasic_list, + std::vector& vstatus, + const std::vector& objective, + std::vector& cB, + std::vector& y, + std::vector& z, + f_t& work_estimate) +{ + for (i_t k = 0; k < lp.num_rows; ++k) { + cB[k] = objective[basic_list[k]]; + } + work_estimate += 3 * lp.num_rows; + ft.b_transpose_solve(cB, y); + compute_reduced_costs(objective, lp.A, y, basic_list, nonbasic_list, z, work_estimate); + work_estimate += lp.num_rows + lp.num_cols; + if (!all_finite(y) || !all_finite(z)) { return false; } + const f_t before = + dual_infeasibility(lp, settings, vstatus, z, settings.tight_tol, settings.dual_tol); + work_estimate += 3 * lp.num_cols; + if (before <= settings.dual_tol) { return true; } + for (i_t j : nonbasic_list) { + if (!std::isfinite(lp.lower[j]) || !std::isfinite(lp.upper[j]) || lp.lower[j] == lp.upper[j]) { + continue; + } + if (vstatus[j] == variable_status_t::NONBASIC_LOWER && z[j] < -settings.dual_tol) { + vstatus[j] = variable_status_t::NONBASIC_UPPER; + } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER && z[j] > settings.dual_tol) { + vstatus[j] = variable_status_t::NONBASIC_LOWER; + } + } + const f_t after = + dual_infeasibility(lp, settings, vstatus, z, settings.tight_tol, settings.dual_tol); + work_estimate += 8 * lp.num_cols; + // The caller must rebuild x and its infeasibilities after these bound changes. + return after <= settings.dual_tol; +} + template f_t primal_infeasibility_breakdown(const lp_problem_t& lp, const simplex_solver_settings_t& settings, @@ -2599,9 +2641,17 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, assert(nonbasic_list.size() == n - m); f_t refactor_start_work = ft.work_estimate(); - i_t refactor_status = ft.refactor_basis( - lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); - refactor_work = ft.work_estimate() - refactor_start_work; + i_t deficient_repaired = 0; + i_t refactor_status = ft.refactor_basis(lp.A, + settings, + lp.lower, + lp.upper, + start_time, + basic_list, + nonbasic_list, + vstatus, + deficient_repaired); + refactor_work = ft.work_estimate() - refactor_start_work; if (refactor_status == CONCURRENT_HALT_RETURN) { return dual_status_t::CONCURRENT_LIMIT; } if (refactor_status == TIME_LIMIT_RETURN) { return dual_status_t::TIME_LIMIT; } if (refactor_status > 0) { return dual_status_t::NUMERICAL; } @@ -2934,8 +2984,16 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, iter, ft.num_updates()); f_t refactor_start_work = ft.work_estimate(); - i_t refactor_status = ft.refactor_basis( - lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); + i_t deficient_repaired = 0; + i_t refactor_status = ft.refactor_basis(lp.A, + settings, + lp.lower, + lp.upper, + start_time, + basic_list, + nonbasic_list, + vstatus, + deficient_repaired); if (refactor_status == CONCURRENT_HALT_RETURN) { return dual_status_t::CONCURRENT_LIMIT; } if (refactor_status == TIME_LIMIT_RETURN) { return dual_status_t::TIME_LIMIT; } if (refactor_status > 0) { return dual_status_t::NUMERICAL; } @@ -2945,6 +3003,20 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, basic_list, nonbasic_list, basic_mark, nonbasic_mark, phase2_work_estimate); compute_initial_nonbasic_end(basic_mark, Arow, nonbasic_end); + if (deficient_repaired > 0 && + !phase2::recover_dual_feasibility_after_repair(lp, + settings, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + c_basic, + y, + z, + phase2_work_estimate)) { + return dual_status_t::NUMERICAL; + } phase2::compute_primal_solution_from_basis( lp, ft, basic_list, nonbasic_list, vstatus, x, xB_workspace, phase2_work_estimate); @@ -2961,6 +3033,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, solve_work = 0.0; if (primal_infeasibility > settings.primal_tol) { + obj = phase2::compute_perturbed_objective(objective, x); + phase2_work_estimate += 2 * n; settings.log.printf( "New infeasibilities found after recompute (primal_inf=%.2e). " "Continuing phase 2.\n", @@ -3591,8 +3665,17 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, num_refactors++; bool should_recompute_x = true; // Needed for numerically difficult problems like cbs-cta f_t refactor_start_work = ft.work_estimate(); - i_t refactor_status = ft.refactor_basis( - lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); + i_t deficient_repaired = 0; + i_t refactor_status = ft.refactor_basis(lp.A, + settings, + lp.lower, + lp.upper, + start_time, + basic_list, + nonbasic_list, + vstatus, + deficient_repaired); + const bool did_basis_repair = deficient_repaired > 0; if (refactor_status == CONCURRENT_HALT_RETURN) { return dual_status_t::CONCURRENT_LIMIT; } if (refactor_status == TIME_LIMIT_RETURN) { return dual_status_t::TIME_LIMIT; } if (refactor_status > 0) { @@ -3602,8 +3685,15 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, i_t count = 0; i_t deficient_size = 0; while (true) { - deficient_size = ft.refactor_basis( - lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); + deficient_size = ft.refactor_basis(lp.A, + settings, + lp.lower, + lp.upper, + start_time, + basic_list, + nonbasic_list, + vstatus, + deficient_repaired); if (deficient_size == CONCURRENT_HALT_RETURN) { return dual_status_t::CONCURRENT_LIMIT; } @@ -3628,6 +3718,20 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2::reset_basis_mark( basic_list, nonbasic_list, basic_mark, nonbasic_mark, phase2_work_estimate); compute_initial_nonbasic_end(basic_mark, Arow, nonbasic_end); + if (did_basis_repair && + !phase2::recover_dual_feasibility_after_repair(lp, + settings, + ft, + basic_list, + nonbasic_list, + vstatus, + objective, + c_basic, + y, + z, + phase2_work_estimate)) { + return dual_status_t::NUMERICAL; + } if (should_recompute_x) { std::vector unperturbed_x(n); phase2_work_estimate += n; @@ -3641,6 +3745,8 @@ dual_status_t dual_phase2_with_advanced_basis(i_t phase, phase2_work_estimate); x = unperturbed_x; phase2_work_estimate += 2 * n; + obj = phase2::compute_perturbed_objective(objective, x); + phase2_work_estimate += 2 * n; } primal_infeasibility_squared = phase2::compute_initial_primal_infeasibilities(lp, diff --git a/cpp/src/linear_algebra/vector_math.hpp b/cpp/src/linear_algebra/vector_math.hpp index 8063bc735b..6c42416435 100644 --- a/cpp/src/linear_algebra/vector_math.hpp +++ b/cpp/src/linear_algebra/vector_math.hpp @@ -13,6 +13,15 @@ namespace cuopt::mathematical_optimization { +template +bool all_finite(const std::vector& in) +{ + for (f_t value : in) { + if (!std::isfinite(value)) { return false; } + } + return true; +} + // Computes || x ||_inf = max_j | x |_j template f_t vector_norm_inf(const std::vector& x) From afa9ab6d93b460304bac7286a656041a358a32cf Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Mon, 21 Sep 2026 16:06:24 -0700 Subject: [PATCH 102/113] Update primal refactor calls for basis repair reporting --- cpp/src/dual_simplex/primal.cpp | 24 ++++++++++++++++++++---- 1 file changed, 20 insertions(+), 4 deletions(-) diff --git a/cpp/src/dual_simplex/primal.cpp b/cpp/src/dual_simplex/primal.cpp index 9f1f5d591f..66fff57374 100644 --- a/cpp/src/dual_simplex/primal.cpp +++ b/cpp/src/dual_simplex/primal.cpp @@ -982,8 +982,16 @@ primal_status_t primal_phase2_with_advanced_basis( // nonbasics exactly on their status bounds, rebuild x_B so Ax = b, and // refresh duals. If that point is not primal/dual feasible, continue. if (basis_update.num_updates() > 0) { - i_t rank = basis_update.refactor_basis( - lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); + i_t deficient_repaired = 0; + i_t rank = basis_update.refactor_basis(lp.A, + settings, + lp.lower, + lp.upper, + start_time, + basic_list, + nonbasic_list, + vstatus, + deficient_repaired); if (rank == CONCURRENT_HALT_RETURN) { return primal_status_t::CONCURRENT_LIMIT; } if (rank == TIME_LIMIT_RETURN) { return primal_status_t::TIME_LIMIT; } if (rank != 0) { @@ -1329,8 +1337,16 @@ primal_status_t primal_phase2_with_advanced_basis( } if (should_refactor) { timers.start_timer(work_estimate + basis_update.work_estimate()); - i_t rank = basis_update.refactor_basis( - lp.A, settings, lp.lower, lp.upper, start_time, basic_list, nonbasic_list, vstatus); + i_t deficient_repaired = 0; + i_t rank = basis_update.refactor_basis(lp.A, + settings, + lp.lower, + lp.upper, + start_time, + basic_list, + nonbasic_list, + vstatus, + deficient_repaired); if (rank == CONCURRENT_HALT_RETURN) { return primal_status_t::CONCURRENT_LIMIT; } if (rank == TIME_LIMIT_RETURN) { return primal_status_t::TIME_LIMIT; } if (rank != 0) { From 5c4c65c2adbd70bd97d840157b34e85fb8336246 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 22 Sep 2026 10:42:17 -0700 Subject: [PATCH 103/113] Format simplex and basis repair changes --- cpp/src/branch_and_bound/branch_and_bound.cpp | 3 ++- cpp/src/dual_simplex/basis_updates.cpp | 2 +- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index de5cabc2c7..ea3e25c694 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3270,7 +3270,8 @@ lp_status_t branch_and_bound_t::solve_root_relaxation( crossover_vstatus_, deficient_repaired); if (refactor_status != 0 || deficient_repaired > 0) { - settings_.log.printf("Failed to refactor basis. %d deficient columns.\n", deficient_repaired); + settings_.log.printf("Failed to refactor basis. %d deficient columns.\n", + deficient_repaired); assert(refactor_status == 0); root_status = lp_status_t::NUMERICAL_ISSUES; } diff --git a/cpp/src/dual_simplex/basis_updates.cpp b/cpp/src/dual_simplex/basis_updates.cpp index 5deaa3f499..2fdda268db 100644 --- a/cpp/src/dual_simplex/basis_updates.cpp +++ b/cpp/src/dual_simplex/basis_updates.cpp @@ -2026,7 +2026,7 @@ template void basis_update_mpf_t::u_multiply(const sparse_vector_t& x, sparse_vector_t& y) const { - const i_t m = L0_.m; + const i_t m = L0_.m; i_t nz = 0; const i_t x_nz = x.i.size(); // The first half of xi_workspace_ holds marks, the second the touched rows. From 08e5df6c1cbc0e4c0248f0ca63776aaec2188562 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 22 Sep 2026 10:48:15 -0700 Subject: [PATCH 104/113] Format basis repair logging --- cpp/src/branch_and_bound/branch_and_bound.cpp | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index de5cabc2c7..ea3e25c694 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3270,7 +3270,8 @@ lp_status_t branch_and_bound_t::solve_root_relaxation( crossover_vstatus_, deficient_repaired); if (refactor_status != 0 || deficient_repaired > 0) { - settings_.log.printf("Failed to refactor basis. %d deficient columns.\n", deficient_repaired); + settings_.log.printf("Failed to refactor basis. %d deficient columns.\n", + deficient_repaired); assert(refactor_status == 0); root_status = lp_status_t::NUMERICAL_ISSUES; } From 14e7e3182f13c046652972968011fd95c3a7a9d2 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 22 Sep 2026 10:51:26 -0700 Subject: [PATCH 105/113] Handle TIME_LIMIT in refactorization --- cpp/src/branch_and_bound/branch_and_bound.cpp | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index ea3e25c694..fc40888cf6 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3269,7 +3269,9 @@ lp_status_t branch_and_bound_t::solve_root_relaxation( nonbasic_list, crossover_vstatus_, deficient_repaired); - if (refactor_status != 0 || deficient_repaired > 0) { + if (refactor_status == TIME_LIMIT_RETURN) { + root_status = lp_status_t::TIME_LIMIT; + } else if (refactor_status != 0 || deficient_repaired > 0) { settings_.log.printf("Failed to refactor basis. %d deficient columns.\n", deficient_repaired); assert(refactor_status == 0); From ba428dac61bf24841d32f01517b15da1ff627140 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 22 Sep 2026 11:10:05 -0700 Subject: [PATCH 106/113] Revert "Stop recursive sub-MIPs when branch-and-bound terminates" This reverts commit bc99b1edef5195d20fc0a3f694c7eeb8f59a159d. --- cpp/src/branch_and_bound/branch_and_bound.cpp | 53 ++----------------- cpp/src/branch_and_bound/branch_and_bound.hpp | 15 ------ 2 files changed, 4 insertions(+), 64 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 990c0b7acf..d0b1c5187c 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -891,7 +891,6 @@ template void branch_and_bound_t::set_solution_at_root(mip_solution_t& solution, const cut_info_t& cut_info) { - request_stop(); mutex_upper_.lock(); incumbent_.set_incumbent_solution(root_objective_, root_relax_soln_.x); upper_bound_ = root_objective_; @@ -921,7 +920,6 @@ template void branch_and_bound_t::set_final_solution(mip_solution_t& solution, f_t lower_bound) { - request_stop(); if (solver_status_ == mip_status_t::HALT) { settings_.log.debug("Stopping the solver...\n"); } if (solver_status_ == mip_status_t::NUMERICAL) { @@ -2070,40 +2068,6 @@ void branch_and_bound_t::work_stealing(bfs_worker_t* worker) } } -template -void branch_and_bound_t::request_stop() -{ - // Protect child lifetimes and acquire locks only from parent to child. - std::lock_guard lock(children_mutex_); - node_concurrent_halt_.store(1, std::memory_order_release); - for (branch_and_bound_t* child : active_children_) { - child->request_stop(); - } -} - -template -branch_and_bound_t::submip_registration_t::submip_registration_t( - branch_and_bound_t& parent, branch_and_bound_t& child) - : parent(parent), child(child) -{ - std::lock_guard lock(parent.children_mutex_); - parent.active_children_.push_back(&child); - if (parent.node_concurrent_halt_.load(std::memory_order_acquire) || - parent.received_halt_signal()) { - child.request_stop(); - } -} - -template -branch_and_bound_t::submip_registration_t::~submip_registration_t() -{ - std::lock_guard lock(parent.children_mutex_); - const typename std::vector::iterator position = - std::find(parent.active_children_.begin(), parent.active_children_.end(), &child); - assert(position != parent.active_children_.end()); - parent.active_children_.erase(position); -} - template void branch_and_bound_t::best_first_search_with(bfs_worker_t* worker) { @@ -2219,7 +2183,9 @@ void branch_and_bound_t::best_first_search_with(bfs_worker_t } } - if (solver_status_ != mip_status_t::UNSET) { request_stop(); } + if (solver_status_ == mip_status_t::TIME_LIMIT || solver_status_ == mip_status_t::OPTIMAL) { + node_concurrent_halt_ = 1; + } // If the worker has still nodes in the queue (this can happen if it was stopped due to // time limit, small gap or other reason), then do not add back to the pool to avoid @@ -2233,7 +2199,6 @@ void branch_and_bound_t::best_first_search_with(bfs_worker_t if (exploration_stats_.nodes_unexplored == 0 && bfs_worker_pool_.num_idle() == bfs_worker_pool_.size()) { is_running_ = false; - request_stop(); } } @@ -2407,10 +2372,6 @@ bool branch_and_bound_t::launch_diving_worker(bfs_worker_t* template bool branch_and_bound_t::launch_submip_worker(const std::vector& sol) { - if (solver_status_ != mip_status_t::UNSET || node_concurrent_halt_.load() || - received_halt_signal()) { - return false; - } if (settings_.submip_settings.rins == 0 && settings_.submip_settings.rens == 0) return false; if (settings_.submip_settings.rens == 0 && !incumbent_.has_incumbent) return false; if (submip_worker_pool_.num_idle() == 0) return false; @@ -2567,7 +2528,6 @@ void branch_and_bound_t::solve_submip(diving_worker_t* worke probing_implied_bound_t empty_probing(submip_problem.num_cols); branch_and_bound_t submip_bnb(submip_problem, submip_settings, tic(), empty_probing); - submip_registration_t registration(*this, submip_bnb); mip_solution_t submip_solution(submip_problem.num_cols); std::vector presolved_incumbent; @@ -5353,10 +5313,6 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut root_lp_current_lower_bound_ = -inf; exploration_stats_.nodes_unexplored = 0; exploration_stats_.nodes_explored = 0; - if (node_concurrent_halt_.load() || received_halt_signal()) { - solver_status_ = mip_status_t::HALT; - return solver_status_; - } original_lp_.A.to_compressed_row(Arow_); settings_.log.debug("Reduced cost strengthening enabled: %d\n", @@ -5726,7 +5682,6 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut mutex_upper_.unlock(); if (cut_pass_action == cut_pass_action_t::RETURN) { - request_stop(); if (settings_.benchmark_info_ptr != nullptr) { settings_.benchmark_info_ptr->cut_generation_time_sec = toc(cut_generation_start_time); } @@ -5882,7 +5837,7 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut } settings_.log.printf("Exploring the B&B tree using %d threads\n\n", settings_.num_threads); - // Do not clear a stop request delivered by an ancestor before tree startup. + node_concurrent_halt_ = 0; exploration_stats_.nodes_explored = 0; exploration_stats_.nodes_unexplored = 2; diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 4a5badf3a4..8330963c4d 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -44,7 +44,6 @@ #include #include #include -#include #include namespace cuopt::mathematical_optimization::mip { @@ -399,20 +398,6 @@ class branch_and_bound_t { bool enable_concurrent_lp_root_solve_{false}; std::atomic root_concurrent_halt_{0}; std::atomic node_concurrent_halt_{0}; - std::mutex children_mutex_; - std::vector active_children_; - - void request_stop(); - - // Declared after the child solver, so unregistration precedes child destruction. - struct submip_registration_t { - branch_and_bound_t& parent; - branch_and_bound_t& child; - submip_registration_t(branch_and_bound_t& parent, branch_and_bound_t& child); - ~submip_registration_t(); - submip_registration_t(const submip_registration_t&) = delete; - submip_registration_t& operator=(const submip_registration_t&) = delete; - }; bool is_root_solution_set{false}; bool has_initial_pseudocost_{false}; From f7c1e5fd658a75745b6c5cbfa981b7cefe3a60d7 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 22 Sep 2026 11:20:54 -0700 Subject: [PATCH 107/113] Update degenerate pump refactor call for repair reporting --- cpp/src/branch_and_bound/branch_and_bound.cpp | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index b2f3b52221..c0700bb77e 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -4233,6 +4233,7 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( nonbasic_list.clear(); simplex::get_basis_from_vstatus(m, vstatus, basic_list, nonbasic_list, superbasic_list); assert(superbasic_list.empty()); + i_t deficient_repaired = 0; const i_t refactor_status = basis_update.refactor_basis(lp.A, settings_, lp.lower, @@ -4240,7 +4241,8 @@ void branch_and_bound_t::dual_degenerate_feasibility_pump( exploration_stats_.start_time, basic_list, nonbasic_list, - vstatus); + vstatus, + deficient_repaired); if (refactor_status == CONCURRENT_HALT_RETURN || refactor_status == TIME_LIMIT_RETURN) { return; } From 9101ece5f7985e4bb5b482a9bf11ab634b513ebf Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 22 Sep 2026 17:21:25 -0700 Subject: [PATCH 108/113] Extract degenerate pivot and reduced-cost bound utilities --- cpp/src/branch_and_bound/CMakeLists.txt | 1 + cpp/src/branch_and_bound/branch_and_bound.cpp | 1264 +--------------- cpp/src/branch_and_bound/branch_and_bound.hpp | 192 +-- .../branch_and_bound/degenerate_pivots.cpp | 1302 +++++++++++++++++ .../branch_and_bound/degenerate_pivots.hpp | 92 ++ cpp/src/branch_and_bound/fractional.hpp | 45 + .../branch_and_bound/reduced_cost_bounds.hpp | 160 ++ 7 files changed, 1642 insertions(+), 1414 deletions(-) create mode 100644 cpp/src/branch_and_bound/degenerate_pivots.cpp create mode 100644 cpp/src/branch_and_bound/degenerate_pivots.hpp create mode 100644 cpp/src/branch_and_bound/fractional.hpp create mode 100644 cpp/src/branch_and_bound/reduced_cost_bounds.hpp diff --git a/cpp/src/branch_and_bound/CMakeLists.txt b/cpp/src/branch_and_bound/CMakeLists.txt index c9f47c4d55..cf5b3eb68f 100644 --- a/cpp/src/branch_and_bound/CMakeLists.txt +++ b/cpp/src/branch_and_bound/CMakeLists.txt @@ -5,6 +5,7 @@ set(BRANCH_AND_BOUND_SRC_FILES ${CMAKE_CURRENT_SOURCE_DIR}/branch_and_bound.cpp + ${CMAKE_CURRENT_SOURCE_DIR}/degenerate_pivots.cpp ${CMAKE_CURRENT_SOURCE_DIR}/pseudo_costs.cpp ${CMAKE_CURRENT_SOURCE_DIR}/diving_heuristics.cpp ) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index c0700bb77e..720c754c18 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -6,7 +6,9 @@ /* clang-format on */ #include +#include #include +#include #include #include #include @@ -80,32 +82,6 @@ using simplex::variable_type_t; namespace { -template -bool is_fractional(f_t x, variable_type_t var_type, f_t integer_tol) -{ - if (var_type == variable_type_t::CONTINUOUS) { - return false; - } else { - f_t x_integer = std::round(x); - return (std::abs(x_integer - x) > integer_tol); - } -} - -template -i_t fractional_variables(const simplex_solver_settings_t& settings, - const std::vector& x, - const std::vector& var_types, - std::vector& fractional) -{ - const i_t n = x.size(); - assert(x.size() == var_types.size()); - fractional.clear(); - for (i_t j = 0; j < n; ++j) { - if (is_fractional(x[j], var_types[j], settings.integer_tol)) { fractional.push_back(j); } - } - return fractional.size(); -} - template void full_variable_types(const user_problem_t& original_problem, const lp_problem_t& original_lp, @@ -1797,16 +1773,25 @@ dual_status_t branch_and_bound_t::solve_node_lp( i_t num_fractional = fractional_variables(settings_, worker->leaf_solution.x, var_types_, fractional); if (settings_.dual_degenerate_pivots != 0) { - pivot_out_integer_variables(worker->leaf_problem, - lp_settings, - worker->new_slacks, - worker->basic_list, - worker->nonbasic_list, - worker->leaf_vstatus, - worker->leaf_solution, - worker->basis_factors, - num_fractional, - fractional); + auto pivot_settings = settings_; + pivot_settings.log = lp_settings.log; + pivot_settings.inside_mip = lp_settings.inside_mip; + pivot_settings.inside_submip = lp_settings.inside_submip; + i_t num_integer_increased = pivot_out_integer_variables(worker->leaf_problem, + pivot_settings, + worker->new_slacks, + var_types_, + exploration_stats_.start_time, + worker->basic_list, + worker->nonbasic_list, + worker->leaf_vstatus, + worker->leaf_solution, + worker->basis_factors, + num_fractional, + fractional); + if (num_integer_increased > 0) { + integer_pivots_.fetch_add(num_integer_increased, std::memory_order_release); + } } } } @@ -3509,6 +3494,8 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t::cut_pass_action_t branch_and_bound_t 0) { + integer_pivots_.fetch_add(num_integer_increased, std::memory_order_release); + } } settings_.log.printf("Pivoted out %d integer variables in %e seconds\n", num_integer_increased, toc(pivot_out_integer_variables_start_time)); if (settings_.dual_degenerate_feasibility_pump != 0) { dual_degenerate_feasibility_pump(original_lp_, + settings_, + var_types_, + edge_norms_, + root_relax_work_estimate_, + exploration_stats_.start_time, basic_list, nonbasic_list, root_vstatus_, @@ -3848,1193 +3843,6 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t -bool branch_and_bound_t::check_for_dual_degeneracy( - const simplex::lp_solution_t& solution, - const std::vector& nonbasic_list, - std::vector& zero_reduced_costs_vars, - std::vector& zero_reduced_costs_vars_nonbasic_index) -{ - const i_t num_nonbasics = nonbasic_list.size(); - for (i_t k = 0; k < num_nonbasics; k++) { - const i_t j = nonbasic_list[k]; - if (std::abs(solution.z[j]) <= settings_.tight_tol) { - zero_reduced_costs_vars.push_back(j); - zero_reduced_costs_vars_nonbasic_index.push_back(k); - } - } - return !zero_reduced_costs_vars.empty(); -} - -template -void branch_and_bound_t::dual_degenerate_feasibility_pump( - const simplex::lp_problem_t& lp, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& vstatus, - simplex::lp_solution_t& soln, - simplex::basis_update_mpf_t& basis_update, - i_t& num_fractional, - std::vector& fractional) -{ - f_t dual_degenerate_feasibility_pump_start_time = tic(); - std::vector zero_reduced_costs_vars; - std::vector zero_reduced_costs_vars_nonbasic_index; - bool dual_degenerate = check_for_dual_degeneracy( - soln, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); - if (!dual_degenerate) { return; } - - // Construct a new LP problem - // minimize p^T x - // subject to B x_B + N_z x_z = b - N x_N - // l_B <= x_B <= u_B - // l_z <= x_z <= u_z - // - // where B is the basic matrix, N is the nonbasic matrix, b is the right-hand side, - - const i_t m = lp.num_rows; - const i_t n = lp.num_rows + zero_reduced_costs_vars.size(); - - i_t nnz = 0; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { - nnz += lp.A.col_start[j + 1] - lp.A.col_start[j]; - } - } - simplex::lp_problem_t lp_reduced(lp.handle_ptr, m, n, nnz); - csc_matrix_t& A_reduced = lp_reduced.A; - std::vector original_col_to_reduced_col(lp.num_cols, -1); - i_t nz = 0; - i_t reduced_col = 0; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { - original_col_to_reduced_col[j] = reduced_col; - A_reduced.col_start[reduced_col] = nz; - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - for (i_t p = col_start; p < col_end; p++) { - const i_t i = lp.A.i[p]; - const f_t value = lp.A.x[p]; - A_reduced.i[nz] = i; - A_reduced.x[nz] = value; - nz++; - } - lp_reduced.lower[reduced_col] = lp.lower[j]; - lp_reduced.upper[reduced_col] = lp.upper[j]; - reduced_col++; - } - } - A_reduced.col_start[reduced_col] = nz; - - std::vector b_reduced = lp.rhs; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { - // PASS - } else { - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - for (i_t p = col_start; p < col_end; p++) { - const i_t i = lp.A.i[p]; - const f_t value = lp.A.x[p]; - b_reduced[i] -= value * soln.x[j]; - } - } - } - lp_reduced.rhs = b_reduced; - lp_reduced.obj_scale = 1.0; - - settings_.log.printf( - "Constructed dual degenerate feasibility pump LP with %d rows and %d columns\n", m, n); - - std::vector reduced_basic_list(m); - std::vector reduced_nonbasic_list(zero_reduced_costs_vars.size()); - std::vector reduced_vstatus(n); - i_t num_basic = 0; - i_t num_nonbasic = 0; - reduced_col = 0; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC) { - reduced_vstatus[reduced_col++] = variable_status_t::BASIC; - } else if (std::abs(soln.z[j]) <= 1e-10) { - reduced_nonbasic_list[num_nonbasic++] = - reduced_col; // Does ordering of nonbasic variables matter? - reduced_vstatus[reduced_col++] = vstatus[j]; - } - } - - simplex::lp_solution_t reduced_solution(m, n); - reduced_col = 0; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { - reduced_solution.x[reduced_col++] = soln.x[j]; - } - } - - std::vector reduced_edge_norms(n); - reduced_col = 0; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { - reduced_edge_norms[reduced_col++] = edge_norms_[j]; - } - } - - simplex::basis_update_mpf_t reduced_basis_update = basis_update; - reduced_basis_update.clear_work_estimate(); - for (i_t k = 0; k < m; k++) { - reduced_basic_list[k] = original_col_to_reduced_col[basic_list[k]]; - } - - f_t primal_work_estimate = 0.0; - i_t iter = 0; - i_t max_pump_iter = 10; - simplex::random_t rng(settings_.random_seed); - i_t best_num_fractional = num_fractional; - std::vector best_reduced_vstatus(n); - bool stalled = false; - for (i_t pump_iter = 0; pump_iter < max_pump_iter; pump_iter++) { - reduced_col = 0; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { - lp_reduced.objective[reduced_col] = 0; - if (var_types_[j] == variable_type_t::INTEGER) { - if (is_fractional( - reduced_solution.x[reduced_col], var_types_[j], settings_.integer_tol)) { - // Default to the exact nearest-integer rounding. Only perturb the - // rounding direction when the previous pass made no progress (a - // zero-pivot solve), to break out of the stall. - const f_t random_value = - stalled ? 0.25 * (2.0 * rng.random() - 1.0) : 0.0; // [-0.25, 0.25] - if (reduced_solution.x[reduced_col] + random_value < - std::floor(reduced_solution.x[reduced_col]) + 0.5) { - lp_reduced.objective[reduced_col] = 1; - } else { - lp_reduced.objective[reduced_col] = -1; - } - } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_LOWER) { - lp_reduced.objective[reduced_col] = 0.1; - } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_UPPER) { - lp_reduced.objective[reduced_col] = -0.1; - } - } - reduced_col++; - } - } - - // Check reduced costs before calling primal simplex. - // Compute y = B^{-T} * c_B (BTRAN with the pump objective on basic variables) - std::vector c_basic_pump(m, 0.0); - for (i_t k = 0; k < m; k++) { - c_basic_pump[k] = lp_reduced.objective[reduced_basic_list[k]]; - } - std::vector y_pump(m); - reduced_basis_update.b_transpose_solve(c_basic_pump, y_pump); - - // Check if any nonbasic has a violated reduced cost - i_t num_violated = 0; - f_t max_violation = 0.0; - const i_t num_nonbasics_reduced = reduced_nonbasic_list.size(); - for (i_t k = 0; k < num_nonbasics_reduced; k++) { - const i_t j = reduced_nonbasic_list[k]; - // z[j] = c[j] - y^T * A(:,j) - f_t zj = lp_reduced.objective[j]; - const i_t col_start = A_reduced.col_start[j]; - const i_t col_end = A_reduced.col_start[j + 1]; - for (i_t p = col_start; p < col_end; p++) { - zj -= y_pump[A_reduced.i[p]] * A_reduced.x[p]; - } - // Check pricing condition - bool violated = false; - if (reduced_vstatus[j] == variable_status_t::NONBASIC_LOWER || - reduced_vstatus[j] == variable_status_t::NONBASIC_FIXED) { - if (zj < -settings_.dual_tol) { violated = true; } - } else if (reduced_vstatus[j] == variable_status_t::NONBASIC_UPPER) { - if (zj > settings_.dual_tol) { violated = true; } - } - if (violated) { - num_violated++; - max_violation = std::max(max_violation, std::abs(zj)); - } - } - - if (num_violated == 0) { - settings_.log.printf( - "Degenerate feasibility pump (%d/%d): skipping primal simplex, no violated reduced costs " - "(%d nonbasics checked)\n", - pump_iter, - max_pump_iter, - num_nonbasics_reduced); - primal_work_estimate += reduced_basis_update.work_estimate(); - reduced_basis_update.clear_work_estimate(); - // Don't count this as a pump iteration, but break if we've skipped twice - // in a row (perturbation isn't helping) - if (stalled) { break; } - stalled = true; - pump_iter--; - continue; - } - settings_.log.printf( - "Degenerate feasibility pump (%d/%d): %d violated reduced costs (max %.2e) out of %d " - "nonbasics\n", - pump_iter, - max_pump_iter, - num_violated, - max_violation, - num_nonbasics_reduced); - - bool recompute_basis = false; - const i_t iter_before = iter; - f_t primal_work_before = primal_work_estimate; - f_t pump_call_start_time = tic(); - simplex_solver_settings_t primal_settings = settings_; - primal_settings.log.log = false; - primal_settings.time_limit = settings_.time_limit; - primal_settings.work_limit = root_relax_work_estimate_ / 10; - settings_.log.printf( - "Degenerate feasibility pump: calling primal simplex with %d rows, %d cols, %d nnz, " - "%d basis updates, work_limit %.2e, primal_work_estimate %.2e\n", - m, - n, - A_reduced.col_start[n], - reduced_basis_update.num_updates(), - primal_settings.work_limit, - primal_work_estimate); - simplex::primal_status_t lp_status = - simplex::primal_phase2_with_advanced_basis(2, - exploration_stats_.start_time, - lp_reduced, - primal_settings, - reduced_vstatus, - reduced_basis_update, - reduced_basic_list, - reduced_nonbasic_list, - reduced_solution, - iter, - primal_work_estimate); - f_t pump_call_time = toc(pump_call_start_time); - f_t pump_call_work = primal_work_estimate - primal_work_before; - i_t pump_call_iters = iter - iter_before; - settings_.log.printf( - "Degenerate feasibility pump: primal simplex returned status %d, %d iters, " - "work %.2e (%.2e/iter), time %.2f (%.2e work/s)\n", - static_cast(lp_status), - pump_call_iters, - pump_call_work, - pump_call_iters > 0 ? pump_call_work / pump_call_iters : 0.0, - pump_call_time, - pump_call_time > 0 ? pump_call_work / pump_call_time : 0.0); - // Detect a stall: the solve made no pivots, so the incumbent vertex was - // already optimal for this objective and x did not move. Perturb next pass. - stalled = (iter == iter_before); - - if (lp_status == simplex::primal_status_t::OPTIMAL) { - std::vector adjusted_solution(lp.num_cols, 0.0); - reduced_col = 0; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { - adjusted_solution[j] = reduced_solution.x[reduced_col++]; - } else { - adjusted_solution[j] = soln.x[j]; - } - } - - // Verify the solution is primal feasible - std::vector residual = lp.rhs; - matrix_vector_multiply(lp.A, 1.0, adjusted_solution, -1.0, residual); - const f_t primal_residual = vector_norm_inf(residual); - - if (primal_residual > 1e-6) { - settings_.log.printf("Reduced LP residual|| A*x - b ||_inf = %.4e\n", primal_residual); - } - - std::vector tmp_fractional; - i_t num_fractional_reduced = - fractional_variables(settings_, adjusted_solution, var_types_, tmp_fractional); - settings_.log.printf( - "Degenerate feasibility pump (%d/%d): primal work estimate %.2e, iter %d, fractional " - "variables %d/%d. Time %.2f\n", - pump_iter, - max_pump_iter, - primal_work_estimate, - iter, - num_fractional_reduced, - num_fractional, - toc(dual_degenerate_feasibility_pump_start_time)); - // Also treat a pass that fails to improve the best as a stall, so we perturb - // the next pass even when the solve pivoted (moved) without reducing the count. - stalled = stalled || (num_fractional_reduced >= best_num_fractional); - if (num_fractional_reduced < best_num_fractional) { - best_num_fractional = num_fractional_reduced; - best_reduced_vstatus = reduced_vstatus; - } - } else { - settings_.log.printf( - "Degenerate feasibility pump: primal simplex returned non-optimal status %d at pump_iter " - "%d. Work estimate %.2e\n", - static_cast(lp_status), - pump_iter, - primal_work_estimate); - // Even if we hit work/time limit, the solution may have improved. - // Check fractional count before breaking. - if (lp_status == simplex::primal_status_t::WORK_LIMIT || - lp_status == simplex::primal_status_t::TIME_LIMIT) { - std::vector adjusted_solution(lp.num_cols, 0.0); - reduced_col = 0; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || - std::abs(soln.z[j]) <= settings_.tight_tol) { - adjusted_solution[j] = reduced_solution.x[reduced_col++]; - } else { - adjusted_solution[j] = soln.x[j]; - } - } - std::vector residual = lp.rhs; - matrix_vector_multiply(lp.A, 1.0, adjusted_solution, -1.0, residual); - const f_t primal_residual = vector_norm_inf(residual); - if (primal_residual <= 1e-6) { - std::vector tmp_fractional; - i_t num_fractional_reduced = - fractional_variables(settings_, adjusted_solution, var_types_, tmp_fractional); - settings_.log.printf( - "Degenerate feasibility pump (%d/%d): after work/time limit, fractional " - "variables %d/%d\n", - pump_iter, - max_pump_iter, - num_fractional_reduced, - num_fractional); - if (num_fractional_reduced < best_num_fractional) { - best_num_fractional = num_fractional_reduced; - best_reduced_vstatus = reduced_vstatus; - } - } - } - break; - } - } - - settings_.log.printf( - "Degenerate feasibility pump: Simplex iterations %d, Best number of fractional variables " - "%d/%d. Work estimate %.2e, Time %.2f, Basis updates %d\n", - iter, - best_num_fractional, - num_fractional, - primal_work_estimate, - toc(dual_degenerate_feasibility_pump_start_time), - reduced_basis_update.num_updates()); - if (best_num_fractional < num_fractional) { - // Translate the vstatus from the reduced problem to the vstatus for the original problem - i_t reduced_cols = 0; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings_.tight_tol) { - vstatus[j] = best_reduced_vstatus[reduced_cols++]; - } - } - - std::vector superbasic_list; - nonbasic_list.clear(); - simplex::get_basis_from_vstatus(m, vstatus, basic_list, nonbasic_list, superbasic_list); - assert(superbasic_list.empty()); - i_t deficient_repaired = 0; - const i_t refactor_status = basis_update.refactor_basis(lp.A, - settings_, - lp.lower, - lp.upper, - exploration_stats_.start_time, - basic_list, - nonbasic_list, - vstatus, - deficient_repaired); - if (refactor_status == CONCURRENT_HALT_RETURN || refactor_status == TIME_LIMIT_RETURN) { - return; - } - if (refactor_status != 0) { - // TODO: On failure vstatus, basic_list, and nonbasic_list are in a bad state. - // We should save copies before the failure and restore them after the failure. - settings_.log.printf( - "Failed to refactor basis after dual degenerate feasibility pump. " - "%d deficient columns.\n", - refactor_status); - return; - } - - // Update the solution - // First set the nonbasic variables on their bounds - for (i_t k = 0; k < lp.num_cols - lp.num_rows; k++) { - const i_t j = nonbasic_list[k]; - if (vstatus[j] == variable_status_t::NONBASIC_LOWER || - vstatus[j] == variable_status_t::NONBASIC_FIXED) { - soln.x[j] = lp.lower[j]; - } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER) { - soln.x[j] = lp.upper[j]; - } else { - soln.x[j] = 0; - } - } - // Then compute the effective rhs - std::vector rhs = lp.rhs; - for (i_t j = 0; j < lp.num_cols; j++) { - if (vstatus[j] == variable_status_t::BASIC) { continue; } - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - - const f_t x_j = soln.x[j]; - for (i_t p = col_start; p < col_end; p++) { - const i_t i = lp.A.i[p]; - const f_t aij = lp.A.x[p]; - rhs[i] -= aij * x_j; - } - } - - // Then solve B xB = rhs - std::vector xB(lp.num_rows); - basis_update.b_solve(rhs, xB); - - // Then update the basic variables - for (i_t k = 0; k < lp.num_rows; k++) { - soln.x[basic_list[k]] = xB[k]; - } - - fractional.clear(); - num_fractional = fractional_variables(settings_, soln.x, var_types_, fractional); - } -} - -template -i_t branch_and_bound_t::apply_delta_x_for_integer_pivot( - const simplex::lp_problem_t& lp, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& nonbasic_index, - std::vector& variable_to_basic, - std::vector& vstatus, - i_t entering_index, - i_t nonbasic_entering, - i_t direction, - const sparse_vector_t& delta_x, - sparse_vector_t& utilde_sparse, - simplex::lp_solution_t& solution, - simplex::basis_update_mpf_t& basis_update, - f_t& work_estimate) -{ - // Keep the existing dense ratio test and full integrality scan. - std::vector delta_x_dense; - delta_x.to_dense(delta_x_dense); - f_t step_length; - i_t basic_leaving; - const i_t leaving_index = simplex::primal_ratio_test(lp, - settings_, - vstatus, - basic_list, - solution.x, - delta_x_dense, - step_length, - basic_leaving, - entering_index, - direction, - work_estimate); - bool binding_integer = - leaving_index != -1 && - is_fractional(solution.x[leaving_index], var_types_[leaving_index], settings_.integer_tol); - if (!binding_integer) { - if (leaving_index == -1) { - return -4; // unbounded or entering hit its own bound - } else if (var_types_[leaving_index] != variable_type_t::INTEGER) { - return -5; // continuous variable won ratio test - } else { - return -6; // integer variable won but it's not fractional (already at integer value) - } - } - - std::vector test_x = solution.x; - i_t integer_destroyed = 0; - for (i_t h = 0; h < lp.num_cols; ++h) { - test_x[h] += step_length * delta_x_dense[h]; - if (var_types_[h] != variable_type_t::INTEGER) { continue; } - const bool was_fractional = is_fractional(solution.x[h], var_types_[h], settings_.integer_tol); - const bool now_fractional = is_fractional(test_x[h], var_types_[h], settings_.integer_tol); - if (now_fractional && !was_fractional) { - integer_destroyed++; - } else if (!now_fractional && was_fractional) { - integer_destroyed--; - } - } - // Require a strict net decrease in fractional integers. - if (integer_destroyed >= 0) { return -2; } - - if (utilde_sparse.i.empty()) { - // Recover B^{-1} abar from the direction before changing the basis: - // delta_x[basic_list[h]] = -direction * (B^{-1} abar)[h]. - // In MPF, utilde = U0 * (B^{-1} abar), since all updates are absorbed into L. - sparse_vector_t b_inv_abar(lp.num_rows, 0); - b_inv_abar.i.reserve(delta_x.i.size()); - b_inv_abar.x.reserve(delta_x.x.size()); - const i_t nz = delta_x.i.size(); - for (i_t k = 0; k < nz; ++k) { - const i_t h = variable_to_basic[delta_x.i[k]]; - if (h >= 0 && delta_x.x[k] != 0) { - b_inv_abar.i.push_back(h); - b_inv_abar.x.push_back(-direction * delta_x.x[k]); - } - } - basis_update.u_multiply(b_inv_abar, utilde_sparse); - } - - solution.x = test_x; - basic_list[basic_leaving] = entering_index; - variable_to_basic[entering_index] = basic_leaving; - variable_to_basic[leaving_index] = -1; - nonbasic_list[nonbasic_entering] = leaving_index; - vstatus[entering_index] = variable_status_t::BASIC; - if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { - vstatus[leaving_index] = variable_status_t::NONBASIC_FIXED; - } else if (delta_x_dense[leaving_index] < 0) { - vstatus[leaving_index] = variable_status_t::NONBASIC_LOWER; - } else { - vstatus[leaving_index] = variable_status_t::NONBASIC_UPPER; - } - - // Keep nonbasic_index consistent with nonbasic_list: entering_index is now basic, - // and leaving_index has taken its slot in nonbasic_list. - nonbasic_index[entering_index] = -1; - nonbasic_index[leaving_index] = nonbasic_entering; - - const i_t m = lp.num_rows; - sparse_vector_t es_sparse(m, 1); - es_sparse.i[0] = basic_leaving; - es_sparse.x[0] = 1.0; - sparse_vector_t UTsol_sparse(m, 1); - sparse_vector_t solution_sparse(m, 1); - basis_update.b_transpose_solve(es_sparse, solution_sparse, UTsol_sparse); - const i_t recommend_refactor = basis_update.update(utilde_sparse, UTsol_sparse, basic_leaving); - if (recommend_refactor == 1) { - csc_matrix_t L(m, m, 1); - csc_matrix_t U(m, m, 1); - std::vector pinv(m); - std::vector p(m); - std::vector q(m); - std::vector deficient; - std::vector slacks_needed; - f_t factorize_work_estimate = 0.0; - const i_t rank = factorize_basis(lp.A, - settings_, - basic_list, - exploration_stats_.start_time, - L, - U, - p, - pinv, - q, - deficient, - slacks_needed, - factorize_work_estimate); - if (rank == CONCURRENT_HALT_RETURN || rank == TIME_LIMIT_RETURN) { return -3; } - if (rank < 0 || rank != lp.num_rows) { return -3; } - simplex::reorder_basic_list(q, basic_list); - for (i_t k = 0; k < m; ++k) { - variable_to_basic[basic_list[k]] = k; - } - basis_update.reset(L, U, p); - } - - return 0; -} - -template -void branch_and_bound_t::fast_slack_integer_pivots( - const simplex::lp_problem_t& lp, - const simplex::simplex_solver_settings_t& settings, - const std::vector& fractional, - const std::vector& row_to_slack, - const simplex::lp_solution_t& solution, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& nonbasic_index, - std::vector& variable_to_basic, - std::vector& vstatus, - simplex::lp_solution_t& soln, - simplex::basis_update_mpf_t& basis_update, - f_t& work_estimate) -{ - std::vector fast_candidates; - std::vector fast_rows; - std::vector fast_nonbasic_slacks; - for (i_t j : fractional) { - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - const i_t num_rows = col_end - col_start; - i_t num_basic_slacks = 0; - i_t num_nonbasic_slacks_with_reduced_cost_zero = 0; - i_t nonbasic_slack = -1; - i_t slack_row = -1; - for (i_t p = col_start; p < col_end; p++) { - const i_t i = lp.A.i[p]; - const i_t slack = row_to_slack[i]; - if (slack >= 0) { - if (vstatus[slack] == variable_status_t::BASIC) { - num_basic_slacks++; - } else if (std::abs(solution.z[slack]) <= 1e-10) { - num_nonbasic_slacks_with_reduced_cost_zero++; - nonbasic_slack = slack; - slack_row = i; - } - } - } - if (num_basic_slacks == num_rows - 1 && num_nonbasic_slacks_with_reduced_cost_zero == 1) { - fast_candidates.push_back(j); - fast_rows.push_back(slack_row); - fast_nonbasic_slacks.push_back(nonbasic_slack); - } - } - - if (fast_candidates.size() > 0 && settings_.inside_mip < 2) { - settings.log.printf("Found %ld fast candidates for pivot out integer variables\n", - fast_candidates.size()); - } - - // Build a reverse index nonbasic_index[v] = position of v in nonbasic_list, or -1 if not - // present. Used to locate the entering variable's slot in the fast-candidate path. - // apply_delta_x_for_integer_pivot keeps this index consistent by applying an O(1) fix-up - // on each successful pivot; the two variables whose (non)basic status changes are the only - // entries that need to be updated. - nonbasic_index.assign(lp.num_cols, -1); - for (i_t p = 0; p < static_cast(nonbasic_list.size()); ++p) { - nonbasic_index[nonbasic_list[p]] = p; - } - - const i_t num_candidates = fast_candidates.size(); - f_t last_log = tic(); - f_t loop_start = tic(); - for (i_t k = 0; k < num_candidates; k++) { - const i_t j = fast_candidates[k]; - const i_t row = fast_rows[k]; - const i_t nonbasic_slack = fast_nonbasic_slacks[k]; - // Skip if state changed by a prior successful pivot. - if (vstatus[j] != variable_status_t::BASIC) { continue; } - if (vstatus[nonbasic_slack] == variable_status_t::BASIC) { continue; } - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - f_t a_ij = 0.0; - for (i_t p = col_start; p < col_end; p++) { - const i_t i = lp.A.i[p]; - if (i == row) { - a_ij = lp.A.x[p]; - break; - } - } - f_t bound = a_ij > 0 ? lp.lower[j] : lp.upper[j]; - if (std::abs(bound) == inf) { continue; } - - const f_t delta_xj = bound - soln.x[j]; - const f_t scale = -delta_xj * a_ij; - if (std::abs(scale) <= 1e-12) { continue; } - - // Build delta_x describing "move x[j] to its bound, let the basic slacks compensate to - // keep A*x = b". This is a feasible direction (A*delta_x = 0). The nonzero pattern lives - // on the entries of column A(:, j) plus j itself. In the nonbasic_slack slot, - // delta_x[nonbasic_slack] = -delta_xj * a_ij > 0, i.e. the entering slack moves up from - // its lower bound 0. We build the sparse version to feed the feasibility scan, then - // normalize so that delta_x[nonbasic_slack] == 1 (the convention primal_ratio_test expects - // for entering variables). - sparse_vector_t delta_x_sparse; - delta_x_sparse.n = lp.num_cols; - delta_x_sparse.i.reserve(col_end - col_start + 1); - delta_x_sparse.x.reserve(col_end - col_start + 1); - delta_x_sparse.i.push_back(j); - delta_x_sparse.x.push_back(delta_xj); - for (i_t p = col_start; p < col_end; p++) { - const i_t r = lp.A.i[p]; - const f_t a_rj = lp.A.x[p]; - const f_t delta_slack_r = -delta_xj * a_rj; - delta_x_sparse.i.push_back(row_to_slack[r]); - delta_x_sparse.x.push_back(delta_slack_r); - } - - // Reject if the full unit step would drive any basic slack below zero. - bool ok = true; - const i_t ndx = delta_x_sparse.i.size(); - for (i_t h = 0; h < ndx; h++) { - const i_t jj = delta_x_sparse.i[h]; - if (jj == j) continue; - const f_t val = delta_x_sparse.x[h]; - const f_t slack_value = soln.x[jj]; - if (val < -slack_value) { - ok = false; - break; - } - } - if (!ok) { continue; } - - // Normalize so that delta_x[nonbasic_slack] == 1 (the standard entering-direction - // convention). Done on the sparse vector, after the feasibility scan above, which reads - // the unnormalized values. - for (f_t& val : delta_x_sparse.x) { - val /= scale; - } - - // Entering variable is the nonbasic slack, moving up from its lower bound 0. - const i_t entering_index = nonbasic_slack; - const i_t nonbasic_entering = nonbasic_index[nonbasic_slack]; - if (nonbasic_entering < 0) { continue; } - const i_t direction = 1; - - // The common helper computes utilde only if the pivot is accepted. - sparse_vector_t utilde_sparse; - - i_t error = apply_delta_x_for_integer_pivot(lp, - basic_list, - nonbasic_list, - nonbasic_index, - variable_to_basic, - vstatus, - entering_index, - nonbasic_entering, - direction, - delta_x_sparse, - utilde_sparse, - soln, - basis_update, - work_estimate); - // apply_delta_x_for_integer_pivot only mutates vstatus when the pivot actually fires, - // so entering_index transitioning to BASIC is a reliable success signal. - if (!error && settings.inside_mip < 2) { - settings.log.printf( - "Fast candidate pivot succeeded: j=%d entering slack=%d row=%d\n", j, entering_index, row); - } - - if (settings.inside_mip < 2 && toc(last_log) > 1.0) { - settings.log.printf("Fast candidates %d/%d processed in %.2f seconds\n", - k + 1, - num_candidates, - toc(loop_start)); - last_log = tic(); - } - } - if (settings.inside_mip < 2) { - settings.log.printf("Fast candidates: %d/%d processed in %.2f seconds\n", - num_candidates, - num_candidates, - toc(loop_start)); - } -} - -template -i_t branch_and_bound_t::pivot_out_integer_variables( - const simplex::lp_problem_t& lp, - const simplex::simplex_solver_settings_t& settings, - const std::vector& new_slacks, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& vstatus, - simplex::lp_solution_t& solution, - simplex::basis_update_mpf_t& basis_update, - i_t& num_fractional, - std::vector& fractional) -{ - if (num_fractional == 0) { return 0; } - f_t pivot_out_integer_variables_start_time = tic(); - std::vector zero_reduced_costs_vars; - std::vector zero_reduced_costs_vars_nonbasic_index; - bool dual_degenerate = check_for_dual_degeneracy( - solution, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); - if (!dual_degenerate) { return 0; } - - lp_solution_t soln_copy = solution; - std::vector basic_list_copy = basic_list; - std::vector nonbasic_list_copy = nonbasic_list; - std::vector vstatus_copy = vstatus; - simplex::basis_update_mpf_t basis_update_copy = basis_update; - - const i_t start_num_fractional = num_fractional; - - const i_t num_zero_reduced_costs_vars = zero_reduced_costs_vars.size(); - - std::vector row_to_slack(lp.num_rows, -1); - for (i_t j : new_slacks) { - if (lp.lower[j] != 0 || lp.upper[j] != inf) { continue; } - const i_t p = lp.A.col_start[j]; - row_to_slack[lp.A.i[p]] = j; - } - - f_t work_estimate = 0.0; - - // Count primal degenerate basic variables - i_t num_degenerate = 0; - i_t num_degenerate_continuous = 0; - i_t num_degenerate_integer = 0; - for (i_t k = 0; k < lp.num_rows; k++) { - const i_t j = basic_list_copy[k]; - const f_t slack_to_lower = soln_copy.x[j] - lp.lower[j]; - const f_t slack_to_upper = lp.upper[j] - soln_copy.x[j]; - if (slack_to_lower <= settings_.primal_tol || slack_to_upper <= settings_.primal_tol) { - num_degenerate++; - if (var_types_[j] == variable_type_t::INTEGER) { - num_degenerate_integer++; - } else { - num_degenerate_continuous++; - } - } - } - const f_t degeneracy_fraction = static_cast(num_degenerate) / lp.num_rows; - if (settings.inside_mip < 2 && settings.inside_submip == 0) { - settings.log.printf( - "Primal degeneracy: %d/%d basic variables are degenerate (%.1f%%), " - "continuous=%d, integer=%d\n", - num_degenerate, - lp.num_rows, - 100.0 * degeneracy_fraction, - num_degenerate_continuous, - num_degenerate_integer); - } - - // Skip pivot_out entirely if primal degeneracy is too high — the ratio test - // will almost always be won by a degenerate variable, making pivots hopeless. - if (degeneracy_fraction > 0.5) { - if (settings.inside_mip < 2) { - settings.log.printf("Skipping pivot_out_integer_variables: degeneracy %.1f%% > 50%%\n", - 100.0 * degeneracy_fraction); - } - return 0; - } - - std::vector nonbasic_index; - std::vector variable_to_basic(lp.num_cols, -1); - for (i_t k = 0; k < lp.num_rows; k++) { - variable_to_basic[basic_list_copy[k]] = k; - } - fast_slack_integer_pivots(lp, - settings, - fractional, - row_to_slack, - solution, - basic_list_copy, - nonbasic_list_copy, - nonbasic_index, - variable_to_basic, - vstatus_copy, - soln_copy, - basis_update_copy, - work_estimate); - - std::vector work_list = fractional; - - sparse_vector_t ep; - ep.n = lp.num_rows; - ep.i.resize(1); - ep.x.resize(1); - ep.x[0] = 1.0; - - std::vector delta_y_dense(lp.num_rows, 0.0); - - // Track which entering variables are actually tried (to detect duplication) - std::vector entering_tried_count(lp.num_cols, 0); - - i_t worklist_total_processed = 0; - i_t worklist_skipped = 0; - i_t worklist_btran_done = 0; - i_t worklist_ftran_done = 0; - i_t worklist_pivots_succeeded = 0; - i_t worklist_readded = 0; - f_t worklist_btran_time = 0.0; - f_t worklist_dot_time = 0.0; - f_t worklist_ftran_time = 0.0; - i_t worklist_no_candidates = 0; // target had no nonzero dot_q - i_t worklist_ratio_test_fail = 0; // ratio test didn't pick a fractional integer (error -1) - i_t worklist_net_increase_fail = 0; // pivot would net-increase fractionals (error -2) - i_t worklist_unbounded = 0; // entering hit its own bound or unbounded (error -4) - i_t worklist_continuous_won = 0; // continuous variable won ratio test (error -5) - i_t worklist_nonfrac_int_won = 0; // non-fractional integer won ratio test (error -6) - - f_t worklist_loop_start = tic(); - f_t worklist_last_log = tic(); - - while (!work_list.empty()) { - const i_t j = work_list.back(); - const i_t p = variable_to_basic[j]; - work_list.pop_back(); - worklist_total_processed++; - - // Skip if j is no longer basic and fractional (may have been fixed by a prior pivot) - if (p < 0) { - worklist_skipped++; - continue; - } - if (vstatus_copy[j] != variable_status_t::BASIC) { - worklist_skipped++; - continue; - } - if (!is_fractional(soln_copy.x[j], var_types_[j], settings_.integer_tol)) { - worklist_skipped++; - continue; - } - - // We want to pivot variable j out of the basis. - // We solve B^T * delta_y = e_p, where p is the position of j in the basis. - // Or delta_y = B^{-T} e_p, or delta_y^T = e_p^T B^{-T} - - ep.i[0] = p; - sparse_vector_t delta_y_sparse; - sparse_vector_t UTsol_sparse; - f_t btran_start = tic(); - basis_update_copy.b_transpose_solve(ep, delta_y_sparse, UTsol_sparse); - worklist_btran_time += toc(btran_start); - worklist_btran_done++; - - // Scatter delta_y_sparse into dense workspace for dot product computation - const i_t delta_y_nz = delta_y_sparse.i.size(); - for (i_t h = 0; h < delta_y_nz; h++) { - delta_y_dense[delta_y_sparse.i[h]] = delta_y_sparse.x[h]; - } - - // We also have that - // B*delta_xB + N*delta_xN = 0 - // So delta_xB = -B^{-1} N * delta_xN - // And delta_xB[p] = e_p^T * delta_xB = -e_p^T B^{-1} N * delta_xN - // = -delta_y^T N * delta_xN - // Recall that delta_xN = e_q where q is the entering variables - // So delta_xB[p] = -delta_y^T A(:, q) - // - // For p to be the leaving variable, we need it to be the binding - // member in the ratio test - // x_B + alpha * delta_xB >= l_B - // x_B + alpha * delta_xB <= u_B - // - // Or alpha <= (l_B[p] - x_B[p]) / delta_xB[p] when delta_xB[p] < 0 - // Or alpha <= (u_B[p] - x_B[p]) / delta_xB[p] when delta_xB[p] > 0 - // - // Thus, if we want to push x_B[p] up to u_B[p], we want - // alpha = (u_B[p] - x_B[p]) / delta_xB[p] to be small - // And if we want to push x_B[p] down to l_B[p], we want - // alpha = (l_B[p] - x_B[p]) / delta_xB[p] to be small - // - // Or equivalently, we want delta_xB[p] to be large - - // Find top 3 candidates by merit = |dot_q| / nnz(A(:,q)) - // Large |dot_q| means the target moves a lot (small step to hit bound). - // Small nnz means the FTRAN result is likely sparse, so fewer competing - // basic variables will have nonzero delta_xB components to block the target. - // Skip entering variables that have already been tried (and failed) by prior targets. - f_t values[3] = {0.0, 0.0, 0.0}; - i_t indices[3] = {-1, -1, -1}; - f_t dot_start = tic(); - for (i_t q : zero_reduced_costs_vars) { - if (var_types_[q] == variable_type_t::INTEGER) { continue; } - if (nonbasic_index[q] < 0) { continue; } - if (entering_tried_count[q] > 0) { continue; } - // Compute dot_q = delta_y^T * A(:, q) using dense delta_y - const i_t col_start = lp.A.col_start[q]; - const i_t col_end = lp.A.col_start[q + 1]; - const i_t col_nnz = col_end - col_start; - f_t dot_q = 0.0; - for (i_t pp = col_start; pp < col_end; pp++) { - dot_q += delta_y_dense[lp.A.i[pp]] * lp.A.x[pp]; - } - const f_t abs_dot_q = std::abs(dot_q); - if (abs_dot_q <= 1e-12) { continue; } - const f_t merit = abs_dot_q / static_cast(col_nnz); - - if (merit > values[0]) { - indices[2] = indices[1]; - values[2] = values[1]; - indices[1] = indices[0]; - values[1] = values[0]; - indices[0] = q; - values[0] = merit; - } else if (merit > values[1]) { - indices[2] = indices[1]; - values[2] = values[1]; - indices[1] = q; - values[1] = merit; - } else if (merit > values[2]) { - indices[2] = q; - values[2] = merit; - } - } - worklist_dot_time += toc(dot_start); - - if (indices[0] == -1) { worklist_no_candidates++; } - - // Try the top 3 candidates - for (i_t h = 0; h < 3; h++) { - if (indices[h] == -1) break; - - const i_t q = indices[h]; - const i_t entering_index = q; - const i_t nonbasic_entering = nonbasic_index[q]; - if (nonbasic_entering < 0) { continue; } - entering_tried_count[q]++; - - // Determine direction based on entering variable's status - const i_t direction = (vstatus_copy[q] == variable_status_t::NONBASIC_LOWER || - vstatus_copy[q] == variable_status_t::NONBASIC_FIXED) - ? 1 - : -1; - - // Solve B * delta_xB = A(:, q) so utilde is valid for the MPF update. - sparse_vector_t rhs(lp.A, q); - sparse_vector_t delta_xB; - sparse_vector_t utilde_sparse; - f_t ftran_start = tic(); - basis_update_copy.b_solve(rhs, delta_xB, utilde_sparse); - worklist_ftran_time += toc(ftran_start); - worklist_ftran_done++; - - sparse_vector_t delta_x(lp.num_cols, 0); - delta_x.i.reserve(delta_xB.i.size() + 1); - delta_x.x.reserve(delta_xB.x.size() + 1); - const i_t nz = delta_xB.i.size(); - for (i_t k = 0; k < nz; ++k) { - delta_x.i.push_back(basic_list_copy[delta_xB.i[k]]); - delta_x.x.push_back(-direction * delta_xB.x[k]); - } - delta_x.i.push_back(q); - delta_x.x.push_back(direction); - - i_t error = apply_delta_x_for_integer_pivot(lp, - basic_list_copy, - nonbasic_list_copy, - nonbasic_index, - variable_to_basic, - vstatus_copy, - entering_index, - nonbasic_entering, - direction, - delta_x, - utilde_sparse, - soln_copy, - basis_update_copy, - work_estimate); - - if (error == -2) { worklist_net_increase_fail++; } - if (error == -4) { - worklist_unbounded++; - worklist_ratio_test_fail++; - } - if (error == -5) { - worklist_continuous_won++; - worklist_ratio_test_fail++; - } - if (error == -6) { - worklist_nonfrac_int_won++; - worklist_ratio_test_fail++; - } - - if (!error) { - worklist_pivots_succeeded++; -#ifdef READD_TO_WORKLIST - // We did a successful pivot; add fractional variables whose values changed to work list. - std::vector delta_x_dense; - delta_x.to_dense(delta_x_dense); - for (i_t k : fractional) { - if (vstatus_copy[k] != variable_status_t::BASIC) { continue; } - if (std::abs(delta_x_dense[k]) > settings_.zero_tol) { - work_list.push_back(k); - worklist_readded++; - } - } -#endif - break; - } - } - - // Clear dense workspace for next target - for (i_t h = 0; h < delta_y_nz; h++) { - delta_y_dense[delta_y_sparse.i[h]] = 0.0; - } - - if (toc(worklist_last_log) > 1.0) { - if (settings.inside_mip < 2) { - settings.log.printf( - "Worklist progress: %d/%d processed, %d pivots, %d ratio_fail (unb=%d cont=%d nfint=%d), " - "%d net_inc_fail, %d no_cand, %.2f seconds\n", - worklist_total_processed, - static_cast(fractional.size()), - worklist_pivots_succeeded, - worklist_ratio_test_fail, - worklist_unbounded, - worklist_continuous_won, - worklist_nonfrac_int_won, - worklist_net_increase_fail, - worklist_no_candidates, - toc(worklist_loop_start)); - } - worklist_last_log = tic(); - } - } - - // Count unique entering variables and duplication - i_t unique_entering = 0; - i_t max_entering_count = 0; - i_t entering_tried_once = 0; - i_t entering_tried_multiple = 0; - for (i_t q = 0; q < lp.num_cols; q++) { - if (entering_tried_count[q] > 0) { - unique_entering++; - max_entering_count = std::max(max_entering_count, entering_tried_count[q]); - if (entering_tried_count[q] == 1) { - entering_tried_once++; - } else { - entering_tried_multiple++; - } - } - } - if (settings.inside_mip < 2) { - settings.log.printf( - "Worklist entering stats: unique=%d, tried_once=%d, tried_multiple=%d, " - "max_count=%d, total_ftran=%d, duplication_ratio=%.1fx\n", - unique_entering, - entering_tried_once, - entering_tried_multiple, - max_entering_count, - worklist_ftran_done, - worklist_ftran_done / std::max(1.0, static_cast(unique_entering))); - - settings.log.printf( - "Worklist stats: processed=%d skipped=%d btran=%d ftran=%d pivots=%d readded=%d " - "btran_time=%.2f dot_time=%.2f ftran_time=%.2f zero_rc_vars=%d " - "no_candidates=%d ratio_test_fail=%d (unbounded=%d continuous_won=%d nonfrac_int_won=%d) " - "net_increase_fail=%d\n", - worklist_total_processed, - worklist_skipped, - worklist_btran_done, - worklist_ftran_done, - worklist_pivots_succeeded, - worklist_readded, - worklist_btran_time, - worklist_dot_time, - worklist_ftran_time, - num_zero_reduced_costs_vars, - worklist_no_candidates, - worklist_ratio_test_fail, - worklist_unbounded, - worklist_continuous_won, - worklist_nonfrac_int_won, - worklist_net_increase_fail); - } - - std::vector new_fractional; - const i_t num_new_fractional = - fractional_variables(settings_, soln_copy.x, var_types_, new_fractional); - if (num_new_fractional < start_num_fractional) { - i_t num_integer_increased = start_num_fractional - num_new_fractional; - integer_pivots_.fetch_add(num_integer_increased, std::memory_order_release); -#if 0 - settings.log.printf("Pivoted out %d integer variables: %d -> %d in %.2f\n", - num_integer_increased, - start_num_fractional, - num_new_fractional, - toc(pivot_out_integer_variables_start_time)); -#endif - num_fractional = num_new_fractional; - fractional = new_fractional; - basic_list = basic_list_copy; - nonbasic_list = nonbasic_list_copy; - vstatus = vstatus_copy; - basis_update = basis_update_copy; - solution = soln_copy; - return num_integer_increased; - } - return 0; -} - template void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( const simplex::lp_problem_t& lp, @@ -5581,6 +4389,8 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut num_integer_increased = pivot_out_integer_variables(original_lp_, settings_, new_slacks_, + var_types_, + exploration_stats_.start_time, basic_list, nonbasic_list, root_vstatus_, @@ -5588,6 +4398,9 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut basis_update, num_fractional, fractional); + if (num_integer_increased > 0) { + integer_pivots_.fetch_add(num_integer_increased, std::memory_order_release); + } } settings_.log.printf("Pivoted out %d integer variables in %e seconds\n", num_integer_increased, @@ -5595,6 +4408,11 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut if (settings_.dual_degenerate_feasibility_pump != 0) { dual_degenerate_feasibility_pump(original_lp_, + settings_, + var_types_, + edge_norms_, + root_relax_work_estimate_, + exploration_stats_.start_time, basic_list, nonbasic_list, root_vstatus_, diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 4066508b56..b05bdfe947 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -12,6 +12,7 @@ #include #include #include +#include #include #include @@ -93,144 +94,6 @@ struct deterministic_bfs_policy_t; template struct deterministic_diving_policy_t; -template -struct objective_bound_pair_t { - objective_bound_pair_t() - : objective(std::numeric_limits::quiet_NaN()), bound(std::numeric_limits::quiet_NaN()) - { - } - objective_bound_pair_t(f_t objective_in, f_t bound_in) : objective(objective_in), bound(bound_in) - { - } - bool is_valid() { return objective == objective && bound == bound; } - f_t objective; - f_t bound; -}; - -template -class reduced_cost_bounds_t { - public: - reduced_cost_bounds_t(i_t original_cols) - : max_objective_(-std::numeric_limits::infinity()), - lower_bounds_(original_cols), - upper_bounds_(original_cols) - { - } - - i_t add_lower_bound(i_t col, f_t objective, f_t bound) - { - if (col < static_cast(lower_bounds_.size())) { - if (!lower_bounds_[col].is_valid()) { - lower_bounds_[col] = objective_bound_pair_t(objective, bound); - if (objective > max_objective_) { max_objective_ = objective; } - return 1; - } else { - if (bound > lower_bounds_[col].bound) { - lower_bounds_[col] = objective_bound_pair_t(objective, bound); - if (objective > max_objective_) { max_objective_ = objective; } - return 2; - } else if (bound == lower_bounds_[col].bound && objective > lower_bounds_[col].objective) { - lower_bounds_[col].objective = objective; - if (objective > max_objective_) { max_objective_ = objective; } - return 1; - } else { - return -2; - } - } - } else { - return -1; - } - } - - i_t add_upper_bound(i_t col, f_t objective, f_t bound) - { - if (col < static_cast(upper_bounds_.size())) { - if (!upper_bounds_[col].is_valid()) { - upper_bounds_[col] = objective_bound_pair_t(objective, bound); - if (objective > max_objective_) { max_objective_ = objective; } - return 1; - } else { - if (bound < upper_bounds_[col].bound) { - upper_bounds_[col] = objective_bound_pair_t(objective, bound); - if (objective > max_objective_) { max_objective_ = objective; } - return 2; - } else if (bound == upper_bounds_[col].bound && objective > upper_bounds_[col].objective) { - upper_bounds_[col].objective = objective; - if (objective > max_objective_) { max_objective_ = objective; } - return 1; - } else { - return -2; - } - } - } else { - return -1; - } - } - - i_t update_bounds_from_new_incumbent(f_t incumbent_objective, - const std::vector& var_types, - std::vector& lower_bounds, - std::vector& upper_bounds) - { - const i_t n = static_cast(lower_bounds_.size()); - f_t max_objective = -std::numeric_limits::infinity(); - i_t integer_bounds_updated = 0; - for (i_t j = 0; j < n; ++j) { - if (lower_bounds_[j].is_valid()) { - if (incumbent_objective <= lower_bounds_[j].objective && - lower_bounds_[j].bound > lower_bounds[j]) { - // printf("RCF Variable %d (%d): lower %e -> %e\n", j, static_cast(var_types[j]), - // lower_bounds[j], lower_bounds_[j].bound); - lower_bounds[j] = lower_bounds_[j].bound; - if (var_types[j] == simplex::variable_type_t::INTEGER) { integer_bounds_updated++; } - lower_bounds_[j].bound = lower_bounds_[j].objective = - std::numeric_limits::quiet_NaN(); - } - if (lower_bounds_[j].objective > max_objective) { - max_objective = lower_bounds_[j].objective; - } - } - if (upper_bounds_[j].is_valid()) { - if (incumbent_objective <= upper_bounds_[j].objective && - upper_bounds_[j].bound < upper_bounds[j]) { - // printf("RCF Variable %d (%d): upper %e -> %e\n", j, static_cast(var_types[j]), - // upper_bounds[j], upper_bounds_[j].bound); - upper_bounds[j] = upper_bounds_[j].bound; - if (var_types[j] == simplex::variable_type_t::INTEGER) { integer_bounds_updated++; } - upper_bounds_[j].bound = upper_bounds_[j].objective = - std::numeric_limits::quiet_NaN(); - } - if (upper_bounds_[j].objective > max_objective) { - max_objective = upper_bounds_[j].objective; - } - } - } - max_objective_ = max_objective; - return integer_bounds_updated; - } - - f_t get_current_lower_bound(i_t col) - { - if (col < static_cast(lower_bounds_.size())) { return lower_bounds_[col].bound; } - return std::numeric_limits::quiet_NaN(); - } - - f_t get_current_upper_bound(i_t col) - { - if (col < static_cast(upper_bounds_.size())) { return upper_bounds_[col].bound; } - return std::numeric_limits::quiet_NaN(); - } - - i_t num_cols() { return static_cast(lower_bounds_.size()); } - - f_t get_max_objective() { return max_objective_; } - - private: - f_t max_objective_; - std::vector> lower_bounds_; - std::vector> upper_bounds_; -}; - template class branch_and_bound_t { public: @@ -492,59 +355,6 @@ class branch_and_bound_t { search_strategy_t thread_type); omp_atomic_t integer_pivots_{0}; - bool check_for_dual_degeneracy(const simplex::lp_solution_t& solution, - const std::vector& nonbasic_list, - std::vector& zero_reduced_costs_vars, - std::vector& zero_reduced_costs_vars_nonbasic_index); - - void fast_slack_integer_pivots(const simplex::lp_problem_t& lp, - const simplex::simplex_solver_settings_t& settings, - const std::vector& fractional, - const std::vector& row_to_slack, - const simplex::lp_solution_t& solution, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& nonbasic_index, - std::vector& variable_to_basic, - std::vector& vstatus, - simplex::lp_solution_t& soln, - simplex::basis_update_mpf_t& basis_update, - f_t& work_estimate); - - i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, - const simplex::simplex_solver_settings_t& settings, - const std::vector& new_slacks, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& vstatus, - simplex::lp_solution_t& soln, - simplex::basis_update_mpf_t& basis_update, - i_t& num_fractional, - std::vector& fractional); - - i_t apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& nonbasic_index, - std::vector& variable_to_basic, - std::vector& vstatus, - i_t entering_index, - i_t nonbasic_entering, - i_t direction, - const sparse_vector_t& delta_x, - sparse_vector_t& utilde_sparse, - simplex::lp_solution_t& solution, - simplex::basis_update_mpf_t& basis_update, - f_t& work_estimate); - - void dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, - std::vector& basic_list, - std::vector& nonbasic_list, - std::vector& vstatus, - simplex::lp_solution_t& soln, - simplex::basis_update_mpf_t& basis_update, - i_t& num_fractional, - std::vector& fractional); void pivot_to_improve_reduced_cost_strengthening( const simplex::lp_problem_t& lp, diff --git a/cpp/src/branch_and_bound/degenerate_pivots.cpp b/cpp/src/branch_and_bound/degenerate_pivots.cpp new file mode 100644 index 0000000000..e47e2ab64a --- /dev/null +++ b/cpp/src/branch_and_bound/degenerate_pivots.cpp @@ -0,0 +1,1302 @@ +/* clang-format off */ +/* + * SPDX-FileCopyrightText: Copyright (c) 2025-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +/* clang-format on */ + +#include +#include + +#include +#include +#include +#include +#include +#include + +#include + +namespace cuopt::mathematical_optimization::mip { + +using simplex::lp_solution_t; +using simplex::simplex_solver_settings_t; +using simplex::variable_status_t; +using simplex::variable_type_t; + +template +bool check_for_dual_degeneracy(const simplex::lp_solution_t& solution, + const simplex::simplex_solver_settings_t& settings, + const std::vector& nonbasic_list, + std::vector& zero_reduced_costs_vars, + std::vector& zero_reduced_costs_vars_nonbasic_index) +{ + const i_t num_nonbasics = nonbasic_list.size(); + for (i_t k = 0; k < num_nonbasics; k++) { + const i_t j = nonbasic_list[k]; + if (std::abs(solution.z[j]) <= settings.tight_tol) { + zero_reduced_costs_vars.push_back(j); + zero_reduced_costs_vars_nonbasic_index.push_back(k); + } + } + return !zero_reduced_costs_vars.empty(); +} + +template +void dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const std::vector& var_types, + const std::vector& edge_norms, + f_t root_relax_work_estimate, + f_t start_time, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + simplex::lp_solution_t& soln, + simplex::basis_update_mpf_t& basis_update, + i_t& num_fractional, + std::vector& fractional) +{ + f_t dual_degenerate_feasibility_pump_start_time = tic(); + std::vector zero_reduced_costs_vars; + std::vector zero_reduced_costs_vars_nonbasic_index; + bool dual_degenerate = check_for_dual_degeneracy( + soln, settings, nonbasic_list, zero_reduced_costs_vars, zero_reduced_costs_vars_nonbasic_index); + if (!dual_degenerate) { return; } + + // Construct a new LP problem + // minimize p^T x + // subject to B x_B + N_z x_z = b - N x_N + // l_B <= x_B <= u_B + // l_z <= x_z <= u_z + // + // where B is the basic matrix, N is the nonbasic matrix, b is the right-hand side, + + const i_t m = lp.num_rows; + const i_t n = lp.num_rows + zero_reduced_costs_vars.size(); + + i_t nnz = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings.tight_tol) { + nnz += lp.A.col_start[j + 1] - lp.A.col_start[j]; + } + } + simplex::lp_problem_t lp_reduced(lp.handle_ptr, m, n, nnz); + csc_matrix_t& A_reduced = lp_reduced.A; + std::vector original_col_to_reduced_col(lp.num_cols, -1); + i_t nz = 0; + i_t reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings.tight_tol) { + original_col_to_reduced_col[j] = reduced_col; + A_reduced.col_start[reduced_col] = nz; + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + const f_t value = lp.A.x[p]; + A_reduced.i[nz] = i; + A_reduced.x[nz] = value; + nz++; + } + lp_reduced.lower[reduced_col] = lp.lower[j]; + lp_reduced.upper[reduced_col] = lp.upper[j]; + reduced_col++; + } + } + A_reduced.col_start[reduced_col] = nz; + + std::vector b_reduced = lp.rhs; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings.tight_tol) { + // PASS + } else { + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + const f_t value = lp.A.x[p]; + b_reduced[i] -= value * soln.x[j]; + } + } + } + lp_reduced.rhs = b_reduced; + lp_reduced.obj_scale = 1.0; + + settings.log.printf( + "Constructed dual degenerate feasibility pump LP with %d rows and %d columns\n", m, n); + + std::vector reduced_basic_list(m); + std::vector reduced_nonbasic_list(zero_reduced_costs_vars.size()); + std::vector reduced_vstatus(n); + i_t num_basic = 0; + i_t num_nonbasic = 0; + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC) { + reduced_vstatus[reduced_col++] = variable_status_t::BASIC; + } else if (std::abs(soln.z[j]) <= 1e-10) { + reduced_nonbasic_list[num_nonbasic++] = + reduced_col; // Does ordering of nonbasic variables matter? + reduced_vstatus[reduced_col++] = vstatus[j]; + } + } + + simplex::lp_solution_t reduced_solution(m, n); + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings.tight_tol) { + reduced_solution.x[reduced_col++] = soln.x[j]; + } + } + + std::vector reduced_edge_norms(n); + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings.tight_tol) { + reduced_edge_norms[reduced_col++] = edge_norms[j]; + } + } + + simplex::basis_update_mpf_t reduced_basis_update = basis_update; + reduced_basis_update.clear_work_estimate(); + for (i_t k = 0; k < m; k++) { + reduced_basic_list[k] = original_col_to_reduced_col[basic_list[k]]; + } + + f_t primal_work_estimate = 0.0; + i_t iter = 0; + i_t max_pump_iter = 10; + simplex::random_t rng(settings.random_seed); + i_t best_num_fractional = num_fractional; + std::vector best_reduced_vstatus(n); + bool stalled = false; + for (i_t pump_iter = 0; pump_iter < max_pump_iter; pump_iter++) { + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings.tight_tol) { + lp_reduced.objective[reduced_col] = 0; + if (var_types[j] == variable_type_t::INTEGER) { + if (is_fractional(reduced_solution.x[reduced_col], var_types[j], settings.integer_tol)) { + // Default to the exact nearest-integer rounding. Only perturb the + // rounding direction when the previous pass made no progress (a + // zero-pivot solve), to break out of the stall. + const f_t random_value = + stalled ? 0.25 * (2.0 * rng.random() - 1.0) : 0.0; // [-0.25, 0.25] + if (reduced_solution.x[reduced_col] + random_value < + std::floor(reduced_solution.x[reduced_col]) + 0.5) { + lp_reduced.objective[reduced_col] = 1; + } else { + lp_reduced.objective[reduced_col] = -1; + } + } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_LOWER) { + lp_reduced.objective[reduced_col] = 0.1; + } else if (reduced_vstatus[reduced_col] == variable_status_t::NONBASIC_UPPER) { + lp_reduced.objective[reduced_col] = -0.1; + } + } + reduced_col++; + } + } + + // Check reduced costs before calling primal simplex. + // Compute y = B^{-T} * c_B (BTRAN with the pump objective on basic variables) + std::vector c_basic_pump(m, 0.0); + for (i_t k = 0; k < m; k++) { + c_basic_pump[k] = lp_reduced.objective[reduced_basic_list[k]]; + } + std::vector y_pump(m); + reduced_basis_update.b_transpose_solve(c_basic_pump, y_pump); + + // Check if any nonbasic has a violated reduced cost + i_t num_violated = 0; + f_t max_violation = 0.0; + const i_t num_nonbasics_reduced = reduced_nonbasic_list.size(); + for (i_t k = 0; k < num_nonbasics_reduced; k++) { + const i_t j = reduced_nonbasic_list[k]; + // z[j] = c[j] - y^T * A(:,j) + f_t zj = lp_reduced.objective[j]; + const i_t col_start = A_reduced.col_start[j]; + const i_t col_end = A_reduced.col_start[j + 1]; + for (i_t p = col_start; p < col_end; p++) { + zj -= y_pump[A_reduced.i[p]] * A_reduced.x[p]; + } + // Check pricing condition + bool violated = false; + if (reduced_vstatus[j] == variable_status_t::NONBASIC_LOWER || + reduced_vstatus[j] == variable_status_t::NONBASIC_FIXED) { + if (zj < -settings.dual_tol) { violated = true; } + } else if (reduced_vstatus[j] == variable_status_t::NONBASIC_UPPER) { + if (zj > settings.dual_tol) { violated = true; } + } + if (violated) { + num_violated++; + max_violation = std::max(max_violation, std::abs(zj)); + } + } + + if (num_violated == 0) { + settings.log.printf( + "Degenerate feasibility pump (%d/%d): skipping primal simplex, no violated reduced costs " + "(%d nonbasics checked)\n", + pump_iter, + max_pump_iter, + num_nonbasics_reduced); + primal_work_estimate += reduced_basis_update.work_estimate(); + reduced_basis_update.clear_work_estimate(); + // Don't count this as a pump iteration, but break if we've skipped twice + // in a row (perturbation isn't helping) + if (stalled) { break; } + stalled = true; + pump_iter--; + continue; + } + settings.log.printf( + "Degenerate feasibility pump (%d/%d): %d violated reduced costs (max %.2e) out of %d " + "nonbasics\n", + pump_iter, + max_pump_iter, + num_violated, + max_violation, + num_nonbasics_reduced); + + bool recompute_basis = false; + const i_t iter_before = iter; + f_t primal_work_before = primal_work_estimate; + f_t pump_call_start_time = tic(); + simplex_solver_settings_t primal_settings = settings; + primal_settings.log.log = false; + primal_settings.time_limit = settings.time_limit; + primal_settings.work_limit = root_relax_work_estimate / 10; + settings.log.printf( + "Degenerate feasibility pump: calling primal simplex with %d rows, %d cols, %d nnz, " + "%d basis updates, work_limit %.2e, primal_work_estimate %.2e\n", + m, + n, + A_reduced.col_start[n], + reduced_basis_update.num_updates(), + primal_settings.work_limit, + primal_work_estimate); + simplex::primal_status_t lp_status = + simplex::primal_phase2_with_advanced_basis(2, + start_time, + lp_reduced, + primal_settings, + reduced_vstatus, + reduced_basis_update, + reduced_basic_list, + reduced_nonbasic_list, + reduced_solution, + iter, + primal_work_estimate); + f_t pump_call_time = toc(pump_call_start_time); + f_t pump_call_work = primal_work_estimate - primal_work_before; + i_t pump_call_iters = iter - iter_before; + settings.log.printf( + "Degenerate feasibility pump: primal simplex returned status %d, %d iters, " + "work %.2e (%.2e/iter), time %.2f (%.2e work/s)\n", + static_cast(lp_status), + pump_call_iters, + pump_call_work, + pump_call_iters > 0 ? pump_call_work / pump_call_iters : 0.0, + pump_call_time, + pump_call_time > 0 ? pump_call_work / pump_call_time : 0.0); + // Detect a stall: the solve made no pivots, so the incumbent vertex was + // already optimal for this objective and x did not move. Perturb next pass. + stalled = (iter == iter_before); + + if (lp_status == simplex::primal_status_t::OPTIMAL) { + std::vector adjusted_solution(lp.num_cols, 0.0); + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings.tight_tol) { + adjusted_solution[j] = reduced_solution.x[reduced_col++]; + } else { + adjusted_solution[j] = soln.x[j]; + } + } + + // Verify the solution is primal feasible + std::vector residual = lp.rhs; + matrix_vector_multiply(lp.A, 1.0, adjusted_solution, -1.0, residual); + const f_t primal_residual = vector_norm_inf(residual); + + if (primal_residual > 1e-6) { + settings.log.printf("Reduced LP residual|| A*x - b ||_inf = %.4e\n", primal_residual); + } + + std::vector tmp_fractional; + i_t num_fractional_reduced = + fractional_variables(settings, adjusted_solution, var_types, tmp_fractional); + settings.log.printf( + "Degenerate feasibility pump (%d/%d): primal work estimate %.2e, iter %d, fractional " + "variables %d/%d. Time %.2f\n", + pump_iter, + max_pump_iter, + primal_work_estimate, + iter, + num_fractional_reduced, + num_fractional, + toc(dual_degenerate_feasibility_pump_start_time)); + // Also treat a pass that fails to improve the best as a stall, so we perturb + // the next pass even when the solve pivoted (moved) without reducing the count. + stalled = stalled || (num_fractional_reduced >= best_num_fractional); + if (num_fractional_reduced < best_num_fractional) { + best_num_fractional = num_fractional_reduced; + best_reduced_vstatus = reduced_vstatus; + } + } else { + settings.log.printf( + "Degenerate feasibility pump: primal simplex returned non-optimal status %d at pump_iter " + "%d. Work estimate %.2e\n", + static_cast(lp_status), + pump_iter, + primal_work_estimate); + // Even if we hit work/time limit, the solution may have improved. + // Check fractional count before breaking. + if (lp_status == simplex::primal_status_t::WORK_LIMIT || + lp_status == simplex::primal_status_t::TIME_LIMIT) { + std::vector adjusted_solution(lp.num_cols, 0.0); + reduced_col = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings.tight_tol) { + adjusted_solution[j] = reduced_solution.x[reduced_col++]; + } else { + adjusted_solution[j] = soln.x[j]; + } + } + std::vector residual = lp.rhs; + matrix_vector_multiply(lp.A, 1.0, adjusted_solution, -1.0, residual); + const f_t primal_residual = vector_norm_inf(residual); + if (primal_residual <= 1e-6) { + std::vector tmp_fractional; + i_t num_fractional_reduced = + fractional_variables(settings, adjusted_solution, var_types, tmp_fractional); + settings.log.printf( + "Degenerate feasibility pump (%d/%d): after work/time limit, fractional " + "variables %d/%d\n", + pump_iter, + max_pump_iter, + num_fractional_reduced, + num_fractional); + if (num_fractional_reduced < best_num_fractional) { + best_num_fractional = num_fractional_reduced; + best_reduced_vstatus = reduced_vstatus; + } + } + } + break; + } + } + + settings.log.printf( + "Degenerate feasibility pump: Simplex iterations %d, Best number of fractional variables " + "%d/%d. Work estimate %.2e, Time %.2f, Basis updates %d\n", + iter, + best_num_fractional, + num_fractional, + primal_work_estimate, + toc(dual_degenerate_feasibility_pump_start_time), + reduced_basis_update.num_updates()); + if (best_num_fractional < num_fractional) { + // Translate the vstatus from the reduced problem to the vstatus for the original problem + i_t reduced_cols = 0; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings.tight_tol) { + vstatus[j] = best_reduced_vstatus[reduced_cols++]; + } + } + + std::vector superbasic_list; + nonbasic_list.clear(); + simplex::get_basis_from_vstatus(m, vstatus, basic_list, nonbasic_list, superbasic_list); + assert(superbasic_list.empty()); + i_t deficient_repaired = 0; + const i_t refactor_status = basis_update.refactor_basis(lp.A, + settings, + lp.lower, + lp.upper, + start_time, + basic_list, + nonbasic_list, + vstatus, + deficient_repaired); + if (refactor_status == CONCURRENT_HALT_RETURN || refactor_status == TIME_LIMIT_RETURN) { + return; + } + if (refactor_status != 0) { + // TODO: On failure vstatus, basic_list, and nonbasic_list are in a bad state. + // We should save copies before the failure and restore them after the failure. + settings.log.printf( + "Failed to refactor basis after dual degenerate feasibility pump. " + "%d deficient columns.\n", + refactor_status); + return; + } + + // Update the solution + // First set the nonbasic variables on their bounds + for (i_t k = 0; k < lp.num_cols - lp.num_rows; k++) { + const i_t j = nonbasic_list[k]; + if (vstatus[j] == variable_status_t::NONBASIC_LOWER || + vstatus[j] == variable_status_t::NONBASIC_FIXED) { + soln.x[j] = lp.lower[j]; + } else if (vstatus[j] == variable_status_t::NONBASIC_UPPER) { + soln.x[j] = lp.upper[j]; + } else { + soln.x[j] = 0; + } + } + // Then compute the effective rhs + std::vector rhs = lp.rhs; + for (i_t j = 0; j < lp.num_cols; j++) { + if (vstatus[j] == variable_status_t::BASIC) { continue; } + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + + const f_t x_j = soln.x[j]; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + const f_t aij = lp.A.x[p]; + rhs[i] -= aij * x_j; + } + } + + // Then solve B xB = rhs + std::vector xB(lp.num_rows); + basis_update.b_solve(rhs, xB); + + // Then update the basic variables + for (i_t k = 0; k < lp.num_rows; k++) { + soln.x[basic_list[k]] = xB[k]; + } + + fractional.clear(); + num_fractional = fractional_variables(settings, soln.x, var_types, fractional); + } +} + +template +i_t apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + i_t entering_index, + i_t nonbasic_entering, + i_t direction, + const sparse_vector_t& delta_x, + const std::vector& var_types, + f_t start_time, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& nonbasic_index, + std::vector& variable_to_basic, + std::vector& vstatus, + sparse_vector_t& utilde_sparse, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + f_t& work_estimate) +{ + // Keep the existing dense ratio test and full integrality scan. + std::vector delta_x_dense; + delta_x.to_dense(delta_x_dense); + f_t step_length; + i_t basic_leaving; + const i_t leaving_index = simplex::primal_ratio_test(lp, + settings, + vstatus, + basic_list, + solution.x, + delta_x_dense, + step_length, + basic_leaving, + entering_index, + direction, + work_estimate); + bool binding_integer = + leaving_index != -1 && + is_fractional(solution.x[leaving_index], var_types[leaving_index], settings.integer_tol); + if (!binding_integer) { + if (leaving_index == -1) { + return -4; // unbounded or entering hit its own bound + } else if (var_types[leaving_index] != variable_type_t::INTEGER) { + return -5; // continuous variable won ratio test + } else { + return -6; // integer variable won but it's not fractional (already at integer value) + } + } + + std::vector test_x = solution.x; + i_t integer_destroyed = 0; + for (i_t h = 0; h < lp.num_cols; ++h) { + test_x[h] += step_length * delta_x_dense[h]; + if (var_types[h] != variable_type_t::INTEGER) { continue; } + const bool was_fractional = is_fractional(solution.x[h], var_types[h], settings.integer_tol); + const bool now_fractional = is_fractional(test_x[h], var_types[h], settings.integer_tol); + if (now_fractional && !was_fractional) { + integer_destroyed++; + } else if (!now_fractional && was_fractional) { + integer_destroyed--; + } + } + // Require a strict net decrease in fractional integers. + if (integer_destroyed >= 0) { return -2; } + + if (utilde_sparse.i.empty()) { + // Recover B^{-1} abar from the direction before changing the basis: + // delta_x[basic_list[h]] = -direction * (B^{-1} abar)[h]. + // In MPF, utilde = U0 * (B^{-1} abar), since all updates are absorbed into L. + sparse_vector_t b_inv_abar(lp.num_rows, 0); + b_inv_abar.i.reserve(delta_x.i.size()); + b_inv_abar.x.reserve(delta_x.x.size()); + const i_t nz = delta_x.i.size(); + for (i_t k = 0; k < nz; ++k) { + const i_t h = variable_to_basic[delta_x.i[k]]; + if (h >= 0 && delta_x.x[k] != 0) { + b_inv_abar.i.push_back(h); + b_inv_abar.x.push_back(-direction * delta_x.x[k]); + } + } + basis_update.u_multiply(b_inv_abar, utilde_sparse); + } + + solution.x = test_x; + basic_list[basic_leaving] = entering_index; + variable_to_basic[entering_index] = basic_leaving; + variable_to_basic[leaving_index] = -1; + nonbasic_list[nonbasic_entering] = leaving_index; + vstatus[entering_index] = variable_status_t::BASIC; + if (std::abs(lp.upper[leaving_index] - lp.lower[leaving_index]) < 1e-12) { + vstatus[leaving_index] = variable_status_t::NONBASIC_FIXED; + } else if (delta_x_dense[leaving_index] < 0) { + vstatus[leaving_index] = variable_status_t::NONBASIC_LOWER; + } else { + vstatus[leaving_index] = variable_status_t::NONBASIC_UPPER; + } + + // Keep nonbasic_index consistent with nonbasic_list: entering_index is now basic, + // and leaving_index has taken its slot in nonbasic_list. + nonbasic_index[entering_index] = -1; + nonbasic_index[leaving_index] = nonbasic_entering; + + const i_t m = lp.num_rows; + sparse_vector_t es_sparse(m, 1); + es_sparse.i[0] = basic_leaving; + es_sparse.x[0] = 1.0; + sparse_vector_t UTsol_sparse(m, 1); + sparse_vector_t solution_sparse(m, 1); + basis_update.b_transpose_solve(es_sparse, solution_sparse, UTsol_sparse); + const i_t recommend_refactor = basis_update.update(utilde_sparse, UTsol_sparse, basic_leaving); + if (recommend_refactor == 1) { + csc_matrix_t L(m, m, 1); + csc_matrix_t U(m, m, 1); + std::vector pinv(m); + std::vector p(m); + std::vector q(m); + std::vector deficient; + std::vector slacks_needed; + f_t factorize_work_estimate = 0.0; + const i_t rank = factorize_basis(lp.A, + settings, + basic_list, + start_time, + L, + U, + p, + pinv, + q, + deficient, + slacks_needed, + factorize_work_estimate); + if (rank == CONCURRENT_HALT_RETURN || rank == TIME_LIMIT_RETURN) { return -3; } + if (rank < 0 || rank != lp.num_rows) { return -3; } + simplex::reorder_basic_list(q, basic_list); + for (i_t k = 0; k < m; ++k) { + variable_to_basic[basic_list[k]] = k; + } + basis_update.reset(L, U, p); + } + + return 0; +} + +template +void fast_slack_integer_pivots(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const std::vector& fractional, + const std::vector& row_to_slack, + const simplex::lp_solution_t& solution, + const std::vector& var_types, + f_t start_time, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& nonbasic_index, + std::vector& variable_to_basic, + std::vector& vstatus, + simplex::lp_solution_t& soln, + simplex::basis_update_mpf_t& basis_update, + f_t& work_estimate) +{ + std::vector fast_candidates; + std::vector fast_rows; + std::vector fast_nonbasic_slacks; + for (i_t j : fractional) { + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + const i_t num_rows = col_end - col_start; + i_t num_basic_slacks = 0; + i_t num_nonbasic_slacks_with_reduced_cost_zero = 0; + i_t nonbasic_slack = -1; + i_t slack_row = -1; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + const i_t slack = row_to_slack[i]; + if (slack >= 0) { + if (vstatus[slack] == variable_status_t::BASIC) { + num_basic_slacks++; + } else if (std::abs(solution.z[slack]) <= 1e-10) { + num_nonbasic_slacks_with_reduced_cost_zero++; + nonbasic_slack = slack; + slack_row = i; + } + } + } + if (num_basic_slacks == num_rows - 1 && num_nonbasic_slacks_with_reduced_cost_zero == 1) { + fast_candidates.push_back(j); + fast_rows.push_back(slack_row); + fast_nonbasic_slacks.push_back(nonbasic_slack); + } + } + + if (fast_candidates.size() > 0 && settings.inside_mip < 2) { + settings.log.printf("Found %ld fast candidates for pivot out integer variables\n", + fast_candidates.size()); + } + + // Build a reverse index nonbasic_index[v] = position of v in nonbasic_list, or -1 if not + // present. Used to locate the entering variable's slot in the fast-candidate path. + // apply_delta_x_for_integer_pivot keeps this index consistent by applying an O(1) fix-up + // on each successful pivot; the two variables whose (non)basic status changes are the only + // entries that need to be updated. + nonbasic_index.assign(lp.num_cols, -1); + for (i_t p = 0; p < static_cast(nonbasic_list.size()); ++p) { + nonbasic_index[nonbasic_list[p]] = p; + } + + const i_t num_candidates = fast_candidates.size(); + f_t last_log = tic(); + f_t loop_start = tic(); + for (i_t k = 0; k < num_candidates; k++) { + const i_t j = fast_candidates[k]; + const i_t row = fast_rows[k]; + const i_t nonbasic_slack = fast_nonbasic_slacks[k]; + // Skip if state changed by a prior successful pivot. + if (vstatus[j] != variable_status_t::BASIC) { continue; } + if (vstatus[nonbasic_slack] == variable_status_t::BASIC) { continue; } + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + f_t a_ij = 0.0; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + if (i == row) { + a_ij = lp.A.x[p]; + break; + } + } + f_t bound = a_ij > 0 ? lp.lower[j] : lp.upper[j]; + if (std::abs(bound) == inf) { continue; } + + const f_t delta_xj = bound - soln.x[j]; + const f_t scale = -delta_xj * a_ij; + if (std::abs(scale) <= 1e-12) { continue; } + + // Build delta_x describing "move x[j] to its bound, let the basic slacks compensate to + // keep A*x = b". This is a feasible direction (A*delta_x = 0). The nonzero pattern lives + // on the entries of column A(:, j) plus j itself. In the nonbasic_slack slot, + // delta_x[nonbasic_slack] = -delta_xj * a_ij > 0, i.e. the entering slack moves up from + // its lower bound 0. We build the sparse version to feed the feasibility scan, then + // normalize so that delta_x[nonbasic_slack] == 1 (the convention primal_ratio_test expects + // for entering variables). + sparse_vector_t delta_x_sparse; + delta_x_sparse.n = lp.num_cols; + delta_x_sparse.i.reserve(col_end - col_start + 1); + delta_x_sparse.x.reserve(col_end - col_start + 1); + delta_x_sparse.i.push_back(j); + delta_x_sparse.x.push_back(delta_xj); + for (i_t p = col_start; p < col_end; p++) { + const i_t r = lp.A.i[p]; + const f_t a_rj = lp.A.x[p]; + const f_t delta_slack_r = -delta_xj * a_rj; + delta_x_sparse.i.push_back(row_to_slack[r]); + delta_x_sparse.x.push_back(delta_slack_r); + } + + // Reject if the full unit step would drive any basic slack below zero. + bool ok = true; + const i_t ndx = delta_x_sparse.i.size(); + for (i_t h = 0; h < ndx; h++) { + const i_t jj = delta_x_sparse.i[h]; + if (jj == j) continue; + const f_t val = delta_x_sparse.x[h]; + const f_t slack_value = soln.x[jj]; + if (val < -slack_value) { + ok = false; + break; + } + } + if (!ok) { continue; } + + // Normalize so that delta_x[nonbasic_slack] == 1 (the standard entering-direction + // convention). Done on the sparse vector, after the feasibility scan above, which reads + // the unnormalized values. + for (f_t& val : delta_x_sparse.x) { + val /= scale; + } + + // Entering variable is the nonbasic slack, moving up from its lower bound 0. + const i_t entering_index = nonbasic_slack; + const i_t nonbasic_entering = nonbasic_index[nonbasic_slack]; + if (nonbasic_entering < 0) { continue; } + const i_t direction = 1; + + // The common helper computes utilde only if the pivot is accepted. + sparse_vector_t utilde_sparse; + + i_t error = apply_delta_x_for_integer_pivot(lp, + settings, + entering_index, + nonbasic_entering, + direction, + delta_x_sparse, + var_types, + start_time, + basic_list, + nonbasic_list, + nonbasic_index, + variable_to_basic, + vstatus, + utilde_sparse, + soln, + basis_update, + work_estimate); + // apply_delta_x_for_integer_pivot only mutates vstatus when the pivot actually fires, + // so entering_index transitioning to BASIC is a reliable success signal. + if (!error && settings.inside_mip < 2) { + settings.log.printf( + "Fast candidate pivot succeeded: j=%d entering slack=%d row=%d\n", j, entering_index, row); + } + + if (settings.inside_mip < 2 && toc(last_log) > 1.0) { + settings.log.printf("Fast candidates %d/%d processed in %.2f seconds\n", + k + 1, + num_candidates, + toc(loop_start)); + last_log = tic(); + } + } + if (settings.inside_mip < 2) { + settings.log.printf("Fast candidates: %d/%d processed in %.2f seconds\n", + num_candidates, + num_candidates, + toc(loop_start)); + } +} + +template +i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const std::vector& new_slacks, + const std::vector& var_types, + f_t start_time, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + i_t& num_fractional, + std::vector& fractional) +{ + if (num_fractional == 0) { return 0; } + f_t pivot_out_integer_variables_start_time = tic(); + std::vector zero_reduced_costs_vars; + std::vector zero_reduced_costs_vars_nonbasic_index; + bool dual_degenerate = check_for_dual_degeneracy(solution, + settings, + nonbasic_list, + zero_reduced_costs_vars, + zero_reduced_costs_vars_nonbasic_index); + if (!dual_degenerate) { return 0; } + + lp_solution_t soln_copy = solution; + std::vector basic_list_copy = basic_list; + std::vector nonbasic_list_copy = nonbasic_list; + std::vector vstatus_copy = vstatus; + simplex::basis_update_mpf_t basis_update_copy = basis_update; + + const i_t start_num_fractional = num_fractional; + + const i_t num_zero_reduced_costs_vars = zero_reduced_costs_vars.size(); + + std::vector row_to_slack(lp.num_rows, -1); + for (i_t j : new_slacks) { + if (lp.lower[j] != 0 || lp.upper[j] != inf) { continue; } + const i_t p = lp.A.col_start[j]; + row_to_slack[lp.A.i[p]] = j; + } + + f_t work_estimate = 0.0; + + // Count primal degenerate basic variables + i_t num_degenerate = 0; + i_t num_degenerate_continuous = 0; + i_t num_degenerate_integer = 0; + for (i_t k = 0; k < lp.num_rows; k++) { + const i_t j = basic_list_copy[k]; + const f_t slack_to_lower = soln_copy.x[j] - lp.lower[j]; + const f_t slack_to_upper = lp.upper[j] - soln_copy.x[j]; + if (slack_to_lower <= settings.primal_tol || slack_to_upper <= settings.primal_tol) { + num_degenerate++; + if (var_types[j] == variable_type_t::INTEGER) { + num_degenerate_integer++; + } else { + num_degenerate_continuous++; + } + } + } + const f_t degeneracy_fraction = static_cast(num_degenerate) / lp.num_rows; + if (settings.inside_mip < 2 && settings.inside_submip == 0) { + settings.log.printf( + "Primal degeneracy: %d/%d basic variables are degenerate (%.1f%%), " + "continuous=%d, integer=%d\n", + num_degenerate, + lp.num_rows, + 100.0 * degeneracy_fraction, + num_degenerate_continuous, + num_degenerate_integer); + } + + // Skip pivot_out entirely if primal degeneracy is too high — the ratio test + // will almost always be won by a degenerate variable, making pivots hopeless. + if (degeneracy_fraction > 0.5) { + if (settings.inside_mip < 2) { + settings.log.printf("Skipping pivot_out_integer_variables: degeneracy %.1f%% > 50%%\n", + 100.0 * degeneracy_fraction); + } + return 0; + } + + std::vector nonbasic_index; + std::vector variable_to_basic(lp.num_cols, -1); + for (i_t k = 0; k < lp.num_rows; k++) { + variable_to_basic[basic_list_copy[k]] = k; + } + fast_slack_integer_pivots(lp, + settings, + fractional, + row_to_slack, + solution, + var_types, + start_time, + basic_list_copy, + nonbasic_list_copy, + nonbasic_index, + variable_to_basic, + vstatus_copy, + soln_copy, + basis_update_copy, + work_estimate); + + std::vector work_list = fractional; + + sparse_vector_t ep; + ep.n = lp.num_rows; + ep.i.resize(1); + ep.x.resize(1); + ep.x[0] = 1.0; + + std::vector delta_y_dense(lp.num_rows, 0.0); + + // Track which entering variables are actually tried (to detect duplication) + std::vector entering_tried_count(lp.num_cols, 0); + + i_t worklist_total_processed = 0; + i_t worklist_skipped = 0; + i_t worklist_btran_done = 0; + i_t worklist_ftran_done = 0; + i_t worklist_pivots_succeeded = 0; + i_t worklist_readded = 0; + f_t worklist_btran_time = 0.0; + f_t worklist_dot_time = 0.0; + f_t worklist_ftran_time = 0.0; + i_t worklist_no_candidates = 0; // target had no nonzero dot_q + i_t worklist_ratio_test_fail = 0; // ratio test didn't pick a fractional integer (error -1) + i_t worklist_net_increase_fail = 0; // pivot would net-increase fractionals (error -2) + i_t worklist_unbounded = 0; // entering hit its own bound or unbounded (error -4) + i_t worklist_continuous_won = 0; // continuous variable won ratio test (error -5) + i_t worklist_nonfrac_int_won = 0; // non-fractional integer won ratio test (error -6) + + f_t worklist_loop_start = tic(); + f_t worklist_last_log = tic(); + + while (!work_list.empty()) { + const i_t j = work_list.back(); + const i_t p = variable_to_basic[j]; + work_list.pop_back(); + worklist_total_processed++; + + // Skip if j is no longer basic and fractional (may have been fixed by a prior pivot) + if (p < 0) { + worklist_skipped++; + continue; + } + if (vstatus_copy[j] != variable_status_t::BASIC) { + worklist_skipped++; + continue; + } + if (!is_fractional(soln_copy.x[j], var_types[j], settings.integer_tol)) { + worklist_skipped++; + continue; + } + + // We want to pivot variable j out of the basis. + // We solve B^T * delta_y = e_p, where p is the position of j in the basis. + // Or delta_y = B^{-T} e_p, or delta_y^T = e_p^T B^{-T} + + ep.i[0] = p; + sparse_vector_t delta_y_sparse; + sparse_vector_t UTsol_sparse; + f_t btran_start = tic(); + basis_update_copy.b_transpose_solve(ep, delta_y_sparse, UTsol_sparse); + worklist_btran_time += toc(btran_start); + worklist_btran_done++; + + // Scatter delta_y_sparse into dense workspace for dot product computation + const i_t delta_y_nz = delta_y_sparse.i.size(); + for (i_t h = 0; h < delta_y_nz; h++) { + delta_y_dense[delta_y_sparse.i[h]] = delta_y_sparse.x[h]; + } + + // We also have that + // B*delta_xB + N*delta_xN = 0 + // So delta_xB = -B^{-1} N * delta_xN + // And delta_xB[p] = e_p^T * delta_xB = -e_p^T B^{-1} N * delta_xN + // = -delta_y^T N * delta_xN + // Recall that delta_xN = e_q where q is the entering variables + // So delta_xB[p] = -delta_y^T A(:, q) + // + // For p to be the leaving variable, we need it to be the binding + // member in the ratio test + // x_B + alpha * delta_xB >= l_B + // x_B + alpha * delta_xB <= u_B + // + // Or alpha <= (l_B[p] - x_B[p]) / delta_xB[p] when delta_xB[p] < 0 + // Or alpha <= (u_B[p] - x_B[p]) / delta_xB[p] when delta_xB[p] > 0 + // + // Thus, if we want to push x_B[p] up to u_B[p], we want + // alpha = (u_B[p] - x_B[p]) / delta_xB[p] to be small + // And if we want to push x_B[p] down to l_B[p], we want + // alpha = (l_B[p] - x_B[p]) / delta_xB[p] to be small + // + // Or equivalently, we want delta_xB[p] to be large + + // Find top 3 candidates by merit = |dot_q| / nnz(A(:,q)) + // Large |dot_q| means the target moves a lot (small step to hit bound). + // Small nnz means the FTRAN result is likely sparse, so fewer competing + // basic variables will have nonzero delta_xB components to block the target. + // Skip entering variables that have already been tried (and failed) by prior targets. + f_t values[3] = {0.0, 0.0, 0.0}; + i_t indices[3] = {-1, -1, -1}; + f_t dot_start = tic(); + for (i_t q : zero_reduced_costs_vars) { + if (var_types[q] == variable_type_t::INTEGER) { continue; } + if (nonbasic_index[q] < 0) { continue; } + if (entering_tried_count[q] > 0) { continue; } + // Compute dot_q = delta_y^T * A(:, q) using dense delta_y + const i_t col_start = lp.A.col_start[q]; + const i_t col_end = lp.A.col_start[q + 1]; + const i_t col_nnz = col_end - col_start; + f_t dot_q = 0.0; + for (i_t pp = col_start; pp < col_end; pp++) { + dot_q += delta_y_dense[lp.A.i[pp]] * lp.A.x[pp]; + } + const f_t abs_dot_q = std::abs(dot_q); + if (abs_dot_q <= 1e-12) { continue; } + const f_t merit = abs_dot_q / static_cast(col_nnz); + + if (merit > values[0]) { + indices[2] = indices[1]; + values[2] = values[1]; + indices[1] = indices[0]; + values[1] = values[0]; + indices[0] = q; + values[0] = merit; + } else if (merit > values[1]) { + indices[2] = indices[1]; + values[2] = values[1]; + indices[1] = q; + values[1] = merit; + } else if (merit > values[2]) { + indices[2] = q; + values[2] = merit; + } + } + worklist_dot_time += toc(dot_start); + + if (indices[0] == -1) { worklist_no_candidates++; } + + // Try the top 3 candidates + for (i_t h = 0; h < 3; h++) { + if (indices[h] == -1) break; + + const i_t q = indices[h]; + const i_t entering_index = q; + const i_t nonbasic_entering = nonbasic_index[q]; + if (nonbasic_entering < 0) { continue; } + entering_tried_count[q]++; + + // Determine direction based on entering variable's status + const i_t direction = (vstatus_copy[q] == variable_status_t::NONBASIC_LOWER || + vstatus_copy[q] == variable_status_t::NONBASIC_FIXED) + ? 1 + : -1; + + // Solve B * delta_xB = A(:, q) so utilde is valid for the MPF update. + sparse_vector_t rhs(lp.A, q); + sparse_vector_t delta_xB; + sparse_vector_t utilde_sparse; + f_t ftran_start = tic(); + basis_update_copy.b_solve(rhs, delta_xB, utilde_sparse); + worklist_ftran_time += toc(ftran_start); + worklist_ftran_done++; + + sparse_vector_t delta_x(lp.num_cols, 0); + delta_x.i.reserve(delta_xB.i.size() + 1); + delta_x.x.reserve(delta_xB.x.size() + 1); + const i_t nz = delta_xB.i.size(); + for (i_t k = 0; k < nz; ++k) { + delta_x.i.push_back(basic_list_copy[delta_xB.i[k]]); + delta_x.x.push_back(-direction * delta_xB.x[k]); + } + delta_x.i.push_back(q); + delta_x.x.push_back(direction); + + i_t error = apply_delta_x_for_integer_pivot(lp, + settings, + entering_index, + nonbasic_entering, + direction, + delta_x, + var_types, + start_time, + basic_list_copy, + nonbasic_list_copy, + nonbasic_index, + variable_to_basic, + vstatus_copy, + utilde_sparse, + soln_copy, + basis_update_copy, + work_estimate); + + if (error == -2) { worklist_net_increase_fail++; } + if (error == -4) { + worklist_unbounded++; + worklist_ratio_test_fail++; + } + if (error == -5) { + worklist_continuous_won++; + worklist_ratio_test_fail++; + } + if (error == -6) { + worklist_nonfrac_int_won++; + worklist_ratio_test_fail++; + } + + if (!error) { + worklist_pivots_succeeded++; +#ifdef READD_TO_WORKLIST + // We did a successful pivot; add fractional variables whose values changed to work list. + std::vector delta_x_dense; + delta_x.to_dense(delta_x_dense); + for (i_t k : fractional) { + if (vstatus_copy[k] != variable_status_t::BASIC) { continue; } + if (std::abs(delta_x_dense[k]) > settings.zero_tol) { + work_list.push_back(k); + worklist_readded++; + } + } +#endif + break; + } + } + + // Clear dense workspace for next target + for (i_t h = 0; h < delta_y_nz; h++) { + delta_y_dense[delta_y_sparse.i[h]] = 0.0; + } + + if (toc(worklist_last_log) > 1.0) { + if (settings.inside_mip < 2) { + settings.log.printf( + "Worklist progress: %d/%d processed, %d pivots, %d ratio_fail (unb=%d cont=%d nfint=%d), " + "%d net_inc_fail, %d no_cand, %.2f seconds\n", + worklist_total_processed, + static_cast(fractional.size()), + worklist_pivots_succeeded, + worklist_ratio_test_fail, + worklist_unbounded, + worklist_continuous_won, + worklist_nonfrac_int_won, + worklist_net_increase_fail, + worklist_no_candidates, + toc(worklist_loop_start)); + } + worklist_last_log = tic(); + } + } + + // Count unique entering variables and duplication + i_t unique_entering = 0; + i_t max_entering_count = 0; + i_t entering_tried_once = 0; + i_t entering_tried_multiple = 0; + for (i_t q = 0; q < lp.num_cols; q++) { + if (entering_tried_count[q] > 0) { + unique_entering++; + max_entering_count = std::max(max_entering_count, entering_tried_count[q]); + if (entering_tried_count[q] == 1) { + entering_tried_once++; + } else { + entering_tried_multiple++; + } + } + } + if (settings.inside_mip < 2) { + settings.log.printf( + "Worklist entering stats: unique=%d, tried_once=%d, tried_multiple=%d, " + "max_count=%d, total_ftran=%d, duplication_ratio=%.1fx\n", + unique_entering, + entering_tried_once, + entering_tried_multiple, + max_entering_count, + worklist_ftran_done, + worklist_ftran_done / std::max(1.0, static_cast(unique_entering))); + + settings.log.printf( + "Worklist stats: processed=%d skipped=%d btran=%d ftran=%d pivots=%d readded=%d " + "btran_time=%.2f dot_time=%.2f ftran_time=%.2f zero_rc_vars=%d " + "no_candidates=%d ratio_test_fail=%d (unbounded=%d continuous_won=%d nonfrac_int_won=%d) " + "net_increase_fail=%d\n", + worklist_total_processed, + worklist_skipped, + worklist_btran_done, + worklist_ftran_done, + worklist_pivots_succeeded, + worklist_readded, + worklist_btran_time, + worklist_dot_time, + worklist_ftran_time, + num_zero_reduced_costs_vars, + worklist_no_candidates, + worklist_ratio_test_fail, + worklist_unbounded, + worklist_continuous_won, + worklist_nonfrac_int_won, + worklist_net_increase_fail); + } + + std::vector new_fractional; + const i_t num_new_fractional = + fractional_variables(settings, soln_copy.x, var_types, new_fractional); + if (num_new_fractional < start_num_fractional) { + i_t num_integer_increased = start_num_fractional - num_new_fractional; +#if 0 + settings.log.printf("Pivoted out %d integer variables: %d -> %d in %.2f\n", + num_integer_increased, + start_num_fractional, + num_new_fractional, + toc(pivot_out_integer_variables_start_time)); +#endif + num_fractional = num_new_fractional; + fractional = new_fractional; + basic_list = basic_list_copy; + nonbasic_list = nonbasic_list_copy; + vstatus = vstatus_copy; + basis_update = basis_update_copy; + solution = soln_copy; + return num_integer_increased; + } + return 0; +} + +template bool check_for_dual_degeneracy( + const simplex::lp_solution_t&, + const simplex::simplex_solver_settings_t&, + const std::vector&, + std::vector&, + std::vector&); + +template void fast_slack_integer_pivots( + const simplex::lp_problem_t&, + const simplex::simplex_solver_settings_t&, + const std::vector&, + const std::vector&, + const simplex::lp_solution_t&, + const std::vector&, + double, + std::vector&, + std::vector&, + std::vector&, + std::vector&, + std::vector&, + simplex::lp_solution_t&, + simplex::basis_update_mpf_t&, + double&); + +template int pivot_out_integer_variables( + const simplex::lp_problem_t&, + const simplex::simplex_solver_settings_t&, + const std::vector&, + const std::vector&, + double, + std::vector&, + std::vector&, + std::vector&, + simplex::lp_solution_t&, + simplex::basis_update_mpf_t&, + int&, + std::vector&); + +template int apply_delta_x_for_integer_pivot( + const simplex::lp_problem_t&, + const simplex::simplex_solver_settings_t&, + int, + int, + int, + const sparse_vector_t&, + const std::vector&, + double, + std::vector&, + std::vector&, + std::vector&, + std::vector&, + std::vector&, + sparse_vector_t&, + simplex::lp_solution_t&, + simplex::basis_update_mpf_t&, + double&); + +template void dual_degenerate_feasibility_pump( + const simplex::lp_problem_t&, + const simplex::simplex_solver_settings_t&, + const std::vector&, + const std::vector&, + double, + double, + std::vector&, + std::vector&, + std::vector&, + simplex::lp_solution_t&, + simplex::basis_update_mpf_t&, + int&, + std::vector&); + +} // namespace cuopt::mathematical_optimization::mip diff --git a/cpp/src/branch_and_bound/degenerate_pivots.hpp b/cpp/src/branch_and_bound/degenerate_pivots.hpp new file mode 100644 index 0000000000..90b75ca049 --- /dev/null +++ b/cpp/src/branch_and_bound/degenerate_pivots.hpp @@ -0,0 +1,92 @@ +/* clang-format off */ +/* + * SPDX-FileCopyrightText: Copyright (c) 2025-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +/* clang-format on */ + +#pragma once + +#include +#include +#include +#include +#include + +#include + +namespace cuopt::mathematical_optimization::mip { + +template +bool check_for_dual_degeneracy(const simplex::lp_solution_t& solution, + const simplex::simplex_solver_settings_t& settings, + const std::vector& nonbasic_list, + std::vector& zero_reduced_costs_vars, + std::vector& zero_reduced_costs_vars_nonbasic_index); + +template +void fast_slack_integer_pivots(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const std::vector& fractional, + const std::vector& row_to_slack, + const simplex::lp_solution_t& solution, + const std::vector& var_types, + f_t start_time, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& nonbasic_index, + std::vector& variable_to_basic, + std::vector& vstatus, + simplex::lp_solution_t& soln, + simplex::basis_update_mpf_t& basis_update, + f_t& work_estimate); + +template +i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const std::vector& new_slacks, + const std::vector& var_types, + f_t start_time, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + i_t& num_fractional, + std::vector& fractional); + +template +i_t apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + i_t entering_index, + i_t nonbasic_entering, + i_t direction, + const sparse_vector_t& delta_x, + const std::vector& var_types, + f_t start_time, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& nonbasic_index, + std::vector& variable_to_basic, + std::vector& vstatus, + sparse_vector_t& utilde_sparse, + simplex::lp_solution_t& solution, + simplex::basis_update_mpf_t& basis_update, + f_t& work_estimate); + +template +void dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const std::vector& var_types, + const std::vector& edge_norms, + f_t root_relax_work_estimate, + f_t start_time, + std::vector& basic_list, + std::vector& nonbasic_list, + std::vector& vstatus, + simplex::lp_solution_t& soln, + simplex::basis_update_mpf_t& basis_update, + i_t& num_fractional, + std::vector& fractional); + +} // namespace cuopt::mathematical_optimization::mip diff --git a/cpp/src/branch_and_bound/fractional.hpp b/cpp/src/branch_and_bound/fractional.hpp new file mode 100644 index 0000000000..a992181ba4 --- /dev/null +++ b/cpp/src/branch_and_bound/fractional.hpp @@ -0,0 +1,45 @@ +/* clang-format off */ +/* + * SPDX-FileCopyrightText: Copyright (c) 2025-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +/* clang-format on */ + +#pragma once + +#include +#include + +#include +#include +#include + +namespace cuopt::mathematical_optimization::mip { + +template +inline bool is_fractional(f_t x, simplex::variable_type_t var_type, f_t integer_tol) +{ + if (var_type == simplex::variable_type_t::CONTINUOUS) { + return false; + } else { + f_t x_integer = std::round(x); + return (std::abs(x_integer - x) > integer_tol); + } +} + +template +inline i_t fractional_variables(const simplex::simplex_solver_settings_t& settings, + const std::vector& x, + const std::vector& var_types, + std::vector& fractional) +{ + const i_t n = x.size(); + assert(x.size() == var_types.size()); + fractional.clear(); + for (i_t j = 0; j < n; ++j) { + if (is_fractional(x[j], var_types[j], settings.integer_tol)) { fractional.push_back(j); } + } + return fractional.size(); +} + +} // namespace cuopt::mathematical_optimization::mip diff --git a/cpp/src/branch_and_bound/reduced_cost_bounds.hpp b/cpp/src/branch_and_bound/reduced_cost_bounds.hpp new file mode 100644 index 0000000000..47483a143f --- /dev/null +++ b/cpp/src/branch_and_bound/reduced_cost_bounds.hpp @@ -0,0 +1,160 @@ +/* clang-format off */ +/* + * SPDX-FileCopyrightText: Copyright (c) 2025-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +/* clang-format on */ + +#pragma once + +#include + +#include +#include + +namespace cuopt::mathematical_optimization::mip { + +template +struct objective_bound_pair_t { + objective_bound_pair_t() + : objective(std::numeric_limits::quiet_NaN()), bound(std::numeric_limits::quiet_NaN()) + { + } + objective_bound_pair_t(f_t objective_in, f_t bound_in) : objective(objective_in), bound(bound_in) + { + } + bool is_valid() { return objective == objective && bound == bound; } + f_t objective; + f_t bound; +}; + +template +class reduced_cost_bounds_t { + public: + static constexpr i_t kBoundDominated = -2; + static constexpr i_t kVariableOutOfBounds = -1; + static constexpr i_t kBoundAccepted = 1; + static constexpr i_t kBoundTightened = 2; + + reduced_cost_bounds_t(i_t original_cols) + : max_objective_(-std::numeric_limits::infinity()), + lower_bounds_(original_cols), + upper_bounds_(original_cols) + { + } + + i_t add_lower_bound(i_t col, f_t objective, f_t bound) + { + if (col < static_cast(lower_bounds_.size())) { + if (!lower_bounds_[col].is_valid()) { + lower_bounds_[col] = objective_bound_pair_t(objective, bound); + if (objective > max_objective_) { max_objective_ = objective; } + return kBoundAccepted; + } else { + if (bound > lower_bounds_[col].bound) { + lower_bounds_[col] = objective_bound_pair_t(objective, bound); + if (objective > max_objective_) { max_objective_ = objective; } + return kBoundTightened; + } else if (bound == lower_bounds_[col].bound && objective > lower_bounds_[col].objective) { + lower_bounds_[col].objective = objective; + if (objective > max_objective_) { max_objective_ = objective; } + return kBoundAccepted; + } else { + return kBoundDominated; + } + } + } else { + return kVariableOutOfBounds; + } + } + + i_t add_upper_bound(i_t col, f_t objective, f_t bound) + { + if (col < static_cast(upper_bounds_.size())) { + if (!upper_bounds_[col].is_valid()) { + upper_bounds_[col] = objective_bound_pair_t(objective, bound); + if (objective > max_objective_) { max_objective_ = objective; } + return kBoundAccepted; + } else { + if (bound < upper_bounds_[col].bound) { + upper_bounds_[col] = objective_bound_pair_t(objective, bound); + if (objective > max_objective_) { max_objective_ = objective; } + return kBoundTightened; + } else if (bound == upper_bounds_[col].bound && objective > upper_bounds_[col].objective) { + upper_bounds_[col].objective = objective; + if (objective > max_objective_) { max_objective_ = objective; } + return kBoundAccepted; + } else { + return kBoundDominated; + } + } + } else { + return kVariableOutOfBounds; + } + } + + i_t update_bounds_from_new_incumbent(f_t incumbent_objective, + const std::vector& var_types, + std::vector& lower_bounds, + std::vector& upper_bounds) + { + const i_t n = static_cast(lower_bounds_.size()); + f_t max_objective = -std::numeric_limits::infinity(); + i_t integer_bounds_updated = 0; + for (i_t j = 0; j < n; ++j) { + if (lower_bounds_[j].is_valid()) { + if (incumbent_objective <= lower_bounds_[j].objective && + lower_bounds_[j].bound > lower_bounds[j]) { + // printf("RCF Variable %d (%d): lower %e -> %e\n", j, static_cast(var_types[j]), + // lower_bounds[j], lower_bounds_[j].bound); + lower_bounds[j] = lower_bounds_[j].bound; + if (var_types[j] == simplex::variable_type_t::INTEGER) { integer_bounds_updated++; } + lower_bounds_[j].bound = lower_bounds_[j].objective = + std::numeric_limits::quiet_NaN(); + } + if (lower_bounds_[j].objective > max_objective) { + max_objective = lower_bounds_[j].objective; + } + } + if (upper_bounds_[j].is_valid()) { + if (incumbent_objective <= upper_bounds_[j].objective && + upper_bounds_[j].bound < upper_bounds[j]) { + // printf("RCF Variable %d (%d): upper %e -> %e\n", j, static_cast(var_types[j]), + // upper_bounds[j], upper_bounds_[j].bound); + upper_bounds[j] = upper_bounds_[j].bound; + if (var_types[j] == simplex::variable_type_t::INTEGER) { integer_bounds_updated++; } + upper_bounds_[j].bound = upper_bounds_[j].objective = + std::numeric_limits::quiet_NaN(); + } + if (upper_bounds_[j].objective > max_objective) { + max_objective = upper_bounds_[j].objective; + } + } + } + max_objective_ = max_objective; + return integer_bounds_updated; + } + + f_t get_current_lower_bound(i_t col) + { + if (col < static_cast(lower_bounds_.size())) { return lower_bounds_[col].bound; } + return std::numeric_limits::quiet_NaN(); + } + + f_t get_current_upper_bound(i_t col) + { + if (col < static_cast(upper_bounds_.size())) { return upper_bounds_[col].bound; } + return std::numeric_limits::quiet_NaN(); + } + + i_t num_cols() { return static_cast(lower_bounds_.size()); } + + f_t get_max_objective() { return max_objective_; } + + private: + f_t max_objective_; + std::vector> lower_bounds_; + std::vector> upper_bounds_; +}; + +} // namespace cuopt::mathematical_optimization::mip From c7fcc6ecd38a6969033b8cbcc17a9e29504de341 Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 22 Sep 2026 17:42:42 -0700 Subject: [PATCH 109/113] Extract reduced-cost strengthening pivot routine --- cpp/src/branch_and_bound/branch_and_bound.cpp | 308 +---------------- cpp/src/branch_and_bound/branch_and_bound.hpp | 12 - .../branch_and_bound/degenerate_pivots.cpp | 313 ++++++++++++++++++ .../branch_and_bound/degenerate_pivots.hpp | 18 + 4 files changed, 339 insertions(+), 312 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 720c754c18..65b60fbd12 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3742,13 +3742,15 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t::cut_pass_action_t branch_and_bound_t -void branch_and_bound_t::pivot_to_improve_reduced_cost_strengthening( - const simplex::lp_problem_t& lp, - const std::vector& basic_list, - const std::vector& nonbasic_list, - const std::vector& vstatus, - const simplex::lp_solution_t& soln, - const simplex::basis_update_mpf_t& basis_update, - const i_t num_fractional, - const std::vector& fractional, - const f_t relaxation_objective, - reduced_cost_bounds_t& reduced_cost_bounds) -{ - const double strengthening_start = tic(); - double btran_time = 0.0; - double reduced_cost_update_time = 0.0; - // Count primal degenerate basic variables - i_t num_degenerate = 0; - i_t num_degenerate_continuous = 0; - i_t num_degenerate_integer = 0; - std::vector degenerate_integer_list; - degenerate_integer_list.reserve(lp.num_rows); - for (i_t k = 0; k < lp.num_rows; k++) { - const i_t j = basic_list[k]; - const f_t slack_to_lower = soln.x[j] - lp.lower[j]; - const f_t slack_to_upper = lp.upper[j] - soln.x[j]; - if (slack_to_lower <= settings_.primal_tol || slack_to_upper <= settings_.primal_tol) { - num_degenerate++; - if (var_types_[j] == variable_type_t::INTEGER) { - num_degenerate_integer++; - degenerate_integer_list.push_back(j); - } else { - num_degenerate_continuous++; - } - } - } - - if (num_degenerate_integer == 0) return; - - settings_.log.printf("RCS timing start: candidates=%d elapsed=%.6f\n", - num_degenerate_integer, - toc(exploration_stats_.start_time)); - std::vector variable_to_basic_position(lp.num_cols, -1); - for (i_t k = 0; k < lp.num_rows; k++) { - variable_to_basic_position[basic_list[k]] = k; - } - // The basis stays fixed across candidates; preserve Arow_'s ordering for cut generation. - csr_matrix_t local_Arow = Arow_; - std::vector nonbasic_end(lp.num_rows); - simplex::compute_initial_nonbasic_end(variable_to_basic_position, local_Arow, nonbasic_end); - std::vector delta_y(lp.num_rows, 0); - std::vector delta_z(lp.num_cols, 0); - std::vector delta_z_mark(lp.num_cols, 0); - std::vector delta_z_indices; - delta_z_indices.reserve(lp.num_cols); - - f_t work_estimate = 0; - const f_t threshold = 100.0 * settings_.integer_tol; - const f_t tol = 1e-2; - const f_t zero_tol = settings_.zero_tol; - const f_t harris_tol = settings_.dual_tol / 10; - - i_t num_bounds_added = 0; - for (i_t j : degenerate_integer_list) { - // x_j is a degenerate integer basic variable. - // We would like a dual-feasible point where x_j is nonbasic with a nonzero - // reduced cost that may be used for reduced cost strengthening. - // We do not need to take the pivot; a dual step along either ray is enough. - // - // One BTRAN: B^T * delta_y = e_p, which matches direction == -1 in - // compute_reduced_cost_update (B^T * delta_y = -direction * e_p). - // The opposite direction is the negated (delta_y, delta_z) ray. - const i_t leaving_index = j; - const i_t p = variable_to_basic_position[j]; - if (p == -1) continue; - - sparse_vector_t ep(lp.num_rows, 1); - ep.i[0] = p; - ep.x[0] = 1.0; - sparse_vector_t delta_y_sparse; - sparse_vector_t UTsol_sparse; - const double btran_start = tic(); - basis_update.b_transpose_solve(ep, delta_y_sparse, UTsol_sparse); - btran_time += toc(btran_start); - - // delta_zN = -N^T * delta_y, delta_z[leaving] = -1 - const double reduced_cost_update_start = tic(); - i_t delta_y_nz0 = 0; - for (const f_t value : delta_y_sparse.x) { - if (std::abs(value) > 1e-12) { delta_y_nz0++; } - } - work_estimate += delta_y_sparse.i.size(); - const f_t delta_y_nz_percentage = delta_y_nz0 / static_cast(lp.num_rows) * 100.0; - if (delta_y_nz_percentage <= 30.0) { - simplex::compute_delta_z(local_Arow, - delta_y_sparse, - leaving_index, - /*direction=*/-1, - nonbasic_end, - delta_z_mark, - delta_z_indices, - delta_z, - work_estimate); - } else { - delta_y_sparse.to_dense(delta_y); - work_estimate += delta_y.size(); - simplex::compute_reduced_cost_update(lp, - basic_list, - nonbasic_list, - delta_y, - leaving_index, - /*direction=*/-1, - delta_z_mark, - delta_z_indices, - delta_z, - work_estimate); - } - reduced_cost_update_time += toc(reduced_cost_update_start); - - const f_t lower_j = lp.lower[j]; - const f_t upper_j = lp.upper[j]; - const bool at_lower = soln.x[j] - lower_j <= settings_.primal_tol; - const bool at_upper = upper_j - soln.x[j] <= settings_.primal_tol; - - // Try both dual rays. scale == +1 uses the computed delta_z (direction -1); - // scale == -1 uses -delta_z (direction +1). Either or both may yield an RCS bound. - for (const f_t scale : {1.0, -1.0}) { - // Maximum dual step-length alpha that keeps dual feasibility on this ray. - // zl_j + alpha * delta_zN_j >= 0 for nonbasic j on lower bound - // zu_j + alpha * delta_zN_j <= 0 for nonbasic j on upper bound - f_t alpha = inf; - for (i_t jj : delta_z_indices) { - if (vstatus[jj] == variable_status_t::NONBASIC_FIXED) { continue; } - const f_t dz = scale * delta_z[jj]; - if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && dz < -zero_tol) { - const f_t ratio = std::max((-harris_tol - soln.z[jj]) / dz, 0.0); - if (ratio < alpha) { alpha = ratio; } - } - if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && dz > zero_tol) { - const f_t ratio = std::max((harris_tol - soln.z[jj]) / dz, 0.0); - if (ratio < alpha) { alpha = ratio; } - } - } - if (alpha == 0.0 || !std::isfinite(alpha)) { continue; } - - // Verify dual feasibility of the new point z_new = z + alpha * scale * delta_z - // For NONBASIC_LOWER: z_new[jj] >= -dual_tol - // For NONBASIC_UPPER: z_new[jj] <= dual_tol - { - f_t max_initial_dual_infeas = 0.0; - f_t max_dual_infeas = 0.0; - f_t worst_old_z = 0.0; - f_t worst_delta_z = 0.0; - f_t worst_step = 0.0; - f_t worst_new_z = 0.0; - i_t num_initial_dual_infeas = 0; - i_t num_dual_infeas = 0; - i_t worst_j = -1; - for (i_t jj : delta_z_indices) { - if (vstatus[jj] == variable_status_t::NONBASIC_FIXED) { continue; } - const f_t old_zj = soln.z[jj]; - const f_t step = alpha * scale * delta_z[jj]; - const f_t new_zj = old_zj + step; - const bool initially_infeasible = - (vstatus[jj] == variable_status_t::NONBASIC_LOWER && old_zj < -settings_.dual_tol) || - (vstatus[jj] == variable_status_t::NONBASIC_UPPER && old_zj > settings_.dual_tol); - if (initially_infeasible) { - num_initial_dual_infeas++; - max_initial_dual_infeas = std::max(max_initial_dual_infeas, std::abs(old_zj)); - } - if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && new_zj < -settings_.dual_tol) { - num_dual_infeas++; - if (std::abs(new_zj) > max_dual_infeas) { - max_dual_infeas = std::abs(new_zj); - worst_j = jj; - worst_old_z = old_zj; - worst_delta_z = scale * delta_z[jj]; - worst_step = step; - worst_new_z = new_zj; - } - } - if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && new_zj > settings_.dual_tol) { - num_dual_infeas++; - if (std::abs(new_zj) > max_dual_infeas) { - max_dual_infeas = std::abs(new_zj); - worst_j = jj; - worst_old_z = old_zj; - worst_delta_z = scale * delta_z[jj]; - worst_step = step; - worst_new_z = new_zj; - } - } - } - // Also check the leaving variable itself - const f_t new_zj_leaving = soln.z[j] + alpha * scale * delta_z[j]; - if (num_dual_infeas > 0) { - settings_.log.printf( - "WARNING pivot_to_improve_rc: dual infeasibility after step! " - "var=%d alpha=%.6e scale=%.0f initial_num_infeas=%d " - "initial_max_infeas=%.6e num_infeas=%d max_infeas=%.6e worst_j=%d " - "worst_status=%d old_z=%.16e delta_z=%.16e step=%.16e new_z=%.16e " - "new_rc_leaving=%.6e\n", - j, - alpha, - scale, - num_initial_dual_infeas, - max_initial_dual_infeas, - num_dual_infeas, - max_dual_infeas, - worst_j, - static_cast(vstatus[worst_j]), - worst_old_z, - worst_delta_z, - worst_step, - worst_new_z, - new_zj_leaving); - } - } - - // Claim: We don't actually need to take a pivot if all we want to do is add a bound - // coming from reduced cost strengthening - const f_t new_reduced_cost = soln.z[j] + alpha * scale * delta_z[j]; - - // x_j <= l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] - // Let u_tilde_j = u_j - epsilon, so that floor(u_tilde_j) = u_j - 1 - // We want to solve for want the incumbent objective needs to be to make - // x_j <= u_tilde_j - // This means l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] <= - // u_tilde_j Or equivalently, incumbent_objective <= relaxation_objective + - // reduced_costs[j] * (u_tilde_j - l_j) when reduced_costs[j] > 0 - if (at_lower && lower_j > -inf && new_reduced_cost > threshold) { - const f_t u_tilde_j = var_types_[j] == variable_type_t::INTEGER - ? upper_j - tol - : std::max(upper_j - 1.0, lower_j); - const f_t bound_j = - var_types_[j] == variable_type_t::INTEGER ? std::floor(u_tilde_j) : u_tilde_j; - const f_t diff = u_tilde_j - lower_j; - const f_t objective_j = relaxation_objective + diff * new_reduced_cost; - if (((var_types_[j] == variable_type_t::INTEGER && bound_j == upper_j - 1.0) || - var_types_[j] != variable_type_t::INTEGER) && - std::isfinite(objective_j) && std::isfinite(bound_j)) { - i_t info = reduced_cost_bounds.add_upper_bound(j, objective_j, bound_j); - if (info > 0) { num_bounds_added++; } - // settings_.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. - // Info %d\n", objective_j, bound_j, j, info); - } - } - - // x_j >= u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] when - // reduced_costs[j] < 0 Let l_tilde_j = l_j + epsilon, so that ceil(l_tilde_j) = l_j + 1 We - // want to solve for want the incumbent objective needs to be to make x_j >= l_tilde_j This - // means u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j Or - // equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * - // (l_tilde_j - u_j) when reduced_costs[j] < 0 - if (at_upper && upper_j < inf && new_reduced_cost < -threshold) { - const f_t l_tilde_j = var_types_[j] == variable_type_t::INTEGER - ? lower_j + tol - : std::min(lower_j + 1.0, upper_j); - const f_t bound_j = - var_types_[j] == variable_type_t::INTEGER ? std::ceil(l_tilde_j) : l_tilde_j; - const f_t diff = l_tilde_j - upper_j; - const f_t objective_j = relaxation_objective + diff * new_reduced_cost; - if (((var_types_[j] == variable_type_t::INTEGER && bound_j == lower_j + 1.0) || - var_types_[j] != variable_type_t::INTEGER) && - std::isfinite(objective_j) && std::isfinite(bound_j)) { - i_t info = reduced_cost_bounds.add_lower_bound(j, objective_j, bound_j); - if (info > 0) { num_bounds_added++; } - // settings_.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. - // Info %d\n", objective_j, bound_j, j, info); - } - } - } - - // Clear arrays for next iteration - for (i_t k : delta_z_indices) { - delta_z_mark[k] = 0; - delta_z[k] = 0.0; - } - delta_z[leaving_index] = 0.0; - delta_z_indices.clear(); - for (i_t k : delta_y_sparse.i) { - delta_y[k] = 0.0; - } - } - settings_.log.printf("Added %d bounds for reduced cost strengthening\n", num_bounds_added); - settings_.log.printf( - "RCS timing end: candidates=%d bounds=%d total=%.6f btran=%.6f reduced_cost_update=%.6f " - "elapsed=%.6f\n", - num_degenerate_integer, - num_bounds_added, - toc(strengthening_start), - btran_time, - reduced_cost_update_time, - toc(exploration_stats_.start_time)); -} - template mip_status_t branch_and_bound_t::solve(mip_solution_t& solution) { @@ -4369,13 +4075,15 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut upper_bound_.load()); if (settings_.primal_degenerate_pivots != 0) { pivot_to_improve_reduced_cost_strengthening(original_lp_, + settings_, basic_list, nonbasic_list, root_vstatus_, root_relax_soln_, basis_update, - num_fractional, - fractional, + var_types_, + Arow_, + exploration_stats_.start_time, root_objective_, reduced_cost_bounds); } diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index b05bdfe947..37e6cc6343 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -356,18 +356,6 @@ class branch_and_bound_t { omp_atomic_t integer_pivots_{0}; - void pivot_to_improve_reduced_cost_strengthening( - const simplex::lp_problem_t& lp, - const std::vector& basic_list, - const std::vector& nonbasic_list, - const std::vector& vstatus, - const simplex::lp_solution_t& soln, - const simplex::basis_update_mpf_t& basis_update, - i_t num_fractional, - const std::vector& fractional, - f_t relaxation_objective, - reduced_cost_bounds_t& reduced_cost_bounds); - // Repairs low-quality solutions from the heuristics, if it is applicable. void repair_heuristic_solutions(); diff --git a/cpp/src/branch_and_bound/degenerate_pivots.cpp b/cpp/src/branch_and_bound/degenerate_pivots.cpp index e47e2ab64a..4c48e009ab 100644 --- a/cpp/src/branch_and_bound/degenerate_pivots.cpp +++ b/cpp/src/branch_and_bound/degenerate_pivots.cpp @@ -10,12 +10,14 @@ #include #include +#include #include #include #include #include #include +#include namespace cuopt::mathematical_optimization::mip { @@ -42,6 +44,303 @@ bool check_for_dual_degeneracy(const simplex::lp_solution_t& solution, return !zero_reduced_costs_vars.empty(); } +template +void pivot_to_improve_reduced_cost_strengthening( + const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const std::vector& basic_list, + const std::vector& nonbasic_list, + const std::vector& vstatus, + const simplex::lp_solution_t& soln, + const simplex::basis_update_mpf_t& basis_update, + const std::vector& var_types, + const csr_matrix_t& Arow, + const f_t start_time, + const f_t relaxation_objective, + reduced_cost_bounds_t& reduced_cost_bounds) +{ + const double strengthening_start = tic(); + double btran_time = 0.0; + double reduced_cost_update_time = 0.0; + // Count primal degenerate basic variables + i_t num_degenerate = 0; + i_t num_degenerate_continuous = 0; + i_t num_degenerate_integer = 0; + std::vector degenerate_integer_list; + degenerate_integer_list.reserve(lp.num_rows); + for (i_t k = 0; k < lp.num_rows; k++) { + const i_t j = basic_list[k]; + const f_t slack_to_lower = soln.x[j] - lp.lower[j]; + const f_t slack_to_upper = lp.upper[j] - soln.x[j]; + if (slack_to_lower <= settings.primal_tol || slack_to_upper <= settings.primal_tol) { + num_degenerate++; + if (var_types[j] == variable_type_t::INTEGER) { + num_degenerate_integer++; + degenerate_integer_list.push_back(j); + } else { + num_degenerate_continuous++; + } + } + } + + if (num_degenerate_integer == 0) return; + + settings.log.printf( + "RCS timing start: candidates=%d elapsed=%.6f\n", num_degenerate_integer, toc(start_time)); + std::vector variable_to_basic_position(lp.num_cols, -1); + for (i_t k = 0; k < lp.num_rows; k++) { + variable_to_basic_position[basic_list[k]] = k; + } + // The basis stays fixed across candidates; preserve Arow's ordering for cut generation. + csr_matrix_t local_Arow = Arow; + std::vector nonbasic_end(lp.num_rows); + simplex::compute_initial_nonbasic_end(variable_to_basic_position, local_Arow, nonbasic_end); + std::vector delta_y(lp.num_rows, 0); + std::vector delta_z(lp.num_cols, 0); + std::vector delta_z_mark(lp.num_cols, 0); + std::vector delta_z_indices; + delta_z_indices.reserve(lp.num_cols); + + f_t work_estimate = 0; + const f_t threshold = 100.0 * settings.integer_tol; + const f_t tol = 1e-2; + const f_t zero_tol = settings.zero_tol; + const f_t harris_tol = settings.dual_tol / 10; + + i_t num_bounds_added = 0; + for (i_t j : degenerate_integer_list) { + // x_j is a degenerate integer basic variable. + // We would like a dual-feasible point where x_j is nonbasic with a nonzero + // reduced cost that may be used for reduced cost strengthening. + // We do not need to take the pivot; a dual step along either ray is enough. + // + // One BTRAN: B^T * delta_y = e_p, which matches direction == -1 in + // compute_reduced_cost_update (B^T * delta_y = -direction * e_p). + // The opposite direction is the negated (delta_y, delta_z) ray. + const i_t leaving_index = j; + const i_t p = variable_to_basic_position[j]; + if (p == -1) continue; + + sparse_vector_t ep(lp.num_rows, 1); + ep.i[0] = p; + ep.x[0] = 1.0; + sparse_vector_t delta_y_sparse; + sparse_vector_t UTsol_sparse; + const double btran_start = tic(); + basis_update.b_transpose_solve(ep, delta_y_sparse, UTsol_sparse); + btran_time += toc(btran_start); + + // delta_zN = -N^T * delta_y, delta_z[leaving] = -1 + const double reduced_cost_update_start = tic(); + i_t delta_y_nz0 = 0; + for (const f_t value : delta_y_sparse.x) { + if (std::abs(value) > 1e-12) { delta_y_nz0++; } + } + work_estimate += delta_y_sparse.i.size(); + const f_t delta_y_nz_percentage = delta_y_nz0 / static_cast(lp.num_rows) * 100.0; + if (delta_y_nz_percentage <= 30.0) { + simplex::compute_delta_z(local_Arow, + delta_y_sparse, + leaving_index, + /*direction=*/-1, + nonbasic_end, + delta_z_mark, + delta_z_indices, + delta_z, + work_estimate); + } else { + delta_y_sparse.to_dense(delta_y); + work_estimate += delta_y.size(); + simplex::compute_reduced_cost_update(lp, + basic_list, + nonbasic_list, + delta_y, + leaving_index, + /*direction=*/-1, + delta_z_mark, + delta_z_indices, + delta_z, + work_estimate); + } + reduced_cost_update_time += toc(reduced_cost_update_start); + + const f_t lower_j = lp.lower[j]; + const f_t upper_j = lp.upper[j]; + const bool at_lower = soln.x[j] - lower_j <= settings.primal_tol; + const bool at_upper = upper_j - soln.x[j] <= settings.primal_tol; + + // Try both dual rays. scale == +1 uses the computed delta_z (direction -1); + // scale == -1 uses -delta_z (direction +1). Either or both may yield an RCS bound. + for (const f_t scale : {1.0, -1.0}) { + // Maximum dual step-length alpha that keeps dual feasibility on this ray. + // zl_j + alpha * delta_zN_j >= 0 for nonbasic j on lower bound + // zu_j + alpha * delta_zN_j <= 0 for nonbasic j on upper bound + f_t alpha = inf; + for (i_t jj : delta_z_indices) { + if (vstatus[jj] == variable_status_t::NONBASIC_FIXED) { continue; } + const f_t dz = scale * delta_z[jj]; + if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && dz < -zero_tol) { + const f_t ratio = std::max((-harris_tol - soln.z[jj]) / dz, 0.0); + if (ratio < alpha) { alpha = ratio; } + } + if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && dz > zero_tol) { + const f_t ratio = std::max((harris_tol - soln.z[jj]) / dz, 0.0); + if (ratio < alpha) { alpha = ratio; } + } + } + if (alpha == 0.0 || !std::isfinite(alpha)) { continue; } + + // Verify dual feasibility of the new point z_new = z + alpha * scale * delta_z + // For NONBASIC_LOWER: z_new[jj] >= -dual_tol + // For NONBASIC_UPPER: z_new[jj] <= dual_tol + { + f_t max_initial_dual_infeas = 0.0; + f_t max_dual_infeas = 0.0; + f_t worst_old_z = 0.0; + f_t worst_delta_z = 0.0; + f_t worst_step = 0.0; + f_t worst_new_z = 0.0; + i_t num_initial_dual_infeas = 0; + i_t num_dual_infeas = 0; + i_t worst_j = -1; + for (i_t jj : delta_z_indices) { + if (vstatus[jj] == variable_status_t::NONBASIC_FIXED) { continue; } + const f_t old_zj = soln.z[jj]; + const f_t step = alpha * scale * delta_z[jj]; + const f_t new_zj = old_zj + step; + const bool initially_infeasible = + (vstatus[jj] == variable_status_t::NONBASIC_LOWER && old_zj < -settings.dual_tol) || + (vstatus[jj] == variable_status_t::NONBASIC_UPPER && old_zj > settings.dual_tol); + if (initially_infeasible) { + num_initial_dual_infeas++; + max_initial_dual_infeas = std::max(max_initial_dual_infeas, std::abs(old_zj)); + } + if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && new_zj < -settings.dual_tol) { + num_dual_infeas++; + if (std::abs(new_zj) > max_dual_infeas) { + max_dual_infeas = std::abs(new_zj); + worst_j = jj; + worst_old_z = old_zj; + worst_delta_z = scale * delta_z[jj]; + worst_step = step; + worst_new_z = new_zj; + } + } + if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && new_zj > settings.dual_tol) { + num_dual_infeas++; + if (std::abs(new_zj) > max_dual_infeas) { + max_dual_infeas = std::abs(new_zj); + worst_j = jj; + worst_old_z = old_zj; + worst_delta_z = scale * delta_z[jj]; + worst_step = step; + worst_new_z = new_zj; + } + } + } + // Also check the leaving variable itself + const f_t new_zj_leaving = soln.z[j] + alpha * scale * delta_z[j]; + if (num_dual_infeas > 0) { + settings.log.printf( + "WARNING pivot_to_improve_rc: dual infeasibility after step! " + "var=%d alpha=%.6e scale=%.0f initial_num_infeas=%d " + "initial_max_infeas=%.6e num_infeas=%d max_infeas=%.6e worst_j=%d " + "worst_status=%d old_z=%.16e delta_z=%.16e step=%.16e new_z=%.16e " + "new_rc_leaving=%.6e\n", + j, + alpha, + scale, + num_initial_dual_infeas, + max_initial_dual_infeas, + num_dual_infeas, + max_dual_infeas, + worst_j, + static_cast(vstatus[worst_j]), + worst_old_z, + worst_delta_z, + worst_step, + worst_new_z, + new_zj_leaving); + } + } + + // Claim: We don't actually need to take a pivot if all we want to do is add a bound + // coming from reduced cost strengthening + const f_t new_reduced_cost = soln.z[j] + alpha * scale * delta_z[j]; + + // x_j <= l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] + // Let u_tilde_j = u_j - epsilon, so that floor(u_tilde_j) = u_j - 1 + // We want to solve for want the incumbent objective needs to be to make + // x_j <= u_tilde_j + // This means l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] <= + // u_tilde_j Or equivalently, incumbent_objective <= relaxation_objective + + // reduced_costs[j] * (u_tilde_j - l_j) when reduced_costs[j] > 0 + if (at_lower && lower_j > -inf && new_reduced_cost > threshold) { + const f_t u_tilde_j = var_types[j] == variable_type_t::INTEGER + ? upper_j - tol + : std::max(upper_j - 1.0, lower_j); + const f_t bound_j = + var_types[j] == variable_type_t::INTEGER ? std::floor(u_tilde_j) : u_tilde_j; + const f_t diff = u_tilde_j - lower_j; + const f_t objective_j = relaxation_objective + diff * new_reduced_cost; + if (((var_types[j] == variable_type_t::INTEGER && bound_j == upper_j - 1.0) || + var_types[j] != variable_type_t::INTEGER) && + std::isfinite(objective_j) && std::isfinite(bound_j)) { + i_t info = reduced_cost_bounds.add_upper_bound(j, objective_j, bound_j); + if (info > 0) { num_bounds_added++; } + // settings.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. + // Info %d\n", objective_j, bound_j, j, info); + } + } + + // x_j >= u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] when + // reduced_costs[j] < 0 Let l_tilde_j = l_j + epsilon, so that ceil(l_tilde_j) = l_j + 1 We + // want to solve for want the incumbent objective needs to be to make x_j >= l_tilde_j This + // means u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j Or + // equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * + // (l_tilde_j - u_j) when reduced_costs[j] < 0 + if (at_upper && upper_j < inf && new_reduced_cost < -threshold) { + const f_t l_tilde_j = var_types[j] == variable_type_t::INTEGER + ? lower_j + tol + : std::min(lower_j + 1.0, upper_j); + const f_t bound_j = + var_types[j] == variable_type_t::INTEGER ? std::ceil(l_tilde_j) : l_tilde_j; + const f_t diff = l_tilde_j - upper_j; + const f_t objective_j = relaxation_objective + diff * new_reduced_cost; + if (((var_types[j] == variable_type_t::INTEGER && bound_j == lower_j + 1.0) || + var_types[j] != variable_type_t::INTEGER) && + std::isfinite(objective_j) && std::isfinite(bound_j)) { + i_t info = reduced_cost_bounds.add_lower_bound(j, objective_j, bound_j); + if (info > 0) { num_bounds_added++; } + // settings.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. + // Info %d\n", objective_j, bound_j, j, info); + } + } + } + + // Clear arrays for next iteration + for (i_t k : delta_z_indices) { + delta_z_mark[k] = 0; + delta_z[k] = 0.0; + } + delta_z[leaving_index] = 0.0; + delta_z_indices.clear(); + for (i_t k : delta_y_sparse.i) { + delta_y[k] = 0.0; + } + } + settings.log.printf("Added %d bounds for reduced cost strengthening\n", num_bounds_added); + settings.log.printf( + "RCS timing end: candidates=%d bounds=%d total=%.6f btran=%.6f reduced_cost_update=%.6f " + "elapsed=%.6f\n", + num_degenerate_integer, + num_bounds_added, + toc(strengthening_start), + btran_time, + reduced_cost_update_time, + toc(start_time)); +} + template void dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, const simplex::simplex_solver_settings_t& settings, @@ -1284,6 +1583,20 @@ template int apply_delta_x_for_integer_pivot( simplex::basis_update_mpf_t&, double&); +template void pivot_to_improve_reduced_cost_strengthening( + const simplex::lp_problem_t&, + const simplex::simplex_solver_settings_t&, + const std::vector&, + const std::vector&, + const std::vector&, + const simplex::lp_solution_t&, + const simplex::basis_update_mpf_t&, + const std::vector&, + const csr_matrix_t&, + double, + double, + reduced_cost_bounds_t&); + template void dual_degenerate_feasibility_pump( const simplex::lp_problem_t&, const simplex::simplex_solver_settings_t&, diff --git a/cpp/src/branch_and_bound/degenerate_pivots.hpp b/cpp/src/branch_and_bound/degenerate_pivots.hpp index 90b75ca049..84edbf3475 100644 --- a/cpp/src/branch_and_bound/degenerate_pivots.hpp +++ b/cpp/src/branch_and_bound/degenerate_pivots.hpp @@ -7,10 +7,13 @@ #pragma once +#include + #include #include #include #include +#include #include #include @@ -74,6 +77,21 @@ i_t apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, simplex::basis_update_mpf_t& basis_update, f_t& work_estimate); +template +void pivot_to_improve_reduced_cost_strengthening( + const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const std::vector& basic_list, + const std::vector& nonbasic_list, + const std::vector& vstatus, + const simplex::lp_solution_t& soln, + const simplex::basis_update_mpf_t& basis_update, + const std::vector& var_types, + const csr_matrix_t& Arow, + f_t start_time, + f_t relaxation_objective, + reduced_cost_bounds_t& reduced_cost_bounds); + template void dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, const simplex::simplex_solver_settings_t& settings, From a59a053612587d246b466b37896cbe70794b298c Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Mon, 28 Sep 2026 14:46:02 -0700 Subject: [PATCH 110/113] Refactor to address review comments --- cpp/src/branch_and_bound/branch_and_bound.cpp | 134 +------------- cpp/src/branch_and_bound/branch_and_bound.hpp | 8 - .../branch_and_bound/degenerate_pivots.cpp | 164 +++++++----------- .../branch_and_bound/degenerate_pivots.hpp | 4 +- .../branch_and_bound/reduced_cost_bounds.hpp | 72 ++++++++ 5 files changed, 139 insertions(+), 243 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index b04da0e7c6..91e2e22692 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -450,132 +450,6 @@ void branch_and_bound_t::report(const lp_problem_t& lp, settings_.log.printf("%s\n", log_line.c_str()); } -template -void branch_and_bound_t::update_reduced_cost_bounds( - f_t relaxation_objective, - const std::vector& reduced_costs, - const std::vector& var_status, - reduced_cost_bounds_t& reduced_cost_bounds) -{ - const i_t n = reduced_cost_bounds.num_cols(); - const f_t threshold = 100.0 * settings_.integer_tol; - const f_t tol = 1e-2; - for (i_t j = 0; j < n; ++j) { - if (std::isfinite(reduced_costs[j]) && std::abs(reduced_costs[j]) > threshold && - var_status[j] != variable_status_t::BASIC) { - const f_t lower_j = original_lp_.lower[j]; - const f_t upper_j = original_lp_.upper[j]; - - // x_j <= l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] - // Let u_tilde_j = u_j - epsilon, so that floor(u_tilde_j) = u_j - 1 - // We want to solve for want the incumbent objective needs to be to make - // x_j <= u_tilde_j - // This means l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] <= - // u_tilde_j Or equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * - // (u_tilde_j - l_j) when reduced_costs[j] > 0 - if (lower_j > -inf && reduced_costs[j] > 0) { - const f_t u_tilde_j = var_types_[j] == variable_type_t::INTEGER - ? upper_j - tol - : std::max(upper_j - 1.0, lower_j); - const f_t bound_j = - var_types_[j] == variable_type_t::INTEGER ? std::floor(u_tilde_j) : u_tilde_j; - const f_t diff = u_tilde_j - lower_j; - const f_t objective_j = relaxation_objective + diff * reduced_costs[j]; - if (((var_types_[j] == variable_type_t::INTEGER && bound_j == upper_j - 1.0) || - var_types_[j] != variable_type_t::INTEGER) && - std::isfinite(objective_j) && std::isfinite(bound_j)) { - i_t info = reduced_cost_bounds.add_upper_bound(j, objective_j, bound_j); - // settings_.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. - // Info %d\n", objective_j, bound_j, j, info); - } - } - - // x_j >= u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] when - // reduced_costs[j] < 0 Let l_tilde_j = l_j + epsilon, so that ceil(l_tilde_j) = l_j + 1 We - // want to solve for want the incumbent objective needs to be to make x_j >= l_tilde_j This - // means u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j Or - // equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * (l_tilde_j - // - u_j) when reduced_costs[j] < 0 - if (upper_j < inf && reduced_costs[j] < 0) { - const f_t l_tilde_j = var_types_[j] == variable_type_t::INTEGER - ? lower_j + tol - : std::min(lower_j + 1.0, upper_j); - const f_t bound_j = - var_types_[j] == variable_type_t::INTEGER ? std::ceil(l_tilde_j) : l_tilde_j; - const f_t diff = l_tilde_j - upper_j; - const f_t objective_j = relaxation_objective + diff * reduced_costs[j]; - if (((var_types_[j] == variable_type_t::INTEGER && bound_j == lower_j + 1.0) || - var_types_[j] != variable_type_t::INTEGER) && - std::isfinite(objective_j) && std::isfinite(bound_j)) { - i_t info = reduced_cost_bounds.add_lower_bound(j, objective_j, bound_j); - // settings_.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. - // Info %d\n", objective_j, bound_j, j, info); - } - } - } - } -} - -template -i_t branch_and_bound_t::find_reduced_cost_fixings(f_t upper_bound, - std::vector& lower_bounds, - std::vector& upper_bounds) -{ - std::vector reduced_costs = root_relax_soln_.z; - lower_bounds = original_lp_.lower; - upper_bounds = original_lp_.upper; - std::vector bounds_changed(original_lp_.num_cols, false); - const f_t root_obj = compute_objective(original_lp_, root_relax_soln_.x); - const f_t threshold = 100.0 * settings_.integer_tol; - const f_t weaken = settings_.integer_tol; - const f_t fixed_tol = settings_.fixed_tol; - i_t num_improved = 0; - i_t num_fixed = 0; - i_t num_cols_to_check = reduced_costs.size(); // Reduced costs will be smaller than the original - // problem because we have added slacks for cuts - for (i_t j = 0; j < num_cols_to_check; j++) { - if (std::isfinite(reduced_costs[j]) && std::abs(reduced_costs[j]) > threshold) { - const f_t lower_j = original_lp_.lower[j]; - const f_t upper_j = original_lp_.upper[j]; - const f_t abs_gap = upper_bound - root_obj; - f_t reduced_cost_upper_bound = upper_j; - f_t reduced_cost_lower_bound = lower_j; - if (lower_j > -inf && reduced_costs[j] > 0) { - const f_t new_upper_bound = lower_j + abs_gap / reduced_costs[j]; - reduced_cost_upper_bound = var_types_[j] == variable_type_t::INTEGER - ? std::floor(new_upper_bound + weaken) - : new_upper_bound; - if (reduced_cost_upper_bound < upper_j && var_types_[j] == variable_type_t::INTEGER) { - num_improved++; - upper_bounds[j] = reduced_cost_upper_bound; - bounds_changed[j] = true; - } - } - if (upper_j < inf && reduced_costs[j] < 0) { - const f_t new_lower_bound = upper_j + abs_gap / reduced_costs[j]; - reduced_cost_lower_bound = var_types_[j] == variable_type_t::INTEGER - ? std::ceil(new_lower_bound - weaken) - : new_lower_bound; - if (reduced_cost_lower_bound > lower_j && var_types_[j] == variable_type_t::INTEGER) { - num_improved++; - lower_bounds[j] = reduced_cost_lower_bound; - bounds_changed[j] = true; - } - } - if (var_types_[j] == variable_type_t::INTEGER && - reduced_cost_upper_bound <= reduced_cost_lower_bound + fixed_tol) { - num_fixed++; - } - } - } - - if (num_fixed > 0 || num_improved > 0) { - settings_.log.printf( - "Reduced costs: Found %d improved bounds and %d fixed variables\n", num_improved, num_fixed); - } - return num_fixed; -} - template void branch_and_bound_t::update_user_bound(const lp_problem_t& lp, f_t lower_bound) @@ -3737,8 +3611,8 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t= 1) { - update_reduced_cost_bounds( - root_objective_, root_relax_soln_.z, root_vstatus_, reduced_cost_bounds); + reduced_cost_bounds.update_reduced_cost_bounds( + original_lp_, settings_, var_types_, root_objective_, root_relax_soln_.z, root_vstatus_); settings_.log.printf("New reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); @@ -4070,8 +3944,8 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut lower_bound_numerical_ = inf; reduced_cost_bounds_t reduced_cost_bounds(original_lp_.num_cols); - update_reduced_cost_bounds( - root_objective_, root_relax_soln_.z, root_vstatus_, reduced_cost_bounds); + reduced_cost_bounds.update_reduced_cost_bounds( + original_lp_, settings_, var_types_, root_objective_, root_relax_soln_.z, root_vstatus_); settings_.log.printf("New reduced cost objective %e (current %e)\n", reduced_cost_bounds.get_max_objective(), upper_bound_.load()); diff --git a/cpp/src/branch_and_bound/branch_and_bound.hpp b/cpp/src/branch_and_bound/branch_and_bound.hpp index 37e6cc6343..c316aa0b36 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.hpp +++ b/cpp/src/branch_and_bound/branch_and_bound.hpp @@ -180,14 +180,6 @@ class branch_and_bound_t { std::vector& edge_norms, f_t& work_estimate); - void update_reduced_cost_bounds(f_t relaxation_objective, - const std::vector& reduced_costs, - const std::vector& var_status, - reduced_cost_bounds_t& reduced_cost_bounds); - i_t find_reduced_cost_fixings(f_t upper_bound, - std::vector& lower_bounds, - std::vector& upper_bounds); - // The main entry routine. Returns the solver status and populates solution with the incumbent. mip_status_t solve(simplex::mip_solution_t& solution); diff --git a/cpp/src/branch_and_bound/degenerate_pivots.cpp b/cpp/src/branch_and_bound/degenerate_pivots.cpp index 4c48e009ab..bec0484692 100644 --- a/cpp/src/branch_and_bound/degenerate_pivots.cpp +++ b/cpp/src/branch_and_bound/degenerate_pivots.cpp @@ -408,15 +408,14 @@ void dual_degenerate_feasibility_pump(const simplex::lp_problem_t& lp, std::vector b_reduced = lp.rhs; for (i_t j = 0; j < lp.num_cols; j++) { if (vstatus[j] == variable_status_t::BASIC || std::abs(soln.z[j]) <= settings.tight_tol) { - // PASS - } else { - const i_t col_start = lp.A.col_start[j]; - const i_t col_end = lp.A.col_start[j + 1]; - for (i_t p = col_start; p < col_end; p++) { - const i_t i = lp.A.i[p]; - const f_t value = lp.A.x[p]; - b_reduced[i] -= value * soln.x[j]; - } + continue; + } + const i_t col_start = lp.A.col_start[j]; + const i_t col_end = lp.A.col_start[j + 1]; + for (i_t p = col_start; p < col_end; p++) { + const i_t i = lp.A.i[p]; + const f_t value = lp.A.x[p]; + b_reduced[i] -= value * soln.x[j]; } } lp_reduced.rhs = b_reduced; @@ -813,15 +812,7 @@ i_t apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, bool binding_integer = leaving_index != -1 && is_fractional(solution.x[leaving_index], var_types[leaving_index], settings.integer_tol); - if (!binding_integer) { - if (leaving_index == -1) { - return -4; // unbounded or entering hit its own bound - } else if (var_types[leaving_index] != variable_type_t::INTEGER) { - return -5; // continuous variable won ratio test - } else { - return -6; // integer variable won but it's not fractional (already at integer value) - } - } + if (!binding_integer) { return 1; } std::vector test_x = solution.x; i_t integer_destroyed = 0; @@ -837,7 +828,7 @@ i_t apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, } } // Require a strict net decrease in fractional integers. - if (integer_destroyed >= 0) { return -2; } + if (integer_destroyed >= 0) { return 1; } if (utilde_sparse.i.empty()) { // Recover B^{-1} abar from the direction before changing the basis: @@ -885,40 +876,27 @@ i_t apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, basis_update.b_transpose_solve(es_sparse, solution_sparse, UTsol_sparse); const i_t recommend_refactor = basis_update.update(utilde_sparse, UTsol_sparse, basic_leaving); if (recommend_refactor == 1) { - csc_matrix_t L(m, m, 1); - csc_matrix_t U(m, m, 1); - std::vector pinv(m); - std::vector p(m); - std::vector q(m); - std::vector deficient; - std::vector slacks_needed; - f_t factorize_work_estimate = 0.0; - const i_t rank = factorize_basis(lp.A, - settings, - basic_list, - start_time, - L, - U, - p, - pinv, - q, - deficient, - slacks_needed, - factorize_work_estimate); - if (rank == CONCURRENT_HALT_RETURN || rank == TIME_LIMIT_RETURN) { return -3; } - if (rank < 0 || rank != lp.num_rows) { return -3; } - simplex::reorder_basic_list(q, basic_list); + i_t deficient_repaired = 0; + const i_t refactor_status = basis_update.refactor_basis(lp.A, + settings, + lp.lower, + lp.upper, + start_time, + basic_list, + nonbasic_list, + vstatus, + deficient_repaired); + if (refactor_status != 0 || deficient_repaired > 0) { return -1; } for (i_t k = 0; k < m; ++k) { variable_to_basic[basic_list[k]] = k; } - basis_update.reset(L, U, p); } return 0; } template -void fast_slack_integer_pivots(const simplex::lp_problem_t& lp, +bool fast_slack_integer_pivots(const simplex::lp_problem_t& lp, const simplex::simplex_solver_settings_t& settings, const std::vector& fractional, const std::vector& row_to_slack, @@ -1076,8 +1054,7 @@ void fast_slack_integer_pivots(const simplex::lp_problem_t& lp, soln, basis_update, work_estimate); - // apply_delta_x_for_integer_pivot only mutates vstatus when the pivot actually fires, - // so entering_index transitioning to BASIC is a reliable success signal. + if (error == -1) { return false; } if (!error && settings.inside_mip < 2) { settings.log.printf( "Fast candidate pivot succeeded: j=%d entering slack=%d row=%d\n", j, entering_index, row); @@ -1097,6 +1074,7 @@ void fast_slack_integer_pivots(const simplex::lp_problem_t& lp, num_candidates, toc(loop_start)); } + return true; } template @@ -1187,21 +1165,23 @@ i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, for (i_t k = 0; k < lp.num_rows; k++) { variable_to_basic[basic_list_copy[k]] = k; } - fast_slack_integer_pivots(lp, - settings, - fractional, - row_to_slack, - solution, - var_types, - start_time, - basic_list_copy, - nonbasic_list_copy, - nonbasic_index, - variable_to_basic, - vstatus_copy, - soln_copy, - basis_update_copy, - work_estimate); + if (!fast_slack_integer_pivots(lp, + settings, + fractional, + row_to_slack, + solution, + var_types, + start_time, + basic_list_copy, + nonbasic_list_copy, + nonbasic_index, + variable_to_basic, + vstatus_copy, + soln_copy, + basis_update_copy, + work_estimate)) { + return 0; + } std::vector work_list = fractional; @@ -1216,21 +1196,17 @@ i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, // Track which entering variables are actually tried (to detect duplication) std::vector entering_tried_count(lp.num_cols, 0); - i_t worklist_total_processed = 0; - i_t worklist_skipped = 0; - i_t worklist_btran_done = 0; - i_t worklist_ftran_done = 0; - i_t worklist_pivots_succeeded = 0; - i_t worklist_readded = 0; - f_t worklist_btran_time = 0.0; - f_t worklist_dot_time = 0.0; - f_t worklist_ftran_time = 0.0; - i_t worklist_no_candidates = 0; // target had no nonzero dot_q - i_t worklist_ratio_test_fail = 0; // ratio test didn't pick a fractional integer (error -1) - i_t worklist_net_increase_fail = 0; // pivot would net-increase fractionals (error -2) - i_t worklist_unbounded = 0; // entering hit its own bound or unbounded (error -4) - i_t worklist_continuous_won = 0; // continuous variable won ratio test (error -5) - i_t worklist_nonfrac_int_won = 0; // non-fractional integer won ratio test (error -6) + i_t worklist_total_processed = 0; + i_t worklist_skipped = 0; + i_t worklist_btran_done = 0; + i_t worklist_ftran_done = 0; + i_t worklist_pivots_succeeded = 0; + i_t worklist_readded = 0; + f_t worklist_btran_time = 0.0; + f_t worklist_dot_time = 0.0; + f_t worklist_ftran_time = 0.0; + i_t worklist_no_candidates = 0; // target had no nonzero dot_q + i_t candidates_rejected = 0; f_t worklist_loop_start = tic(); f_t worklist_last_log = tic(); @@ -1395,19 +1371,8 @@ i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, basis_update_copy, work_estimate); - if (error == -2) { worklist_net_increase_fail++; } - if (error == -4) { - worklist_unbounded++; - worklist_ratio_test_fail++; - } - if (error == -5) { - worklist_continuous_won++; - worklist_ratio_test_fail++; - } - if (error == -6) { - worklist_nonfrac_int_won++; - worklist_ratio_test_fail++; - } + if (error == -1) { return 0; } + if (error == 1) { candidates_rejected++; } if (!error) { worklist_pivots_succeeded++; @@ -1435,16 +1400,12 @@ i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, if (toc(worklist_last_log) > 1.0) { if (settings.inside_mip < 2) { settings.log.printf( - "Worklist progress: %d/%d processed, %d pivots, %d ratio_fail (unb=%d cont=%d nfint=%d), " - "%d net_inc_fail, %d no_cand, %.2f seconds\n", + "Worklist progress: %d/%d processed, %d pivots, %d candidates_rejected, " + "%d no_cand, %.2f seconds\n", worklist_total_processed, static_cast(fractional.size()), worklist_pivots_succeeded, - worklist_ratio_test_fail, - worklist_unbounded, - worklist_continuous_won, - worklist_nonfrac_int_won, - worklist_net_increase_fail, + candidates_rejected, worklist_no_candidates, toc(worklist_loop_start)); } @@ -1482,8 +1443,7 @@ i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, settings.log.printf( "Worklist stats: processed=%d skipped=%d btran=%d ftran=%d pivots=%d readded=%d " "btran_time=%.2f dot_time=%.2f ftran_time=%.2f zero_rc_vars=%d " - "no_candidates=%d ratio_test_fail=%d (unbounded=%d continuous_won=%d nonfrac_int_won=%d) " - "net_increase_fail=%d\n", + "no_candidates=%d candidates_rejected=%d\n", worklist_total_processed, worklist_skipped, worklist_btran_done, @@ -1495,11 +1455,7 @@ i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, worklist_ftran_time, num_zero_reduced_costs_vars, worklist_no_candidates, - worklist_ratio_test_fail, - worklist_unbounded, - worklist_continuous_won, - worklist_nonfrac_int_won, - worklist_net_increase_fail); + candidates_rejected); } std::vector new_fractional; @@ -1533,7 +1489,7 @@ template bool check_for_dual_degeneracy( std::vector&, std::vector&); -template void fast_slack_integer_pivots( +template bool fast_slack_integer_pivots( const simplex::lp_problem_t&, const simplex::simplex_solver_settings_t&, const std::vector&, diff --git a/cpp/src/branch_and_bound/degenerate_pivots.hpp b/cpp/src/branch_and_bound/degenerate_pivots.hpp index 84edbf3475..91bd57bc96 100644 --- a/cpp/src/branch_and_bound/degenerate_pivots.hpp +++ b/cpp/src/branch_and_bound/degenerate_pivots.hpp @@ -28,7 +28,7 @@ bool check_for_dual_degeneracy(const simplex::lp_solution_t& solution, std::vector& zero_reduced_costs_vars_nonbasic_index); template -void fast_slack_integer_pivots(const simplex::lp_problem_t& lp, +bool fast_slack_integer_pivots(const simplex::lp_problem_t& lp, const simplex::simplex_solver_settings_t& settings, const std::vector& fractional, const std::vector& row_to_slack, @@ -58,6 +58,8 @@ i_t pivot_out_integer_variables(const simplex::lp_problem_t& lp, i_t& num_fractional, std::vector& fractional); +// Returns 0 on success, 1 on candidate rejection, or -1 when the trial state is invalid +// and the caller must abort the trial. template i_t apply_delta_x_for_integer_pivot(const simplex::lp_problem_t& lp, const simplex::simplex_solver_settings_t& settings, diff --git a/cpp/src/branch_and_bound/reduced_cost_bounds.hpp b/cpp/src/branch_and_bound/reduced_cost_bounds.hpp index 47483a143f..2ec9de1079 100644 --- a/cpp/src/branch_and_bound/reduced_cost_bounds.hpp +++ b/cpp/src/branch_and_bound/reduced_cost_bounds.hpp @@ -7,8 +7,13 @@ #pragma once +#include +#include +#include #include +#include +#include #include #include @@ -93,6 +98,73 @@ class reduced_cost_bounds_t { } } + void update_reduced_cost_bounds(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const std::vector& var_types, + f_t relaxation_objective, + const std::vector& reduced_costs, + const std::vector& var_status) + { + const i_t n = num_cols(); + const f_t threshold = 100.0 * settings.integer_tol; + const f_t tol = 1e-2; + for (i_t j = 0; j < n; ++j) { + if (std::isfinite(reduced_costs[j]) && std::abs(reduced_costs[j]) > threshold && + var_status[j] != simplex::variable_status_t::BASIC) { + const f_t lower_j = lp.lower[j]; + const f_t upper_j = lp.upper[j]; + + // x_j <= l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] + // Let u_tilde_j = u_j - epsilon, so that floor(u_tilde_j) = u_j - 1 + // We want to solve for what the incumbent objective needs to be to make + // x_j <= u_tilde_j + // This means l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] <= + // u_tilde_j Or equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] + // * (u_tilde_j - l_j) when reduced_costs[j] > 0 + if (lower_j > -inf && reduced_costs[j] > 0) { + const f_t u_tilde_j = var_types[j] == simplex::variable_type_t::INTEGER + ? upper_j - tol + : std::max(upper_j - 1.0, lower_j); + const f_t bound_j = + var_types[j] == simplex::variable_type_t::INTEGER ? std::floor(u_tilde_j) : u_tilde_j; + const f_t diff = u_tilde_j - lower_j; + const f_t objective_j = relaxation_objective + diff * reduced_costs[j]; + if (((var_types[j] == simplex::variable_type_t::INTEGER && bound_j == upper_j - 1.0) || + var_types[j] != simplex::variable_type_t::INTEGER) && + std::isfinite(objective_j) && std::isfinite(bound_j)) { + i_t info = add_upper_bound(j, objective_j, bound_j); + // settings.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. + // Info %d\n", objective_j, bound_j, j, info); + } + } + + // x_j >= u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] when + // reduced_costs[j] < 0 Let l_tilde_j = l_j + epsilon, so that ceil(l_tilde_j) = l_j + 1 We + // want to solve for what the incumbent objective needs to be to make x_j >= l_tilde_j This + // means u_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] >= l_tilde_j + // Or equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * + // (l_tilde_j + // - u_j) when reduced_costs[j] < 0 + if (upper_j < inf && reduced_costs[j] < 0) { + const f_t l_tilde_j = var_types[j] == simplex::variable_type_t::INTEGER + ? lower_j + tol + : std::min(lower_j + 1.0, upper_j); + const f_t bound_j = + var_types[j] == simplex::variable_type_t::INTEGER ? std::ceil(l_tilde_j) : l_tilde_j; + const f_t diff = l_tilde_j - upper_j; + const f_t objective_j = relaxation_objective + diff * reduced_costs[j]; + if (((var_types[j] == simplex::variable_type_t::INTEGER && bound_j == lower_j + 1.0) || + var_types[j] != simplex::variable_type_t::INTEGER) && + std::isfinite(objective_j) && std::isfinite(bound_j)) { + i_t info = add_lower_bound(j, objective_j, bound_j); + // settings.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. + // Info %d\n", objective_j, bound_j, j, info); + } + } + } + } + } + i_t update_bounds_from_new_incumbent(f_t incumbent_objective, const std::vector& var_types, std::vector& lower_bounds, From bafd54e314fcd6cd6e2b0ee4ddc198fd164c494b Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Mon, 28 Sep 2026 14:59:37 -0700 Subject: [PATCH 111/113] Fix test failures --- cpp/src/branch_and_bound/branch_and_bound.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 91e2e22692..16ba38948c 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -1645,7 +1645,7 @@ dual_status_t branch_and_bound_t::solve_node_lp( if (lp_status == dual_status_t::OPTIMAL) { std::vector fractional; i_t num_fractional = - fractional_variables(settings_, worker->leaf_solution.x, var_types_, fractional); + fractional_variables(settings_, worker->leaf_solution.x, worker->var_types, fractional); if (settings_.dual_degenerate_pivots != 0) { auto pivot_settings = settings_; pivot_settings.log = lp_settings.log; @@ -1654,7 +1654,7 @@ dual_status_t branch_and_bound_t::solve_node_lp( i_t num_integer_increased = pivot_out_integer_variables(worker->leaf_problem, pivot_settings, worker->new_slacks, - var_types_, + worker->var_types, exploration_stats_.start_time, worker->basic_list, worker->nonbasic_list, From 1718c29462c5f01655fbec6a78a67d1a0f19ee5a Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Mon, 28 Sep 2026 18:04:59 -0700 Subject: [PATCH 112/113] Apply eligible reduced-cost bounds and restore general-integer tightening --- cpp/src/branch_and_bound/branch_and_bound.cpp | 28 ++++++++--- .../branch_and_bound/reduced_cost_bounds.hpp | 49 +++++++++++++++++-- 2 files changed, 67 insertions(+), 10 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index 571bf0558d..f42d64fc89 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3478,14 +3478,22 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t= 1 && upper_bound_.load() < last_upper_bound) { + if (settings_.reduced_cost_strengthening >= 1 && + (upper_bound_.load() < last_upper_bound || + upper_bound_.load() <= reduced_cost_bounds.get_max_objective())) { mutex_upper_.lock(); last_upper_bound = upper_bound_.load(); std::vector lower_bounds = original_lp_.lower; std::vector upper_bounds = original_lp_.upper; f_t previous_max_objective = reduced_cost_bounds.get_max_objective(); - i_t new_bounds = reduced_cost_bounds.update_bounds_from_new_incumbent( - upper_bound_.load(), var_types_, lower_bounds, upper_bounds); + i_t new_bounds = reduced_cost_bounds.update_bounds_from_new_incumbent(original_lp_, + settings_, + root_relax_soln_, + root_objective_, + upper_bound_.load(), + var_types_, + lower_bounds, + upper_bounds); mutex_upper_.unlock(); mutex_original_lp_.lock(); original_lp_.lower = lower_bounds; @@ -4191,12 +4199,20 @@ mip_status_t branch_and_bound_t::solve(mip_solution_t& solut return solver_status_; } - if (settings_.reduced_cost_strengthening >= 2 && upper_bound_.load() < last_upper_bound) { + if (settings_.reduced_cost_strengthening >= 2 && + (upper_bound_.load() < last_upper_bound || + upper_bound_.load() <= reduced_cost_bounds.get_max_objective())) { std::vector lower_bounds = original_lp_.lower; std::vector upper_bounds = original_lp_.upper; f_t previous_max_objective = reduced_cost_bounds.get_max_objective(); - i_t num_changed = reduced_cost_bounds.update_bounds_from_new_incumbent( - upper_bound_.load(), var_types_, lower_bounds, upper_bounds); + i_t num_changed = reduced_cost_bounds.update_bounds_from_new_incumbent(original_lp_, + settings_, + root_relax_soln_, + root_objective_, + upper_bound_.load(), + var_types_, + lower_bounds, + upper_bounds); settings_.log.printf( "Updated %d integer bounds using reduced cost strengthening from new incumbent. Max " "objective %e Current objective %e Previous max objective %e\n", diff --git a/cpp/src/branch_and_bound/reduced_cost_bounds.hpp b/cpp/src/branch_and_bound/reduced_cost_bounds.hpp index 2ec9de1079..8d59a3171b 100644 --- a/cpp/src/branch_and_bound/reduced_cost_bounds.hpp +++ b/cpp/src/branch_and_bound/reduced_cost_bounds.hpp @@ -10,6 +10,7 @@ #include #include #include +#include #include #include @@ -165,14 +166,50 @@ class reduced_cost_bounds_t { } } - i_t update_bounds_from_new_incumbent(f_t incumbent_objective, + i_t update_bounds_from_new_incumbent(const simplex::lp_problem_t& lp, + const simplex::simplex_solver_settings_t& settings, + const simplex::lp_solution_t& solution, + f_t relaxation_objective, + f_t incumbent_objective, const std::vector& var_types, std::vector& lower_bounds, std::vector& upper_bounds) { - const i_t n = static_cast(lower_bounds_.size()); - f_t max_objective = -std::numeric_limits::infinity(); - i_t integer_bounds_updated = 0; + const i_t n = static_cast(lower_bounds_.size()); + f_t max_objective = -std::numeric_limits::infinity(); + i_t integer_bounds_updated = 0; + const std::vector& reduced_costs = solution.z; + const f_t threshold = 100.0 * settings.integer_tol; + const f_t weaken = settings.integer_tol; + if (std::isfinite(incumbent_objective) && std::isfinite(relaxation_objective) && + incumbent_objective >= relaxation_objective) { + const f_t abs_gap = incumbent_objective - relaxation_objective; + for (i_t j = 0; j < n; ++j) { + if (var_types[j] != simplex::variable_type_t::INTEGER || lp.upper[j] - lp.lower[j] <= 1.0) { + continue; + } + const f_t rc = reduced_costs[j]; + if (!std::isfinite(rc) || std::abs(rc) <= threshold) { continue; } + const f_t lower_j = lp.lower[j]; + const f_t upper_j = lp.upper[j]; + if (lower_j > -inf && reduced_costs[j] > 0) { + const f_t new_upper_bound = lower_j + abs_gap / reduced_costs[j]; + const f_t reduced_cost_upper_bound = std::floor(new_upper_bound + weaken); + if (reduced_cost_upper_bound < upper_j) { + upper_bounds[j] = reduced_cost_upper_bound; + ++integer_bounds_updated; + } + } + if (upper_j < inf && reduced_costs[j] < 0) { + const f_t new_lower_bound = upper_j + abs_gap / reduced_costs[j]; + const f_t reduced_cost_lower_bound = std::ceil(new_lower_bound - weaken); + if (reduced_cost_lower_bound > lower_j) { + lower_bounds[j] = reduced_cost_lower_bound; + ++integer_bounds_updated; + } + } + } + } for (i_t j = 0; j < n; ++j) { if (lower_bounds_[j].is_valid()) { if (incumbent_objective <= lower_bounds_[j].objective && @@ -181,6 +218,8 @@ class reduced_cost_bounds_t { // lower_bounds[j], lower_bounds_[j].bound); lower_bounds[j] = lower_bounds_[j].bound; if (var_types[j] == simplex::variable_type_t::INTEGER) { integer_bounds_updated++; } + } + if (lower_bounds_[j].bound <= lower_bounds[j]) { lower_bounds_[j].bound = lower_bounds_[j].objective = std::numeric_limits::quiet_NaN(); } @@ -195,6 +234,8 @@ class reduced_cost_bounds_t { // upper_bounds[j], upper_bounds_[j].bound); upper_bounds[j] = upper_bounds_[j].bound; if (var_types[j] == simplex::variable_type_t::INTEGER) { integer_bounds_updated++; } + } + if (upper_bounds_[j].bound >= upper_bounds[j]) { upper_bounds_[j].bound = upper_bounds_[j].objective = std::numeric_limits::quiet_NaN(); } From 81ff49a59ceeb47ff9e84fbee83b5780bb0591db Mon Sep 17 00:00:00 2001 From: Christopher Maes Date: Tue, 29 Sep 2026 17:38:16 -0700 Subject: [PATCH 113/113] Skip dense reduced-cost strengthening directions and track work --- cpp/src/branch_and_bound/branch_and_bound.cpp | 2 + .../branch_and_bound/degenerate_pivots.cpp | 324 +++++++++++++++--- .../branch_and_bound/degenerate_pivots.hpp | 1 + 3 files changed, 285 insertions(+), 42 deletions(-) diff --git a/cpp/src/branch_and_bound/branch_and_bound.cpp b/cpp/src/branch_and_bound/branch_and_bound.cpp index f42d64fc89..a8025cd969 100644 --- a/cpp/src/branch_and_bound/branch_and_bound.cpp +++ b/cpp/src/branch_and_bound/branch_and_bound.cpp @@ -3636,6 +3636,7 @@ typename branch_and_bound_t::cut_pass_action_t branch_and_bound_t::solve(mip_solution_t& solut Arow_, exploration_stats_.start_time, root_objective_, + root_relax_work_estimate_, reduced_cost_bounds); } settings_.log.printf("After pivoting: new reduced cost objective %e (current %e)\n", diff --git a/cpp/src/branch_and_bound/degenerate_pivots.cpp b/cpp/src/branch_and_bound/degenerate_pivots.cpp index bec0484692..5ca2a19c4d 100644 --- a/cpp/src/branch_and_bound/degenerate_pivots.cpp +++ b/cpp/src/branch_and_bound/degenerate_pivots.cpp @@ -57,17 +57,59 @@ void pivot_to_improve_reduced_cost_strengthening( const csr_matrix_t& Arow, const f_t start_time, const f_t relaxation_objective, + const f_t root_relax_work_estimate, reduced_cost_bounds_t& reduced_cost_bounds) { const double strengthening_start = tic(); - double btran_time = 0.0; - double reduced_cost_update_time = 0.0; + // The const basis object accumulates solve work in a mutable counter. Never reset it here. + const f_t basis_work_start = basis_update.work_estimate(); + f_t work_estimate = 0.0; + const i_t entry_num_updates = basis_update.num_updates(); + double btran_time = 0.0; + double density_time = 0.0; + double reduced_cost_update_time = 0.0; + double ratio_time = 0.0; + double validation_time = 0.0; + double bound_insertion_time = 0.0; + double cleanup_time = 0.0; + double dy_sparse_sum = 0.0; + double dy_sparse_max = 0.0; + double dy_significant_sum = 0.0; + double dy_significant_max = 0.0; + double dz_indices_sum = 0.0; + double dz_indices_max = 0.0; + i_t fixedwidth_candidates = 0; + i_t processed_candidates = 0; + i_t btran_candidates = 0; + i_t skipped_dense = 0; + i_t sparse_update_calls = 0; + i_t dense_update_calls = 0; + double zero_steps[2] = {}; + double infinite_steps[2] = {}; + double positive_steps[2] = {}; + double rejected_dual[2] = {}; + double bound_attempts[2] = {}; + double accepted_bounds[2] = {}; + // Include all BTRANs in the significant dy density bins, even skipped candidates. + const double density_limits[5] = {0.1, 1.0, 5.0, 10.0, 30.0}; + const char* density_labels[6] = {"[0,0.1)", "[0.1,1)", "[1,5)", "[5,10)", "[10,30)", "[30,inf)"}; + struct density_stats_t { + double candidates = 0.0; + double skipped_dense = 0.0; + double btran = 0.0; + double density = 0.0; + double update = 0.0; + double scan = 0.0; + double additions = 0.0; + } density_stats[6]; // Count primal degenerate basic variables i_t num_degenerate = 0; i_t num_degenerate_continuous = 0; i_t num_degenerate_integer = 0; std::vector degenerate_integer_list; degenerate_integer_list.reserve(lp.num_rows); + // Approximate scalar work, including reserved/initialized storage and scan passes. + work_estimate += lp.num_rows + 7.0 * lp.num_rows; for (i_t k = 0; k < lp.num_rows; k++) { const i_t j = basic_list[k]; const f_t slack_to_lower = soln.x[j] - lp.lower[j]; @@ -77,16 +119,45 @@ void pivot_to_improve_reduced_cost_strengthening( if (var_types[j] == variable_type_t::INTEGER) { num_degenerate_integer++; degenerate_integer_list.push_back(j); + work_estimate += 3; + if (lp.lower[j] == lp.upper[j]) { fixedwidth_candidates++; } } else { num_degenerate_continuous++; } } } - if (num_degenerate_integer == 0) return; - settings.log.printf( - "RCS timing start: candidates=%d elapsed=%.6f\n", num_degenerate_integer, toc(start_time)); + "RCS timing start: candidates=%d fixedwidth=%d nonfixed=%d num_updates=%d " + "m=%d n=%d nnz=%d factor_nnz=unavailable elapsed=%.6f\n", + num_degenerate_integer, + fixedwidth_candidates, + num_degenerate_integer - fixedwidth_candidates, + entry_num_updates, + lp.num_rows, + lp.num_cols, + lp.A.nnz(), + toc(start_time)); + if (num_degenerate_integer == 0) { + const double total_time = toc(strengthening_start); + settings.log.printf( + "RCS timing end: candidates=0 bounds=0 total=%.6f setup=%.6f btran=0 density=0 " + "reduced_cost_update=0 ratio=0 validation=0 bound_insertion=0 cleanup=0 elapsed=%.6f\n", + total_time, + total_time, + toc(start_time)); + const f_t basis_work = basis_update.work_estimate() - basis_work_start; + const f_t total_work = work_estimate + basis_work; + settings.log.printf( + "RCS work: processed=0 skipped=0 skipped_dense=0 local=%.6e basis=%.6e total=%.6e " + "root=%.6e root_ratio=%.6e\n", + work_estimate, + basis_work, + total_work, + root_relax_work_estimate, + root_relax_work_estimate > 0 ? total_work / root_relax_work_estimate : 0.0); + return; + } std::vector variable_to_basic_position(lp.num_cols, -1); for (i_t k = 0; k < lp.num_rows; k++) { variable_to_basic_position[basic_list[k]] = k; @@ -95,19 +166,27 @@ void pivot_to_improve_reduced_cost_strengthening( csr_matrix_t local_Arow = Arow; std::vector nonbasic_end(lp.num_rows); simplex::compute_initial_nonbasic_end(variable_to_basic_position, local_Arow, nonbasic_end); + work_estimate += lp.num_cols + 2.0 * lp.num_rows; + work_estimate += Arow.row_start.size() + Arow.j.size() + Arow.x.size(); + // Row partitioning visits each coefficient once and swaps each basic coefficient. + work_estimate += 4.0 * lp.num_rows + 3.0 * Arow.row_start[Arow.m]; + for (i_t k = 0; k < lp.num_rows; k++) { + work_estimate += 3 + 6.0 * (local_Arow.row_start[k + 1] - 1 - nonbasic_end[k]); + } std::vector delta_y(lp.num_rows, 0); std::vector delta_z(lp.num_cols, 0); std::vector delta_z_mark(lp.num_cols, 0); std::vector delta_z_indices; delta_z_indices.reserve(lp.num_cols); + work_estimate += lp.num_rows + 3.0 * lp.num_cols; - f_t work_estimate = 0; const f_t threshold = 100.0 * settings.integer_tol; const f_t tol = 1e-2; const f_t zero_tol = settings.zero_tol; const f_t harris_tol = settings.dual_tol / 10; - i_t num_bounds_added = 0; + i_t num_bounds_added = 0; + const double setup_time = toc(strengthening_start); for (i_t j : degenerate_integer_list) { // x_j is a degenerate integer basic variable. // We would like a dual-feasible point where x_j is nonbasic with a nonzero @@ -119,26 +198,55 @@ void pivot_to_improve_reduced_cost_strengthening( // The opposite direction is the negated (delta_y, delta_z) ray. const i_t leaving_index = j; const i_t p = variable_to_basic_position[j]; + work_estimate += 3; if (p == -1) continue; + btran_candidates++; sparse_vector_t ep(lp.num_rows, 1); ep.i[0] = p; ep.x[0] = 1.0; + work_estimate += 4; // Singleton RHS allocation/initialization and assignments. sparse_vector_t delta_y_sparse; sparse_vector_t UTsol_sparse; const double btran_start = tic(); basis_update.b_transpose_solve(ep, delta_y_sparse, UTsol_sparse); - btran_time += toc(btran_start); + const double candidate_btran_time = toc(btran_start); + btran_time += candidate_btran_time; - // delta_zN = -N^T * delta_y, delta_z[leaving] = -1 - const double reduced_cost_update_start = tic(); - i_t delta_y_nz0 = 0; + const double density_start = tic(); + i_t delta_y_nz0 = 0; for (const f_t value : delta_y_sparse.x) { if (std::abs(value) > 1e-12) { delta_y_nz0++; } } - work_estimate += delta_y_sparse.i.size(); + work_estimate += 3.0 * delta_y_sparse.x.size() + 2; const f_t delta_y_nz_percentage = delta_y_nz0 / static_cast(lp.num_rows) * 100.0; + dy_sparse_sum += static_cast(delta_y_sparse.i.size()); + dy_sparse_max = std::max(dy_sparse_max, static_cast(delta_y_sparse.i.size())); + dy_significant_sum += delta_y_nz0; + dy_significant_max = std::max(dy_significant_max, static_cast(delta_y_nz0)); + i_t density_bin = 0; + while (density_bin < 5 && delta_y_nz_percentage >= density_limits[density_bin]) { + density_bin++; + } + work_estimate += 8 + 2 * density_bin; + auto& bin = density_stats[density_bin]; + bin.candidates++; + bin.btran += candidate_btran_time; + const double candidate_density_time = toc(density_start); + density_time += candidate_density_time; + bin.density += candidate_density_time; + if (delta_y_nz_percentage > 5.0) { + skipped_dense++; + bin.skipped_dense++; + // BTRAN restored its own workspace; no delta_z marks or dense delta_y were touched. + continue; + } + processed_candidates++; + + // delta_zN = -N^T * delta_y, delta_z[leaving] = -1 + const double reduced_cost_update_start = tic(); if (delta_y_nz_percentage <= 30.0) { + sparse_update_calls++; simplex::compute_delta_z(local_Arow, delta_y_sparse, leaving_index, @@ -149,8 +257,9 @@ void pivot_to_improve_reduced_cost_strengthening( delta_z, work_estimate); } else { + dense_update_calls++; delta_y_sparse.to_dense(delta_y); - work_estimate += delta_y.size(); + work_estimate += delta_y.size() + 2.0 * delta_y_sparse.i.size(); simplex::compute_reduced_cost_update(lp, basic_list, nonbasic_list, @@ -162,33 +271,57 @@ void pivot_to_improve_reduced_cost_strengthening( delta_z, work_estimate); } - reduced_cost_update_time += toc(reduced_cost_update_start); + const double candidate_update_time = toc(reduced_cost_update_start); + reduced_cost_update_time += candidate_update_time; + dz_indices_sum += static_cast(delta_z_indices.size()); + dz_indices_max = std::max(dz_indices_max, static_cast(delta_z_indices.size())); + bin.update += candidate_update_time; const f_t lower_j = lp.lower[j]; const f_t upper_j = lp.upper[j]; const bool at_lower = soln.x[j] - lower_j <= settings.primal_tol; const bool at_upper = upper_j - soln.x[j] <= settings.primal_tol; + work_estimate += 8; // Try both dual rays. scale == +1 uses the computed delta_z (direction -1); // scale == -1 uses -delta_z (direction +1). Either or both may yield an RCS bound. for (const f_t scale : {1.0, -1.0}) { + const i_t sign_index = scale == 1.0 ? 0 : 1; + const double ratio_start = tic(); // Maximum dual step-length alpha that keeps dual feasibility on this ray. // zl_j + alpha * delta_zN_j >= 0 for nonbasic j on lower bound // zu_j + alpha * delta_zN_j <= 0 for nonbasic j on upper bound f_t alpha = inf; + work_estimate += 3; for (i_t jj : delta_z_indices) { + work_estimate += 2; if (vstatus[jj] == variable_status_t::NONBASIC_FIXED) { continue; } + work_estimate += 5; const f_t dz = scale * delta_z[jj]; if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && dz < -zero_tol) { const f_t ratio = std::max((-harris_tol - soln.z[jj]) / dz, 0.0); + work_estimate += 4; if (ratio < alpha) { alpha = ratio; } } if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && dz > zero_tol) { const f_t ratio = std::max((harris_tol - soln.z[jj]) / dz, 0.0); + work_estimate += 4; if (ratio < alpha) { alpha = ratio; } } } - if (alpha == 0.0 || !std::isfinite(alpha)) { continue; } + const double ray_ratio_time = toc(ratio_start); + ratio_time += ray_ratio_time; + bin.scan += ray_ratio_time; + if (alpha == 0.0 || !std::isfinite(alpha)) { + if (alpha == 0.0) { + zero_steps[sign_index]++; + } else { + infinite_steps[sign_index]++; + } + continue; + } + positive_steps[sign_index]++; + const double validation_start = tic(); // Verify dual feasibility of the new point z_new = z + alpha * scale * delta_z // For NONBASIC_LOWER: z_new[jj] >= -dual_tol @@ -204,7 +337,9 @@ void pivot_to_improve_reduced_cost_strengthening( i_t num_dual_infeas = 0; i_t worst_j = -1; for (i_t jj : delta_z_indices) { + work_estimate += 2; if (vstatus[jj] == variable_status_t::NONBASIC_FIXED) { continue; } + work_estimate += 12; const f_t old_zj = soln.z[jj]; const f_t step = alpha * scale * delta_z[jj]; const f_t new_zj = old_zj + step; @@ -212,12 +347,14 @@ void pivot_to_improve_reduced_cost_strengthening( (vstatus[jj] == variable_status_t::NONBASIC_LOWER && old_zj < -settings.dual_tol) || (vstatus[jj] == variable_status_t::NONBASIC_UPPER && old_zj > settings.dual_tol); if (initially_infeasible) { + work_estimate += 3; num_initial_dual_infeas++; max_initial_dual_infeas = std::max(max_initial_dual_infeas, std::abs(old_zj)); } if (vstatus[jj] == variable_status_t::NONBASIC_LOWER && new_zj < -settings.dual_tol) { num_dual_infeas++; if (std::abs(new_zj) > max_dual_infeas) { + work_estimate += 7; max_dual_infeas = std::abs(new_zj); worst_j = jj; worst_old_z = old_zj; @@ -229,6 +366,7 @@ void pivot_to_improve_reduced_cost_strengthening( if (vstatus[jj] == variable_status_t::NONBASIC_UPPER && new_zj > settings.dual_tol) { num_dual_infeas++; if (std::abs(new_zj) > max_dual_infeas) { + work_estimate += 7; max_dual_infeas = std::abs(new_zj); worst_j = jj; worst_old_z = old_zj; @@ -240,33 +378,44 @@ void pivot_to_improve_reduced_cost_strengthening( } // Also check the leaving variable itself const f_t new_zj_leaving = soln.z[j] + alpha * scale * delta_z[j]; + work_estimate += 4; + const double ray_validation_time = toc(validation_start); + validation_time += ray_validation_time; + bin.scan += ray_validation_time; if (num_dual_infeas > 0) { - settings.log.printf( - "WARNING pivot_to_improve_rc: dual infeasibility after step! " - "var=%d alpha=%.6e scale=%.0f initial_num_infeas=%d " - "initial_max_infeas=%.6e num_infeas=%d max_infeas=%.6e worst_j=%d " - "worst_status=%d old_z=%.16e delta_z=%.16e step=%.16e new_z=%.16e " - "new_rc_leaving=%.6e\n", - j, - alpha, - scale, - num_initial_dual_infeas, - max_initial_dual_infeas, - num_dual_infeas, - max_dual_infeas, - worst_j, - static_cast(vstatus[worst_j]), - worst_old_z, - worst_delta_z, - worst_step, - worst_new_z, - new_zj_leaving); + rejected_dual[sign_index]++; + // Keep one diagnostic sample per call; report all rejections in the summary. + if (rejected_dual[0] + rejected_dual[1] == 1.0) { + settings.log.printf( + "WARNING pivot_to_improve_rc: dual infeasibility after step! " + "var=%d alpha=%.6e scale=%.0f initial_num_infeas=%d " + "initial_max_infeas=%.6e num_infeas=%d max_infeas=%.6e worst_j=%d " + "worst_status=%d old_z=%.16e delta_z=%.16e step=%.16e new_z=%.16e " + "new_rc_leaving=%.6e\n", + j, + alpha, + scale, + num_initial_dual_infeas, + max_initial_dual_infeas, + num_dual_infeas, + max_dual_infeas, + worst_j, + static_cast(vstatus[worst_j]), + worst_old_z, + worst_delta_z, + worst_step, + worst_new_z, + new_zj_leaving); + } + continue; } } // Claim: We don't actually need to take a pivot if all we want to do is add a bound // coming from reduced cost strengthening - const f_t new_reduced_cost = soln.z[j] + alpha * scale * delta_z[j]; + const double bound_insertion_start = tic(); + const f_t new_reduced_cost = soln.z[j] + alpha * scale * delta_z[j]; + work_estimate += 10; // x_j <= l_j + (incumbent_objective - relaxation_objective) / reduced_costs[j] // Let u_tilde_j = u_j - epsilon, so that floor(u_tilde_j) = u_j - 1 @@ -276,6 +425,7 @@ void pivot_to_improve_reduced_cost_strengthening( // u_tilde_j Or equivalently, incumbent_objective <= relaxation_objective + // reduced_costs[j] * (u_tilde_j - l_j) when reduced_costs[j] > 0 if (at_lower && lower_j > -inf && new_reduced_cost > threshold) { + work_estimate += 12; const f_t u_tilde_j = var_types[j] == variable_type_t::INTEGER ? upper_j - tol : std::max(upper_j - 1.0, lower_j); @@ -287,7 +437,13 @@ void pivot_to_improve_reduced_cost_strengthening( var_types[j] != variable_type_t::INTEGER) && std::isfinite(objective_j) && std::isfinite(bound_j)) { i_t info = reduced_cost_bounds.add_upper_bound(j, objective_j, bound_j); - if (info > 0) { num_bounds_added++; } + work_estimate += 10; // Constant-size bound-table lookup, comparisons and writes. + bound_attempts[sign_index]++; + if (info > 0) { + num_bounds_added++; + accepted_bounds[sign_index]++; + bin.additions++; + } // settings.log.printf("Added objective bound pair (%e, %e) for variable %d upper bound. // Info %d\n", objective_j, bound_j, j, info); } @@ -300,6 +456,7 @@ void pivot_to_improve_reduced_cost_strengthening( // equivalently, incumbent_objective <= relaxation_objective + reduced_costs[j] * // (l_tilde_j - u_j) when reduced_costs[j] < 0 if (at_upper && upper_j < inf && new_reduced_cost < -threshold) { + work_estimate += 12; const f_t l_tilde_j = var_types[j] == variable_type_t::INTEGER ? lower_j + tol : std::min(lower_j + 1.0, upper_j); @@ -311,14 +468,23 @@ void pivot_to_improve_reduced_cost_strengthening( var_types[j] != variable_type_t::INTEGER) && std::isfinite(objective_j) && std::isfinite(bound_j)) { i_t info = reduced_cost_bounds.add_lower_bound(j, objective_j, bound_j); - if (info > 0) { num_bounds_added++; } + work_estimate += 10; + bound_attempts[sign_index]++; + if (info > 0) { + num_bounds_added++; + accepted_bounds[sign_index]++; + bin.additions++; + } // settings.log.printf("Added objective bound pair (%e, %e) for variable %d lower bound. // Info %d\n", objective_j, bound_j, j, info); } } + bound_insertion_time += toc(bound_insertion_start); } - // Clear arrays for next iteration + // Clear arrays for next iteration, including when either or both rays were rejected. + const double cleanup_start = tic(); + work_estimate += 3.0 * delta_z_indices.size() + 2.0 * delta_y_sparse.i.size() + 2; for (i_t k : delta_z_indices) { delta_z_mark[k] = 0; delta_z[k] = 0.0; @@ -328,17 +494,90 @@ void pivot_to_improve_reduced_cost_strengthening( for (i_t k : delta_y_sparse.i) { delta_y[k] = 0.0; } + cleanup_time += toc(cleanup_start); } + const double total_time = toc(strengthening_start); + const double candidate_count = std::max(1.0, static_cast(processed_candidates)); + const double btran_count = std::max(1.0, static_cast(btran_candidates)); settings.log.printf("Added %d bounds for reduced cost strengthening\n", num_bounds_added); settings.log.printf( - "RCS timing end: candidates=%d bounds=%d total=%.6f btran=%.6f reduced_cost_update=%.6f " - "elapsed=%.6f\n", + "RCS timing end: candidates=%d bounds=%d total=%.6f setup=%.6f btran=%.6f density=%.6f " + "reduced_cost_update=%.6f ratio=%.6f validation=%.6f bound_insertion=%.6f cleanup=%.6f " + "other=%.6f seconds_per_candidate=%.9f elapsed=%.6f\n", num_degenerate_integer, num_bounds_added, - toc(strengthening_start), + total_time, + setup_time, btran_time, + density_time, reduced_cost_update_time, + ratio_time, + validation_time, + bound_insertion_time, + cleanup_time, + total_time - setup_time - btran_time - density_time - reduced_cost_update_time - ratio_time - + validation_time - bound_insertion_time - cleanup_time, + total_time / btran_count, toc(start_time)); + settings.log.printf( + "RCS sparsity: processed=%d skipped=%d skipped_dense=%d btran_candidates=%d " + "dy_sparse_avg=%.3f dy_sparse_max=%.0f " + "dy_significant_avg=%.3f dy_significant_max=%.0f dz_indices_avg=%.3f dz_indices_max=%.0f " + "sparse_update_calls=%d dense_update_calls=%d significant_tol=1e-12 sparse_max_pct=30 " + "skip_dense_above_pct=5\n", + processed_candidates, + num_degenerate_integer - processed_candidates, + skipped_dense, + btran_candidates, + dy_sparse_sum / btran_count, + dy_sparse_max, + dy_significant_sum / btran_count, + dy_significant_max, + dz_indices_sum / candidate_count, + dz_indices_max, + sparse_update_calls, + dense_update_calls); + for (i_t sign_index = 0; sign_index < 2; sign_index++) { + settings.log.printf( + "RCS rays: scale=%d zero_steps=%.0f infinite_steps=%.0f positive_steps=%.0f " + "rejected_dual=%.0f dual_feasible_steps=%.0f bound_attempts=%.0f accepted_bounds=%.0f\n", + sign_index == 0 ? 1 : -1, + zero_steps[sign_index], + infinite_steps[sign_index], + positive_steps[sign_index], + rejected_dual[sign_index], + positive_steps[sign_index] - rejected_dual[sign_index], + bound_attempts[sign_index], + accepted_bounds[sign_index]); + } + for (i_t density_bin = 0; density_bin < 6; density_bin++) { + const auto& bin = density_stats[density_bin]; + settings.log.printf( + "RCS density: dy_significant_pct=%s candidates=%.0f skipped_dense=%.0f " + "btran=%.6f density=%.6f update=%.6f " + "scan=%.6f accepted_bounds=%.0f\n", + density_labels[density_bin], + bin.candidates, + bin.skipped_dense, + bin.btran, + bin.density, + bin.update, + bin.scan, + bin.additions); + } + const f_t basis_work = basis_update.work_estimate() - basis_work_start; + const f_t total_work = work_estimate + basis_work; + settings.log.printf( + "RCS work: processed=%d skipped=%d skipped_dense=%d local=%.6e basis=%.6e total=%.6e " + "root=%.6e root_ratio=%.6e\n", + processed_candidates, + num_degenerate_integer - processed_candidates, + skipped_dense, + work_estimate, + basis_work, + total_work, + root_relax_work_estimate, + root_relax_work_estimate > 0 ? total_work / root_relax_work_estimate : 0.0); } template @@ -1551,6 +1790,7 @@ template void pivot_to_improve_reduced_cost_strengthening( const csr_matrix_t&, double, double, + double, reduced_cost_bounds_t&); template void dual_degenerate_feasibility_pump( diff --git a/cpp/src/branch_and_bound/degenerate_pivots.hpp b/cpp/src/branch_and_bound/degenerate_pivots.hpp index 91bd57bc96..f9a40aad9e 100644 --- a/cpp/src/branch_and_bound/degenerate_pivots.hpp +++ b/cpp/src/branch_and_bound/degenerate_pivots.hpp @@ -92,6 +92,7 @@ void pivot_to_improve_reduced_cost_strengthening( const csr_matrix_t& Arow, f_t start_time, f_t relaxation_objective, + f_t root_relax_work_estimate, reduced_cost_bounds_t& reduced_cost_bounds); template