diff --git a/docs/data/operating_point_parity_51.json b/docs/data/operating_point_parity_51.json new file mode 100644 index 00000000..ffd3518e --- /dev/null +++ b/docs/data/operating_point_parity_51.json @@ -0,0 +1,582 @@ +{ + "control": { + "agrees": true, + "delta": -0.0025, + "f1_op_cache": 0.8245, + "f1_published_bundle": 0.827, + "per_split": { + "annapolis": { + "agrees": true, + "delta": 0.0005, + "f1_op_cache": 0.8395, + "f1_published_bundle": 0.839, + "paths_agree_by_construction": true, + "tolerance": 0.002 + }, + "bend": { + "agrees": true, + "delta": 0.0032, + "f1_op_cache": 0.8532, + "f1_published_bundle": 0.85, + "paths_agree_by_construction": false, + "tolerance": 0.025 + }, + "budapest_district5": { + "agrees": true, + "delta": 0.0002, + "f1_op_cache": 0.6442, + "f1_published_bundle": 0.644, + "paths_agree_by_construction": true, + "tolerance": 0.002 + }, + "clovis": { + "agrees": true, + "delta": 0.0002, + "f1_op_cache": 0.8012, + "f1_published_bundle": 0.801, + "paths_agree_by_construction": true, + "tolerance": 0.002 + }, + "gainesville": { + "agrees": true, + "delta": -0.0159, + "f1_op_cache": 0.7871, + "f1_published_bundle": 0.803, + "paths_agree_by_construction": false, + "tolerance": 0.025 + }, + "manual_gold": { + "agrees": true, + "delta": -0.009, + "f1_op_cache": 0.899, + "f1_published_bundle": 0.908, + "paths_agree_by_construction": false, + "tolerance": 0.025 + }, + "morgantown": { + "agrees": true, + "delta": 0.0001, + "f1_op_cache": 0.8351, + "f1_published_bundle": 0.835, + "paths_agree_by_construction": true, + "tolerance": 0.002 + }, + "paterson": { + "agrees": true, + "delta": -0.0044, + "f1_op_cache": 0.8006, + "f1_published_bundle": 0.805, + "paths_agree_by_construction": false, + "tolerance": 0.025 + }, + "richmond": { + "agrees": true, + "delta": -0.0004, + "f1_op_cache": 0.8546, + "f1_published_bundle": 0.855, + "paths_agree_by_construction": true, + "tolerance": 0.002 + }, + "sao_paulo": { + "agrees": true, + "delta": -0.0192, + "f1_op_cache": 0.7578, + "f1_published_bundle": 0.777, + "paths_agree_by_construction": false, + "tolerance": 0.025 + } + }, + "per_split_agrees": true, + "threshold": 0.55, + "tolerance": 0.005 + }, + "dev_split": "sao_paulo", + "grid_rampnet": [ + 0.05, + 0.1, + 0.15, + 0.2, + 0.25, + 0.3, + 0.35, + 0.4, + 0.45, + 0.5, + 0.55, + 0.6, + 0.65, + 0.7, + 0.75, + 0.8, + 0.85, + 0.9, + 0.95 + ], + "models": { + "RampNet": { + "at_floor": false, + "per_split": { + "annapolis": { + "f1": 0.853, + "fn": 56, + "fp": 26, + "p": 0.9015, + "r": 0.8095, + "tp": 238 + }, + "bend": { + "f1": 0.8706, + "fn": 58, + "fp": 22, + "p": 0.9244, + "r": 0.8226, + "tp": 269 + }, + "budapest_district5": { + "f1": 0.6736, + "fn": 107, + "fp": 80, + "p": 0.707, + "r": 0.6433, + "tp": 193 + }, + "clovis": { + "f1": 0.8355, + "fn": 35, + "fp": 28, + "p": 0.8511, + "r": 0.8205, + "tp": 160 + }, + "gainesville": { + "f1": 0.8124, + "fn": 62, + "fp": 35, + "p": 0.8571, + "r": 0.7721, + "tp": 210 + }, + "manual_gold": { + "f1": 0.9018, + "fn": 423, + "fp": 338, + "p": 0.9118, + "r": 0.8921, + "tp": 3496 + }, + "morgantown": { + "f1": 0.8448, + "fn": 52, + "fp": 27, + "p": 0.8884, + "r": 0.8052, + "tp": 215 + }, + "paterson": { + "f1": 0.8184, + "fn": 111, + "fp": 15, + "p": 0.9498, + "r": 0.719, + "tp": 284 + }, + "richmond": { + "f1": 0.8639, + "fn": 53, + "fp": 28, + "p": 0.9018, + "r": 0.829, + "tp": 257 + }, + "sao_paulo": { + "f1": 0.8, + "fn": 57, + "fp": 55, + "p": 0.8029, + "r": 0.7972, + "tp": 224 + } + }, + "pooled": { + "f1": 0.8427, + "n_splits": 7, + "p": 0.8963, + "r": 0.7969, + "threshold": 0.3 + }, + "published_point": { + "f1": 0.8245, + "n_splits": 7, + "p": 0.9608, + "r": 0.7226, + "threshold": 0.55 + }, + "selected_threshold": 0.3 + }, + "y11x_pano": { + "at_floor": false, + "per_split": { + "annapolis": { + "f1": 0.723, + "fn": 110, + "fp": 31, + "p": 0.856, + "r": 0.626, + "tp": 184 + }, + "bend": { + "f1": 0.789, + "fn": 78, + "fp": 55, + "p": 0.819, + "r": 0.761, + "tp": 249 + }, + "budapest_district5": { + "f1": 0.528, + "fn": 169, + "fp": 65, + "p": 0.668, + "r": 0.437, + "tp": 131 + }, + "clovis": { + "f1": 0.704, + "fn": 52, + "fp": 68, + "p": 0.678, + "r": 0.733, + "tp": 143 + }, + "gainesville": { + "f1": 0.673, + "fn": 103, + "fp": 61, + "p": 0.735, + "r": 0.621, + "tp": 169 + }, + "manual_gold": { + "f1": 0.844, + "fn": 318, + "fp": 1013, + "p": 0.78, + "r": 0.919, + "tp": 3601 + }, + "morgantown": { + "f1": 0.791, + "fn": 59, + "fp": 51, + "p": 0.803, + "r": 0.779, + "tp": 208 + }, + "paterson": { + "f1": 0.794, + "fn": 100, + "fp": 53, + "p": 0.848, + "r": 0.747, + "tp": 295 + }, + "richmond": { + "f1": 0.791, + "fn": 57, + "fp": 77, + "p": 0.767, + "r": 0.816, + "tp": 253 + }, + "sao_paulo": { + "f1": 0.744, + "fn": 63, + "fp": 87, + "p": 0.715, + "r": 0.776, + "tp": 218 + } + }, + "pooled": { + "f1": 0.7521, + "n_splits": 7, + "p": 0.7866, + "r": 0.7261, + "threshold": 0.1 + }, + "published_point": { + "f1": 0.6231, + "n_splits": 7, + "p": 0.9417, + "r": 0.4743, + "threshold": 0.25 + }, + "selected_threshold": 0.1 + }, + "y11x_pano_h200": { + "at_floor": false, + "per_split": { + "annapolis": { + "f1": 0.689, + "fn": 130, + "fp": 18, + "p": 0.901, + "r": 0.558, + "tp": 164 + }, + "bend": { + "f1": 0.799, + "fn": 79, + "fp": 46, + "p": 0.844, + "r": 0.758, + "tp": 248 + }, + "budapest_district5": { + "f1": 0.51, + "fn": 180, + "fp": 51, + "p": 0.702, + "r": 0.4, + "tp": 120 + }, + "clovis": { + "f1": 0.734, + "fn": 61, + "fp": 36, + "p": 0.788, + "r": 0.687, + "tp": 134 + }, + "gainesville": { + "f1": 0.668, + "fn": 110, + "fp": 51, + "p": 0.761, + "r": 0.596, + "tp": 162 + }, + "manual_gold": { + "f1": 0.872, + "fn": 282, + "fp": 782, + "p": 0.823, + "r": 0.928, + "tp": 3637 + }, + "morgantown": { + "f1": 0.813, + "fn": 69, + "fp": 22, + "p": 0.9, + "r": 0.742, + "tp": 198 + }, + "paterson": { + "f1": 0.776, + "fn": 104, + "fp": 64, + "p": 0.82, + "r": 0.737, + "tp": 291 + }, + "richmond": { + "f1": 0.777, + "fn": 91, + "fp": 35, + "p": 0.862, + "r": 0.706, + "tp": 219 + }, + "sao_paulo": { + "f1": 0.754, + "fn": 62, + "fp": 81, + "p": 0.73, + "r": 0.779, + "tp": 219 + } + }, + "pooled": { + "f1": 0.7509, + "n_splits": 7, + "p": 0.8394, + "r": 0.6834, + "threshold": 0.1 + }, + "published_point": { + "f1": 0.575, + "n_splits": 7, + "p": 0.9686, + "r": 0.4156, + "threshold": 0.25 + }, + "selected_threshold": 0.1 + }, + "y11x_tiles": { + "at_floor": false, + "per_split": { + "annapolis": { + "f1": 0.783, + "fn": 87, + "fp": 28, + "p": 0.881, + "r": 0.704, + "tp": 207 + }, + "bend": { + "f1": 0.87, + "fn": 57, + "fp": 24, + "p": 0.918, + "r": 0.826, + "tp": 270 + }, + "budapest_district5": { + "f1": 0.551, + "fn": 164, + "fp": 58, + "p": 0.701, + "r": 0.453, + "tp": 136 + }, + "clovis": { + "f1": 0.77, + "fn": 39, + "fp": 54, + "p": 0.743, + "r": 0.8, + "tp": 156 + }, + "gainesville": { + "f1": 0.778, + "fn": 79, + "fp": 31, + "p": 0.862, + "r": 0.71, + "tp": 193 + }, + "manual_gold": { + "f1": 0.911, + "fn": 420, + "fp": 263, + "p": 0.93, + "r": 0.893, + "tp": 3499 + }, + "morgantown": { + "f1": 0.862, + "fn": 51, + "fp": 18, + "p": 0.923, + "r": 0.809, + "tp": 216 + }, + "paterson": { + "f1": 0.761, + "fn": 143, + "fp": 15, + "p": 0.944, + "r": 0.638, + "tp": 252 + }, + "richmond": { + "f1": 0.803, + "fn": 75, + "fp": 40, + "p": 0.855, + "r": 0.758, + "tp": 235 + }, + "sao_paulo": { + "f1": 0.776, + "fn": 70, + "fp": 52, + "p": 0.802, + "r": 0.751, + "tp": 211 + } + }, + "pooled": { + "f1": 0.8039, + "n_splits": 7, + "p": 0.8751, + "r": 0.7493, + "threshold": 0.1 + }, + "published_point": { + "f1": 0.6669, + "n_splits": 7, + "p": 0.9446, + "r": 0.5197, + "threshold": 0.25 + }, + "selected_threshold": 0.1 + } + }, + "pool": [ + "richmond", + "bend", + "clovis", + "morgantown", + "annapolis", + "paterson", + "gainesville" + ], + "sensitivity": { + "budapest_district5": { + "RampNet": { + "f1": 0.842, + "thr": 0.35 + }, + "y11x_pano": { + "f1": 0.6873, + "thr": 0.05 + }, + "y11x_pano_h200": { + "f1": 0.7509, + "thr": 0.1 + }, + "y11x_tiles": { + "f1": 0.8096, + "thr": 0.05 + } + }, + "manual_gold": { + "RampNet": { + "f1": 0.842, + "thr": 0.35 + }, + "y11x_pano": { + "f1": 0.7393, + "thr": 0.15 + }, + "y11x_pano_h200": { + "f1": 0.7203, + "thr": 0.15 + }, + "y11x_tiles": { + "f1": 0.8039, + "thr": 0.1 + } + }, + "sao_paulo": { + "RampNet": { + "f1": 0.8427, + "thr": 0.3 + }, + "y11x_pano": { + "f1": 0.7521, + "thr": 0.1 + }, + "y11x_pano_h200": { + "f1": 0.7509, + "thr": 0.1 + }, + "y11x_tiles": { + "f1": 0.8039, + "thr": 0.1 + } + } + } +} diff --git a/docs/data/yolo_geometry_51.json b/docs/data/yolo_geometry_51.json new file mode 100644 index 00000000..dc8f0710 --- /dev/null +++ b/docs/data/yolo_geometry_51.json @@ -0,0 +1,524 @@ +{ + "best_f1_sweep_tune_on_test": { + "y11x_pano": { + "annapolis": { + "f1": 0.723, + "thr": 0.1 + }, + "bend": { + "f1": 0.789, + "thr": 0.1 + }, + "budapest_district5": { + "f1": 0.536, + "thr": 0.05 + }, + "clovis": { + "f1": 0.734, + "thr": 0.15 + }, + "gainesville": { + "f1": 0.673, + "thr": 0.1 + }, + "manual_gold": { + "f1": 0.873, + "thr": 0.15 + }, + "morgantown": { + "f1": 0.793, + "thr": 0.15 + }, + "paterson": { + "f1": 0.794, + "thr": 0.1 + }, + "richmond": { + "f1": 0.799, + "thr": 0.15 + }, + "sao_paulo": { + "f1": 0.744, + "thr": 0.1 + } + }, + "y11x_pano_h200": { + "annapolis": { + "f1": 0.69, + "thr": 0.05 + }, + "bend": { + "f1": 0.802, + "thr": 0.15 + }, + "budapest_district5": { + "f1": 0.51, + "thr": 0.1 + }, + "clovis": { + "f1": 0.734, + "thr": 0.1 + }, + "gainesville": { + "f1": 0.668, + "thr": 0.1 + }, + "manual_gold": { + "f1": 0.893, + "thr": 0.15 + }, + "morgantown": { + "f1": 0.813, + "thr": 0.1 + }, + "paterson": { + "f1": 0.776, + "thr": 0.1 + }, + "richmond": { + "f1": 0.777, + "thr": 0.1 + }, + "sao_paulo": { + "f1": 0.754, + "thr": 0.1 + } + }, + "y11x_tiles": { + "annapolis": { + "f1": 0.799, + "thr": 0.05 + }, + "bend": { + "f1": 0.87, + "thr": 0.1 + }, + "budapest_district5": { + "f1": 0.605, + "thr": 0.05 + }, + "clovis": { + "f1": 0.775, + "thr": 0.15 + }, + "gainesville": { + "f1": 0.797, + "thr": 0.05 + }, + "manual_gold": { + "f1": 0.911, + "thr": 0.1 + }, + "morgantown": { + "f1": 0.862, + "thr": 0.1 + }, + "paterson": { + "f1": 0.807, + "thr": 0.05 + }, + "richmond": { + "f1": 0.806, + "thr": 0.05 + }, + "sao_paulo": { + "f1": 0.776, + "thr": 0.1 + } + } + }, + "control_reproduced": true, + "decomposition": { + "budget": 0.0481, + "geometry": 0.0437, + "published_gap": 0.252, + "residual": 0.1601, + "total": 0.0919 + }, + "held_out": { + "budapest_district5": "single-rater GT at low reviewer confidence (docs/model_comparison.md: do not pool)", + "manual_gold": "in-distribution GSV + independently-labelled GT (in-domain reference, not a deployment city)", + "sao_paulo": "non-US city \u2014 the pooled recommendation is a US-deployment basis (GT is HIGH reviewer confidence; held out for geography, not GT quality)" + }, + "legs": { + "y11x_pano": { + "epoch": 38, + "geometry": "whole pano, imgsz 1280" + }, + "y11x_pano_h200": { + "epoch": 60, + "geometry": "whole pano, imgsz 1280" + }, + "y11x_tiles": { + "epoch": 44, + "geometry": "perspective tiles, imgsz 1024" + } + }, + "operating_point": 0.25, + "per_split": { + "y11x_pano": { + "annapolis": { + "ap": 0.678, + "f1": 0.452, + "fn": 207, + "fp": 4, + "ign": 0, + "p": 0.956, + "r": 0.296, + "tp": 87 + }, + "bend": { + "ap": 0.775, + "f1": 0.728, + "fn": 137, + "fp": 5, + "ign": 2, + "p": 0.974, + "r": 0.581, + "tp": 190 + }, + "budapest_district5": { + "ap": 0.444, + "f1": 0.176, + "fn": 270, + "fp": 11, + "ign": 1, + "p": 0.732, + "r": 0.1, + "tp": 30 + }, + "clovis": { + "ap": 0.72, + "f1": 0.61, + "fn": 106, + "fp": 8, + "ign": 1, + "p": 0.918, + "r": 0.456, + "tp": 89 + }, + "gainesville": { + "ap": 0.649, + "f1": 0.501, + "fn": 178, + "fp": 9, + "ign": 2, + "p": 0.913, + "r": 0.346, + "tp": 94 + }, + "manual_gold": { + "ap": 0.916, + "f1": 0.84, + "fn": 975, + "fp": 150, + "ign": 0, + "p": 0.952, + "r": 0.751, + "tp": 2944 + }, + "morgantown": { + "ap": 0.796, + "f1": 0.728, + "fn": 108, + "fp": 11, + "ign": 0, + "p": 0.935, + "r": 0.596, + "tp": 159 + }, + "paterson": { + "ap": 0.816, + "f1": 0.635, + "fn": 209, + "fp": 5, + "ign": 3, + "p": 0.974, + "r": 0.471, + "tp": 186 + }, + "richmond": { + "ap": 0.827, + "f1": 0.708, + "fn": 132, + "fp": 15, + "ign": 12, + "p": 0.922, + "r": 0.574, + "tp": 178 + }, + "sao_paulo": { + "ap": 0.761, + "f1": 0.65, + "fn": 139, + "fp": 14, + "ign": 8, + "p": 0.91, + "r": 0.505, + "tp": 142 + } + }, + "y11x_pano_h200": { + "annapolis": { + "ap": 0.662, + "f1": 0.397, + "fn": 221, + "fp": 1, + "ign": 0, + "p": 0.986, + "r": 0.248, + "tp": 73 + }, + "bend": { + "ap": 0.781, + "f1": 0.71, + "fn": 146, + "fp": 2, + "ign": 2, + "p": 0.989, + "r": 0.554, + "tp": 181 + }, + "budapest_district5": { + "ap": 0.427, + "f1": 0.221, + "fn": 262, + "fp": 6, + "ign": 2, + "p": 0.864, + "r": 0.127, + "tp": 38 + }, + "clovis": { + "ap": 0.711, + "f1": 0.551, + "fn": 119, + "fp": 5, + "ign": 0, + "p": 0.938, + "r": 0.39, + "tp": 76 + }, + "gainesville": { + "ap": 0.616, + "f1": 0.499, + "fn": 180, + "fp": 5, + "ign": 1, + "p": 0.948, + "r": 0.338, + "tp": 92 + }, + "manual_gold": { + "ap": 0.931, + "f1": 0.851, + "fn": 914, + "fp": 139, + "ign": 0, + "p": 0.956, + "r": 0.767, + "tp": 3005 + }, + "morgantown": { + "ap": 0.79, + "f1": 0.686, + "fn": 127, + "fp": 1, + "ign": 0, + "p": 0.993, + "r": 0.524, + "tp": 140 + }, + "paterson": { + "ap": 0.8, + "f1": 0.635, + "fn": 209, + "fp": 5, + "ign": 2, + "p": 0.974, + "r": 0.471, + "tp": 186 + }, + "richmond": { + "ap": 0.748, + "f1": 0.547, + "fn": 191, + "fp": 6, + "ign": 4, + "p": 0.952, + "r": 0.384, + "tp": 119 + }, + "sao_paulo": { + "ap": 0.783, + "f1": 0.659, + "fn": 137, + "fp": 12, + "ign": 12, + "p": 0.923, + "r": 0.512, + "tp": 144 + } + }, + "y11x_tiles": { + "annapolis": { + "ap": 0.738, + "f1": 0.64, + "fn": 150, + "fp": 12, + "ign": 2, + "p": 0.923, + "r": 0.49, + "tp": 144 + }, + "bend": { + "ap": 0.844, + "f1": 0.75, + "fn": 128, + "fp": 5, + "ign": 2, + "p": 0.975, + "r": 0.609, + "tp": 199 + }, + "budapest_district5": { + "ap": 0.458, + "f1": 0.319, + "fn": 241, + "fp": 11, + "ign": 4, + "p": 0.843, + "r": 0.197, + "tp": 59 + }, + "clovis": { + "ap": 0.769, + "f1": 0.712, + "fn": 79, + "fp": 15, + "ign": 4, + "p": 0.885, + "r": 0.595, + "tp": 116 + }, + "gainesville": { + "ap": 0.751, + "f1": 0.576, + "fn": 160, + "fp": 5, + "ign": 2, + "p": 0.957, + "r": 0.412, + "tp": 112 + }, + "manual_gold": { + "ap": 0.909, + "f1": 0.82, + "fn": 1133, + "fp": 86, + "ign": 0, + "p": 0.97, + "r": 0.711, + "tp": 2786 + }, + "morgantown": { + "ap": 0.824, + "f1": 0.728, + "fn": 112, + "fp": 4, + "ign": 2, + "p": 0.975, + "r": 0.581, + "tp": 155 + }, + "paterson": { + "ap": 0.715, + "f1": 0.595, + "fn": 226, + "fp": 4, + "ign": 3, + "p": 0.977, + "r": 0.428, + "tp": 169 + }, + "richmond": { + "ap": 0.773, + "f1": 0.667, + "fn": 148, + "fp": 14, + "ign": 8, + "p": 0.92, + "r": 0.523, + "tp": 162 + }, + "sao_paulo": { + "ap": 0.751, + "f1": 0.625, + "fn": 146, + "fp": 16, + "ign": 9, + "p": 0.894, + "r": 0.48, + "tp": 135 + } + } + }, + "pooled": { + "y11x_pano": { + "fn": 1077, + "fp": 57, + "macro_ap": 0.7516, + "macro_f1": 0.6231, + "macro_p": 0.9417, + "macro_r": 0.4743, + "micro_f1": 0.6342, + "micro_p": 0.9452, + "micro_r": 0.4772, + "n_splits": 7, + "tp": 983 + }, + "y11x_pano_h200": { + "fn": 1193, + "fp": 25, + "macro_ap": 0.7297, + "macro_f1": 0.575, + "macro_p": 0.9686, + "macro_r": 0.4156, + "micro_f1": 0.5874, + "micro_p": 0.972, + "micro_r": 0.4209, + "n_splits": 7, + "tp": 867 + }, + "y11x_tiles": { + "fn": 1003, + "fp": 59, + "macro_ap": 0.7734, + "macro_f1": 0.6669, + "macro_p": 0.9446, + "macro_r": 0.5197, + "micro_f1": 0.6656, + "micro_p": 0.9471, + "micro_r": 0.5131, + "n_splits": 7, + "tp": 1057 + } + }, + "pooled_splits": [ + "richmond", + "bend", + "clovis", + "morgantown", + "annapolis", + "paterson", + "gainesville" + ], + "published_control": { + "f1": 0.575, + "model": "YOLO11x (pano)", + "p": 0.969, + "r": 0.416 + }, + "published_rampnet_pooled_f1": 0.827, + "what": "#51 geometry-pair eval: tiles vs pano at near-matched budget, plus the published pano arm re-scored as a control" +} diff --git a/docs/data/yolo_geometry_51/annapolis_pano.txt b/docs/data/yolo_geometry_51/annapolis_pano.txt new file mode 100644 index 00000000..6d2c68e7 --- /dev/null +++ b/docs/data/yolo_geometry_51/annapolis_pano.txt @@ -0,0 +1,58 @@ +Bundle: benchmark/annapolis (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +[y11x_pano_h200] all 125 panos already cached; model load skipped +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_pano 0.956 (0.892-0.983) 0.296 (0.247-0.350) 0.452 0.678 87/4/207/0 +y11x_pano_h200 0.986 (0.927-0.998) 0.248 (0.202-0.301) 0.397 0.662 73/1/221/0 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_pano] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.596 0.731 0.656 215/146/79 + 0.10 0.856 0.626 0.723 184/31/110 <- best F1 + 0.15 0.928 0.524 0.670 154/12/140 + 0.20 0.946 0.415 0.577 122/7/172 + 0.25 0.956 0.296 0.452 87/4/207 + 0.30 0.971 0.231 0.374 68/2/226 + 0.40 0.977 0.143 0.249 42/1/252 + 0.50 1.000 0.078 0.145 23/0/271 + 0.60 1.000 0.048 0.091 14/0/280 + 0.70 1.000 0.014 0.027 4/0/290 + +[y11x_pano_h200] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.676 0.704 0.690 207/99/87 <- best F1 + 0.10 0.901 0.558 0.689 164/18/130 + 0.15 0.963 0.446 0.609 131/5/163 + 0.20 0.963 0.354 0.517 104/4/190 + 0.25 0.986 0.248 0.397 73/1/221 + 0.30 1.000 0.173 0.296 51/0/243 + 0.40 1.000 0.092 0.168 27/0/267 + 0.50 1.000 0.051 0.097 15/0/279 + 0.60 1.000 0.020 0.040 6/0/288 + 0.70 1.000 0.003 0.007 1/0/293 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_annapolis_pano (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 222 (correct 214, incorrect 8) +Detections duplicate: 3 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 5 (abstained — not in precision or recall) +Missed ramps marked: 80 (+34 unsure, abstained) +Precision: 0.964 (95% CI 0.931-0.982) +Recall: 0.728 (95% CI 0.674-0.776) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 222 0.964 0.728 + 0.60 208 0.976 0.690 + 0.65 193 0.990 0.650 + 0.70 181 0.989 0.609 + 0.75 164 0.994 0.554 + 0.80 145 0.993 0.490 + 0.85 117 0.991 0.395 + 0.90 68 1.000 0.231 + 0.95 26 1.000 0.088 diff --git a/docs/data/yolo_geometry_51/annapolis_tiles.txt b/docs/data/yolo_geometry_51/annapolis_tiles.txt new file mode 100644 index 00000000..471f44e2 --- /dev/null +++ b/docs/data/yolo_geometry_51/annapolis_tiles.txt @@ -0,0 +1,43 @@ +Bundle: benchmark/annapolis (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_tiles 0.923 (0.870-0.955) 0.490 (0.433-0.547) 0.640 0.738 144/12/150/2 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_tiles] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.823 0.776 0.799 228/49/66 <- best F1 + 0.10 0.881 0.704 0.783 207/28/87 + 0.15 0.914 0.650 0.759 191/18/103 + 0.20 0.914 0.578 0.708 170/16/124 + 0.25 0.923 0.490 0.640 144/12/150 + 0.30 0.953 0.415 0.578 122/6/172 + 0.40 1.000 0.299 0.461 88/0/206 + 0.50 1.000 0.167 0.286 49/0/245 + 0.60 1.000 0.095 0.174 28/0/266 + 0.70 1.000 0.003 0.007 1/0/293 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_annapolis_tiles (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 222 (correct 214, incorrect 8) +Detections duplicate: 3 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 5 (abstained — not in precision or recall) +Missed ramps marked: 80 (+34 unsure, abstained) +Precision: 0.964 (95% CI 0.931-0.982) +Recall: 0.728 (95% CI 0.674-0.776) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 222 0.964 0.728 + 0.60 208 0.976 0.690 + 0.65 193 0.990 0.650 + 0.70 181 0.989 0.609 + 0.75 164 0.994 0.554 + 0.80 145 0.993 0.490 + 0.85 117 0.991 0.395 + 0.90 68 1.000 0.231 + 0.95 26 1.000 0.088 diff --git a/docs/data/yolo_geometry_51/bend_pano.txt b/docs/data/yolo_geometry_51/bend_pano.txt new file mode 100644 index 00000000..330d044c --- /dev/null +++ b/docs/data/yolo_geometry_51/bend_pano.txt @@ -0,0 +1,60 @@ +Bundle: benchmark/bend (110 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +[y11x_pano_h200] all 110 panos already cached; model load skipped +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_pano 0.974 (0.941-0.989) 0.581 (0.527-0.633) 0.728 0.775 190/5/137/2 +y11x_pano_h200 0.989 (0.961-0.997) 0.554 (0.499-0.606) 0.710 0.781 181/2/146/2 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_pano] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.650 0.810 0.721 265/143/62 + 0.10 0.819 0.761 0.789 249/55/78 <- best F1 + 0.15 0.898 0.703 0.789 230/26/97 + 0.20 0.932 0.633 0.754 207/15/120 + 0.25 0.974 0.581 0.728 190/5/137 + 0.30 0.987 0.468 0.635 153/2/174 + 0.40 0.990 0.306 0.467 100/1/227 + 0.50 1.000 0.205 0.340 67/0/260 + 0.60 1.000 0.147 0.256 48/0/279 + 0.70 1.000 0.101 0.183 33/0/294 + 0.80 1.000 0.003 0.006 1/0/326 + +[y11x_pano_h200] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.688 0.807 0.743 264/120/63 + 0.10 0.844 0.758 0.799 248/46/79 + 0.15 0.939 0.700 0.802 229/15/98 <- best F1 + 0.20 0.964 0.648 0.775 212/8/115 + 0.25 0.989 0.554 0.710 181/2/146 + 0.30 0.987 0.456 0.623 149/2/178 + 0.40 0.990 0.315 0.478 103/1/224 + 0.50 0.985 0.199 0.331 65/1/262 + 0.60 1.000 0.138 0.242 45/0/282 + 0.70 1.000 0.089 0.163 29/0/298 + 0.80 1.000 0.037 0.071 12/0/315 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_bend_pano (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 110 (of 110 seen) +Detections judged: 260 (correct 248, incorrect 12) +Detections duplicate: 1 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 5 (abstained — not in precision or recall) +Missed ramps marked: 79 (+72 unsure, abstained) +Precision: 0.954 (95% CI 0.921-0.973) +Recall: 0.758 (95% CI 0.709-0.802) [vs ramps visible in the 110 recall-pool panos] + +threshold kept precision recall + 0.55 260 0.954 0.758 + 0.60 252 0.952 0.734 + 0.65 244 0.959 0.716 + 0.70 221 0.973 0.657 + 0.75 195 0.974 0.581 + 0.80 171 0.988 0.517 + 0.85 138 0.986 0.416 + 0.90 81 0.975 0.242 + 0.95 31 0.968 0.092 diff --git a/docs/data/yolo_geometry_51/bend_tiles.txt b/docs/data/yolo_geometry_51/bend_tiles.txt new file mode 100644 index 00000000..7b2fd28a --- /dev/null +++ b/docs/data/yolo_geometry_51/bend_tiles.txt @@ -0,0 +1,43 @@ +Bundle: benchmark/bend (110 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_tiles 0.975 (0.944-0.989) 0.609 (0.555-0.660) 0.750 0.844 199/5/128/2 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_tiles] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.870 0.859 0.865 281/42/46 + 0.10 0.918 0.826 0.870 270/24/57 <- best F1 + 0.15 0.958 0.758 0.846 248/11/79 + 0.20 0.961 0.679 0.796 222/9/105 + 0.25 0.975 0.609 0.750 199/5/128 + 0.30 0.994 0.505 0.669 165/1/162 + 0.40 1.000 0.355 0.524 116/0/211 + 0.50 1.000 0.232 0.377 76/0/251 + 0.60 1.000 0.107 0.193 35/0/292 + 0.70 1.000 0.018 0.036 6/0/321 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_bend_tiles (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 110 (of 110 seen) +Detections judged: 260 (correct 248, incorrect 12) +Detections duplicate: 1 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 5 (abstained — not in precision or recall) +Missed ramps marked: 79 (+72 unsure, abstained) +Precision: 0.954 (95% CI 0.921-0.973) +Recall: 0.758 (95% CI 0.709-0.802) [vs ramps visible in the 110 recall-pool panos] + +threshold kept precision recall + 0.55 260 0.954 0.758 + 0.60 252 0.952 0.734 + 0.65 244 0.959 0.716 + 0.70 221 0.973 0.657 + 0.75 195 0.974 0.581 + 0.80 171 0.988 0.517 + 0.85 138 0.986 0.416 + 0.90 81 0.975 0.242 + 0.95 31 0.968 0.092 diff --git a/docs/data/yolo_geometry_51/budapest_district5_pano.txt b/docs/data/yolo_geometry_51/budapest_district5_pano.txt new file mode 100644 index 00000000..c3de51d1 --- /dev/null +++ b/docs/data/yolo_geometry_51/budapest_district5_pano.txt @@ -0,0 +1,56 @@ +Bundle: benchmark/budapest_district5 (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +[y11x_pano_h200] all 125 panos already cached; model load skipped +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_pano 0.732 (0.581-0.843) 0.100 (0.071-0.139) 0.176 0.444 30/11/270/1 +y11x_pano_h200 0.864 (0.733-0.936) 0.127 (0.094-0.169) 0.221 0.427 38/6/262/2 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_pano] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.467 0.630 0.536 189/216/111 <- best F1 + 0.10 0.668 0.437 0.528 131/65/169 + 0.15 0.720 0.317 0.440 95/37/205 + 0.20 0.726 0.177 0.284 53/20/247 + 0.25 0.732 0.100 0.176 30/11/270 + 0.30 0.750 0.070 0.128 21/7/279 + 0.40 0.733 0.037 0.070 11/4/289 + 0.50 0.750 0.030 0.058 9/3/291 + 0.60 0.778 0.023 0.045 7/2/293 + +[y11x_pano_h200] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.444 0.570 0.499 171/214/129 + 0.10 0.702 0.400 0.510 120/51/180 <- best F1 + 0.15 0.786 0.270 0.402 81/22/219 + 0.20 0.810 0.170 0.281 51/12/249 + 0.25 0.864 0.127 0.221 38/6/262 + 0.30 0.828 0.080 0.146 24/5/276 + 0.40 0.727 0.027 0.051 8/3/292 + 0.50 0.833 0.017 0.033 5/1/295 + 0.60 1.000 0.010 0.020 3/0/297 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_budapest_district5_pano (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 173 (correct 151, incorrect 22) +Detections duplicate: 7 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 16 (abstained — not in precision or recall) +Missed ramps marked: 149 (+48 unsure, abstained) +Precision: 0.873 (95% CI 0.815-0.915) +Recall: 0.503 (95% CI 0.447-0.560) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 173 0.873 0.503 + 0.60 139 0.914 0.423 + 0.65 119 0.924 0.367 + 0.70 99 0.919 0.303 + 0.75 73 0.904 0.220 + 0.80 50 0.920 0.153 + 0.85 23 1.000 0.077 + 0.90 9 1.000 0.030 + 0.95 2 1.000 0.007 diff --git a/docs/data/yolo_geometry_51/budapest_district5_tiles.txt b/docs/data/yolo_geometry_51/budapest_district5_tiles.txt new file mode 100644 index 00000000..79e4ecf6 --- /dev/null +++ b/docs/data/yolo_geometry_51/budapest_district5_tiles.txt @@ -0,0 +1,42 @@ +Bundle: benchmark/budapest_district5 (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_tiles 0.843 (0.740-0.910) 0.197 (0.156-0.245) 0.319 0.458 59/11/241/4 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_tiles] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.645 0.570 0.605 171/94/129 <- best F1 + 0.10 0.701 0.453 0.551 136/58/164 + 0.15 0.757 0.343 0.472 103/33/197 + 0.20 0.784 0.267 0.398 80/22/220 + 0.25 0.843 0.197 0.319 59/11/241 + 0.30 0.881 0.123 0.216 37/5/263 + 0.40 0.958 0.077 0.142 23/1/277 + 0.50 0.900 0.030 0.058 9/1/291 + 0.60 1.000 0.013 0.026 4/0/296 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_budapest_district5_tiles (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 173 (correct 151, incorrect 22) +Detections duplicate: 7 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 16 (abstained — not in precision or recall) +Missed ramps marked: 149 (+48 unsure, abstained) +Precision: 0.873 (95% CI 0.815-0.915) +Recall: 0.503 (95% CI 0.447-0.560) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 173 0.873 0.503 + 0.60 139 0.914 0.423 + 0.65 119 0.924 0.367 + 0.70 99 0.919 0.303 + 0.75 73 0.904 0.220 + 0.80 50 0.920 0.153 + 0.85 23 1.000 0.077 + 0.90 9 1.000 0.030 + 0.95 2 1.000 0.007 diff --git a/docs/data/yolo_geometry_51/clovis_pano.txt b/docs/data/yolo_geometry_51/clovis_pano.txt new file mode 100644 index 00000000..11133121 --- /dev/null +++ b/docs/data/yolo_geometry_51/clovis_pano.txt @@ -0,0 +1,57 @@ +Bundle: benchmark/clovis (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +[y11x_pano_h200] all 125 panos already cached; model load skipped +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_pano 0.918 (0.846-0.958) 0.456 (0.388-0.526) 0.610 0.720 89/8/106/1 +y11x_pano_h200 0.938 (0.864-0.973) 0.390 (0.324-0.460) 0.551 0.711 76/5/119/0 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_pano] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.520 0.815 0.635 159/147/36 + 0.10 0.678 0.733 0.704 143/68/52 + 0.15 0.809 0.672 0.734 131/31/64 <- best F1 + 0.20 0.868 0.574 0.691 112/17/83 + 0.25 0.918 0.456 0.610 89/8/106 + 0.30 0.905 0.390 0.545 76/8/119 + 0.40 0.935 0.221 0.357 43/3/152 + 0.50 1.000 0.133 0.235 26/0/169 + 0.60 1.000 0.087 0.160 17/0/178 + 0.70 1.000 0.031 0.060 6/0/189 + +[y11x_pano_h200] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.573 0.785 0.662 153/114/42 + 0.10 0.788 0.687 0.734 134/36/61 <- best F1 + 0.15 0.866 0.595 0.705 116/18/79 + 0.20 0.917 0.508 0.653 99/9/96 + 0.25 0.938 0.390 0.551 76/5/119 + 0.30 0.948 0.282 0.435 55/3/140 + 0.40 1.000 0.164 0.282 32/0/163 + 0.50 1.000 0.103 0.186 20/0/175 + 0.60 1.000 0.046 0.088 9/0/186 + 0.70 1.000 0.005 0.010 1/0/194 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_clovis_pano (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 152 (correct 139, incorrect 13) +Detections unsure: 12 (abstained — not in precision or recall) +Missed ramps marked: 56 (+21 unsure, abstained) +Precision: 0.914 (95% CI 0.859-0.949) +Recall: 0.713 (95% CI 0.646-0.772) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 152 0.914 0.713 + 0.60 140 0.936 0.672 + 0.65 126 0.937 0.605 + 0.70 115 0.948 0.559 + 0.75 96 0.938 0.462 + 0.80 75 0.973 0.374 + 0.85 47 0.979 0.236 + 0.90 25 0.960 0.123 + 0.95 6 1.000 0.031 diff --git a/docs/data/yolo_geometry_51/clovis_tiles.txt b/docs/data/yolo_geometry_51/clovis_tiles.txt new file mode 100644 index 00000000..79f6deb2 --- /dev/null +++ b/docs/data/yolo_geometry_51/clovis_tiles.txt @@ -0,0 +1,42 @@ +Bundle: benchmark/clovis (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_tiles 0.885 (0.820-0.929) 0.595 (0.525-0.661) 0.712 0.769 116/15/79/4 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_tiles] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.673 0.856 0.754 167/81/28 + 0.10 0.743 0.800 0.770 156/54/39 + 0.15 0.802 0.749 0.775 146/36/49 <- best F1 + 0.20 0.826 0.682 0.747 133/28/62 + 0.25 0.885 0.595 0.712 116/15/79 + 0.30 0.895 0.482 0.627 94/11/101 + 0.40 0.940 0.323 0.481 63/4/132 + 0.50 0.919 0.174 0.293 34/3/161 + 0.60 1.000 0.056 0.107 11/0/184 + 0.70 1.000 0.015 0.030 3/0/192 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_clovis_tiles (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 152 (correct 139, incorrect 13) +Detections unsure: 12 (abstained — not in precision or recall) +Missed ramps marked: 56 (+21 unsure, abstained) +Precision: 0.914 (95% CI 0.859-0.949) +Recall: 0.713 (95% CI 0.646-0.772) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 152 0.914 0.713 + 0.60 140 0.936 0.672 + 0.65 126 0.937 0.605 + 0.70 115 0.948 0.559 + 0.75 96 0.938 0.462 + 0.80 75 0.973 0.374 + 0.85 47 0.979 0.236 + 0.90 25 0.960 0.123 + 0.95 6 1.000 0.031 diff --git a/docs/data/yolo_geometry_51/driver.log b/docs/data/yolo_geometry_51/driver.log new file mode 100644 index 00000000..3c85f9df --- /dev/null +++ b/docs/data/yolo_geometry_51/driver.log @@ -0,0 +1,53 @@ +run started : 2026-08-30T17:10:09-07:00 +host : makelab2.cs.washington.edu +repo HEAD : d964d5db2037451ef4060efc8419d90d1794190b Merge pull request #142 from ProjectSidewalk/docs/seam-amendment +repo dirty : 10 paths +torch 2.13.0+cu130 cuda True +ultralytics 8.4.120 +gpu : NVIDIA A40 +checkpoint sha256: +63bbe5d6ed773b04b79a46e2ab6b3c216abea8f369aaa18dc53404269abde077 /homes/gws/jonf/RampNet/yolo_ckpts/y11x_tiles.pt +0653409f9e1b309e479153c5967fadadf74f88913dceb5c1527db97dd0d958fe /homes/gws/jonf/RampNet/yolo_ckpts/y11x_pano.pt +8ee4f7bfc372270071dd1f94ae3130adbefe4fec27acac8b4e8887ede6600159 /homes/gws/jonf/RampNet/yolo_ckpts/y11x_pano_h200.pt +splits : manual_gold bend richmond annapolis budapest_district5 clovis gainesville morgantown paterson sao_paulo +=== manual_gold tiles start 2026-08-30T17:10:13-07:00 +=== manual_gold tiles exit=0 elapsed=1229s +=== manual_gold pano start 2026-08-30T17:30:42-07:00 +=== manual_gold pano exit=0 elapsed=140s +=== bend tiles start 2026-08-30T17:33:02-07:00 +=== bend tiles exit=0 elapsed=241s +=== bend pano start 2026-08-30T17:37:03-07:00 +=== bend pano exit=0 elapsed=143s +=== richmond tiles start 2026-08-30T17:39:26-07:00 +=== richmond tiles exit=0 elapsed=175s +=== richmond pano start 2026-08-30T17:42:21-07:00 +=== richmond pano exit=0 elapsed=67s +=== annapolis tiles start 2026-08-30T17:43:28-07:00 +=== annapolis tiles exit=0 elapsed=166s +=== annapolis pano start 2026-08-30T17:46:14-07:00 +=== annapolis pano exit=0 elapsed=51s +=== budapest_district5 tiles start 2026-08-30T17:47:05-07:00 +=== budapest_district5 tiles exit=0 elapsed=120s +=== budapest_district5 pano start 2026-08-30T17:49:05-07:00 +=== budapest_district5 pano exit=0 elapsed=32s +=== clovis tiles start 2026-08-30T17:49:37-07:00 +=== clovis tiles exit=0 elapsed=153s +=== clovis pano start 2026-08-30T17:52:10-07:00 +=== clovis pano exit=0 elapsed=50s +=== gainesville tiles start 2026-08-30T17:53:00-07:00 +=== gainesville tiles exit=0 elapsed=276s +=== gainesville pano start 2026-08-30T17:57:36-07:00 +=== gainesville pano exit=0 elapsed=154s +=== morgantown tiles start 2026-08-30T18:00:10-07:00 +=== morgantown tiles exit=0 elapsed=120s +=== morgantown pano start 2026-08-30T18:02:10-07:00 +=== morgantown pano exit=0 elapsed=22s +=== paterson tiles start 2026-08-30T18:02:32-07:00 +=== paterson tiles exit=0 elapsed=286s +=== paterson pano start 2026-08-30T18:07:18-07:00 +=== paterson pano exit=0 elapsed=173s +=== sao_paulo tiles start 2026-08-30T18:10:11-07:00 +=== sao_paulo tiles exit=0 elapsed=287s +=== sao_paulo pano start 2026-08-30T18:14:58-07:00 +=== sao_paulo pano exit=0 elapsed=170s +ALL_SPLITS_DONE 2026-08-30T18:17:48-07:00 diff --git a/docs/data/yolo_geometry_51/env.txt b/docs/data/yolo_geometry_51/env.txt new file mode 100644 index 00000000..0ad4a94c --- /dev/null +++ b/docs/data/yolo_geometry_51/env.txt @@ -0,0 +1,12 @@ +run started : 2026-08-30T17:10:09-07:00 +host : makelab2.cs.washington.edu +repo HEAD : d964d5db2037451ef4060efc8419d90d1794190b Merge pull request #142 from ProjectSidewalk/docs/seam-amendment +repo dirty : 10 paths +torch 2.13.0+cu130 cuda True +ultralytics 8.4.120 +gpu : NVIDIA A40 +checkpoint sha256: +63bbe5d6ed773b04b79a46e2ab6b3c216abea8f369aaa18dc53404269abde077 /homes/gws/jonf/RampNet/yolo_ckpts/y11x_tiles.pt +0653409f9e1b309e479153c5967fadadf74f88913dceb5c1527db97dd0d958fe /homes/gws/jonf/RampNet/yolo_ckpts/y11x_pano.pt +8ee4f7bfc372270071dd1f94ae3130adbefe4fec27acac8b4e8887ede6600159 /homes/gws/jonf/RampNet/yolo_ckpts/y11x_pano_h200.pt +splits : manual_gold bend richmond annapolis budapest_district5 clovis gainesville morgantown paterson sao_paulo diff --git a/docs/data/yolo_geometry_51/gainesville_pano.txt b/docs/data/yolo_geometry_51/gainesville_pano.txt new file mode 100644 index 00000000..908bc1b9 --- /dev/null +++ b/docs/data/yolo_geometry_51/gainesville_pano.txt @@ -0,0 +1,60 @@ +Bundle: benchmark/gainesville (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +[y11x_pano_h200] all 125 panos already cached; model load skipped +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_pano 0.913 (0.842-0.953) 0.346 (0.292-0.404) 0.501 0.649 94/9/178/2 +y11x_pano_h200 0.948 (0.885-0.978) 0.338 (0.285-0.396) 0.499 0.616 92/5/180/1 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_pano] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.529 0.746 0.619 203/181/69 + 0.10 0.735 0.621 0.673 169/61/103 <- best F1 + 0.15 0.859 0.515 0.644 140/23/132 + 0.20 0.895 0.438 0.588 119/14/153 + 0.25 0.913 0.346 0.501 94/9/178 + 0.30 0.923 0.265 0.411 72/6/200 + 0.40 0.935 0.158 0.270 43/3/229 + 0.50 0.964 0.099 0.180 27/1/245 + 0.60 1.000 0.051 0.098 14/0/258 + 0.70 1.000 0.018 0.036 5/0/267 + 0.80 1.000 0.004 0.007 1/0/271 + +[y11x_pano_h200] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.559 0.680 0.614 185/146/87 + 0.10 0.761 0.596 0.668 162/51/110 <- best F1 + 0.15 0.867 0.504 0.637 137/21/135 + 0.20 0.922 0.434 0.590 118/10/154 + 0.25 0.948 0.338 0.499 92/5/180 + 0.30 0.970 0.235 0.379 64/2/208 + 0.40 0.974 0.140 0.244 38/1/234 + 0.50 1.000 0.074 0.137 20/0/252 + 0.60 1.000 0.040 0.078 11/0/261 + 0.70 1.000 0.029 0.057 8/0/264 + 0.80 1.000 0.004 0.007 1/0/271 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_gainesville_pano (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 200 (correct 189, incorrect 11) +Detections duplicate: 2 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 5 (abstained — not in precision or recall) +Missed ramps marked: 83 (+31 unsure, abstained) +Precision: 0.945 (95% CI 0.904-0.969) +Recall: 0.695 (95% CI 0.638-0.747) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 200 0.945 0.695 + 0.60 180 0.961 0.636 + 0.65 164 0.970 0.585 + 0.70 150 0.973 0.537 + 0.75 128 0.984 0.463 + 0.80 103 1.000 0.379 + 0.85 79 1.000 0.290 + 0.90 45 1.000 0.165 + 0.95 16 1.000 0.059 diff --git a/docs/data/yolo_geometry_51/gainesville_tiles.txt b/docs/data/yolo_geometry_51/gainesville_tiles.txt new file mode 100644 index 00000000..727b9b61 --- /dev/null +++ b/docs/data/yolo_geometry_51/gainesville_tiles.txt @@ -0,0 +1,42 @@ +Bundle: benchmark/gainesville (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_tiles 0.957 (0.904-0.982) 0.412 (0.355-0.471) 0.576 0.751 112/5/160/2 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_tiles] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.800 0.794 0.797 216/54/56 <- best F1 + 0.10 0.862 0.710 0.778 193/31/79 + 0.15 0.904 0.592 0.716 161/17/111 + 0.20 0.926 0.504 0.652 137/11/135 + 0.25 0.957 0.412 0.576 112/5/160 + 0.30 0.959 0.342 0.504 93/4/179 + 0.40 1.000 0.191 0.321 52/0/220 + 0.50 1.000 0.129 0.228 35/0/237 + 0.60 1.000 0.055 0.105 15/0/257 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_gainesville_tiles (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 200 (correct 189, incorrect 11) +Detections duplicate: 2 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 5 (abstained — not in precision or recall) +Missed ramps marked: 83 (+31 unsure, abstained) +Precision: 0.945 (95% CI 0.904-0.969) +Recall: 0.695 (95% CI 0.638-0.747) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 200 0.945 0.695 + 0.60 180 0.961 0.636 + 0.65 164 0.970 0.585 + 0.70 150 0.973 0.537 + 0.75 128 0.984 0.463 + 0.80 103 1.000 0.379 + 0.85 79 1.000 0.290 + 0.90 45 1.000 0.165 + 0.95 16 1.000 0.059 diff --git a/docs/data/yolo_geometry_51/manual_gold_pano.txt b/docs/data/yolo_geometry_51/manual_gold_pano.txt new file mode 100644 index 00000000..00447958 --- /dev/null +++ b/docs/data/yolo_geometry_51/manual_gold_pano.txt @@ -0,0 +1,42 @@ +Bundle: benchmark/manual_gold (1000 scored panos) match radius 0.022 ground truth: independent manual labels (YOLO box centers) +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +[y11x_pano_h200] all 1000 panos already cached; model load skipped +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_pano 0.952 (0.943-0.959) 0.751 (0.737-0.764) 0.840 0.916 2944/150/975/0 +y11x_pano_h200 0.956 (0.948-0.962) 0.767 (0.753-0.780) 0.851 0.931 3005/139/914/0 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_pano] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.638 0.958 0.766 3754/2129/165 + 0.10 0.780 0.919 0.844 3601/1013/318 + 0.15 0.869 0.878 0.873 3440/520/479 <- best F1 + 0.20 0.916 0.819 0.864 3208/295/711 + 0.25 0.952 0.751 0.840 2944/150/975 + 0.30 0.965 0.675 0.794 2644/97/1275 + 0.40 0.976 0.522 0.680 2044/51/1875 + 0.50 0.983 0.379 0.547 1487/26/2432 + 0.60 0.986 0.278 0.433 1088/15/2831 + 0.70 0.995 0.192 0.321 751/4/3168 + 0.80 0.996 0.057 0.108 224/1/3695 + +[y11x_pano_h200] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.665 0.961 0.786 3766/1901/153 + 0.10 0.823 0.928 0.872 3637/782/282 + 0.15 0.896 0.891 0.893 3491/407/428 <- best F1 + 0.20 0.937 0.833 0.882 3264/218/655 + 0.25 0.956 0.767 0.851 3005/139/914 + 0.30 0.971 0.691 0.808 2709/81/1210 + 0.40 0.986 0.536 0.695 2101/30/1818 + 0.50 0.991 0.384 0.554 1506/13/2413 + 0.60 0.994 0.281 0.438 1102/7/2817 + 0.70 0.998 0.214 0.353 840/2/3079 + 0.80 0.997 0.093 0.170 364/1/3555 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_manual_gold_pano (JSON + pr_curves.png) + +Manual-GT bundle: no verdict-based cross-check (validate the rampnet row against the published gold-set numbers instead; see docs/model_comparison.md). diff --git a/docs/data/yolo_geometry_51/manual_gold_tiles.txt b/docs/data/yolo_geometry_51/manual_gold_tiles.txt new file mode 100644 index 00000000..c601976b --- /dev/null +++ b/docs/data/yolo_geometry_51/manual_gold_tiles.txt @@ -0,0 +1,26 @@ +Bundle: benchmark/manual_gold (1000 scored panos) match radius 0.022 ground truth: independent manual labels (YOLO box centers) +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_tiles 0.970 (0.963-0.976) 0.711 (0.697-0.725) 0.820 0.909 2786/86/1133/0 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_tiles] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.884 0.925 0.904 3627/478/292 + 0.10 0.930 0.893 0.911 3499/263/420 <- best F1 + 0.15 0.949 0.845 0.894 3311/178/608 + 0.20 0.964 0.777 0.860 3046/115/873 + 0.25 0.970 0.711 0.820 2786/86/1133 + 0.30 0.978 0.645 0.777 2527/58/1392 + 0.40 0.986 0.508 0.671 1992/29/1927 + 0.50 0.995 0.373 0.543 1462/8/2457 + 0.60 0.997 0.221 0.362 866/3/3053 + 0.70 0.997 0.073 0.136 287/1/3632 + 0.80 1.000 0.000 0.001 1/0/3918 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_manual_gold_tiles (JSON + pr_curves.png) + +Manual-GT bundle: no verdict-based cross-check (validate the rampnet row against the published gold-set numbers instead; see docs/model_comparison.md). diff --git a/docs/data/yolo_geometry_51/morgantown_pano.txt b/docs/data/yolo_geometry_51/morgantown_pano.txt new file mode 100644 index 00000000..74947d7f --- /dev/null +++ b/docs/data/yolo_geometry_51/morgantown_pano.txt @@ -0,0 +1,57 @@ +Bundle: benchmark/morgantown (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +[y11x_pano_h200] all 125 panos already cached; model load skipped +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_pano 0.935 (0.888-0.963) 0.596 (0.536-0.653) 0.728 0.796 159/11/108/0 +y11x_pano_h200 0.993 (0.961-0.999) 0.524 (0.465-0.583) 0.686 0.790 140/1/127/0 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_pano] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.652 0.850 0.738 227/121/40 + 0.10 0.803 0.779 0.791 208/51/59 + 0.15 0.867 0.730 0.793 195/30/72 <- best F1 + 0.20 0.926 0.652 0.765 174/14/93 + 0.25 0.935 0.596 0.728 159/11/108 + 0.30 0.952 0.524 0.676 140/7/127 + 0.40 0.950 0.360 0.522 96/5/171 + 0.50 0.987 0.288 0.446 77/1/190 + 0.60 1.000 0.202 0.336 54/0/213 + 0.70 1.000 0.097 0.177 26/0/241 + +[y11x_pano_h200] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.714 0.813 0.760 217/87/50 + 0.10 0.900 0.742 0.813 198/22/69 <- best F1 + 0.15 0.963 0.678 0.796 181/7/86 + 0.20 0.982 0.599 0.744 160/3/107 + 0.25 0.993 0.524 0.686 140/1/127 + 0.30 0.991 0.419 0.589 112/1/155 + 0.40 0.988 0.296 0.455 79/1/188 + 0.50 0.984 0.228 0.371 61/1/206 + 0.60 1.000 0.154 0.266 41/0/226 + 0.70 1.000 0.034 0.065 9/0/258 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_morgantown_pano (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 200 (correct 195, incorrect 5) +Detections unsure: 9 (abstained — not in precision or recall) +Missed ramps marked: 72 (+20 unsure, abstained) +Precision: 0.975 (95% CI 0.943-0.989) +Recall: 0.730 (95% CI 0.674-0.780) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 200 0.975 0.730 + 0.60 189 0.984 0.697 + 0.65 169 0.988 0.625 + 0.70 159 0.994 0.592 + 0.75 140 1.000 0.524 + 0.80 123 1.000 0.461 + 0.85 99 1.000 0.371 + 0.90 57 1.000 0.213 + 0.95 16 1.000 0.060 diff --git a/docs/data/yolo_geometry_51/morgantown_tiles.txt b/docs/data/yolo_geometry_51/morgantown_tiles.txt new file mode 100644 index 00000000..b104a332 --- /dev/null +++ b/docs/data/yolo_geometry_51/morgantown_tiles.txt @@ -0,0 +1,42 @@ +Bundle: benchmark/morgantown (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_tiles 0.975 (0.937-0.990) 0.581 (0.521-0.638) 0.728 0.824 155/4/112/2 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_tiles] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.839 0.839 0.839 224/43/43 + 0.10 0.923 0.809 0.862 216/18/51 <- best F1 + 0.15 0.948 0.757 0.842 202/11/65 + 0.20 0.973 0.678 0.799 181/5/86 + 0.25 0.975 0.581 0.728 155/4/112 + 0.30 0.986 0.517 0.678 138/2/129 + 0.40 0.991 0.416 0.586 111/1/156 + 0.50 1.000 0.266 0.420 71/0/196 + 0.60 1.000 0.124 0.220 33/0/234 + 0.70 1.000 0.022 0.044 6/0/261 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_morgantown_tiles (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 200 (correct 195, incorrect 5) +Detections unsure: 9 (abstained — not in precision or recall) +Missed ramps marked: 72 (+20 unsure, abstained) +Precision: 0.975 (95% CI 0.943-0.989) +Recall: 0.730 (95% CI 0.674-0.780) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 200 0.975 0.730 + 0.60 189 0.984 0.697 + 0.65 169 0.988 0.625 + 0.70 159 0.994 0.592 + 0.75 140 1.000 0.524 + 0.80 123 1.000 0.461 + 0.85 99 1.000 0.371 + 0.90 57 1.000 0.213 + 0.95 16 1.000 0.060 diff --git a/docs/data/yolo_geometry_51/paterson_pano.txt b/docs/data/yolo_geometry_51/paterson_pano.txt new file mode 100644 index 00000000..a70e3d8c --- /dev/null +++ b/docs/data/yolo_geometry_51/paterson_pano.txt @@ -0,0 +1,60 @@ +Bundle: benchmark/paterson (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +[y11x_pano_h200] all 125 panos already cached; model load skipped +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_pano 0.974 (0.940-0.989) 0.471 (0.422-0.520) 0.635 0.816 186/5/209/3 +y11x_pano_h200 0.974 (0.940-0.989) 0.471 (0.422-0.520) 0.635 0.800 186/5/209/2 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_pano] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.653 0.876 0.748 346/184/49 + 0.10 0.848 0.747 0.794 295/53/100 <- best F1 + 0.15 0.912 0.630 0.746 249/24/146 + 0.20 0.934 0.539 0.684 213/15/182 + 0.25 0.974 0.471 0.635 186/5/209 + 0.30 0.976 0.408 0.575 161/4/234 + 0.40 1.000 0.251 0.401 99/0/296 + 0.50 1.000 0.159 0.275 63/0/332 + 0.60 1.000 0.111 0.200 44/0/351 + 0.70 1.000 0.061 0.115 24/0/371 + 0.80 1.000 0.005 0.010 2/0/393 + +[y11x_pano_h200] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.660 0.856 0.745 338/174/57 + 0.10 0.820 0.737 0.776 291/64/104 <- best F1 + 0.15 0.919 0.635 0.751 251/22/144 + 0.20 0.957 0.559 0.706 221/10/174 + 0.25 0.974 0.471 0.635 186/5/209 + 0.30 0.982 0.405 0.573 160/3/235 + 0.40 0.990 0.261 0.413 103/1/292 + 0.50 1.000 0.182 0.308 72/0/323 + 0.60 1.000 0.104 0.188 41/0/354 + 0.70 1.000 0.068 0.128 27/0/368 + 0.80 1.000 0.020 0.040 8/0/387 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_paterson_pano (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 277 (correct 270, incorrect 7) +Detections duplicate: 1 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 7 (abstained — not in precision or recall) +Missed ramps marked: 125 (+36 unsure, abstained) +Precision: 0.975 (95% CI 0.949-0.988) +Recall: 0.684 (95% CI 0.636-0.727) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 277 0.975 0.684 + 0.60 270 0.974 0.666 + 0.65 259 0.977 0.641 + 0.70 248 0.984 0.618 + 0.75 222 0.982 0.552 + 0.80 199 0.980 0.494 + 0.85 156 1.000 0.395 + 0.90 88 1.000 0.223 + 0.95 25 1.000 0.063 diff --git a/docs/data/yolo_geometry_51/paterson_tiles.txt b/docs/data/yolo_geometry_51/paterson_tiles.txt new file mode 100644 index 00000000..43d10b41 --- /dev/null +++ b/docs/data/yolo_geometry_51/paterson_tiles.txt @@ -0,0 +1,43 @@ +Bundle: benchmark/paterson (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_tiles 0.977 (0.942-0.991) 0.428 (0.380-0.477) 0.595 0.715 169/4/226/3 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_tiles] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.900 0.732 0.807 289/32/106 <- best F1 + 0.10 0.944 0.638 0.761 252/15/143 + 0.15 0.964 0.544 0.696 215/8/180 + 0.20 0.970 0.486 0.648 192/6/203 + 0.25 0.977 0.428 0.595 169/4/226 + 0.30 0.979 0.362 0.529 143/3/252 + 0.40 1.000 0.276 0.433 109/0/286 + 0.50 1.000 0.182 0.308 72/0/323 + 0.60 1.000 0.106 0.192 42/0/353 + 0.70 1.000 0.028 0.054 11/0/384 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_paterson_tiles (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 277 (correct 270, incorrect 7) +Detections duplicate: 1 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 7 (abstained — not in precision or recall) +Missed ramps marked: 125 (+36 unsure, abstained) +Precision: 0.975 (95% CI 0.949-0.988) +Recall: 0.684 (95% CI 0.636-0.727) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 277 0.975 0.684 + 0.60 270 0.974 0.666 + 0.65 259 0.977 0.641 + 0.70 248 0.984 0.618 + 0.75 222 0.982 0.552 + 0.80 199 0.980 0.494 + 0.85 156 1.000 0.395 + 0.90 88 1.000 0.223 + 0.95 25 1.000 0.063 diff --git a/docs/data/yolo_geometry_51/richmond_pano.txt b/docs/data/yolo_geometry_51/richmond_pano.txt new file mode 100644 index 00000000..a5ce05d7 --- /dev/null +++ b/docs/data/yolo_geometry_51/richmond_pano.txt @@ -0,0 +1,59 @@ +Bundle: benchmark/richmond (124 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +[y11x_pano_h200] all 124 panos already cached; model load skipped +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_pano 0.922 (0.876-0.952) 0.574 (0.519-0.628) 0.708 0.827 178/15/132/12 +y11x_pano_h200 0.952 (0.899-0.978) 0.384 (0.331-0.439) 0.547 0.748 119/6/191/4 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_pano] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.566 0.897 0.694 278/213/32 + 0.10 0.767 0.816 0.791 253/77/57 + 0.15 0.862 0.745 0.799 231/37/79 <- best F1 + 0.20 0.889 0.671 0.765 208/26/102 + 0.25 0.922 0.574 0.708 178/15/132 + 0.30 0.951 0.497 0.653 154/8/156 + 0.40 0.972 0.342 0.506 106/3/204 + 0.50 0.974 0.245 0.392 76/2/234 + 0.60 0.982 0.177 0.301 55/1/255 + 0.70 1.000 0.100 0.182 31/0/279 + 0.80 1.000 0.003 0.006 1/0/309 + +[y11x_pano_h200] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.621 0.803 0.700 249/152/61 + 0.10 0.862 0.706 0.777 219/35/91 <- best F1 + 0.15 0.941 0.613 0.742 190/12/120 + 0.20 0.949 0.484 0.641 150/8/160 + 0.25 0.952 0.384 0.547 119/6/191 + 0.30 0.943 0.268 0.417 83/5/227 + 0.40 0.915 0.139 0.241 43/4/267 + 0.50 0.955 0.068 0.127 21/1/289 + 0.60 1.000 0.029 0.056 9/0/301 + 0.70 1.000 0.010 0.019 3/0/307 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_richmond_pano (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 124 (of 124 seen) +Detections judged: 247 (correct 237, incorrect 10) +Detections duplicate: 1 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 20 (abstained — not in precision or recall) +Missed ramps marked: 73 (+78 unsure, abstained) +Precision: 0.960 (95% CI 0.927-0.978) +Recall: 0.765 (95% CI 0.714-0.808) [vs ramps visible in the 124 recall-pool panos] + +threshold kept precision recall + 0.55 247 0.960 0.765 + 0.60 240 0.963 0.745 + 0.65 226 0.982 0.716 + 0.70 210 0.986 0.668 + 0.75 195 0.990 0.623 + 0.80 178 0.989 0.568 + 0.85 151 0.993 0.484 + 0.90 90 1.000 0.290 + 0.95 31 1.000 0.100 diff --git a/docs/data/yolo_geometry_51/richmond_tiles.txt b/docs/data/yolo_geometry_51/richmond_tiles.txt new file mode 100644 index 00000000..d86a801a --- /dev/null +++ b/docs/data/yolo_geometry_51/richmond_tiles.txt @@ -0,0 +1,43 @@ +Bundle: benchmark/richmond (124 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_tiles 0.920 (0.871-0.952) 0.523 (0.467-0.578) 0.667 0.773 162/14/148/8 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_tiles] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.789 0.823 0.806 255/68/55 <- best F1 + 0.10 0.855 0.758 0.803 235/40/75 + 0.15 0.889 0.671 0.765 208/26/102 + 0.20 0.911 0.594 0.719 184/18/126 + 0.25 0.920 0.523 0.667 162/14/148 + 0.30 0.934 0.455 0.612 141/10/169 + 0.40 0.980 0.319 0.482 99/2/211 + 0.50 0.985 0.206 0.341 64/1/246 + 0.60 0.969 0.100 0.181 31/1/279 + 0.70 1.000 0.032 0.062 10/0/300 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_richmond_tiles (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 124 (of 124 seen) +Detections judged: 247 (correct 237, incorrect 10) +Detections duplicate: 1 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 20 (abstained — not in precision or recall) +Missed ramps marked: 73 (+78 unsure, abstained) +Precision: 0.960 (95% CI 0.927-0.978) +Recall: 0.765 (95% CI 0.714-0.808) [vs ramps visible in the 124 recall-pool panos] + +threshold kept precision recall + 0.55 247 0.960 0.765 + 0.60 240 0.963 0.745 + 0.65 226 0.982 0.716 + 0.70 210 0.986 0.668 + 0.75 195 0.990 0.623 + 0.80 178 0.989 0.568 + 0.85 151 0.993 0.484 + 0.90 90 1.000 0.290 + 0.95 31 1.000 0.100 diff --git a/docs/data/yolo_geometry_51/sao_paulo_pano.txt b/docs/data/yolo_geometry_51/sao_paulo_pano.txt new file mode 100644 index 00000000..a9dc90d0 --- /dev/null +++ b/docs/data/yolo_geometry_51/sao_paulo_pano.txt @@ -0,0 +1,60 @@ +Bundle: benchmark/sao_paulo (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +[y11x_pano_h200] all 125 panos already cached; model load skipped +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_pano 0.910 (0.855-0.946) 0.505 (0.447-0.563) 0.650 0.761 142/14/139/8 +y11x_pano_h200 0.923 (0.870-0.955) 0.512 (0.454-0.570) 0.659 0.783 144/12/137/12 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_pano] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.507 0.861 0.639 242/235/39 + 0.10 0.715 0.776 0.744 218/87/63 <- best F1 + 0.15 0.820 0.680 0.743 191/42/90 + 0.20 0.874 0.594 0.708 167/24/114 + 0.25 0.910 0.505 0.650 142/14/139 + 0.30 0.932 0.441 0.599 124/9/157 + 0.40 0.939 0.331 0.489 93/6/188 + 0.50 0.932 0.242 0.384 68/5/213 + 0.60 0.913 0.149 0.257 42/4/239 + 0.70 0.960 0.085 0.157 24/1/257 + 0.80 1.000 0.007 0.014 2/0/279 + +[y11x_pano_h200] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.501 0.872 0.636 245/244/36 + 0.10 0.730 0.779 0.754 219/81/62 <- best F1 + 0.15 0.825 0.669 0.739 188/40/93 + 0.20 0.894 0.601 0.719 169/20/112 + 0.25 0.923 0.512 0.659 144/12/137 + 0.30 0.933 0.445 0.602 125/9/156 + 0.40 0.966 0.306 0.465 86/3/195 + 0.50 1.000 0.203 0.337 57/0/224 + 0.60 1.000 0.121 0.216 34/0/247 + 0.70 1.000 0.060 0.114 17/0/264 + 0.80 1.000 0.004 0.007 1/0/280 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_sao_paulo_pano (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 214 (correct 190, incorrect 24) +Detections duplicate: 1 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 37 (abstained — not in precision or recall) +Missed ramps marked: 91 (+43 unsure, abstained) +Precision: 0.888 (95% CI 0.839-0.923) +Recall: 0.676 (95% CI 0.619-0.728) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 214 0.888 0.676 + 0.60 197 0.909 0.637 + 0.65 185 0.930 0.612 + 0.70 167 0.928 0.552 + 0.75 146 0.938 0.488 + 0.80 128 0.953 0.434 + 0.85 96 0.979 0.335 + 0.90 59 1.000 0.210 + 0.95 23 1.000 0.082 diff --git a/docs/data/yolo_geometry_51/sao_paulo_tiles.txt b/docs/data/yolo_geometry_51/sao_paulo_tiles.txt new file mode 100644 index 00000000..7a244161 --- /dev/null +++ b/docs/data/yolo_geometry_51/sao_paulo_tiles.txt @@ -0,0 +1,43 @@ +Bundle: benchmark/sao_paulo (125 scored panos) match radius 0.022 ground truth: reviewer-confirmed ramps + missed marks +Detection cache: /homes/gws/jonf/RampNet/.model_cache + +Operating point: predictions with confidence < 0.25 dropped (models without confidences are unaffected). +model P 95% CI R 95% CI F1 AP tp/fp/fn/ign +------------------------------------------------------------------------------------------------------------------- +y11x_tiles 0.894 (0.835-0.934) 0.480 (0.423-0.539) 0.625 0.751 135/16/146/9 +AP: all-point interpolated, over the recall-confirmed panos, from the full confidence range (--op-threshold does not truncate it); '-' = no calibrated per-box score. + +[y11x_tiles] threshold sweep (re-scored from cached detections) + thr P R F1 tp/fp/fn + 0.05 0.722 0.815 0.766 229/88/52 + 0.10 0.802 0.751 0.776 211/52/70 <- best F1 + 0.15 0.844 0.676 0.751 190/35/91 + 0.20 0.894 0.601 0.719 169/20/112 + 0.25 0.894 0.480 0.625 135/16/146 + 0.30 0.921 0.413 0.570 116/10/165 + 0.40 0.974 0.267 0.419 75/2/206 + 0.50 0.979 0.167 0.286 47/1/234 + 0.60 1.000 0.075 0.139 21/0/260 + 0.70 1.000 0.004 0.007 1/0/280 + +PR curves written to /homes/gws/jonf/RampNet/yolo_eval_results_geometry_51/pr_sao_paulo_tiles (JSON + pr_curves.png) + +--- RampNet verdict-based cross-check --- +Panos fully judged: 125 (of 125 seen) +Detections judged: 214 (correct 190, incorrect 24) +Detections duplicate: 1 (redundant hits on an already-counted ramp; folded into the numbers above per this run's duplicate scoring) +Detections unsure: 37 (abstained — not in precision or recall) +Missed ramps marked: 91 (+43 unsure, abstained) +Precision: 0.888 (95% CI 0.839-0.923) +Recall: 0.676 (95% CI 0.619-0.728) [vs ramps visible in the 125 recall-pool panos] + +threshold kept precision recall + 0.55 214 0.888 0.676 + 0.60 197 0.909 0.637 + 0.65 185 0.930 0.612 + 0.70 167 0.928 0.552 + 0.75 146 0.938 0.488 + 0.80 128 0.953 0.434 + 0.85 96 0.979 0.335 + 0.90 59 1.000 0.210 + 0.95 23 1.000 0.082 diff --git a/docs/model_comparison.md b/docs/model_comparison.md index 46c3971b..70610585 100644 --- a/docs/model_comparison.md +++ b/docs/model_comparison.md @@ -144,7 +144,7 @@ sweep from a 0.05 peak floor across all nine splits, per-imagery-tier curves, th confidence-calibration tables, and the recommendation to lower the deployment threshold from 0.55 to 0.30. -**Supervised baseline (issue #51): the pano trio is benchmarked; tiles still training.** +**Supervised baseline (issue #51): the pano trio is benchmarked, and the tiles arm has since been scored too.** The roster above is all zero-shot except RampNet. The supervised **YOLO** baseline — the architecture-vs-data ablation (does a generic detector trained on the RampNet dataset *also* beat the zero-shot field, or is RampNet's keypoint architecture doing the work?) — completed @@ -182,9 +182,17 @@ verify-identical) are in the [training record](../scripts/model_comparison/yolo_baseline/README.md) and its `benchmark_eval/` directory. Training-side history (the warmup-LR collapse at epoch 3 across all arms, the ckpt slice ceiling, the `y26_tiles` fork) stays in that record; the stabilized -rerun remains tracked in #70 and the caveat write-up in #72. The tiles arms — the -resolution-controlled half of the ablation, and the geometry the VLM rows are scored with — -are still training and are deliberately absent from every table above. +rerun remains tracked in #70 and the caveat write-up in #72. + +The tiles arms — the resolution-controlled half of the ablation, and the geometry the VLM +rows are scored with — are **absent from every table above because their detections are not +published**, not because they are unmeasured. `y11x_tiles` (ep44) was scored on all ten +splits on 2026-08-30: the equirect handicap is real and worth about **0.044 F1**, and the +tiles arm is the best YOLO cell in the grid +([`yolo_geometry_51.md`](yolo_geometry_51.md)). Two further caveats belong with the YOLO +rows above: they are all read at conf **0.25**, the Ultralytics default that nobody +selected, and at a threshold selected the same way as RampNet's the residual to RampNet is +**0.039**, not 0.252 ([`operating_point_parity_51.md`](operating_point_parity_51.md)). Three classes of challenger, which fail differently and are worth keeping distinct: diff --git a/docs/model_scoreboard.md b/docs/model_scoreboard.md index 34707dd3..7d1629bb 100644 --- a/docs/model_scoreboard.md +++ b/docs/model_scoreboard.md @@ -25,6 +25,18 @@ Macro-mean over the seven pooled US city splits, each city weighted equally. **Read the operating-point column before comparing rows** — it is not the same for every model, and the reasons are in "How to read this" below. +> **Read the `op` column before comparing two rows — and know what it costs.** The +> operating points in this table were **not chosen by one procedure**: RampNet is at its +> shipped deployment threshold 0.55, and every YOLO row is at 0.25, the Ultralytics +> `predict()` default that nobody selected. Give each model one uniform threshold picked +> the same way, on a split the headline is never reported over, and the RampNet-vs-YOLO +> residual falls from 0.252 to **0.039** — because parity is worth +0.018 F1 to RampNet +> and +0.137 to the best YOLO leg. The measurement, its controls and its caveats are in +> [`operating_point_parity_51.md`](operating_point_parity_51.md); the geometry half of +> the same question is in [`yolo_geometry_51.md`](yolo_geometry_51.md). **The rows below +> are unchanged and still correct at the operating points they name** — what changes is +> how much of the gap between two of them you may attribute to the models. + | model | class | op | P | R | F1 | ΔF1 vs RampNet | AP (macro) | FP/pano | F1 range | @@ -359,8 +371,11 @@ Omissions are content, so they are named rather than left as blanks: Including them here is deliberate: this page's job is to show everything that has been measured, and `standing` governs which roster tables a leg appears in, not whether its numbers are real. -- **The YOLO tiles arms are absent** — still training. The three pano arms are the completed - half of the #51 ablation; the resolution-controlled half is not done. +- **The YOLO tiles arms are absent from this board**, but they are no longer unmeasured: + `y11x_tiles` (ep44) was scored on all ten splits on 2026-08-30 and it is the best YOLO + cell in the grid. See [`yolo_geometry_51.md`](yolo_geometry_51.md). It is absent here + because its detections are not published and it is not in `rampnet/roster.py`, not + because it is untested. - **`manual_gold` has no null-recall pass** (O(n²) in panos), so the open detectors' recall discount is unmeasured on that split. - **Six legs have one split each**, so they are in the partial table rather than the diff --git a/docs/operating_point_parity_51.md b/docs/operating_point_parity_51.md new file mode 100644 index 00000000..ba6f5932 --- /dev/null +++ b/docs/operating_point_parity_51.md @@ -0,0 +1,228 @@ +# The RampNet-vs-YOLO gap at matched operating points (#51) + +**Status: measured 2026-09-01, CPU only, from already-committed detections. The published +gap is mostly an operating-point artifact. At parity it is 0.039 F1, not 0.160 — and on +3 of the 10 splits the best YOLO leg is level with RampNet or ahead of it, though every +one of those three margins (0.001–0.017) is inside this benchmark's own noise rule.** + +## What was wrong with the comparison + +[`model_scoreboard.md`](model_scoreboard.md) and [`yolo_geometry_51.md`](yolo_geometry_51.md) +compare F1 across rows whose operating points were chosen by *different procedures*: + +| | operating point | how it was chosen | +|---|---|---| +| RampNet | 0.55 | its shipped deployment threshold | +| every YOLO leg | 0.25 | the Ultralytics `predict()` default | + +Nobody selected 0.25 for YOLO. It is what you get when you do not pass `conf`. So the +published gap mixes *which model is better* with *whose default happened to suit F1 on +this benchmark*. The scoreboard already tells the reader to check the op column before +comparing rows; this measures what that warning was worth. It turns out to be worth a +lot. + +## What parity means here + +One threshold per model, chosen the same way for every model, on data the headline is +never reported over: + +1. **Select** each model's single uniform threshold on a **dev split**, by F1. +2. **Report** every model at that threshold, macro-meaned over the seven pooled US splits. + +The dev split is `sao_paulo` — already outside the pooled seven, and unremarkable, unlike +`budapest_district5` (the benchmark's only ranking inversion, low reviewer confidence) or +`manual_gold` (the only independently-labelled split). The threshold is **uniform across +splits**; a per-split best would be tune-on-test and the script does not offer it. + +Selection never touches a reported split, and +`test_no_candidate_dev_split_is_ever_reported_over` asserts it rather than trusting +review — on the output of `build()` for *every* candidate dev split, not on the +definition of the candidate list, which would be a tautology. + +## Results + +Macro-mean over the seven pooled US splits: + +| model | sel thr | P | R | F1 | ΔF1 vs RampNet | at published op | gain from parity | +|---|--:|--:|--:|--:|--:|--:|--:| +| **RampNet** | 0.30 | 0.896 | 0.797 | **0.843** | — | 0.824 | +0.018 | +| `y11x_tiles` (ep44) | 0.10 | 0.875 | 0.749 | **0.804** | **−0.039** | 0.667 | **+0.137** | +| `y11x_pano` (ep38) | 0.10 | 0.787 | 0.726 | 0.752 | −0.091 | 0.623 | +0.129 | +| `y11x_pano_h200` (ep60) | 0.10 | 0.839 | 0.683 | 0.751 | −0.092 | 0.575 | +0.176 | + +**The asymmetry in the last column is the whole finding.** Parity moves RampNet by 0.018 +because 0.55 was nearly right for it. It moves the tiles arm by 0.137 because 0.25 was +nowhere near right. The published comparison was scoring one model at its operating point +and the other 0.137 F1 away from its own. + +Per split, F1 at each model's selected threshold: + +| split | RampNet | `y11x_tiles` | `y11x_pano` | `y11x_pano_h200` | +|---|--:|--:|--:|--:| +| richmond | **0.864** | 0.803 | 0.791 | 0.777 | +| bend | **0.871** | 0.870 | 0.789 | 0.799 | +| clovis | **0.836** | 0.770 | 0.704 | 0.734 | +| morgantown | 0.845 | **0.862** | 0.791 | 0.813 | +| annapolis | **0.853** | 0.783 | 0.723 | 0.689 | +| paterson | **0.818** | 0.761 | 0.794 | 0.776 | +| gainesville | **0.812** | 0.778 | 0.673 | 0.668 | +| *budapest_district5* | **0.674** | 0.551 | 0.528 | 0.510 | +| *sao_paulo* (dev) | **0.800** | 0.776 | 0.744 | 0.754 | +| *manual_gold* | 0.902 | **0.911** | 0.844 | 0.872 | + +### The dev split does not carry the result + +| dev split | RampNet | `y11x_tiles` | gap | +|---|--:|--:|--:| +| `budapest_district5` | 0.842 @0.35 | 0.810 @0.05 | 0.032 | +| `sao_paulo` | 0.843 @0.30 | 0.804 @0.10 | 0.039 | +| `manual_gold` | 0.842 @0.35 | 0.804 @0.10 | 0.038 | + +The gap is 0.032–0.039 whichever of the three is used. + +## What it means + +### 1. The headline shrinks by a factor of four, and it is not architecture that moved + +Published residual **0.160**; at parity **0.039**. Nothing about either model changed — +only the procedure for picking where to read them. #51's claim that the keypoint +architecture is what RampNet contributes still points the right way, but 0.039 is a very +different claim from 0.252 or 0.160, and it is close enough to the noise floor to need +the seed work below before it carries weight. + +### 2. RampNet's lead is not uniform — but the splits it does not win are all inside the noise + +The tiles arm is **ahead** on `morgantown` (0.862 vs 0.845, +0.017) and on `manual_gold` +(0.911 vs 0.902, +0.009), and level on `bend` (0.870 vs 0.871, −0.001). That is a +materially different picture from "RampNet wins everywhere". + +**Read all three as ties, not as wins.** Every margin is below the ~0.02 F1 this document's +own noise rule says must not be read (see *Gaps*), and one of them has a second, measured +reason to be discounted: + +- **`manual_gold`'s margin is the size of a known instrument difference on that split.** + RampNet's committed `manual_gold` detections were exported *with* horizontal-flip TTA + (`benchmark/manual_gold/detections_meta.json`); the `op_cache` this document sweeps is + the no-TTA deployment path, and at 0.55 the two differ by 0.009 F1 — the same size as + the 0.009 margin. With TTA at 0.30 RampNet would plausibly be level. +- **A paired test would settle it and needs no GPU** — both models' per-pano detections + exist. It cannot be run from a clean clone, because the tiles arm's detections are not + published (see *Gaps*). + +So the defensible claim is the weaker one: at a defensible operating point the supervised +baseline is **no longer distinguishable from RampNet** on three of ten splits, including +`manual_gold` — the only split whose ground truth was labelled independently of RampNet's +own outputs, and so the one place the comparison is furthest from circular. That is +already a first for this benchmark. Calling it an outright win needs the seed work below. + +### 3. What does NOT change: AP + +AP is threshold-free, so parity cannot move it, and it still favours RampNet: + +| | AP (macro, US7) | source | +|---|--:|---| +| RampNet | **0.849** | [`model_scoreboard.md`](model_scoreboard.md) AP column, `op_cache`-derived | +| `y11x_tiles` | 0.773 | [`yolo_geometry_51.json`](data/yolo_geometry_51.json) `pooled.*.macro_ap` | +| `y11x_pano` | 0.752 | as above | +| `y11x_pano_h200` | 0.730 | as above | + +RampNet's 0.849 is the **only number on this page that this document's own script does not +compute** — it is carried from the scoreboard, whose AP column is `op_cache`-derived like +everything here, so it is on the same footing as the three YOLO rows. + +**Both sides are floored at 0.05** (`op_cache` `meta.score_floor` and +`YoloDetector.score_threshold`), so this comparison was already like-for-like — an +earlier reading of this analysis claimed the AP columns were asymmetric and that was +wrong. RampNet's PR curve genuinely dominates by **0.076 AP**. The honest summary is that +RampNet has the better curve, while at each model's own best point on that curve the F1s +are much closer than published. + +## Gaps, stated + +- **The tiles arm's parity F1 is a lower bound.** With `budapest_district5` as dev its + selected threshold is 0.05, the cache floor — the edge of what was ever measured. Its + true optimum may be lower and is unmeasured. Both models would have to be re-dumped at + a lower floor to close this, and doing it for only one of them would reintroduce + exactly the asymmetry this document removes. +- **One seed per cell.** This is now the binding limitation, not a footnote. At a 0.160 + gap seed variance was irrelevant; at 0.039 it is the entire question, and #51's own + rule is that differences under ~0.02 should not be read. Three seeds of `y11x_tiles` + is the experiment that would settle this. +- **`y11x_tiles` is at ep44 of a 60-epoch schedule**, and the pano lineage suggests more + epochs would *hurt* it out of distribution ([`yolo_geometry_51.md`](yolo_geometry_51.md)), + so this is not simply "the undertrained arm". +- **The tiles arm's detections are not published**, so the paired test that would settle + the three near-tied splits in §2 cannot be run from a clean clone. `y11x_tiles` and + `y11x_pano` are not in `rampnet/roster.py` and have no files under + `benchmark/model_detections/`; their checkpoints live only on cluster storage + (`/gscratch/makelab/jonf/rampnet_yolo_baseline_51//weights/best.pt`, sha256s in + [`env.txt`](data/yolo_geometry_51/env.txt)). Registering them as roster legs and + exporting their detections is what unblocks it. +- **Only the `x` architecture.** `y11l_*` and `y26_*` legs are not swept here; their + committed reports do not carry a sweep. +- **The VLM and pointing challengers cannot be included at all** — they emit no calibrated + confidence, so they have one operating point rather than a curve. Their rows in + `model_comparison.md` are unaffected by this document, in either direction. + +## Reproducing + +CPU only, no GPU, no network — everything it reads is committed: + +```bash +python scripts/analysis/operating_point_parity_51.py --sensitivity +python scripts/analysis/operating_point_parity_51.py --check # fails if the artifact drifted +pytest tests/test_operating_point_parity_51.py -q +``` + +Artifact: [`docs/data/operating_point_parity_51.json`](data/operating_point_parity_51.json). +Inputs: `analysis_out/op_cache/*.json` (RampNet, re-scored at each threshold) and +`docs/data/yolo_geometry_51/*.txt` (the YOLO legs' committed sweeps). + +**The control.** RampNet at 0.55 re-scores to 0.824 from `op_cache` against a published, +bundle-derived 0.827. + +That −0.0025 is **not** peak extraction. An earlier version of this section said it was — +that lowering `peak_local_max`'s floor to 0.05 changes which maxima survive `min_distance` +suppression — and that is wrong: `peak_local_max` suppresses on a maximum filter, so a +pixel that is the maximum of its neighbourhood at 0.55 is still the maximum at 0.05. +Lowering the floor can only *add* peaks; the ≥ 0.55 subset of a 0.05-floor extraction is +exactly a 0.55-floor extraction. + +The two sources differ because they are **two different heatmaps**, and +[`operating_point.md`](operating_point.md) already measured which and why: + +| split | op_cache @0.55 | published bundle | Δ | why | +|---|--:|--:|--:|---| +| richmond | 0.8546 | 0.855 | −0.0004 | same computation — Mapillary, bit-exact | +| clovis | 0.8012 | 0.801 | +0.0002 | same computation | +| morgantown | 0.8351 | 0.835 | +0.0001 | same computation | +| annapolis | 0.8395 | 0.839 | +0.0005 | same computation | +| *budapest_district5* | 0.6442 | 0.644 | +0.0002 | same computation | +| bend | 0.8532 | 0.850 | **+0.0032** | GSV: production used a 4096×2048 resample | +| paterson | 0.8006 | 0.805 | **−0.0044** | GSV resample | +| gainesville | 0.7871 | 0.803 | **−0.0159** | GSV resample | +| *sao_paulo* (dev) | 0.7578 | 0.777 | **−0.0192** | GSV resample | +| *manual_gold* | 0.8990 | 0.908 | **−0.0090** | committed detections carry flip-TTA; `op_cache` does not | + +Three things follow, and they qualify every number above: + +1. **The pooled macro clears its 0.005 tolerance partly by cancellation** (+0.003 on bend + against −0.004 and −0.016 on paterson and gainesville). A macro-only control could not + see a split-level divergence, so the script now asserts the control **per split** too: + the five same-computation splits to 0.002, the five with a documented reason to differ + to 0.025. The five exact splits are the real regression guard. +2. **The dev split has the largest discrepancy of the ten** (−0.019). RampNet's threshold + is selected on the split where its `op_cache` curve is least like its shipped + detections. That is what the sensitivity table answers: the selection lands on + 0.30–0.35 whichever of the three candidates is used. +3. **Three of the seven pooled splits are GSV, so RampNet's parity F1 is about 0.002 low** + relative to a bundle-derived path. The reported 0.039 gap is very slightly conservative, + not flattering. + +## Cost + +CPU only, seconds, $0 — this document needed no GPU. The YOLO sweeps it reads came from +the 2026-08-30 geometry run: 67 min 39 s on one A40 on makelab2, ≈ 1.13 GPU-hours at no +cost (lab hardware). See [`yolo_geometry_51.md`](yolo_geometry_51.md) for that run's +provenance. A ledger row follows once [#147](https://github.com/ProjectSidewalk/RampNet/pull/147) +lands; until then this paragraph is the record. diff --git a/docs/yolo_geometry_51.md b/docs/yolo_geometry_51.md new file mode 100644 index 00000000..692311f4 --- /dev/null +++ b/docs/yolo_geometry_51.md @@ -0,0 +1,218 @@ +# Is the RampNet-vs-YOLO gap architecture, or equirectangular input? (#51) + +**Status: measured 2026-08-30. The gap is real but it is smaller than published. +About a third of it is not architecture.** + +Issue [#51](https://github.com/ProjectSidewalk/RampNet/issues/51) reports that RampNet beats +a supervised YOLO baseline trained on the same data by **0.252 F1**, macro-meaned over the +seven pooled US city splits (RampNet 0.827 vs YOLO11x pano 0.575, see +[`model_scoreboard.md`](model_scoreboard.md)). That number is the basis for the claim that +the keypoint architecture — not the dataset — is what RampNet contributes. + +The standing objection is that this is not a fair fight. The YOLO arms are fed 2048×4096 +**equirectangular** panoramas, which is not the geometry a COCO-shaped detector was designed +for: straight edges bow, scale varies with latitude, and a ramp far from the camera occupies +very few pixels. Under that reading the gap is a handicap, not a result. + +The `tiles` arms exist precisely to answer this — same data, same schedule, but fed through +the same perspective-view rig the VLM challengers get. **Until now no tiles checkpoint had +ever been scored on the benchmark**, so the control for #51's headline was unmeasured. This +document is that measurement. + +## What was run + +Three legs, each in its **training** geometry per the pre-registered protocol +([#71](https://github.com/ProjectSidewalk/RampNet/issues/71), `yolo_baseline/README.md`): +`best.pt` as-saved, headline **F1 at conf 0.25**, sweep reported but never selected on. + +| leg | epoch | geometry | note | +|---|---:|---|---| +| `y11x_tiles` | 44 | perspective tiles, imgsz 1024 | never scored before | +| `y11x_pano` | 38 | whole pano, imgsz 1280 | never scored before | +| `y11x_pano_h200` | 60 | whole pano, imgsz 1280 | **control** — already published | + +Run on makelab2 (A40, ultralytics 8.4.120) at repo `d964d5d`. Driver: +[`run_yolo_geometry_eval.sh`](../scripts/model_comparison/yolo_baseline/run_yolo_geometry_eval.sh). +Reports and checkpoint sha256s: [`docs/data/yolo_geometry_51/`](data/yolo_geometry_51). +Table regenerated by [`yolo_geometry_51.py`](../scripts/analysis/yolo_geometry_51.py); `--check` +turns drift into a failure. + +### The control is the load-bearing part + +`y11x_pano_h200`'s published numbers were produced 2026-08-14 against a repo that **predates +the [#132](https://github.com/ProjectSidewalk/RampNet/issues/132) seam fix**, which changed +how the matcher wraps the 360° seam and therefore changes scores. Scoring a fresh tiles +number against a stale pano number would have attributed a code change to geometry. So the +published checkpoint was re-scored under the same commit as the two new legs: + +| | P | R | F1 | AP | +|---|--:|--:|--:|--:| +| published 2026-08-14 | 0.969 | 0.416 | 0.575 | 0.730 | +| re-scored 2026-08-30 | 0.969 | 0.416 | 0.575 | 0.730 | + +**Identical to three decimals.** The seam fix did not move YOLO pano scoring — which both +validates this comparison and means the published #51 table needs no amendment. The assertion +is in the script, so the day that stops being true, it fails loudly. + +**What the control does and does not test.** All ten `*_pano.txt` reports record +`[y11x_pano_h200] all N panos already cached; model load skipped`: the control leg was +**re-scored from the detections cached by the 2026-08-14 publishing run, not re-inferred**. +That is the right instrument for the question — it isolates the matcher and the scorer, +which is what #132 changed, and holds the detections fixed. It is *not* a test that YOLO +inference reproduces across ultralytics versions or hardware; nothing here measures that. + +## Results + +Per-split F1 at conf 0.25: + +| split | `y11x_tiles` (44) | `y11x_pano` (38) | `y11x_pano_h200` (60) | +|---|--:|--:|--:| +| richmond | **0.667** | 0.708 | 0.547 | +| bend | **0.750** | 0.728 | 0.710 | +| clovis | **0.712** | 0.610 | 0.551 | +| morgantown | **0.728** | 0.728 | 0.686 | +| annapolis | **0.640** | 0.452 | 0.397 | +| paterson | 0.595 | **0.635** | 0.635 | +| gainesville | **0.576** | 0.501 | 0.499 | +| *budapest_district5* (held out) | 0.319 | 0.176 | 0.221 | +| *sao_paulo* (held out) | 0.625 | 0.650 | 0.659 | +| *manual_gold* (held out) | 0.820 | 0.840 | 0.851 | + +Macro-mean over the seven pooled US splits: + +| leg | epoch | P | R | F1 | AP | ΔF1 vs RampNet | +|---|--:|--:|--:|--:|--:|--:| +| `y11x_tiles` | 44 | 0.945 | **0.520** | **0.667** | 0.773 | **−0.160** | +| `y11x_pano` | 38 | 0.942 | 0.474 | 0.623 | 0.752 | −0.204 | +| `y11x_pano_h200` | 60 | **0.969** | 0.416 | 0.575 | 0.730 | −0.252 | + +## What it means + +### 1. The published gap does not decompose into "geometry" alone + +A raw tiles-minus-published difference (+0.092) would be the wrong number to quote, because +the tiles arm is at ep44 and the published arm at ep60 — it mixes geometry with training +budget. Split on the pano lineage, where only budget moves: + +| term | contrast | ΔF1 | +|---|---|--:| +| over-training | pano ep60 → ep38 (same geometry) | **+0.048** | +| geometry | pano ep38 → tiles ep44 (~matched budget) | **+0.044** | +| | **best YOLO cell we have** | **+0.092** | + +So the equirect handicap is **real and roughly half the recoverable difference** — worth about +0.044 F1, not the whole 0.092 and certainly not the whole 0.252. + +**The geometry half is recall**, as the [Vistas parity arm](model_comparison.md) also found: +R 0.474 → 0.520 (**+0.045**) at essentially unchanged precision (0.942 → 0.945). Perspective +views find ramps the equirect view loses; they do not make the detector more careful. + +> **Caveat, and it matters.** `y11x_pano` and `y11x_pano_h200` are **divergent continuations +> of one lineage** — h200 forked from `y11x_pano/best.pt` (`MANIFEST-2026-08-03`) and +> continued on different hardware. The budget term is therefore confounded with the fork and +> is *suggestive, not clean*. The geometry term is the better-controlled of the two, and it is +> the smaller one. + +### 2. #51's headline survives, restated + +After removing the handicap, the best YOLO cell that exists is **0.667 against RampNet's +0.827 — a residual 0.160 F1**. The claim "the architecture is what RampNet contributes" +still holds; the honest magnitude is **0.16, not 0.25**, and #51's write-up should say so. + +> **Superseded, 2026-09-01.** Both numbers in that sentence are read at operating points +> chosen by different procedures — RampNet at its shipped 0.55, every YOLO leg at the +> Ultralytics default 0.25. Give each model a threshold selected the same way and the +> residual falls to **0.039**. See +> [`operating_point_parity_51.md`](operating_point_parity_51.md). The geometry decomposition +> below is unaffected — it compares YOLO legs to each other, all at the same 0.25 — but the +> *headline against RampNet* on this page should be read as 0.039, not 0.160. + +Note the shape is unchanged: RampNet still wins on **recall** (0.728 vs 0.520) while YOLO +holds comparable precision. Every version of this comparison lands in the same place — +finding the ramps is the hard part. + +### 3. More training made the pano arm *worse* out of distribution + +This is the result nobody was looking for. On the seven-city pool the **ep38** arm beats the +**ep60** arm by 0.048 F1, and it does so on 6 of the 7 splits — while the ordering **reverses +in-distribution**, where ep60 wins on `manual_gold` (0.851 vs 0.840). + +That is overfitting to the training distribution, and it lines up with the already-recorded +#51 finding that the 1-epoch RampNet release beats all three 60-epoch YOLO arms. Subject to +the fork caveat above, **more epochs bought in-distribution fit and cost OOD generalization.** + +## What this says about finishing the tiles arms + +The open question on #51 was whether to spend ~400 GPU-hours taking the tiles family to ep60. +Two things now argue against it: + +- **The tiles curve is flat.** `y11x_tiles` val mAP50-95 went 0.46686 (ep8) → 0.47804 (ep44): + **+0.011 over 36 epochs**, and +0.0054 over the last 12. Sixteen more epochs buy roughly + +0.007 on the metric the run selects on. +- **The pano lineage says the extra epochs may be actively harmful OOD**, which is the regime + every city split measures and the one the paper's claim rests on. + +The tiles arm at ep44 — undertrained, on free preempted GPU — is already the best YOLO cell +in the grid on the pooled US splits. Finishing it is unlikely to change any conclusion in #51, +and the decomposition it would sharpen is the *budget* term, which is the confounded one. + +**A cheaper experiment was proposed here and it cannot be run.** The original version of this +section suggested scoring "the tiles checkpoints that already exist at several epochs" to turn +the fork-confounded budget term into a measured one. **Those checkpoints do not exist.** Every +arm was trained with `save_period: -1`, so Ultralytics kept only `best.pt` and `last.pt`; +`find` over both the durable snapshot and `/gscratch/scrubbed/jfroehli/yolo_runs` returns zero +`epoch*.pt` files, and `best.pt`/`last.pt` are 0–4 epochs apart (tiles 43 vs 44, pano 35 vs 38, +h200 60 vs 60). The `y11x_pano_h200` arm was a *resume*, so it inherited `save_period=-1` and +the Tillicum launcher's `SAVE_PERIOD=5` never applied to it. + +`retarget_yolo_checkpoint.py` already says so in as many words — "per-epoch weights cannot be +recovered for an arm that did not start with it" — which is the docstring this section +contradicted. **The budget term cannot be de-confounded cheaply**: per-epoch weights require +*starting* a run with `save_period` set, and a resume honours the value saved in the +checkpoint, so a fresh-start run would reset the LR schedule and stop being a continuation. +It is the full retrain or nothing. Given that the geometry term is the better-controlled half +and is already measured, leaving the budget term confounded is the right call. + +**What did turn out to be cheap** is fixing the operating-point asymmetry — see +[`operating_point_parity_51.md`](operating_point_parity_51.md), which needed no GPU at all and +moved the headline four times further than any of this. + +## Provenance and cost + +- **Compute.** The eval ran 2026-08-30 17:10:09 → 18:17:48 on one A40 on makelab2: + **67 min 39 s wall, ≈ 1.13 GPU-hours, $0** (lab hardware, not billed). Per-split + timings are in [`driver.log`](data/yolo_geometry_51/driver.log). The analysis on top of + it ([`yolo_geometry_51.py`](../scripts/analysis/yolo_geometry_51.py)) is CPU-only and + takes seconds. The label rescue and the durable snapshot that preceded it were CPU-only + klone jobs on the free `ckpt-all` partition; their Slurm job ids are not recorded in + this repo, which is a gap. A ledger row follows once + [#147](https://github.com/ProjectSidewalk/RampNet/pull/147) lands. +- **The working tree was dirty.** [`env.txt`](data/yolo_geometry_51/env.txt) records + `repo dirty : 10 paths` at `d964d5d` — a **count**, not a list, so a reader cannot tell + whether those were untracked output directories or modified code. The driver now logs + `git status --porcelain` in full, so this is answerable for every future run and + unanswerable for this one. The control leg reproducing its published row to three + decimals is the evidence that the scoring code was not among them. +- **The checkpoints are not published.** `env.txt` records sha256s of the copies under + makelab2's `yolo_ckpts/`. The durable copies carrying those hashes are on klone at + `/gscratch/makelab/jonf/rampnet_yolo_baseline_51//weights/best.pt` + ([`run_durable_snapshot.slurm`](../scripts/model_comparison/yolo_baseline/run_durable_snapshot.slurm)). + **So this run is reproducible by someone with cluster access, and not from a clean + clone.** Publishing the three checkpoints, or their per-pano detections, is what would + change that; neither is done. +- **The rescue's outcome is recorded only in the pull request that introduced it** + ([#154](https://github.com/ProjectSidewalk/RampNet/pull/154)) — 557,413 train records / + 968,227 boxes / 59,923 background, `--verify PASS`. The expected counts are pinned in + the two scripts' docstrings and enforced by `check_yolo_dataset_loads.py --expect-*`, + so a re-run proves them; the job ids and dates are not in the repo. + +## Gaps, stated + +- **Only the `x` architecture is covered.** `y11l_tiles` (ep11) and `y26_tiles` (ep12) are far + too undertrained to compare, so this is a single-architecture read on geometry. +- **ep44 vs ep38 is "roughly matched", not matched.** They are the epochs that exist. +- **One seed per cell.** As with the rest of #51, seed variance is unmeasured, so differences + below roughly 0.02 F1 should not be read. +- **Not re-scored:** `y11l_pano` and `y26_pano`, whose published numbers stand on the control + above (the seam fix moved nothing for `y11x_pano_h200`, so it almost certainly moved nothing + for them either — but that is an inference, not a measurement). diff --git a/scripts/analysis/README.md b/scripts/analysis/README.md index 2ca0526d..106c0732 100644 --- a/scripts/analysis/README.md +++ b/scripts/analysis/README.md @@ -57,6 +57,8 @@ checkout. | `miss_gallery.py` | no | Crops for the misses geometry cannot explain (#46 gallery half). **Checks the instrument before rendering**: `geom()` sizes ramps at the model's 4096-px input, but stored panos run 4096–16384 px wide, so it classifies each crop `parity` vs `advantaged` and renders a third "as the model saw it" panel so a reviewer compares pixel budgets instead of inferring. Needs `benchmark//panos` (`--panos-root` if run from a worktree). | | `fp_gallery.py` | no | The FP half of the gallery: worst-N `isolated` false positives per model, through the same instrument and manifest as `miss_gallery.py`. Ranked by the model's own confidence where it has one; the sample size and what was left out are always printed, never silent. Reads `.model_cache` + `panos/`. | | `make_tagger.py ` | no | Turns a rendered gallery into a keyboard-driven `tagger.html` beside it — one keystroke per crop, auto-advance, `localStorage` autosave, and an export keyed exactly like `benchmark//incremental_fp_tags.json`. Picks the verdict scheme from the manifest's own contents (miss vs FP). Local page by design: the crops are git-ignored files on disk. | +| `yolo_geometry_51.py` | no | The #51 equirect control: reads the committed geometry-pair reports under `docs/data/yolo_geometry_51/` and decomposes the tiles-vs-pano difference into geometry and training budget. `--check` fails on artifact drift. Its split population is **pinned to the 2026-08-30 run**, not the live registry. | +| `operating_point_parity_51.py` | no | RampNet vs the YOLO legs at **matched** operating points (#51): one uniform threshold per model, selected on a split the headline is never reported over. `--sensitivity` re-runs the selection on every candidate dev split; `--check` fails on artifact drift. | | `plot_operating_point.py` | no | The headline figure: PR response per split + F1-vs-threshold → `docs/figures/operating_point_pr.png`. | | `plot_storage_floor.py` | no | Storage-floor cost + recall ceiling → `docs/figures/storage_floor_ceiling.png`. | diff --git a/scripts/analysis/operating_point_parity_51.py b/scripts/analysis/operating_point_parity_51.py new file mode 100644 index 00000000..1ec57d0b --- /dev/null +++ b/scripts/analysis/operating_point_parity_51.py @@ -0,0 +1,487 @@ +"""RampNet vs the YOLO baselines at MATCHED operating points (#51). + +THE PROBLEM THIS FIXES +---------------------- +``docs/model_scoreboard.md`` and ``docs/yolo_geometry_51.md`` compare F1 across rows +whose operating points were chosen by different procedures: + + * every YOLO leg is scored at **conf 0.25** -- the Ultralytics default. Nobody + selected it; it is what ``predict()`` uses when you do not say otherwise. + * RampNet is scored at **0.55** -- its shipped deployment threshold, which #54/#55 + already established is not F1-optimal (0.30 is the recommendation). + +So the published gap mixes "which model is better" with "whose default happened to +suit this metric". The scoreboard warns to read the op column before comparing rows; +this script measures what that warning is worth. + +WHAT PARITY MEANS HERE +---------------------- +One threshold per model, chosen the same way for every model, on data the headline is +not reported over: + + 1. **Select** each model's single uniform threshold on a DEV split, by F1. + 2. **Report** every model at that threshold, macro-meaned over ``POOLED_SPLITS``. + +The dev split defaults to ``sao_paulo``: it is already outside the seven-city pool the +headline is computed over, so selection never touches a reported split, and unlike +``budapest_district5`` (the benchmark's only ranking inversion, low reviewer +confidence) or ``manual_gold`` (the only independently-labelled split) it is +unremarkable. ``--sensitivity`` re-runs the whole selection with each of the three +non-pooled splits as dev, so the choice can be seen not to carry the result. + +The threshold is UNIFORM across splits. Picking a per-split best would be tune-on-test +and is not offered. + +DELIBERATELY GENEROUS TO RAMPNET +-------------------------------- +RampNet is swept on a full 0.05-step grid from its cache floor to 0.95, while the YOLO +legs are limited to the grid their committed reports carry (0.05-step to 0.30, then +0.40/0.50/0.60/0.70). Both models' optima are interior to their own dense regions, so +this changes nothing -- but where it could, it favours RampNet, which is the +conservative direction for the finding. + +BOTH SIDES ARE FLOORED AT 0.05, AND THAT IS NOT AN ASYMMETRY +------------------------------------------------------------ +``analysis_out/op_cache/*.json`` carries ``meta.score_floor = 0.05`` and +``YoloDetector.score_threshold = 0.05`` is the YOLO cache floor. The AP columns are +therefore already like-for-like. What the shared floor *does* mean is that a model +whose best grid point is 0.05 is reported at the edge of what was measured, so its +parity F1 is a LOWER BOUND -- flagged per model in the output. + +SELF-CHECK, AND WHY IT IS NOT AN EQUALITY +----------------------------------------- +RampNet is swept from ``op_cache`` because that is the only source that goes below the +deployment threshold at all -- the committed bundle records stop at ~0.55 (measured +floor 0.5501 over the pooled splits). The two sources do not agree exactly at 0.55: + + op_cache filtered to >= 0.55 F1 0.824 + bundle records as shipped F1 0.827 <- docs/model_scoreboard.md + +That -0.0025 is not a bug, not a floor effect (the bundle floor is *above* 0.55), and +NOT a peak-extraction artifact. Lowering ``threshold_abs`` can only ADD candidates: +``peak_local_max`` suppresses on a maximum filter, so a pixel that is the maximum of its +``min_distance`` neighbourhood at 0.55 is still that maximum at 0.05, and the >= 0.55 +subset of a 0.05-floor extraction is identical to a 0.55-floor extraction. (An earlier +version of this docstring gave that as the mechanism. It was wrong.) + +The two sources differ because they are two different HEATMAPS, and +``docs/operating_point.md`` already measured which and why: + + * FIVE splits agree, to the 3 decimals the published per-split values carry: + richmond, clovis, morgantown, annapolis, budapest_district5. These are the Mapillary + splits, recorded there as reproducing bit-exactly. They are the real regression + guard -- ``CONTROL_EXACT_SPLITS``, tolerance ``CONTROL_TOL_EXACT``. + * The four GSV splits (bend, paterson, gainesville, sao_paulo) differ by -0.019 to + +0.003. The production path assembled tiles into a 4096x2048 intermediate, so the + shipped detections came from a DIFFERENT RESAMPLE of the same panorama than these + native-resolution caches; detections sit up to 0.44 match radii apart. + * ``manual_gold`` differs by -0.009 because its committed detections were exported + WITH horizontal-flip TTA (``benchmark/manual_gold/detections_meta.json``) and + op_cache is the no-TTA deployment path. + +Two consequences travel with every number below, and they are stated in +``docs/operating_point_parity_51.md`` too: + + 1. The pooled macro passes ``CONTROL_TOL`` partly by CANCELLATION (+0.003 on bend + against -0.004 and -0.016 on paterson and gainesville). A macro-only control would + have hidden a split-level divergence, so the control is asserted per split as well. + 2. The default dev split, ``sao_paulo``, has the LARGEST discrepancy of the ten + (-0.019). ``--sensitivity`` is what answers that: the selection lands on 0.30-0.35 + for RampNet whichever candidate is used. + +Everything here is op_cache-derived end to end, so the comparison is internally +consistent -- and because three of the seven pooled splits are GSV, RampNet's parity F1 +is about 0.002 LOW relative to a bundle-derived path, i.e. the reported gap is very +slightly conservative rather than flattering. ``--check`` turns artifact drift into a +non-zero exit. + +USAGE + python scripts/analysis/operating_point_parity_51.py + python scripts/analysis/operating_point_parity_51.py --dev-split manual_gold + python scripts/analysis/operating_point_parity_51.py --sensitivity + python scripts/analysis/operating_point_parity_51.py --check +""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import sys + +HERE = os.path.dirname(os.path.abspath(__file__)) +ROOT = os.path.dirname(HERE) +REPO = os.path.dirname(ROOT) +sys.path.insert(0, ROOT) +sys.path.insert(0, REPO) + +from analysis.yolo_geometry_51 import ( # noqa: E402 + ALL_SPLITS_AS_RUN, IN_DIR, LEGS, POOLED_SPLITS, PUBLISHED_RAMPNET_F1, rnd) +from rampnet.detection_eval import aggregate, radius_sq_for, score_pano # noqa: E402 + +OP_CACHE = os.path.join(REPO, "analysis_out", "op_cache") +OUT_JSON = os.path.join(REPO, "docs", "data", "operating_point_parity_51.json") + +RAMPNET = "RampNet" +DEPLOYED_THRESHOLD = 0.55 + +# The split population is PINNED to the 2026-08-30 YOLO run, via yolo_geometry_51 -- +# NOT the live registry in ``analysis.low_floor_sweep``. A split added to the benchmark +# later has no YOLO report, so following the registry would drop every YOLO leg (each +# needs a full pool) while RampNet quietly re-pooled over a different population, and +# the two sides of the comparison would stop being the same comparison. + +# ``PUBLISHED_RAMPNET_F1`` (0.827, bundle-derived at 0.55) is imported rather than +# restated so the two scripts cannot drift apart on the number they both check against. + +# RampNet's per-split published F1 at 0.55, bundle-derived, from the by-split table in +# docs/model_scoreboard.md. The pooled 0.827 is the macro-mean of the first seven. +PUBLISHED_RAMPNET_PER_SPLIT_F1 = { + "richmond": 0.855, "bend": 0.850, "clovis": 0.801, "morgantown": 0.835, + "annapolis": 0.839, "paterson": 0.805, "gainesville": 0.803, + "budapest_district5": 0.644, "sao_paulo": 0.777, "manual_gold": 0.908, +} + +# Splits where the op_cache path and the published bundle are the SAME computation, so +# any disagreement beyond 3-dp rounding of the published value is a scoring regression. +# docs/operating_point.md records these five as reproducing bit-exactly. +CONTROL_EXACT_SPLITS = ("richmond", "clovis", "morgantown", "annapolis", "budapest_district5") +CONTROL_TOL_EXACT = 0.002 # 3-dp rounding of the published per-split value +# Splits with a DOCUMENTED reason for the two paths to differ (docs/operating_point.md): +# the four GSV splits were produced from a different resample of the panorama, and +# manual_gold's committed detections carry horizontal-flip TTA that op_cache does not. +CONTROL_TOL_DIVERGENT = 0.025 +CONTROL_TOL = 0.005 # pooled macro; see SELF-CHECK in the module docstring + +# Candidate dev splits: everything the headline is NOT macro-meaned over. +NON_POOLED = tuple(s for s in ALL_SPLITS_AS_RUN if s not in POOLED_SPLITS) +DEFAULT_DEV = "sao_paulo" + +# A sweep row: " 0.05 0.789 0.823 0.806 255/68/55 <- best F1" +SWEEP_ROW_RE = re.compile( + r"^\s*(?P\d\.\d+)\s+(?P

\d\.\d+)\s+(?P\d\.\d+)\s+(?P\d\.\d+)\s+" + r"(?P\d+)/(?P\d+)/(?P\d+)\s*(?:<- best F1)?\s*$" +) +SWEEP_HDR_RE = re.compile(r"^\[(?P\S+)\] threshold sweep") + + +# --------------------------------------------------------------------------- # +# YOLO: the committed driver reports already carry a full sweep per leg +# --------------------------------------------------------------------------- # +def parse_sweeps(path): + """``{model: {threshold: {p, r, f1, tp, fp, fn}}}`` from one driver report. + + ``yolo_geometry_51.parse_report`` keeps only each sweep's best row, because the + pre-registered headline is conf 0.25 and it may not select on the sweep. Parity + needs the whole curve, so this reads every row. + """ + out, current = {}, None + with open(path, encoding="utf-8") as fh: + for line in fh: + m = SWEEP_HDR_RE.match(line) + if m: + current = m.group("model") + continue + if current is None: + continue + m = SWEEP_ROW_RE.match(line.rstrip("\n")) + if m: + d = m.groupdict() + out.setdefault(current, {})[float(d["thr"])] = { + "p": float(d["p"]), "r": float(d["r"]), "f1": float(d["f1"]), + "tp": int(d["tp"]), "fp": int(d["fp"]), "fn": int(d["fn"]), + } + elif line.strip() and not line.startswith(" "): + current = None # a new section ends the sweep block + return out + + +def collect_yolo(): + """``{model: {split: {threshold: metrics}}}`` over every committed report.""" + cells = {} + for split in ALL_SPLITS_AS_RUN: + for kind in ("tiles", "pano"): + path = os.path.join(IN_DIR, f"{split}_{kind}.txt") + if not os.path.exists(path): + continue + for model, sweep in parse_sweeps(path).items(): + cells.setdefault(model, {})[split] = sweep + return cells + + +# --------------------------------------------------------------------------- # +# RampNet: re-scored from the committed low-floor cache, no GPU +# --------------------------------------------------------------------------- # +def rampnet_grid(floor=0.05, step=0.05, top=0.95): + n = int(round((top - floor) / step)) + return [round(floor + i * step, 10) for i in range(n + 1)] + + +def rampnet_sweep(split, grid, radius_sq=None): + """``{threshold: metrics}`` for RampNet on one split, or None if uncached. + + Re-scores ``analysis_out/op_cache/.json`` at each threshold with the same + ``score_pano``/``aggregate`` path ``scoreboard.py`` uses, so the numbers are the + published ones by construction rather than by coincidence. + """ + path = os.path.join(OP_CACHE, f"{split}.json") + if not os.path.exists(path): + return None + if radius_sq is None: + radius_sq = radius_sq_for() + with open(path, encoding="utf-8") as fh: + payload = json.load(fh) + panos = [(p["preds"], p["gt"]) for p in payload["panos"]] + + from rampnet.detection_eval import GroundTruth + gts = [GroundTruth([tuple(q) for q in g["gt_points"]], + [tuple(q) for q in g["ignore_points"]], + bool(g["fn_confirmed"])) for _, g in panos] + + out = {} + for thr in grid: + rep = aggregate([ + score_pano([tuple(q) for q in preds if q[2] >= thr], gt, radius_sq=radius_sq) + for (preds, _), gt in zip(panos, gts)]) + out[thr] = {"p": rep.precision, "r": rep.recall, "f1": rep.f1, + "tp": rep.tp, "fp": rep.fp, "fn": rep.fn} + return out + + +def collect_rampnet(grid): + per_split = {} + for split in ALL_SPLITS_AS_RUN: + sweep = rampnet_sweep(split, grid) + if sweep is not None: + per_split[split] = sweep + return per_split + + +# --------------------------------------------------------------------------- # +# selection + reporting +# --------------------------------------------------------------------------- # +def select_threshold(per_split, dev_split): + """The uniform threshold maximising F1 on ``dev_split``. Ties -> higher threshold, + which is the precision-favouring side and the one a deployment would pick.""" + sweep = per_split.get(dev_split) + if not sweep: + return None + best = max(sweep.items(), key=lambda kv: (kv[1]["f1"], kv[0])) + return best[0] + + +def macro_at(per_split, thr, pool=POOLED_SPLITS): + """Macro-mean P/R/F1 over ``pool`` at ``thr``, or None if any split is missing. + + Macro, not micro, to match ``yolo_geometry_51.pooled`` and the scoreboard: each + city weighted equally so the largest split cannot dominate. + """ + got = [] + for s in pool: + sweep = per_split.get(s) + if not sweep or thr not in sweep: + return None + got.append(sweep[thr]) + n = len(got) + return {"threshold": thr, + "p": sum(c["p"] for c in got) / n, + "r": sum(c["r"] for c in got) / n, + "f1": sum(c["f1"] for c in got) / n, + "n_splits": n} + + +def control_per_split(rampnet_per_split): + """Per-split op_cache-vs-bundle agreement for RampNet at the deployed threshold. + + The pooled macro control clears its tolerance partly by cancellation, so it cannot + see a split-level divergence. This can. Each split carries its own tolerance and the + reason it gets that tolerance: ``CONTROL_EXACT_SPLITS`` are the ones where the two + paths are the same computation and must agree, while the rest have a measured, + written-down reason to differ (``docs/operating_point.md``). + """ + out = {} + for split, published in PUBLISHED_RAMPNET_PER_SPLIT_F1.items(): + sweep = rampnet_per_split.get(split) + if not sweep or DEPLOYED_THRESHOLD not in sweep: + continue + got = sweep[DEPLOYED_THRESHOLD]["f1"] + exact = split in CONTROL_EXACT_SPLITS + tol = CONTROL_TOL_EXACT if exact else CONTROL_TOL_DIVERGENT + out[split] = { + "f1_op_cache": got, + "f1_published_bundle": published, + "delta": got - published, + "tolerance": tol, + "paths_agree_by_construction": exact, + "agrees": abs(got - published) <= tol, + } + return out + + +def build(dev_split=DEFAULT_DEV): + grid = rampnet_grid() + models = {RAMPNET: collect_rampnet(grid)} + models.update(collect_yolo()) + + # Self-check before anything is reported off this path. + at_deployed = macro_at(models[RAMPNET], DEPLOYED_THRESHOLD) + delta = None if at_deployed is None else at_deployed["f1"] - PUBLISHED_RAMPNET_F1 + per_split_control = control_per_split(models[RAMPNET]) + control = { + "threshold": DEPLOYED_THRESHOLD, + "f1_op_cache": at_deployed["f1"] if at_deployed else None, + "f1_published_bundle": PUBLISHED_RAMPNET_F1, + "delta": delta, + "tolerance": CONTROL_TOL, + "agrees": bool(delta is not None and abs(delta) <= CONTROL_TOL), + "per_split": per_split_control, + "per_split_agrees": all(c["agrees"] for c in per_split_control.values()), + } + + rows = {} + for model, per_split in models.items(): + thr = select_threshold(per_split, dev_split) + if thr is None: + continue + pooled = macro_at(per_split, thr) + if pooled is None: + continue + floor = min(per_split[dev_split]) + rows[model] = { + "selected_threshold": thr, + "at_floor": thr <= floor, # reported value is a lower bound + "pooled": pooled, + "published_point": macro_at( + per_split, DEPLOYED_THRESHOLD if model == RAMPNET else 0.25), + "per_split": {s: per_split[s][thr] for s in ALL_SPLITS_AS_RUN + if s in per_split and thr in per_split[s]}, + } + return {"dev_split": dev_split, "control": control, "models": rows, + "pool": list(POOLED_SPLITS), "grid_rampnet": grid} + + +def sensitivity(): + """Selected threshold + pooled F1 per model, for each candidate dev split.""" + return {dev: {m: {"thr": r["selected_threshold"], "f1": r["pooled"]["f1"]} + for m, r in build(dev)["models"].items()} + for dev in NON_POOLED} + + +def artifact(dev_split=DEFAULT_DEV): + """The committed payload: the run, plus the sensitivity table that shows the dev + split does not carry it. + + Sensitivity is always included rather than being a flag, so ``--check`` compares + the same object every time -- an artifact whose content depends on which switches + the last person typed cannot be checked at all. + """ + result = build(dev_split) + result["sensitivity"] = sensitivity() + return result + + +# --------------------------------------------------------------------------- # +def _render(result): + lines = [] + c = result["control"] + lines.append(f"Dev split (selection only): {result['dev_split']}") + lines.append(f"Reported over {len(result['pool'])} pooled US splits: " + f"{', '.join(result['pool'])}") + status = "OK" if c["agrees"] else "MISMATCH" + lines.append(f"Control -- RampNet @{c['threshold']:.2f}: op_cache {c['f1_op_cache']:.3f} " + f"vs published bundle {c['f1_published_bundle']:.3f} " + f"(delta {c['delta']:+.3f}, tol {c['tolerance']:.3f}) [{status}]") + if c.get("per_split"): + lines.append(" per split -- the macro above clears its tolerance partly by " + "cancellation, so it is also asserted here:") + lines.append(f" {'split':<20}{'op_cache':>10}{'bundle':>9}{'delta':>9}{'tol':>8} why") + for split, r in c["per_split"].items(): + why = ("same computation, must agree" if r["paths_agree_by_construction"] + else "documented divergence (docs/operating_point.md)") + flag = "" if r["agrees"] else " <- MISMATCH" + lines.append(f" {split:<20}{r['f1_op_cache']:>10.4f}" + f"{r['f1_published_bundle']:>9.3f}{r['delta']:>+9.4f}" + f"{r['tolerance']:>8.3f} {why}{flag}") + lines.append("") + hdr = (f"{'model':<18}{'sel thr':>9}{'P':>8}{'R':>8}{'F1':>8}" + f"{'dF1 vs RampNet':>16}{'published F1':>14}{'gain':>8}") + lines.append(hdr) + lines.append("-" * len(hdr)) + base = result["models"].get(RAMPNET, {}).get("pooled", {}).get("f1") + order = sorted(result["models"], key=lambda m: -result["models"][m]["pooled"]["f1"]) + for m in order: + r = result["models"][m] + p = r["pooled"] + pub = r["published_point"]["f1"] if r["published_point"] else float("nan") + d = "" if base is None else f"{p['f1'] - base:>+16.3f}" + mark = " *" if r["at_floor"] else "" + lines.append(f"{m:<18}{r['selected_threshold']:>9.2f}{p['p']:>8.3f}{p['r']:>8.3f}" + f"{p['f1']:>8.3f}{d}{pub:>14.3f}{p['f1'] - pub:>+8.3f}{mark}") + if any(r["at_floor"] for r in result["models"].values()): + lines.append("") + lines.append("* selected threshold is the cache floor -- the true optimum may be " + "lower and unmeasured, so this F1 is a LOWER BOUND.") + return "\n".join(lines) + + +def main(): + ap = argparse.ArgumentParser(description=__doc__.split("\n")[0]) + ap.add_argument("--dev-split", default=DEFAULT_DEV, choices=sorted(NON_POOLED), + help="split the uniform threshold is selected on (never reported over)") + ap.add_argument("--sensitivity", action="store_true", + help="re-run selection with every candidate dev split") + ap.add_argument("--json", default=OUT_JSON, help="artifact path") + ap.add_argument("--check", action="store_true", + help="exit non-zero if the artifact or the control has drifted") + args = ap.parse_args() + + result = artifact(args.dev_split) + print(_render(result)) + + if args.sensitivity: + print("\nSensitivity -- the dev split does not carry the result:\n") + sens = result["sensitivity"] + models = sorted({m for v in sens.values() for m in v}) + print(f"{'dev split':<22}" + "".join(f"{m:>22}" for m in models)) + for dev in NON_POOLED: + row = "".join(f"{sens[dev][m]['thr']:>10.2f} -> {sens[dev][m]['f1']:<9.3f}" + if m in sens[dev] else f"{'-':>22}" for m in models) + print(f"{dev:<22}{row}") + + + if not result["control"]["agrees"]: + print("\nFATAL: RampNet no longer reproduces its published 0.827 at 0.55.", + file=sys.stderr) + return 2 + + if not result["control"]["per_split_agrees"]: + bad = ", ".join(k for k, r in result["control"]["per_split"].items() + if not r["agrees"]) + print(f"\nFATAL: the per-split control failed on {bad} -- op_cache and" + " the published bundle have diverged on a split where they should not.", + file=sys.stderr) + return 2 + + payload = rnd(result) + if args.check: + if not os.path.exists(args.json): + print(f"\n--check: {args.json} does not exist", file=sys.stderr) + return 1 + with open(args.json, encoding="utf-8") as fh: + if json.load(fh) != payload: + print(f"\n--check: {args.json} is stale", file=sys.stderr) + return 1 + print(f"\n--check: {os.path.relpath(args.json, REPO)} is current") + return 0 + + os.makedirs(os.path.dirname(args.json), exist_ok=True) + with open(args.json, "w", encoding="utf-8", newline="") as fh: + json.dump(payload, fh, indent=2, sort_keys=True) + fh.write("\n") + print(f"\nwrote {os.path.relpath(args.json, REPO)}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/analysis/yolo_geometry_51.py b/scripts/analysis/yolo_geometry_51.py new file mode 100644 index 00000000..6786eac3 --- /dev/null +++ b/scripts/analysis/yolo_geometry_51.py @@ -0,0 +1,305 @@ +#!/usr/bin/env python3 +"""Read the #51 geometry-pair eval and answer the equirect objection. + +WHAT QUESTION THIS ANSWERS + #51 reports that RampNet beats a supervised YOLO baseline by 0.252 F1 macro-meaned + over the seven pooled US city splits. The standing objection is that this is not + architecture: the YOLO arms are fed 2048x4096 equirectangular panoramas, which is + not the geometry a COCO-shaped detector expects. The tiles arms are the control -- + same data, same schedule, fed through the perspective-view rig the VLMs get -- and + until 2026-08-30 no tiles checkpoint had ever been scored, so the objection was + live and unmeasured. + +WHAT IT READS + docs/data/yolo_geometry_51/_{tiles,pano}.txt, the captured stdout of + scripts/model_comparison/yolo_baseline/run_yolo_geometry_eval.sh (makelab2, A40). + Parsing the driver's own output rather than re-deriving means this script cannot + silently disagree with the run that produced the numbers; the run is the artifact. + +THE CONTROL IS THE LOAD-BEARING PART + y11x_pano_h200 is already published, scored 2026-08-14 against a repo predating the + #132 seam fix. It was re-run here under the same commit as the two new legs. If + the geometry comparison is sound, this leg must reproduce the committed scoreboard + row for "YOLO11x (pano)" -- P 0.969 / R 0.416 / F1 0.575. ``--check`` asserts + exactly that, so a code change that moves YOLO scoring fails here loudly instead of + being read as a geometry effect. + +USAGE + python scripts/analysis/yolo_geometry_51.py # table + rewrite the JSON + python scripts/analysis/yolo_geometry_51.py --check # verify, write nothing +""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import sys + +HERE = os.path.dirname(os.path.abspath(__file__)) +ROOT = os.path.dirname(HERE) +sys.path.insert(0, ROOT) + +from analysis.low_floor_sweep import HELD_OUT # noqa: E402 + +# The split population this study was run over, PINNED -- deliberately NOT the live +# registry in ``analysis.low_floor_sweep``. The 2026-08-30 run scored these ten splits +# and no others, and the seven-split pool is the population every macro-mean in +# docs/yolo_geometry_51.md is taken over. A split added to the benchmark afterwards has +# no report here, so following the registry would silently turn every pooled number into +# ``None`` and take the committed artifact with it. Adding a split to this study means +# running the driver over it, not editing a tuple. +POOLED_SPLITS = ("richmond", "bend", "clovis", "morgantown", "annapolis", "paterson", + "gainesville") +ALL_SPLITS_AS_RUN = POOLED_SPLITS + ("budapest_district5", "sao_paulo", "manual_gold") + +# Why each non-pooled split is scored but not pooled. The KEYS are pinned to the run; +# only the reason text follows the registry, so the two cannot disagree about a split +# this study never saw. +HELD_OUT_AS_RUN = {s: HELD_OUT[s] for s in ALL_SPLITS_AS_RUN if s in HELD_OUT} + +IN_DIR = os.path.join(ROOT, "..", "docs", "data", "yolo_geometry_51") +OUT_JSON = os.path.join(ROOT, "..", "docs", "data", "yolo_geometry_51.json") + +# The committed scoreboard row this run's control leg must reproduce. Restated here +# (not imported) on purpose: it is the PUBLISHED number as of 2026-08-14, and the point +# of the check is to catch the day the generator stops agreeing with it. +PUBLISHED_CONTROL = {"model": "YOLO11x (pano)", "p": 0.969, "r": 0.416, "f1": 0.575} +PUBLISHED_RAMPNET_F1 = 0.827 + +LEGS = { + "y11x_tiles": {"geometry": "perspective tiles, imgsz 1024", "epoch": 44}, + "y11x_pano": {"geometry": "whole pano, imgsz 1280", "epoch": 38}, + "y11x_pano_h200": {"geometry": "whole pano, imgsz 1280", "epoch": 60}, +} + +ROW_RE = re.compile( + r"^(?P\S+)\s+" + r"(?P

[\d.]+)\s+\([\d.]+-[\d.]+\)\s+" + r"(?P[\d.]+)\s+\([\d.]+-[\d.]+\)\s+" + r"(?P[\d.]+)\s+" + r"(?P[\d.]+|-)\s+" + r"(?P\d+)/(?P\d+)/(?P\d+)/(?P\d+)\s*$" +) +BEST_RE = re.compile(r"^\s*(?P[\d.]+)\s+[\d.]+\s+[\d.]+\s+(?P[\d.]+)\s+\S+\s+<- best F1") + + +def parse_report(path): + """Rows from the operating-point table, plus each model's tuned-on-test best-F1. + + The sweep's best row is parsed but never used for a headline: the pre-registered + operating point is conf 0.25 (#71). It is carried so the write-up can say by how + much a tuned threshold would have flattered each leg, which is the honest way to + report a number nobody is allowed to select on. + """ + rows, best, current = {}, {}, None + with open(path, encoding="utf-8") as fh: + for line in fh: + m = ROW_RE.match(line.rstrip("\n")) + if m and m.group("model") not in ("model",): + d = m.groupdict() + rows[d["model"]] = { + "p": float(d["p"]), "r": float(d["r"]), "f1": float(d["f1"]), + "ap": None if d["ap"] == "-" else float(d["ap"]), + "tp": int(d["tp"]), "fp": int(d["fp"]), + "fn": int(d["fn"]), "ign": int(d["ign"]), + } + continue + m = re.match(r"^\[(\S+)\] threshold sweep", line) + if m: + current = m.group(1) + continue + m = BEST_RE.match(line) + if m and current: + best[current] = {"thr": float(m.group("thr")), "f1": float(m.group("f1"))} + return rows, best + + +def collect(): + cells, best = {}, {} + for split in ALL_SPLITS_AS_RUN: + for kind in ("tiles", "pano"): + path = os.path.join(IN_DIR, f"{split}_{kind}.txt") + if not os.path.exists(path): + continue + rows, bests = parse_report(path) + for model, vals in rows.items(): + cells.setdefault(model, {})[split] = vals + for model, vals in bests.items(): + best.setdefault(model, {})[split] = vals + return cells, best + + +def pooled(per_split): + """Macro-mean over POOLED_SPLITS, and count-pooled P/R alongside it. + + Macro is the published convention (each city weighted equally, so a big split + cannot dominate); micro is carried because a macro-mean of ratios hides how many + ramps are actually behind each city. + """ + got = [per_split[s] for s in POOLED_SPLITS if s in per_split] + if len(got) != len(POOLED_SPLITS): + return None + tp = sum(c["tp"] for c in got) + fp = sum(c["fp"] for c in got) + fn = sum(c["fn"] for c in got) + micro_p = tp / (tp + fp) if tp + fp else 0.0 + micro_r = tp / (tp + fn) if tp + fn else 0.0 + return { + "n_splits": len(got), + "macro_p": sum(c["p"] for c in got) / len(got), + "macro_r": sum(c["r"] for c in got) / len(got), + "macro_f1": sum(c["f1"] for c in got) / len(got), + "macro_ap": sum(c["ap"] for c in got) / len(got) if all(c["ap"] is not None for c in got) else None, + "micro_p": micro_p, + "micro_r": micro_r, + "micro_f1": 2 * micro_p * micro_r / (micro_p + micro_r) if micro_p + micro_r else 0.0, + "tp": tp, "fp": fp, "fn": fn, + } + + +def rnd(o, n=4): + if isinstance(o, float): + return round(o, n) + if isinstance(o, dict): + return {k: rnd(v, n) for k, v in o.items()} + if isinstance(o, list): + return [rnd(v, n) for v in o] + return o + + +def main(): + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--check", action="store_true", + help="Verify the control leg and the committed JSON; write nothing.") + args = ap.parse_args() + + cells, best = collect() + if not cells: + sys.exit(f"no eval reports found under {IN_DIR}") + + pools = {m: pooled(v) for m, v in cells.items()} + + # --- the control check, before any number is reported ------------------------ + ctrl = pools.get("y11x_pano_h200") + problems = [] + if ctrl is None: + problems.append("control leg y11x_pano_h200 did not cover all seven US splits") + else: + for key, pub in (("macro_p", "p"), ("macro_r", "r"), ("macro_f1", "f1")): + got, want = round(ctrl[key], 3), PUBLISHED_CONTROL[pub] + if abs(got - want) > 0.0005: + problems.append( + f"control {key}={got:.3f} != published {PUBLISHED_CONTROL['model']} " + f"{pub}={want:.3f} -- YOLO scoring moved; the geometry read is NOT safe" + ) + + print("=" * 78) + print("#51 geometry pair: does the equirect input explain the RampNet-YOLO gap?") + print("=" * 78) + print(f"\nControl: y11x_pano_h200 re-scored under this commit vs its published row") + if ctrl: + print(f" published P {PUBLISHED_CONTROL['p']:.3f} R {PUBLISHED_CONTROL['r']:.3f} F1 {PUBLISHED_CONTROL['f1']:.3f}") + print(f" re-scored P {ctrl['macro_p']:.3f} R {ctrl['macro_r']:.3f} F1 {ctrl['macro_f1']:.3f}") + print(" " + ("REPRODUCED - the #132 seam fix did not move YOLO pano scoring" + if not problems else "MISMATCH:\n " + "\n ".join(problems))) + + print("\nPer-split F1 at the pre-registered conf 0.25:\n") + order = ["y11x_tiles", "y11x_pano", "y11x_pano_h200"] + hdr = f"{'split':<20}" + "".join(f"{m:>17}" for m in order) + " tiles-pano60" + print(hdr) + print("-" * len(hdr)) + for s in ALL_SPLITS_AS_RUN: + line = f"{s:<20}" + for m in order: + c = cells.get(m, {}).get(s) + line += f"{c['f1']:>17.3f}" if c else f"{'-':>17}" + t = cells.get("y11x_tiles", {}).get(s) + h = cells.get("y11x_pano_h200", {}).get(s) + line += f"{t['f1'] - h['f1']:>15.3f}" if t and h else f"{'-':>15}" + if s in HELD_OUT_AS_RUN: + line += " (held out)" + print(line) + + print(f"\nMacro-mean over the seven pooled US splits ({', '.join(POOLED_SPLITS)}):\n") + print(f"{'leg':<20}{'epoch':>7}{'P':>9}{'R':>9}{'F1':>9}{'AP':>9} vs RampNet {PUBLISHED_RAMPNET_F1}") + print("-" * 78) + for m in order: + p = pools.get(m) + if not p: + continue + meta = LEGS.get(m, {}) + print(f"{m:<20}{meta.get('epoch', '?'):>7}{p['macro_p']:>9.3f}{p['macro_r']:>9.3f}" + f"{p['macro_f1']:>9.3f}{(p['macro_ap'] or 0):>9.3f}" + f"{p['macro_f1'] - PUBLISHED_RAMPNET_F1:>15.3f}") + + t = pools.get("y11x_tiles") # tiles, ep44 + p38 = pools.get("y11x_pano") # pano, ep38 + h = pools.get("y11x_pano_h200") # pano, ep60 -- the published arm + decomposition = None + if t and p38 and h: + # The published gap does NOT decompose into "geometry" alone. The tiles arm is + # at ep44 and the published pano arm at ep60, so a raw tiles-minus-published + # difference mixes geometry with training budget. Split it on the pano lineage, + # where budget is the only thing that moves: + # budget = pano ep38 - pano ep60 (same geometry, different budget) + # geometry = tiles ep44 - pano ep38 (different geometry, ~matched budget) + budget = p38["macro_f1"] - h["macro_f1"] + geometry = t["macro_f1"] - p38["macro_f1"] + total = t["macro_f1"] - h["macro_f1"] + gap = PUBLISHED_RAMPNET_F1 - h["macro_f1"] + decomposition = {"budget": budget, "geometry": geometry, "total": total, + "published_gap": gap, "residual": PUBLISHED_RAMPNET_F1 - t["macro_f1"]} + print(f"\nThe published {gap:.3f} F1 gap, decomposed (pooled US splits):") + print(f" over-training pano ep60 -> ep38 {budget:+.3f} same geometry, less budget") + print(f" geometry pano ep38 -> tiles {geometry:+.3f} ~matched budget (38 vs 44)") + print(f" {'':<33}{'-' * 7}") + print(f" best YOLO cell we have {total:+.3f} ({total / gap * 100:.0f}% of the gap)") + print(f"\n Residual still to RampNet: {decomposition['residual']:.3f} F1.") + print(f" The geometry half is RECALL: R {p38['macro_r']:.3f} -> {t['macro_r']:.3f} " + f"({t['macro_r'] - p38['macro_r']:+.3f}) at P {p38['macro_p']:.3f} -> " + f"{t['macro_p']:.3f} ({t['macro_p'] - p38['macro_p']:+.3f}).") + print("\n CAVEAT: y11x_pano and y11x_pano_h200 are divergent continuations of one\n" + " lineage (h200 forked from y11x_pano/best.pt, MANIFEST-2026-08-03), trained\n" + " on different hardware. The budget term is therefore suggestive, not clean:\n" + " it is confounded with the fork. The geometry term is the better-controlled\n" + " of the two, and it is the smaller one.") + + payload = rnd({ + "what": "#51 geometry-pair eval: tiles vs pano at near-matched budget, " + "plus the published pano arm re-scored as a control", + "operating_point": 0.25, + "pooled_splits": list(POOLED_SPLITS), + "held_out": HELD_OUT_AS_RUN, + "legs": LEGS, + "published_control": PUBLISHED_CONTROL, + "published_rampnet_pooled_f1": PUBLISHED_RAMPNET_F1, + "control_reproduced": not problems, + "per_split": cells, + "best_f1_sweep_tune_on_test": best, + "pooled": pools, + "decomposition": decomposition, + }) + + if args.check: + if not os.path.exists(OUT_JSON): + sys.exit(f"--check: {OUT_JSON} does not exist") + with open(OUT_JSON, encoding="utf-8") as fh: + if json.load(fh) != payload: + sys.exit("--check: committed JSON does not match a fresh read of the reports") + print(f"\n--check: {os.path.relpath(OUT_JSON, ROOT)} matches.") + else: + # newline="" so the committed bytes are LF on every platform. + with open(OUT_JSON, "w", encoding="utf-8", newline="") as fh: + json.dump(payload, fh, indent=2, sort_keys=True) + fh.write("\n") + print(f"\nwrote {os.path.relpath(OUT_JSON, ROOT)}") + + if problems: + sys.exit(1) + + +if __name__ == "__main__": + main() diff --git a/scripts/model_comparison/check_yolo_dataset_loads.py b/scripts/model_comparison/check_yolo_dataset_loads.py new file mode 100644 index 00000000..9a82a795 --- /dev/null +++ b/scripts/model_comparison/check_yolo_dataset_loads.py @@ -0,0 +1,120 @@ +#!/usr/bin/env python3 +"""Acceptance test for the YOLO label rebuild: does Ultralytics load the dataset again? (#51) + +The rebuild in `rebuild_yolo_labels_from_cache.py` verifies its own output against the +cache it read, which proves the round trip but *not* the thing that actually broke. The +failure was in Ultralytics' dataset init: + + ValueError: train: No labels found in .../yolo/tiles/labels/train.cache + +so the only test that closes the loop is making Ultralytics build the dataset and report +a non-zero `nf` (labels found). This does that and nothing else -- it builds the dataset, +checks what the scan produced, and exits. No model, no GPU, no training step. + +Three things it checks that a bare "did it crash" run would not: + +- **Missing label files against empty ones.** This is the one that needs care, because + Ultralytics makes the two look identical downstream: `verify_image_label` counts a + missing file as `nm` and an empty file as `ne`, but BOTH append a record with zero + boxes to `dataset.labels`. So a rebuild that skipped the 59,923 label-less background + tiles would produce the same image count, the same box count and the same number of + zero-box records as a correct one -- and the dataset would still be wrong, because the + original failure (`No labels found ... nf == 0`) is raised off `nf`, which only counts + files that exist. `dataset.labels` therefore cannot answer this; the check stats the + label paths directly and fails if any is absent. +- **Total boxes against the expected count.** Loading is not the same as loading + everything. The cache the labels came from holds 968,227 train boxes; if the rebuilt + dataset scans to a different number, the labels are wrong in a way that trains fine and + scores wrong. +- **Background count against the expected count.** 59,923 of the 557,413 train tiles are + background and must be present as zero-byte files. + +Note this builds the dataset with `augment=False`, where Ultralytics only *warns* on +`nf == 0` instead of raising -- so the original ValueError is not reproduced verbatim. +The `--expect-boxes` and missing-file checks are what catch that state here. + +Usage: + + python check_yolo_dataset_loads.py --data-root /gscratch/scrubbed/jfroehli/yolo/tiles \\ + --split train --expect-images 557413 --expect-boxes 968227 \\ + --expect-background 59923 + +Exit status is 0 only when every check passes, so it can be run unattended. +""" +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__.split("\n")[0]) + ap.add_argument("--data-root", type=Path, required=True, + help="dataset root holding images/ and labels/") + ap.add_argument("--split", default="train", help="split to load (default: %(default)s)") + ap.add_argument("--expect-images", type=int, default=None, + help="fail unless this many image/label pairs are found") + ap.add_argument("--expect-boxes", type=int, default=None, + help="fail unless the scanned labels hold this many boxes") + ap.add_argument("--expect-background", type=int, default=None, + help="fail unless this many scanned tiles have zero boxes") + ap.add_argument("--imgsz", type=int, default=640, help="only affects the scan, not results") + args = ap.parse_args() + + from ultralytics.data.dataset import YOLODataset + + img_dir = args.data_root / "images" / args.split + if not img_dir.is_dir(): + print(f"FATAL: no image directory at {img_dir}") + return 1 + + print(f"data-root : {args.data_root}") + print(f"split : {args.split}") + print("building the dataset (this is the call that raised the original ValueError) ...", + flush=True) + + ds = YOLODataset(img_path=str(img_dir), imgsz=args.imgsz, augment=False, + data={"names": {0: "curb_ramp"}, "channels": 3}) + + n_images = len(ds.labels) + n_boxes = sum(len(rec["bboxes"]) for rec in ds.labels) + n_empty = sum(1 for rec in ds.labels if len(rec["bboxes"]) == 0) + + # A record with zero boxes is either an empty label file (correct) or a MISSING one + # (the rebuild skipped it), and ds.labels cannot tell them apart. Stat the paths. + from ultralytics.data.utils import img2label_paths + + label_files = img2label_paths(ds.im_files) + missing = [f for f in label_files if not Path(f).is_file()] + + print(f"images : {n_images}") + print(f"boxes : {n_boxes}") + print(f"background: {n_empty} (zero-box tiles, which must be PRESENT as empty files)") + print(f"missing : {len(missing)} label files absent from disk") + + ok = True + if n_images == 0: + print("FAIL: zero image/label pairs -- this is the original failure, unfixed") + ok = False + if missing: + print(f"FAIL: {len(missing)} label files are MISSING, not empty -- Ultralytics " + f"counts these as nm, not nf, and nf == 0 is what raised the original " + f"ValueError. First few: {[str(m) for m in missing[:3]]}") + ok = False + if args.expect_images is not None and n_images != args.expect_images: + print(f"FAIL: expected {args.expect_images} images, scanned {n_images}") + ok = False + if args.expect_boxes is not None and n_boxes != args.expect_boxes: + print(f"FAIL: expected {args.expect_boxes} boxes, scanned {n_boxes}") + ok = False + if args.expect_background is not None and n_empty != args.expect_background: + print(f"FAIL: expected {args.expect_background} background tiles, scanned {n_empty}") + ok = False + + print("PASS -- Ultralytics builds the dataset and the counts match" if ok else "FAIL") + return 0 if ok else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/model_comparison/rebuild_yolo_labels_from_cache.py b/scripts/model_comparison/rebuild_yolo_labels_from_cache.py new file mode 100644 index 00000000..0c320dfd --- /dev/null +++ b/scripts/model_comparison/rebuild_yolo_labels_from_cache.py @@ -0,0 +1,170 @@ +#!/usr/bin/env python3 +"""Rebuild YOLO label .txt files from an Ultralytics label cache (#51). + +WHY THIS EXISTS +--------------- +On 2026-08-19 the `y11x_tiles` arm failed at dataset init with + + ValueError: train: No labels found in .../yolo/tiles/labels/train.cache + +Every label directory under `/gscratch/scrubbed/jfroehli/yolo/` was empty -- all five +dataset variants, train and val -- while every image directory was fully intact +(557,413 train tiles, 161,002 val). `/gscratch/scrubbed` purges by access time, and +the `.cache` files are exactly what stopped anything from reading the individual +label `.txt` files after 2026-07-25. Their atime froze while the images kept being +read every epoch, so the purge took the labels and left the images. + +**The cache that made training fast is what got the labels deleted.** That is the +generalisable part, and it is the same shape as the partially-purged conda package +cache that broke the env build in #51 -- a cache is not a backup, and a *populated* +cache actively hides the thing it caches from anything that decides what is cold. + +WHY THE CACHE IS A SUFFICIENT SOURCE +------------------------------------ +Ultralytics' cache is not a digest: it stores the fully parsed labels. Each record +carries `cls` (n,1) and `bboxes` (n,4) as normalized xywh -- which is the on-disk +`.txt` format itself. So the labels are recoverable exactly, with no re-derivation +from the source panoramas and no GPU. + +That matters for provenance as much as for cost: rebuilding from the cache reproduces +the labels the published `y11x_tiles` / `y26_pano` arms actually trained on, whereas +re-running `prepare_yolo_dataset.py` would produce labels from today's code and today's +geometry constants. Those should agree, but "should" is not "do", and the arms in +flight were trained on these. + +The write format below is lifted from `prepare_yolo_dataset.py::_write_pair` and the +line construction beside it, so a rebuilt file is byte-identical to the original for +any value that survives the float32 round trip through the cache -- with ONE exception, +which is training-equivalent but not byte-identical. Ultralytics' `verify_image_label` +runs `np.unique(lb, axis=0, return_index=True)` and, when a file held duplicate rows, +keeps `lb[i]`: the duplicates are dropped AND the survivors come back in `np.unique`'s +sorted order. So a source file that had a duplicated box is rebuilt deduplicated and +row-sorted. That is exactly what Ultralytics would itself have trained on, but do not +expect `diff` against a pre-purge backup to be empty for such a file. + + line = f"0 {u:.6f} {v:.6f} {w:.6f} {h:.6f}" + body = "\n".join(lines) + ("\n" if lines else "") + +Background tiles therefore get a **zero-byte file, not a missing one**. That is +deliberate: Ultralytics counts a missing label as `nm` (missing) and an empty one as +`nf` (found, no objects), and the failure above is raised when `nf == 0`. + +USAGE +----- + python rebuild_yolo_labels_from_cache.py \ + --cache /gscratch/makelab/jonf/rampnet_yolo_baseline_51/label_cache_rescue/train.cache \ + --labels-dir /gscratch/scrubbed/jfroehli/yolo/tiles/labels/train \ + --verify + +Prefer the durable copy of the cache under `/gscratch/makelab` (purchased, never +purged) over the one on `scrubbed`, which is one purge window from being the same +problem again. + +`--verify` re-reads every file it wrote and compares the parsed boxes back against the +cache, which is the only check that actually proves the round trip rather than assuming +it. It roughly doubles the runtime; run it at least once per rebuilt split. +""" +from __future__ import annotations + +import argparse +import os +import sys +from pathlib import Path + + +def load_cache(path: Path): + """Return the list of label records from an Ultralytics *.cache file.""" + import numpy as np + obj = np.load(str(path), allow_pickle=True).item() + labels = obj.get("labels") + if labels is None: + raise SystemExit(f"{path}: no 'labels' key -- not an Ultralytics label cache") + return labels + + +def format_record(rec) -> str: + """Render one cache record as the .txt body prepare_yolo_dataset.py would have written.""" + cls, boxes = rec["cls"], rec["bboxes"] + lines = [ + f"{int(cls[i][0])} {boxes[i][0]:.6f} {boxes[i][1]:.6f} {boxes[i][2]:.6f} {boxes[i][3]:.6f}" + for i in range(len(boxes)) + ] + return "\n".join(lines) + ("\n" if lines else "") + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__.split("\n")[0]) + ap.add_argument("--cache", type=Path, required=True, help="Ultralytics *.cache to read") + ap.add_argument("--labels-dir", type=Path, required=True, help="directory to write *.txt into") + ap.add_argument("--verify", action="store_true", + help="re-read every written file and compare boxes back to the cache") + ap.add_argument("--dry-run", action="store_true", help="report what would be written, write nothing") + ap.add_argument("--progress-every", type=int, default=50000, help="progress line cadence") + args = ap.parse_args() + + labels = load_cache(args.cache) + print(f"cache : {args.cache}") + print(f"records : {len(labels)}") + total_boxes = sum(len(r["bboxes"]) for r in labels) + print(f"boxes : {total_boxes}") + print(f"labels-dir : {args.labels_dir}") + if args.dry_run: + print("DRY RUN -- nothing written") + return 0 + + args.labels_dir.mkdir(parents=True, exist_ok=True) + written = empty = 0 + for i, rec in enumerate(labels, 1): + stem = Path(rec["im_file"]).stem + body = format_record(rec) + # newline='\n' so a rebuild run from Windows cannot write CRLF into a + # dataset the cluster reads. + (args.labels_dir / f"{stem}.txt").write_text(body, newline="\n") + written += 1 + if not body: + empty += 1 + if args.progress_every and i % args.progress_every == 0: + print(f" ... {i}/{len(labels)}", flush=True) + + print(f"written : {written} ({empty} background/empty, {written - empty} with boxes)") + + if not args.verify: + print("OK (unverified -- pass --verify to prove the round trip)") + return 0 + + print("verifying ...", flush=True) + bad = 0 + seen_boxes = 0 + for i, rec in enumerate(labels, 1): + stem = Path(rec["im_file"]).stem + text = (args.labels_dir / f"{stem}.txt").read_text() + got = [ln.split() for ln in text.splitlines() if ln.strip()] + exp = rec["bboxes"] + if len(got) != len(exp): + print(f" MISMATCH count {stem}: file {len(got)} vs cache {len(exp)}") + bad += 1 + continue + for j, parts in enumerate(got): + vals = [float(p) for p in parts[1:5]] + for k in range(4): + # .6f is the on-disk precision, so agreement is bounded by rounding, + # not by anything about the data. + if abs(vals[k] - float(exp[j][k])) > 1e-6: + print(f" MISMATCH value {stem} box {j} coord {k}: " + f"{vals[k]} vs {float(exp[j][k])}") + bad += 1 + break + seen_boxes += len(got) + if args.progress_every and i % args.progress_every == 0: + print(f" ... verified {i}/{len(labels)}", flush=True) + + print(f"verified : {len(labels)} files, {seen_boxes} boxes, {bad} mismatched") + if bad or seen_boxes != total_boxes: + print("FAIL") + return 1 + print("PASS -- every file round-trips to the cache it came from") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/model_comparison/run_rebuild_yolo_labels.slurm b/scripts/model_comparison/run_rebuild_yolo_labels.slurm new file mode 100644 index 00000000..504cfd50 --- /dev/null +++ b/scripts/model_comparison/run_rebuild_yolo_labels.slurm @@ -0,0 +1,82 @@ +#!/bin/bash +# Rebuild the purged YOLO tile labels from the rescued Ultralytics caches (#51). +# +# This is a batch job rather than a login-node one-liner for one reason: it creates +# ~718,000 small files on GPFS (557,413 train + 161,002 val), which is metadata-heavy +# and sustained. klone reaps heavy login processes and that reap also kills the SSH +# control master. +# +# It reads the DURABLE cache copies under /gscratch/makelab (purchased, never purged), +# not the ones on /gscratch/scrubbed -- those are the copies whose siblings were already +# lost once, and rebuilding from them would make the recovery depend on the thing that +# failed. The rescued copies were staged 2026-08-17 with sha256sums.txt beside them. +# +# sbatch scripts/model_comparison/run_rebuild_yolo_labels.slurm +# +# Env overrides: RAMPNET_REPO, YOLO_CACHE_DIR, YOLO_DATA_ROOT. +# +#SBATCH --job-name=yolo_label_rebuild_51 +#SBATCH --account=ckpt-makelab +#SBATCH --partition=ckpt-all +#SBATCH --nodes=1 +#SBATCH --ntasks=1 +#SBATCH --cpus-per-task=4 +#SBATCH --mem=16G +#SBATCH --time=6:00:00 +#SBATCH --output=logs/yolo_label_rebuild_%j.out +#SBATCH --error=logs/yolo_label_rebuild_%j.err +#SBATCH --requeue + +set -eo pipefail + +REPO="${RAMPNET_REPO:-$HOME/RampNet}" +CACHE_DIR="${YOLO_CACHE_DIR:-/gscratch/makelab/jonf/rampnet_yolo_baseline_51/label_cache_rescue}" +DATA_ROOT="${YOLO_DATA_ROOT:-/gscratch/scrubbed/jfroehli/yolo/tiles}" +PY="${YOLO_PY:-/gscratch/makelab/jonf/envs/yolo/bin/python}" +SCRIPT="$REPO/scripts/model_comparison/rebuild_yolo_labels_from_cache.py" + +echo "--- YOLO label rebuild (#51) ---" +echo "Job ID : ${SLURM_JOBID}" +echo "Node : ${SLURMD_NODENAME:-?}" +echo "Cache dir : $CACHE_DIR" +echo "Data root : $DATA_ROOT" +echo "Script : $SCRIPT" +echo "Python : $PY" +echo "-------------------------------" + +# The rescued caches carry their own checksums. Verify before trusting them as the +# single source for 968,227 boxes. +if [ -f "$CACHE_DIR/sha256sums.txt" ]; then + echo "=== verifying rescued caches against sha256sums.txt ===" + ( cd "$CACHE_DIR" && sha256sum -c sha256sums.txt ) || { echo "FATAL: cache checksum mismatch"; exit 1; } +else + echo "WARNING: no sha256sums.txt beside the caches -- proceeding unverified" +fi + +for split in train val; do + echo + echo "=== $split $(date -Is) ===" + "$PY" "$SCRIPT" \ + --cache "$CACHE_DIR/${split}.cache" \ + --labels-dir "$DATA_ROOT/labels/${split}" \ + --verify +done + +echo +echo "=== stale .cache files ===" +# Ultralytics rebuilds a cache whose hash no longer matches, so these are not harmful -- +# but they are what made the failure confusing, so move them aside rather than trust them. +for split in train val; do + if [ -f "$DATA_ROOT/labels/${split}.cache" ]; then + mv -v "$DATA_ROOT/labels/${split}.cache" "$DATA_ROOT/labels/${split}.cache.pre-rebuild" + fi +done + +echo +echo "=== final counts ===" +for split in train val; do + echo " labels/$split : $(ls "$DATA_ROOT/labels/$split" | wc -l) files" + echo " images/$split : $(ls "$DATA_ROOT/images/$split" | wc -l) files" +done + +echo "--- done $(date -Is) ---" diff --git a/scripts/model_comparison/yolo_baseline/README.md b/scripts/model_comparison/yolo_baseline/README.md index ab281ba0..ed895add 100644 --- a/scripts/model_comparison/yolo_baseline/README.md +++ b/scripts/model_comparison/yolo_baseline/README.md @@ -85,7 +85,8 @@ rather than from a login session. Metrics are Ultralytics **validation** metrics auto-labelled val split at each arm's best epoch. They are the *selection* metric only — **not** the issue #51 headline, which is F1@conf0.25 against the benchmark. (As of 2026-08-14 that benchmark eval **has now run for the three pano arms** — see "Benchmark -eval — the pano trio" below; the tiles arms remain unevaluated.) +eval — the pano trio" below; `y11x_tiles` was scored on 2026-08-30, see +`docs/yolo_geometry_51.md`. `y11l_tiles` and `y26_tiles` remain unevaluated.) | arm | epochs | best ep | mAP50 | mAP50-95 | state | |---|---:|---:|---:|---:|---| @@ -133,8 +134,13 @@ The three pano-geometry arms are the first checkpoints of this grid to be scored the benchmark, under the pre-registered protocol below with nothing changed: `best.pt` as saved, one `compare.py` run per bundle, `--tiling none --yolo-imgsz 1280`, headline **F1 at conf 0.25**, match radius 0.022, all ten splits (nine cities + `manual_gold`). This was -each test bundle's first and only contact with any YOLO checkpoint. The tiles arms are -still training and remain unevaluated. +each test bundle's first and only contact with any YOLO checkpoint. + +**Since superseded in part.** `y11x_tiles` (ep44) and `y11x_pano` (ep38) were scored on all +ten splits on 2026-08-30 under the same protocol, with `y11x_pano_h200` re-scored alongside +them as a control; it reproduced its published row to three decimals. Results, decomposition +and caveats: `docs/yolo_geometry_51.md`. `y11l_tiles` and `y26_tiles` are still unevaluated +and are too undertrained to compare. **Provenance.** Run on makelab2 (A40), torch 2.13.0+cu130, **ultralytics 8.4.120 at inference vs 8.4.105 at training** (recorded, not assumed equivalent). Checkpoints are the diff --git a/scripts/model_comparison/yolo_baseline/rescue_label_caches.sh b/scripts/model_comparison/yolo_baseline/rescue_label_caches.sh new file mode 100644 index 00000000..c153d633 --- /dev/null +++ b/scripts/model_comparison/yolo_baseline/rescue_label_caches.sh @@ -0,0 +1,165 @@ +#!/usr/bin/env bash +# +# rescue_label_caches.sh - copy Ultralytics label caches to durable storage, verified. +# Issue #51. +# +# WHY THIS EXISTS +# /gscratch/scrubbed purges by ACCESS time, and an Ultralytics label cache is precisely +# what stops anything from reading the individual label .txt files. Their atime froze +# while the images kept being read every epoch, so the 2026-08 purge took all 718,415 +# tile labels and left every image. The cache that made training fast is what got the +# labels deleted. +# +# The recovery works because the cache is not a digest: it stores parsed `cls` (n,1) +# and `bboxes` (n,4) as normalized xywh, which is the on-disk .txt format itself, so +# rebuild_yolo_labels_from_cache.py can reconstitute the labels exactly. That makes +# the cache the single most valuable small file in the dataset -- and it was sitting on +# the same volume, on the same purge clock, as the thing it is the backup for. +# +# The tiles pair was staged by hand on 2026-08-17. A hand-assembled backup cannot be +# re-run by someone else, which is the test this repo applies to everything else +# (CLAUDE.md, "replicable from a clean clone"). This is that staging, as a script. +# +# WHAT IT GUARANTEES +# - Never destroys a good copy: each file lands as .tmp and is renamed only after +# its hash matches the source. +# - Idempotent: a destination that already matches is reported unchanged and left alone. +# - Refuses to overwrite a DIFFERING existing copy unless FORCE=1. This is deliberate. +# A cache regenerated by a later Ultralytics run is not the cache the published arms +# trained on, and silently replacing the durable copy with it would lose the original +# while looking like a successful backup. +# +# LAYOUT +# Caches land at $DST//{train,val}.cache with sha256sums.txt beside them. +# The tiles pair already lives FLAT at $DST/{train,val}.cache, which is what +# run_rebuild_yolo_labels.slurm defaults YOLO_CACHE_DIR to; that legacy path is left +# exactly as it is. For any other variant, point YOLO_CACHE_DIR at its subdirectory: +# +# YOLO_CACHE_DIR=$DST/pano YOLO_DATA_ROOT=$SRC/pano \ +# sbatch scripts/model_comparison/run_rebuild_yolo_labels.slurm +# +# USAGE +# ./rescue_label_caches.sh # every variant under SRC that has caches +# ./rescue_label_caches.sh pano # only the named variants +# SRC=... DST=... ./rescue_label_caches.sh # override the committed defaults +# +set -uo pipefail + +SRC="${SRC:-/gscratch/scrubbed/jfroehli/yolo}" +DST="${DST:-/gscratch/makelab/jonf/rampnet_yolo_baseline_51/label_cache_rescue}" +FORCE="${FORCE:-0}" +MAX_TRIES="${MAX_TRIES:-3}" + +SPLITS=(train val) + +n_copied=0; n_same=0; n_failed=0; n_refused=0 + +log() { printf '%s\n' "$*"; } + +human() { numfmt --to=iec --suffix=B "$1" 2>/dev/null || printf '%s' "$1"; } + +# copy_verified