From 9cf5f7ab7d4042b3c17209a37bdad9510dcbb852 Mon Sep 17 00:00:00 2001 From: linxi123-A <3951574582@qq.com> Date: Fri, 18 Sep 2026 00:21:09 +0800 Subject: [PATCH] fix: exclude unscored submissions from phi_grad half averages --- tests/test_maturity_phi_grad.py | 99 +++++++++++++++++++++++++++++++++ utils/maturity_calculator.py | 24 +++++--- 2 files changed, 115 insertions(+), 8 deletions(-) create mode 100644 tests/test_maturity_phi_grad.py diff --git a/tests/test_maturity_phi_grad.py b/tests/test_maturity_phi_grad.py new file mode 100644 index 0000000..bf27d34 --- /dev/null +++ b/tests/test_maturity_phi_grad.py @@ -0,0 +1,99 @@ +"""Regression tests for ``phi_grad`` in ``utils/maturity_calculator.py``. + +The growth-gradient component averages the first and second halves of a +student's chronological submissions. ``Submission.score`` is a percentage +(0–100) and nullable (``models.py`` declares ``score = db.Column(db.Integer)`` +with no default), because a submitted-but-not-yet-evaluated submission has no +score. The maturity formula works on the historical 0–5 scale, so scores are +converted inside the formula via ``normalize_mixed_score(...) / 20``. + +The buggy implementation filtered unscored rows out of the *numerator* but +divided by the full half length, so an unscored submission silently counted as +zero. When the two halves have different missing-score rates, the reported +gradient was not just off but could have the wrong sign (e.g. real improvement +40 -> 43 reported as a decline). + +These tests use lightweight in-memory fakes: no Flask app, database, Redis, +or network. +""" +from datetime import datetime, timedelta + +import pytest + +from utils.maturity_calculator import calculate_maturity_components + + +class _Sub: + """Minimal stand-in for the Submission rows the function reads.""" + + def __init__(self, score, submitted_at): + self.score = score + self.submitted_at = submitted_at + + +def _subs(scores): + # Spaced one day apart; only relative order and values matter. + start = datetime.now() - timedelta(days=len(scores)) + return [_Sub(score, start + timedelta(days=i)) for i, score in enumerate(scores)] + + +def _phi_grad(scores): + return calculate_maturity_components(_subs(scores))["phi_grad"] + + +# --------------------------------------------------------------------------- +# regression: unscored submissions must be excluded, not counted as zero +# --------------------------------------------------------------------------- + +def test_phi_grad_excludes_unscored_submission_in_recent_half(): + # Real trajectory: scored halves are 40 and 43 (percent). In the 0–5 + # scale that is 2.00 vs 2.15, growth +0.15 -> 50 + 1.5 = 51.5. + # Before the fix the recent half divided by 2 (None counted as zero), + # averaging 1.075, growth -0.925, and phi_grad read 40.75 (a decline). + assert _phi_grad([40, 40, None, 43]) == pytest.approx(51.5) + + +def test_phi_grad_excludes_unscored_submission_in_first_half(): + # Mirror case: scored first half 43, recent half 40, growth -0.15 -> 48.5. + # Before the fix the first half averaged 1.075, so a mild decline was + # reported as growth (59.25). + assert _phi_grad([None, 43, 40, 40]) == pytest.approx(48.5) + + +def test_phi_grad_normal_path_unchanged(): + # All submissions scored: behavior must be identical to before. + # 0–5 scale: 2.0 vs 4.0, growth +2 -> 70. + assert _phi_grad([40, 40, 80, 80]) == 70 + + +def test_phi_grad_flat_trajectory_is_neutral(): + # No change -> the neutral gradient value 50. + assert _phi_grad([60, 60, 60, 60]) == 50 + + +# --------------------------------------------------------------------------- +# boundary: nothing to compare against -> neutral default, never a crash +# --------------------------------------------------------------------------- + +def test_phi_grad_neutral_when_no_scored_submissions(): + # Division by the scored count must be guarded. The function has no + # evidence for growth, so it must keep the neutral 50 instead of raising + # ZeroDivisionError on the empty score lists. + assert _phi_grad([None, None, None, None]) == 50 + + +def test_phi_grad_neutral_when_one_half_has_no_scores(): + # First half entirely unscored: no valid first-half mean exists. + # Before the fix it was treated as 0 vs a 4.0 recent mean -> phi 90. + assert _phi_grad([None, None, 80, 80]) == 50 + + +def test_phi_grad_neutral_when_fewer_than_four_submissions(): + # Pre-existing rule: the gradient needs two comparable halves. + assert _phi_grad([40, 90]) == 50 + + +def test_zero_scores_are_real_scores_not_dropped(): + # 0 is a legitimate value, not a missing score. It participates as zero: + # first-half mean (0+4)/2 = 2.0 vs recent 4.0, growth +2 -> 70. + assert _phi_grad([0, 80, 80, 80]) == 70 diff --git a/utils/maturity_calculator.py b/utils/maturity_calculator.py index a7077e7..573d916 100644 --- a/utils/maturity_calculator.py +++ b/utils/maturity_calculator.py @@ -70,18 +70,26 @@ def calculate_maturity_components(all_subs, ability_scores=None, class_averages= half_len = len(all_subs) // 2 first_half = all_subs[:half_len] second_half = all_subs[half_len:] - avg_init = sum( - (normalize_mixed_score(s.score) or 0) / 20 + # score 可空(已提交但尚未评分),且公式内部按历史 0–5 制计算 + # (百分制 / 20 换算)。分子与分母必须基于同一批“已评分”提交: + # 不能把 None 排除出分子却计入整半长度(那等价于把未评分当 0 分, + # 两半缺失率不同时会让梯度方向都反掉)。任一半没有可评分提交时, + # 没有可比较的均值,保持中性默认 50。 + init_scores = [ + normalize_mixed_score(s.score) / 20 for s in first_half if s.score is not None - ) / len(first_half) - avg_recent = sum( - (normalize_mixed_score(s.score) or 0) / 20 + ] + recent_scores = [ + normalize_mixed_score(s.score) / 20 for s in second_half if s.score is not None - ) / len(second_half) - growth = avg_recent - avg_init - result['phi_grad'] = min(100, max(0, 50 + growth * 10)) + ] + if init_scores and recent_scores: + avg_init = sum(init_scores) / len(init_scores) + avg_recent = sum(recent_scores) / len(recent_scores) + growth = avg_recent - avg_init + result['phi_grad'] = min(100, max(0, 50 + growth * 10)) # 计算总分 result['maturity_score'] = round(min(100, (