Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
99 changes: 99 additions & 0 deletions tests/test_maturity_phi_grad.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,99 @@
"""Regression tests for ``phi_grad`` in ``utils/maturity_calculator.py``.

The growth-gradient component averages the first and second halves of a
student's chronological submissions. ``Submission.score`` is a percentage
(0–100) and nullable (``models.py`` declares ``score = db.Column(db.Integer)``
with no default), because a submitted-but-not-yet-evaluated submission has no
score. The maturity formula works on the historical 0–5 scale, so scores are
converted inside the formula via ``normalize_mixed_score(...) / 20``.

The buggy implementation filtered unscored rows out of the *numerator* but
divided by the full half length, so an unscored submission silently counted as
zero. When the two halves have different missing-score rates, the reported
gradient was not just off but could have the wrong sign (e.g. real improvement
40 -> 43 reported as a decline).

These tests use lightweight in-memory fakes: no Flask app, database, Redis,
or network.
"""
from datetime import datetime, timedelta

import pytest

from utils.maturity_calculator import calculate_maturity_components


class _Sub:
"""Minimal stand-in for the Submission rows the function reads."""

def __init__(self, score, submitted_at):
self.score = score
self.submitted_at = submitted_at


def _subs(scores):
# Spaced one day apart; only relative order and values matter.
start = datetime.now() - timedelta(days=len(scores))
return [_Sub(score, start + timedelta(days=i)) for i, score in enumerate(scores)]


def _phi_grad(scores):
return calculate_maturity_components(_subs(scores))["phi_grad"]


# ---------------------------------------------------------------------------
# regression: unscored submissions must be excluded, not counted as zero
# ---------------------------------------------------------------------------

def test_phi_grad_excludes_unscored_submission_in_recent_half():
# Real trajectory: scored halves are 40 and 43 (percent). In the 0–5
# scale that is 2.00 vs 2.15, growth +0.15 -> 50 + 1.5 = 51.5.
# Before the fix the recent half divided by 2 (None counted as zero),
# averaging 1.075, growth -0.925, and phi_grad read 40.75 (a decline).
assert _phi_grad([40, 40, None, 43]) == pytest.approx(51.5)


def test_phi_grad_excludes_unscored_submission_in_first_half():
# Mirror case: scored first half 43, recent half 40, growth -0.15 -> 48.5.
# Before the fix the first half averaged 1.075, so a mild decline was
# reported as growth (59.25).
assert _phi_grad([None, 43, 40, 40]) == pytest.approx(48.5)


def test_phi_grad_normal_path_unchanged():
# All submissions scored: behavior must be identical to before.
# 0–5 scale: 2.0 vs 4.0, growth +2 -> 70.
assert _phi_grad([40, 40, 80, 80]) == 70


def test_phi_grad_flat_trajectory_is_neutral():
# No change -> the neutral gradient value 50.
assert _phi_grad([60, 60, 60, 60]) == 50


# ---------------------------------------------------------------------------
# boundary: nothing to compare against -> neutral default, never a crash
# ---------------------------------------------------------------------------

def test_phi_grad_neutral_when_no_scored_submissions():
# Division by the scored count must be guarded. The function has no
# evidence for growth, so it must keep the neutral 50 instead of raising
# ZeroDivisionError on the empty score lists.
assert _phi_grad([None, None, None, None]) == 50


def test_phi_grad_neutral_when_one_half_has_no_scores():
# First half entirely unscored: no valid first-half mean exists.
# Before the fix it was treated as 0 vs a 4.0 recent mean -> phi 90.
assert _phi_grad([None, None, 80, 80]) == 50


def test_phi_grad_neutral_when_fewer_than_four_submissions():
# Pre-existing rule: the gradient needs two comparable halves.
assert _phi_grad([40, 90]) == 50


def test_zero_scores_are_real_scores_not_dropped():
# 0 is a legitimate value, not a missing score. It participates as zero:
# first-half mean (0+4)/2 = 2.0 vs recent 4.0, growth +2 -> 70.
assert _phi_grad([0, 80, 80, 80]) == 70
24 changes: 16 additions & 8 deletions utils/maturity_calculator.py
Original file line number Diff line number Diff line change
Expand Up @@ -70,18 +70,26 @@ def calculate_maturity_components(all_subs, ability_scores=None, class_averages=
half_len = len(all_subs) // 2
first_half = all_subs[:half_len]
second_half = all_subs[half_len:]
avg_init = sum(
(normalize_mixed_score(s.score) or 0) / 20
# score 可空(已提交但尚未评分),且公式内部按历史 0–5 制计算
# (百分制 / 20 换算)。分子与分母必须基于同一批“已评分”提交:
# 不能把 None 排除出分子却计入整半长度(那等价于把未评分当 0 分,
# 两半缺失率不同时会让梯度方向都反掉)。任一半没有可评分提交时,
# 没有可比较的均值,保持中性默认 50。
init_scores = [
normalize_mixed_score(s.score) / 20
for s in first_half
if s.score is not None
) / len(first_half)
avg_recent = sum(
(normalize_mixed_score(s.score) or 0) / 20
]
recent_scores = [
normalize_mixed_score(s.score) / 20
for s in second_half
if s.score is not None
) / len(second_half)
growth = avg_recent - avg_init
result['phi_grad'] = min(100, max(0, 50 + growth * 10))
]
if init_scores and recent_scores:
avg_init = sum(init_scores) / len(init_scores)
avg_recent = sum(recent_scores) / len(recent_scores)
growth = avg_recent - avg_init
result['phi_grad'] = min(100, max(0, 50 + growth * 10))

# 计算总分
result['maturity_score'] = round(min(100, (
Expand Down
Loading