AdvancedPython · Lesson 9 of 9

Project: A Results Analysis Tool

Combine everything — dataclasses, validation, statistics, ranking and reports — into a small, tested package.

This project builds the core of a school results system as a small package of modules, each with one job: models.py defines the data, loading.py reads and validates CSV rows (collecting errors instead of crashing), analysis.py computes statistics and rankings, and main.py ties them together and prints the report.

Rankings use competition ranking: equal scores share a position and the next position is skipped (1, 2, 2, 4) — the way schools usually rank. Statistics come from the standard statistics module.

Because the logic lives in small pure functions, it's easy to test: the solution adds pytest tests for loading, grading, ranking and statistics. Grow it from here: a command-line interface (previous lesson), a FastAPI endpoint, or a PostgreSQL store.

models.pyPython
from dataclasses import dataclass

GRADE_BANDS = ((75, "A"), (65, "B"), (45, "C"), (30, "D"), (0, "F"))


@dataclass(frozen=True)
class Result:
    student: str
    form: int
    subject: str
    score: int

    @property
    def grade(self) -> str:
        return next(letter for minimum, letter in GRADE_BANDS if self.score >= minimum)
loading.pyPython
import csv
from io import StringIO

from models import Result


def load_results(text: str) -> tuple[list[Result], list[str]]:
    """Parse CSV text; return the valid results and a list of readable errors."""
    results: list[Result] = []
    errors: list[str] = []
    for line_no, row in enumerate(csv.DictReader(StringIO(text)), start=2):
        try:
            score = int(row["score"])
            form = int(row["form"])
            if not 0 <= score <= 100:
                raise ValueError(f"score {score} is outside 0-100")
            if not row["student"].strip():
                raise ValueError("student name is empty")
            results.append(Result(row["student"].strip(), form, row["subject"].strip(), score))
        except (ValueError, KeyError, TypeError) as err:
            errors.append(f"line {line_no}: {err}")
    return results, errors
analysis.pyPython
import statistics
from collections import Counter, defaultdict

from models import Result


def subject_stats(results: list[Result]) -> dict[str, dict[str, float]]:
    by_subject: dict[str, list[int]] = defaultdict(list)
    for r in results:
        by_subject[r.subject].append(r.score)
    return {
        subject: {
            "mean": round(statistics.mean(scores), 1),
            "median": statistics.median(scores),
            "pass_rate": round(100 * sum(s >= 30 for s in scores) / len(scores), 1),
        }
        for subject, scores in sorted(by_subject.items())
    }


def student_averages(results: list[Result]) -> dict[str, float]:
    by_student: dict[str, list[int]] = defaultdict(list)
    for r in results:
        by_student[r.student].append(r.score)
    return {name: round(statistics.mean(s), 1) for name, s in by_student.items()}


def rank(averages: dict[str, float]) -> list[tuple[int, str, float]]:
    """Competition ranking: ties share a position, the next one is skipped (1, 2, 2, 4)."""
    ordered = sorted(averages.items(), key=lambda item: (-item[1], item[0]))
    ranked: list[tuple[int, str, float]] = []
    for i, (name, avg) in enumerate(ordered):
        position = ranked[-1][0] if ranked and ranked[-1][2] == avg else i + 1
        ranked.append((position, name, avg))
    return ranked


def grade_distribution(results: list[Result]) -> dict[str, int]:
    counts = Counter(r.grade for r in results)
    return {grade: counts.get(grade, 0) for grade in "ABCDF"}
main.pyPython
from analysis import grade_distribution, rank, student_averages, subject_stats
from loading import load_results

DATA = """student,form,subject,score
Amina Hassan,4,Maths,88
Amina Hassan,4,Biology,79
Juma Said,4,Maths,29
Juma Said,4,Biology,48
Neema Kimaro,4,Maths,71
Neema Kimaro,4,Biology,84
Ali Mohamed,4,Maths,95
Ali Mohamed,4,Biology,60
Rehema Mollel,4,Maths,abc
Zawadi Njau,4,Biology,120
"""


def main() -> None:
    results, errors = load_results(DATA)
    print(f"Loaded {len(results)} results, skipped {len(errors)}:")
    for e in errors:
        print("  -", e)

    print("\nSubject   Mean  Median  Pass%")
    for subject, s in subject_stats(results).items():
        print(f"{subject:<9}{s['mean']:>5}{s['median']:>8}{s['pass_rate']:>7}")

    print("\nPos  Student          Average")
    for position, name, avg in rank(student_averages(results)):
        print(f"{position:>3}  {name:<16}{avg:>7}")

    print("\nGrades:", grade_distribution(results))


if __name__ == "__main__":
    main()
Runs in your browser · Python

Key points

  • Split a program into small modules with one job each: models, loading, analysis, output.
  • Collect validation errors with line numbers instead of stopping at the first bad row.
  • Pure functions (stats, ranking) are easy to test and reuse in a CLI, API or web page.

Exercise

Add class_report(results, student) that returns each subject's score, grade and the student's position in that subject. Write pytest tests for rank (including a three-way tie), load_results (a bad score and a missing column) and grade_distribution.

Show solution

Try the exercise yourself first — then compare your approach with this one.

class_report reuses rank per subject, so ties are handled the same way everywhere. The tests cover a three-way tie (positions 1, 1, 1, 4), a bad score and a missing column in the CSV, and the grade distribution.

report_card.pyPython
from analysis import rank
from models import Result


def class_report(results: list[Result], student: str) -> list[tuple[str, int, str, int]]:
    """(subject, score, grade, position in that subject) for one student."""
    rows = []
    for subject in sorted({r.subject for r in results if r.student == student}):
        scores = {r.student: float(r.score) for r in results if r.subject == subject}
        position = next(pos for pos, name, _ in rank(scores) if name == student)
        mine = next(r for r in results if r.student == student and r.subject == subject)
        rows.append((subject, mine.score, mine.grade, position))
    return rows
test_project.pyPython
from analysis import grade_distribution, rank
from loading import load_results
from models import Result
from report_card import class_report


def test_rank_three_way_tie() -> None:
    ranked = rank({"A": 80.0, "B": 80.0, "C": 80.0, "D": 70.0})
    assert [pos for pos, _, _ in ranked] == [1, 1, 1, 4]


def test_load_results_collects_errors() -> None:
    text = "student,form,subject,score\nAmina,4,Maths,88\nJuma,4,Maths,abc\n"
    results, errors = load_results(text)
    assert len(results) == 1 and errors[0].startswith("line 3")

    results, errors = load_results("student,form,subject\nAmina,4,Maths\n")   # no score column
    assert results == [] and len(errors) == 1


def test_grade_distribution() -> None:
    rs = [Result("A", 4, "Maths", s) for s in (90, 70, 50, 35, 10, 80)]
    assert grade_distribution(rs) == {"A": 2, "B": 1, "C": 1, "D": 1, "F": 1}


def test_class_report_positions() -> None:
    rs = [Result("Amina", 4, "Maths", 88), Result("Ali", 4, "Maths", 95), Result("Amina", 4, "Biology", 79)]
    assert class_report(rs, "Amina") == [("Biology", 79, "A", 1), ("Maths", 88, "A", 2)]

Check your understanding

  1. With competition ranking, what positions do averages 90, 85, 85, 70 get?

  2. Why does load_results return errors instead of raising on the first bad row?

  3. What makes rank and subject_stats easy to test?

  4. Why is Result a frozen dataclass?

Ask AI