diff --git a/README.md b/README.md index 0496cb1..2671d39 100644 --- a/README.md +++ b/README.md @@ -79,7 +79,7 @@ project_root/ ├── run_pipeline.py # Entrypoint: orchestrates ETL ├── src/ │ ├── config.py # Paths, constants, env vars -│ ├── db/connection.py # DB abstraction (DuckDB now, BQ later) +│ ├── db/connection.py # DB abstraction (DuckDB now, BigQuery later) │ ├── pipeline/ │ │ ├── step_01_ingest.py # CSV OULAD → raw DuckDB tables │ │ ├── step_02_transform.py # Raw tables → analytical views @@ -171,7 +171,7 @@ contains 32,593 course enrollments by 28,785 distinct students across | studentVle | Clickstream (daily clicks per resource) | id_site, date, sum_click | | studentAssessment | Assessment scores | id_assessment, score | | assessments | Assessment metadata | assessment_type, date, weight | -| vle | VLE resource metadata | activity_type | +| vle | Virtual Learning Environment (VLE) resource metadata | activity_type | | courses | Course metadata | module_presentation_length | **Target variable**: `final_result` ∈ {Pass, Distinction, Fail, Withdrawn}, diff --git a/it/README.md b/it/README.md index 2f31023..7142c09 100644 --- a/it/README.md +++ b/it/README.md @@ -79,7 +79,7 @@ project_root/ ├── run_pipeline.py # Entrypoint: orchestra l'ETL ├── src/ │ ├── config.py # Path, costanti, variabili d'ambiente -│ ├── db/connection.py # Astrazione DB (DuckDB ora, BQ in futuro) +│ ├── db/connection.py # Astrazione DB (DuckDB ora, BigQuery in futuro) │ ├── pipeline/ │ │ ├── step_01_ingest.py # CSV OULAD → tabelle raw DuckDB │ │ ├── step_02_transform.py # Tabelle raw → viste analitiche @@ -171,7 +171,7 @@ su 7 moduli (22 presentazioni) presso la Open University (UK). | studentVle | Clickstream (click giornalieri per risorsa) | id_site, date, sum_click | | studentAssessment | Punteggi delle valutazioni | id_assessment, score | | assessments | Metadati delle valutazioni | assessment_type, date, weight | -| vle | Metadati risorse VLE | activity_type | +| vle | Metadati risorse VLE (Virtual Learning Environment) | activity_type | | courses | Metadati dei corsi | module_presentation_length | **Variabile target**: `final_result` ∈ {Pass, Distinction, Fail, Withdrawn}, diff --git a/notebooks/01_eda_student_base.ipynb b/notebooks/01_eda_student_base.ipynb index 4227887..4184c6c 100644 --- a/notebooks/01_eda_student_base.ipynb +++ b/notebooks/01_eda_student_base.ipynb @@ -5,7 +5,7 @@ "id": "0", "metadata": {}, "source": [ - "# 01 — EDA: Student Base\n", + "# 01. EDA: Student Base\n", "\n", "> **Notebook 01 of 7** | Learning Retention Analytics \n", "> First exploratory analysis: who are the students, what are their outcomes, and what does the data look like?" @@ -30,7 +30,7 @@ "- Brief engagement baseline (preview)\n", "\n", "**What comes next:**\n", - "- **Notebook 02** (`02_eda_engagement_patterns.ipynb`): detailed behavioral analysis — daily clickstream patterns, early engagement signals, temporal trends\n", + "- **Notebook 02** (`02_eda_engagement_patterns.ipynb`): detailed behavioral analysis (daily clickstream patterns, early engagement signals, temporal trends)\n", "\n", "**Connection to business questions:** \n", "This EDA does not answer BQ1–BQ5 directly. Instead, it establishes the population profile and data quality baseline that all subsequent analyses depend on. Think of it as *understanding the terrain before navigating*.\n", @@ -46,14 +46,14 @@ "## Table of Contents\n", "\n", "1. [Environment Setup](#1.-Environment-Setup)\n", - "2. [Dataset Overview — Shape and Scale](#2.-Dataset-Overview-—-Shape-and-Scale)\n", + "2. [Dataset Overview: Shape and Scale](#2.-Dataset-Overview:-Shape-and-Scale)\n", "3. [Outcome Distribution](#3.-Outcome-Distribution)\n", "4. [Outcome by Course](#4.-Outcome-by-Course)\n", "5. [Demographic Profile](#5.-Demographic-Profile)\n", "6. [Enrollment Patterns](#6.-Enrollment-Patterns)\n", "7. [Course Landscape](#7.-Course-Landscape)\n", "8. [Data Quality Assessment](#8.-Data-Quality-Assessment)\n", - "9. [Engagement Baseline — Preview](#9.-Engagement-Baseline-—-Preview)\n", + "9. [Engagement Baseline: Preview](#9.-Engagement-Baseline:-Preview)\n", "10. [Key Takeaways and Next Steps](#10.-Key-Takeaways-and-Next-Steps)\n", "\n", "---\n", @@ -62,7 +62,7 @@ "- The ETL pipeline must have been run: `python -m run_pipeline`\n", "- The DuckDB database at `data/db/oulad.duckdb` must contain all 5 analytical views\n", "\n", - "**Dataset:** Open University Learning Analytics Dataset (OULAD) — ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." + "**Dataset:** Open University Learning Analytics Dataset (OULAD), ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." ] }, { @@ -76,7 +76,7 @@ "\n", "**Technical notes for readers:**\n", "- Notebooks live in `notebooks/` but project modules are in `src/` at the project root. We add the project root to `sys.path` so that `from src.config import ...` works. The linter rule `E402` (imports not at top of file) is suppressed for notebooks in `pyproject.toml`.\n", - "- All database queries go through `src.db.connection.execute_query()` — the project's DB abstraction layer. This returns a `pandas.DataFrame` and ensures we never call `duckdb.connect()` directly (see [ADR-003](../docs/ADR.md)).\n", + "- All database queries go through `src.db.connection.execute_query()`, the project's DB abstraction layer. This returns a `pandas.DataFrame` and ensures we never call `duckdb.connect()` directly (see [ADR-003](../docs/ADR.md)).\n", "- Figures are saved to `reports/figures/` at 150 DPI. Since `nbstripout` removes notebook outputs before commit, the saved PNGs are the persistent visual record." ] }, @@ -137,10 +137,10 @@ "\n", "# Semantic color palette: one color per outcome category\n", "PALETTE_OUTCOME = {\n", - " 'Pass': '#4C72B0', # blue — neutral positive\n", - " 'Distinction': '#55A868', # green — strong positive\n", - " 'Fail': '#C44E52', # red — negative\n", - " 'Withdrawn': '#8172B3', # purple — departed (distinct from failed)\n", + " 'Pass': '#4C72B0', # blue: neutral positive\n", + " 'Distinction': '#55A868', # green: strong positive\n", + " 'Fail': '#C44E52', # red: negative\n", + " 'Withdrawn': '#8172B3', # purple: departed (distinct from failed)\n", "}\n", "# Binary version: completed (1) vs not completed (0)\n", "PALETTE_BINARY = {1: '#55A868', 0: '#C44E52'}\n", @@ -149,7 +149,7 @@ "PALETTE_BINARY_LABELS = {'Completed': '#55A868', 'Not completed': '#C44E52'}\n", "# Sequential palette for heatmaps and continuous scales\n", "PALETTE_SEQUENTIAL = 'YlOrRd'\n", - "# Shared axis label — the unit of analysis is the enrollment, not the student\n", + "# Shared axis label: the unit of analysis is the enrollment, not the student\n", "LABEL_NUM_ENROLLMENTS = 'Number of enrollments'\n", "\n", "FIG_DPI = 150\n", @@ -174,7 +174,7 @@ " Column names are standardized so that plot_demo_completion() works\n", " regardless of which demographic variable we are analyzing.\n", "\n", - " COALESCE handles NULL values (e.g. imd_band) — SQL NULLs become\n", + " COALESCE handles NULL values (e.g. imd_band): SQL NULLs become\n", " pandas NaN which matplotlib cannot use as categorical axis labels.\n", " Converting to 'Unknown' at the SQL level keeps downstream code clean.\n", " \"\"\"\n", @@ -244,7 +244,7 @@ " _n_rows = _check['n'].iloc[0]\n", " if _n_rows == 0:\n", " raise RuntimeError('v_student_enriched is empty')\n", - " print(f'Database OK — v_student_enriched has {_n_rows:,} rows')\n", + " print(f'Database OK: v_student_enriched has {_n_rows:,} rows')\n", "except Exception as exc:\n", " raise RuntimeError(\n", " 'Cannot query v_student_enriched. '\n", @@ -257,12 +257,12 @@ "id": "5", "metadata": {}, "source": [ - "## 2. Dataset Overview — Shape and Scale\n", + "## 2. Dataset Overview: Shape and Scale\n", "\n", "The first step in any EDA is understanding **how much data we have** and **what it looks like at a high level**.\n", "\n", "**Key concept: the unit of analysis.** \n", - "In OULAD, one student can enroll in multiple courses (modules). The unit of analysis is the **student-module enrollment** — identified by the composite key `(id_student, code_module, code_presentation)` — not the unique student.\n", + "In OULAD, one student can enroll in multiple courses (modules). The unit of analysis is the **student-module enrollment**, identified by the composite key `(id_student, code_module, code_presentation)`, not the unique student.\n", "\n", "This means the total number of rows in `v_student_enriched` is the number of **enrollments**, which is larger than the number of unique students. A student who completes Course A but withdraws from Course B appears as two rows: one *completed* and one *not completed*. Both are valid data points for retention analysis." ] @@ -348,9 +348,9 @@ "source": [ "> **What this table tells us:**\n", "> - Each row is a unique **course-presentation** (a specific module offered in a specific semester, e.g. AAA-2013J).\n", - "> - **Completion rate** varies across courses — some are notably higher or lower, which BQ4 will investigate.\n", + "> - **Completion rate** varies across courses: some are notably higher or lower, which BQ4 will investigate.\n", "> - **Course design** also varies: some courses have more assessments, others more VLE resources.\n", - "> - The `course_length_days` column shows that courses range from short to long — this affects withdrawal timing (BQ1).\n", + "> - The `course_length_days` column shows that courses range from short to long: this affects withdrawal timing (BQ1).\n", ">\n", "> We will return to this table in [Section 7 (Course Landscape)](#7.-Course-Landscape) for visual analysis." ] @@ -443,7 +443,7 @@ " ORDER BY completed DESC\n", "''')\n", "\n", - "# Donut chart — simple, clean, conveys the key ratio at a glance\n", + "# Donut chart: simple, clean, conveys the key ratio at a glance\n", "fig, ax = plt.subplots(figsize=(7, 7))\n", "labels = [LABEL_BINARY[c] for c in df_binary['completed']]\n", "colors = [PALETTE_BINARY[c] for c in df_binary['completed']]\n", @@ -470,7 +470,7 @@ "metadata": {}, "source": [ "> **Interpretation:**\n", - "> - **Withdrawn** is the largest non-completion category. These students actively left — they are the primary target for retention interventions.\n", + "> - **Withdrawn** is the largest non-completion category. These students actively left: they are the primary target for retention interventions.\n", "> - **Fail** represents students who stayed engaged but did not pass. They need a different type of support (academic, not motivational).\n", "> - From a platform perspective: a significant share of students who enroll do **not** complete. This is the central problem this project investigates.\n", ">\n", @@ -548,7 +548,7 @@ "> - Modules at the bottom of the chart have the highest combined Pass + Distinction rates; those at the top have the most withdrawals.\n", "> - The Withdrawn segment (purple) varies across courses, suggesting that some course designs may be more prone to triggering early departures.\n", ">\n", - "> **Implication:** Any retention analysis that ignores course identity risks [Simpson's paradox](https://en.wikipedia.org/wiki/Simpson%27s_paradox) — where an aggregate trend reverses within subgroups. We will keep course awareness throughout." + "> **Implication:** Any retention analysis that ignores course identity risks [Simpson's paradox](https://en.wikipedia.org/wiki/Simpson%27s_paradox), where an aggregate trend reverses within subgroups. We will keep course awareness throughout." ] }, { @@ -561,11 +561,11 @@ "Understanding **who the students are** is essential before analyzing **what they did**. This section profiles the student population across key demographic variables and examines whether completion rates differ by group.\n", "\n", "**Variables examined:**\n", - "- **Gender** — binary in OULAD (M/F)\n", - "- **Age band** — categorized as 0-35, 35-55, or 55<=\n", - "- **Highest education** — prior qualification level\n", - "- **IMD band** — Index of Multiple Deprivation (socio-economic indicator, UK-specific)\n", - "- **Numeric variables** — `num_of_prev_attempts` and `studied_credits`\n", + "- **Gender**: binary in OULAD (M/F)\n", + "- **Age band**: categorized as 0-35, 35-55, or 55<=\n", + "- **Highest education**: prior qualification level\n", + "- **IMD band**: Index of Multiple Deprivation (socio-economic indicator, UK-specific)\n", + "- **Numeric variables**: `num_of_prev_attempts` and `studied_credits`\n", "\n", "**Important caveat:** Observing that a demographic group has a lower completion rate does **not** mean that demographic factor *causes* lower completion. Correlation is not causation. BQ3 will formally compare the explanatory power of demographics vs. behavior.\n", "\n", @@ -736,9 +736,9 @@ "**Key observations from this section:**\n", "- Completion rates **vary by demographic group**, but the magnitude of the differences varies.\n", "- Education level and IMD band tend to show the largest spread in completion rates.\n", - "- Students with more previous attempts may have lower completion rates — but this could reflect inherent course difficulty rather than student ability.\n", + "- Students with more previous attempts may have lower completion rates, but this could reflect inherent course difficulty rather than student ability.\n", "\n", - "> **Causation warning:** Observing that Group A completes at a higher rate than Group B does NOT mean that *being in Group A causes* higher completion. There are confounding factors (e.g., course choice, motivation, prior knowledge) that we cannot isolate from demographics alone. BQ3 (Notebook 05) will formally compare the explanatory power of demographics versus behavior — the answer has direct implications for where a platform should invest its resources." + "> **Causation warning:** Observing that Group A completes at a higher rate than Group B does NOT mean that *being in Group A causes* higher completion. There are confounding factors (e.g., course choice, motivation, prior knowledge) that we cannot isolate from demographics alone. BQ3 (Notebook 05) will formally compare the explanatory power of demographics versus behavior: the answer has direct implications for where a platform should invest its resources." ] }, { @@ -750,7 +750,7 @@ "\n", "When do students register relative to the course start?\n", "\n", - "In OULAD, `date_registration` is measured in **days relative to course start** (day 0). Negative values mean the student registered *before* the course officially began — this is common in university systems where enrollment opens weeks or months in advance.\n", + "In OULAD, `date_registration` is measured in **days relative to course start** (day 0). Negative values mean the student registered *before* the course officially began: this is common in university systems where enrollment opens weeks or months in advance.\n", "\n", "**Why this matters:** Early registration may signal higher motivation or better planning. If early registrants complete at higher rates, this is a behavioral marker (though not necessarily a causal one) that could inform intervention timing." ] @@ -802,7 +802,7 @@ "outputs": [], "source": [ "# --- Registration day summary statistics by outcome ---\n", - "# PERCENTILE_CONT is the ANSI SQL standard for computing medians —\n", + "# PERCENTILE_CONT is the ANSI SQL standard for computing medians;\n", "# MEDIAN() is a DuckDB shortcut that would break on BigQuery migration\n", "df_reg_stats = execute_query('''\n", " SELECT\n", @@ -829,7 +829,7 @@ "source": [ "> **Interpretation:**\n", "> - Most students register **before** the course starts (negative registration day), which is expected.\n", - "> - If there is a difference between completers and non-completers in registration timing, it suggests that **enrollment timing** may be a weak early signal. However, this is observational — early registration could simply proxy for motivation or institutional support, not directly cause completion.\n", + "> - If there is a difference between completers and non-completers in registration timing, it suggests that **enrollment timing** may be a weak early signal. However, this is observational: early registration could simply proxy for motivation or institutional support, not directly cause completion.\n", "> - Students who register very late (positive registration day, i.e. after course start) may already be at a disadvantage due to missed content." ] }, @@ -956,13 +956,13 @@ "## 8. Data Quality Assessment\n", "\n", "**Why check data quality?** \n", - "Even well-curated public datasets can have surprises: missing values, unexpected categories, outliers, or coverage gaps. Professional EDA always includes a quality check — *trust but verify*.\n", + "Even well-curated public datasets can have surprises: missing values, unexpected categories, outliers, or coverage gaps. Professional EDA always includes a quality check: *trust but verify*.\n", "\n", "What we check:\n", - "1. **Missing values** — which columns have NULLs, and how many?\n", - "2. **Cardinality** — how many distinct values does each categorical column have?\n", - "3. **Range checks** — are numeric values within expected bounds?\n", - "4. **Coverage gaps** — do all views contain the expected number of students?" + "1. **Missing values**: which columns have NULLs, and how many?\n", + "2. **Cardinality**: how many distinct values does each categorical column have?\n", + "3. **Range checks**: are numeric values within expected bounds?\n", + "4. **Coverage gaps**: do all views contain the expected number of students?" ] }, { @@ -1090,7 +1090,7 @@ "print(f' \"Ghost students\" (zero activity): {n_ghosts:>8,} ({ghost_pct:.1f}%)')\n", "print(f'\\n → {n_ghosts:,} enrollments had zero VLE activity in the first 28 days.')\n", "print(' These enrollments are ABSENT from v_engagement_early by design.')\n", - "print(' This is not a data quality issue — it is an analytical finding (BQ5 segment).')" + "print(' This is not a data quality issue: it is an analytical finding (BQ5 segment).')" ] }, { @@ -1100,9 +1100,9 @@ "source": [ "> **Data quality verdict:**\n", "> - **Missing values**: `imd_band` has known NULLs (documented in OULAD). Other columns are generally complete.\n", - "> - **Cardinality**: all categorical columns have the expected number of distinct values — no unexpected categories.\n", + "> - **Cardinality**: all categorical columns have the expected number of distinct values: no unexpected categories.\n", "> - **Ranges**: numeric columns are within plausible bounds. Negative registration days are expected (pre-course enrollment). Dropout days are positive (relative to course start).\n", - "> - **Ghost students**: a measurable share of enrollments have zero VLE activity in the first 28 days. This is not missing data — these students simply never engaged with the platform. They represent a distinct segment for intervention (BQ5).\n", + "> - **Ghost students**: a measurable share of enrollments have zero VLE activity in the first 28 days. This is not missing data: these students simply never engaged with the platform. They represent a distinct segment for intervention (BQ5).\n", ">\n", "> **Conclusion:** The dataset is clean enough for analysis. The only caveat is `imd_band` nulls, which we handle by including the NULL category in analyses or excluding it where noted." ] @@ -1112,11 +1112,11 @@ "id": "43", "metadata": {}, "source": [ - "## 9. Engagement Baseline — Preview\n", + "## 9. Engagement Baseline: Preview\n", "\n", - "Before closing this notebook, we take a **brief look** at early engagement. The full behavioral analysis is in Notebook 02 — here we just establish a baseline.\n", + "Before closing this notebook, we take a **brief look** at early engagement. The full behavioral analysis is in Notebook 02; here we just establish a baseline.\n", "\n", - "We use `v_engagement_early`, which aggregates clickstream activity for the first 28 days of each course. **Important:** this view only includes students with at least one click — the \"ghost students\" identified in Section 8 are not in this view." + "We use `v_engagement_early`, which aggregates clickstream activity for the first 28 days of each course. **Important:** this view only includes students with at least one click: the \"ghost students\" identified in Section 8 are not in this view." ] }, { @@ -1173,11 +1173,11 @@ "\n", "1. **Scale**: The OULAD dataset contains ~32K enrollments across ~28K unique students in 22 course-presentations. The unit of analysis is the enrollment, not the student.\n", "\n", - "2. **Completion challenge**: A significant share of enrollments do not result in completion. Withdrawn students (who actively left) outnumber those who stayed but failed — suggesting that **retention**, not academic support, is the primary lever.\n", + "2. **Completion challenge**: A significant share of enrollments do not result in completion. Withdrawn students (who actively left) outnumber those who stayed but failed, suggesting that **retention**, not academic support, is the primary lever.\n", "\n", "3. **Course variation**: Completion rates vary substantially across modules. Any analysis that ignores course identity risks Simpson's paradox.\n", "\n", - "4. **Demographics matter — but how much?** Completion rates differ by education level, IMD band, and other demographics. BQ3 will test whether these differences are large enough to be actionable, or whether behavior is a stronger signal.\n", + "4. **Demographics matter, but how much?** Completion rates differ by education level, IMD band, and other demographics. BQ3 will test whether these differences are large enough to be actionable, or whether behavior is a stronger signal.\n", "\n", "5. **Early engagement signal**: Even in this brief preview, completed students show notably higher VLE activity in the first 28 days. BQ2 will quantify the predictive strength of these early signals.\n", "\n", @@ -1189,10 +1189,10 @@ "\n", "| Notebook | Business Question | Focus |\n", "|----------|------------------|-------|\n", - "| **02** | — | EDA: engagement patterns (daily clickstream, temporal trends) |\n", + "| **02** | - | EDA: engagement patterns (daily clickstream, temporal trends) |\n", "| **03** | BQ1 | Where and when do students drop out? |\n", "| **04** | BQ2 | Which early behavioral signals predict drop-out? |\n", - "| **05** | BQ3 | Demographics vs. behavior — what predicts outcome more? |\n", + "| **05** | BQ3 | Demographics vs. behavior: what predicts outcome more? |\n", "| **06** | BQ4 | How do course characteristics affect retention? |\n", "| **07** | BQ5 | Top 3 actionable interventions |\n", "\n", @@ -1206,7 +1206,7 @@ "id": "46", "metadata": {}, "source": [ - "> **Preview finding:** Completed students have visibly **higher early engagement** than non-completers — both in total clicks and in the shape of the distribution (longer tail towards high engagement).\n", + "> **Preview finding:** Completed students have visibly **higher early engagement** than non-completers, both in total clicks and in the shape of the distribution (longer tail towards high engagement).\n", ">\n", "> This is a **descriptive observation**, not a causal claim. It will be quantified with statistical tests (t-test, effect size) in Notebook 04 (BQ2: early behavioral signals).\n", ">\n", diff --git a/notebooks/02_eda_engagement_patterns.ipynb b/notebooks/02_eda_engagement_patterns.ipynb index 7eb0e81..fbfc805 100644 --- a/notebooks/02_eda_engagement_patterns.ipynb +++ b/notebooks/02_eda_engagement_patterns.ipynb @@ -5,7 +5,7 @@ "id": "0", "metadata": {}, "source": [ - "# 02 — EDA: Engagement Patterns\n", + "# 02. EDA: Engagement Patterns\n", "\n", "> **Notebook 02 of 7** | Learning Retention Analytics \n", "> Second exploratory analysis: how do students interact with the platform, and what do engagement patterns reveal?" @@ -18,15 +18,15 @@ "source": [ "## Purpose and Scope\n", "\n", - "This notebook is the **second of two EDA notebooks**. Notebook 01 profiled the *student base* (demographics, outcomes, course landscape). Here we shift to **behavioral engagement** — the clickstream patterns that reflect what students actually *do* on the platform.\n", + "This notebook is the **second of two EDA notebooks**. Notebook 01 profiled the *student base* (demographics, outcomes, course landscape). Here we shift to **behavioral engagement**: the clickstream patterns that reflect what students actually *do* on the platform.\n", "\n", "**What this notebook covers:**\n", - "- Ghost students — enrollments with zero VLE activity\n", + "- Ghost students: enrollments with zero VLE activity\n", "- Distribution of early engagement metrics (first 28 days)\n", "- Dose-response relationship between engagement decile and outcome\n", - "- Daily engagement trajectory — when does the gap open?\n", - "- Engagement typology — binge vs. steady patterns\n", - "- Activity type diversity — what resources do students use?\n", + "- Daily engagement trajectory: when does the gap open?\n", + "- Engagement typology: binge vs. steady patterns\n", + "- Activity type diversity: what resources do students use?\n", "- Course-level engagement profiles\n", "- Engagement-outcome correlation summary\n", "\n", @@ -36,7 +36,7 @@ "**Connection to business questions:** \n", "This EDA does not answer BQ1–BQ5 directly. Instead, it builds the **behavioral foundation** for BQ2 (early signals predicting dropout), BQ3 (demographics vs. behavior), and BQ5 (actionable interventions). Think of it as *mapping the behavioral landscape before testing specific hypotheses*.\n", "\n", - "> **Methodological transferability:** The engagement patterns explored here — ghost users, onboarding-window signals, binge vs. steady usage, activity diversity — are directly portable to SaaS retention, subscription churn, and fitness app engagement. The *domain* is education; the *analytical framework* is product analytics." + "> **Methodological transferability:** The engagement patterns explored here (ghost users, onboarding-window signals, binge vs. steady usage, activity diversity) are directly portable to SaaS retention, subscription churn, and fitness app engagement. The *domain* is education; the *analytical framework* is product analytics." ] }, { @@ -47,11 +47,11 @@ "## Table of Contents\n", "\n", "1. [Environment Setup](#1.-Environment-Setup)\n", - "2. [Ghost Students — Zero Engagement](#2.-Ghost-Students-—-Zero-Engagement)\n", + "2. [Ghost Students: Zero Engagement](#2.-Ghost-Students:-Zero-Engagement)\n", "3. [Early Engagement Distributions](#3.-Early-Engagement-Distributions)\n", "4. [Engagement Deciles vs. Outcome](#4.-Engagement-Deciles-vs.-Outcome)\n", "5. [Daily Engagement Trajectory](#5.-Daily-Engagement-Trajectory)\n", - "6. [Engagement Typology — Binge vs. Steady](#6.-Engagement-Typology-—-Binge-vs.-Steady)\n", + "6. [Engagement Typology: Binge vs. Steady](#6.-Engagement-Typology:-Binge-vs.-Steady)\n", "7. [Activity Type Diversity](#7.-Activity-Type-Diversity)\n", "8. [Course-Level Engagement Profiles](#8.-Course-Level-Engagement-Profiles)\n", "9. [Engagement-Outcome Correlation Matrix](#9.-Engagement-Outcome-Correlation-Matrix)\n", @@ -63,7 +63,7 @@ "- The ETL pipeline must have been run: `python -m run_pipeline`\n", "- The DuckDB database at `data/db/oulad.duckdb` must contain all 5 analytical views\n", "\n", - "**Dataset:** Open University Learning Analytics Dataset (OULAD) — ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." + "**Dataset:** Open University Learning Analytics Dataset (OULAD), ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." ] }, { @@ -77,7 +77,7 @@ "\n", "**Technical notes for readers:**\n", "- Notebooks live in `notebooks/` but project modules are in `src/` at the project root. We add the project root to `sys.path` so that `from src.config import ...` works. The linter rule `E402` (imports not at top of file) is suppressed for notebooks in `pyproject.toml`.\n", - "- All database queries go through `src.db.connection.execute_query()` — the project's DB abstraction layer. This returns a `pandas.DataFrame` and ensures we never call `duckdb.connect()` directly (see ADR-003).\n", + "- All database queries go through `src.db.connection.execute_query()`, the project's DB abstraction layer. This returns a `pandas.DataFrame` and ensures we never call `duckdb.connect()` directly (see ADR-003).\n", "- Figures are saved to `reports/figures/` at 150 DPI. Since `nbstripout` removes notebook outputs before commit, the saved PNGs are the persistent visual record." ] }, @@ -138,12 +138,12 @@ "\n", "# Semantic color palette: one color per outcome category\n", "PALETTE_OUTCOME = {\n", - " 'Pass': '#4C72B0', # blue — neutral positive\n", - " 'Distinction': '#55A868', # green — strong positive\n", - " 'Fail': '#C44E52', # red — negative\n", - " 'Withdrawn': '#8172B3', # purple — departed (distinct from failed)\n", + " 'Pass': '#4C72B0', # blue: neutral positive\n", + " 'Distinction': '#55A868', # green: strong positive\n", + " 'Fail': '#C44E52', # red: negative\n", + " 'Withdrawn': '#8172B3', # purple: departed (distinct from failed)\n", "}\n", - "# Outcome label constants — single source of truth for the binary labels\n", + "# Outcome label constants: single source of truth for the binary labels\n", "# used in SQL output mapping, palette keys, and plot iterations\n", "LABEL_COMPLETED = 'Completed'\n", "LABEL_NOT_COMPLETED = 'Not completed'\n", @@ -154,7 +154,7 @@ "PALETTE_BINARY_LABELS = {LABEL_COMPLETED: '#55A868', LABEL_NOT_COMPLETED: '#C44E52'}\n", "# Sequential palette for heatmaps and continuous scales\n", "PALETTE_SEQUENTIAL = 'YlOrRd'\n", - "# Shared axis labels — avoids cross-cell string literal duplication\n", + "# Shared axis labels: avoids cross-cell string literal duplication\n", "LABEL_NUM_ENROLLMENTS = 'Number of enrollments'\n", "LABEL_ACTIVE_DAYS = 'Active days (first 28)'\n", "LABEL_COMPLETION_RATE = 'Completion rate (%)'\n", @@ -202,13 +202,13 @@ "id": "5", "metadata": {}, "source": [ - "## 2. Ghost Students — Zero Engagement\n", + "## 2. Ghost Students: Zero Engagement\n", "\n", - "The most extreme behavioral signal is **no behavior at all**. A \"ghost student\" is an enrollment present in `v_student_enriched` but absent from `v_engagement_early` — meaning zero VLE clicks in the first 28 days.\n", + "The most extreme behavioral signal is **no behavior at all**. A \"ghost student\" is an enrollment present in `v_student_enriched` but absent from `v_engagement_early`, meaning zero VLE clicks in the first 28 days.\n", "\n", "**Why start here?** Ghost students establish the *floor* of engagement. Before analyzing how much active students click, we need to know how many students never clicked at all, and what happened to them.\n", "\n", - "**SaaS parallel:** Ghost students are the education equivalent of *dormant users* — people who signed up but never activated. In product analytics, activation rate is the first retention lever: you cannot retain users who never started using the product." + "**SaaS parallel:** Ghost students are the education equivalent of *dormant users*: people who signed up but never activated. In product analytics, activation rate is the first retention lever: you cannot retain users who never started using the product." ] }, { @@ -220,7 +220,7 @@ "source": [ "# --- Ghost student identification ---\n", "# v_engagement_early only includes enrollments with >= 1 VLE click in days 0-28.\n", - "# Enrollments absent from that view had zero activity — these are \"ghost students\".\n", + "# Enrollments absent from that view had zero activity: these are \"ghost students\".\n", "# The LEFT JOIN + IS NULL pattern identifies them precisely per enrollment\n", "# (a student can be a ghost in one course but active in another).\n", "df_ghost_summary = execute_query('''\n", @@ -322,7 +322,7 @@ "ax1.invert_yaxis()\n", "sns.despine(ax=ax1)\n", "\n", - "# Right panel: completion rate — ghost vs active per module\n", + "# Right panel: completion rate, ghost vs active per module\n", "x = np.arange(len(modules))\n", "bar_width = 0.35\n", "ax2.barh(x - bar_width / 2, df_ghost_by_course['active_completion_pct'],\n", @@ -386,7 +386,7 @@ " )\n", "\n", "ax.set_xlabel(LABEL_NUM_ENROLLMENTS)\n", - "ax.set_title('Outcome Distribution — Ghost Students Only (zero VLE activity)')\n", + "ax.set_title('Outcome Distribution: Ghost Students Only (zero VLE activity)')\n", "ax.invert_yaxis()\n", "sns.despine(left=True)\n", "fig.tight_layout()\n", @@ -399,13 +399,13 @@ "id": "9", "metadata": {}, "source": [ - "> **Key finding:** Ghost students have a near-zero completion rate. The vast majority are **Withdrawn** — they enrolled, never interacted with the VLE, and eventually departed.\n", + "> **Key finding:** Ghost students have a near-zero completion rate. The vast majority are **Withdrawn**: they enrolled, never interacted with the VLE, and eventually departed.\n", ">\n", "> - Ghost students are present across all modules, not concentrated in a single course.\n", - "> - The completion rate gap between active and ghost students is enormous — this is the clearest engagement-outcome signal in the dataset.\n", + "> - The completion rate gap between active and ghost students is enormous: this is the clearest engagement-outcome signal in the dataset.\n", "> - A small number of ghost students may still Pass or Fail (e.g., through in-person exams or banked credits), but they are a tiny minority.\n", ">\n", - "> **Implication for BQ5:** Ghost students are the lowest-hanging fruit for intervention. They need *activation*, not academic support. In SaaS terms, this is the \"onboarding gap\" — the user signed up but never experienced the core product value.\n", + "> **Implication for BQ5:** Ghost students are the lowest-hanging fruit for intervention. They need *activation*, not academic support. In SaaS terms, this is the \"onboarding gap\": the user signed up but never experienced the core product value.\n", ">\n", "> **From here on**, Sections 3–9 analyze only **active students** (those with at least one VLE click in the first 28 days), unless explicitly noted otherwise." ] @@ -477,7 +477,7 @@ "outputs": [], "source": [ "# --- 2x2 violin plot grid: all 4 early metrics split by outcome ---\n", - "# Violins show the full distribution shape — more informative than box plots\n", + "# Violins show the full distribution shape: more informative than box plots\n", "# for detecting bimodality or heavy tails in engagement data.\n", "df_early = execute_query('''\n", " SELECT\n", @@ -571,11 +571,11 @@ "source": [ "> **Interpretation:**\n", "> - All four metrics show **visible separation** between completers and non-completers. Completers tend to have more active days, more total clicks, and their last active day is later in the 28-day window.\n", - "> - The **total clicks** distribution is heavily right-skewed — a few power users generate disproportionate click volumes. The mean is substantially higher than the median, confirming this skewness.\n", + "> - The **total clicks** distribution is heavily right-skewed: a few power users generate disproportionate click volumes. The mean is substantially higher than the median, confirming this skewness.\n", "> - The **last active day** metric is particularly telling: non-completers' activity tends to trail off earlier in the window, while completers remain active closer to day 28.\n", - "> - The **active days** histogram shows that many non-completers are active for only 1–5 days out of 28 — brief engagement before disappearing.\n", + "> - The **active days** histogram shows that many non-completers are active for only 1–5 days out of 28: brief engagement before disappearing.\n", ">\n", - "> **Causation caveat:** Higher engagement is *associated with* completion, but we cannot say engagement *causes* completion from this data alone. Motivated students may both engage more and complete more — engagement could be a proxy for motivation, not a cause of success. Notebook 04 (BQ2) will quantify these associations with formal statistical tests and effect sizes." + "> **Causation caveat:** Higher engagement is *associated with* completion, but we cannot say engagement *causes* completion from this data alone. Motivated students may both engage more and complete more: engagement could be a proxy for motivation, not a cause of success. Notebook 04 (BQ2) will quantify these associations with formal statistical tests and effect sizes." ] }, { @@ -587,7 +587,7 @@ "\n", "Is there a **dose-response** relationship between engagement and completion? If more engagement monotonically predicts better outcomes, this strengthens the case for engagement-based interventions.\n", "\n", - "`v_engagement_early` includes an `engagement_decile_in_course` column: each student is ranked 1–10 within their own course-presentation using `NTILE(10)` over `total_clicks_first_28`. This normalization is critical — it ensures that decile 1 in a high-engagement course and decile 1 in a low-engagement course both mean \"bottom 10% of *that* course.\"" + "`v_engagement_early` includes an `engagement_decile_in_course` column: each student is ranked 1–10 within their own course-presentation using `NTILE(10)` over `total_clicks_first_28`. This normalization is critical: it ensures that decile 1 in a high-engagement course and decile 1 in a low-engagement course both mean \"bottom 10% of *that* course.\"" ] }, { @@ -618,7 +618,7 @@ "# --- Bar chart with gradient coloring ---\n", "# Color gradient from red (low decile) to green (high decile) reinforces\n", "# the dose-response visual pattern and matches the binary palette semantics.\n", - "# Weighted by decile size — unweighted mean would misstate the rate if NTILE\n", + "# Weighted by decile size: unweighted mean would misstate the rate if NTILE\n", "# produces groups of slightly different size.\n", "overall_rate = (df_decile['completion_rate_pct'] * df_decile['n']).sum() / df_decile['n'].sum()\n", "n_deciles = len(df_decile)\n", @@ -712,7 +712,7 @@ "source": [ "> **Interpretation:**\n", "> - The dose-response pattern is clear: completion rate increases monotonically (or near-monotonically) from the lowest to the highest engagement decile.\n", - "> - The bottom 2–3 deciles are an **extreme-risk segment** — their completion rates are far below average.\n", + "> - The bottom 2–3 deciles are an **extreme-risk segment**: their completion rates are far below average.\n", "> - In the stacked bar chart, the Withdrawn segment (purple) shrinks dramatically as engagement increases, while Distinction (green) grows. This suggests that higher engagement is associated with both lower dropout and higher academic achievement.\n", "> - Because deciles are **within-course normalized**, this pattern is robust to course-level differences in click volume. A student in decile 1 of any course is at high risk.\n", ">\n", @@ -743,7 +743,7 @@ "# --- Average daily clicks by outcome group (days 0-28) ---\n", "# We compute the mean clicks per day for each outcome group.\n", "# Students who did not click on a given day are NOT in v_engagement_daily\n", - "# for that day — so this average is \"clicks per active student on that day\",\n", + "# for that day, so this average is \"clicks per active student on that day\",\n", "# not \"clicks per enrolled student\". The active student count plot below\n", "# provides the complementary view.\n", "df_trajectory = execute_query('''\n", @@ -804,7 +804,7 @@ "# This contextualizes the trajectory above: does the gap widen because\n", "# non-completers click less, or because they stop clicking entirely?\n", "# A declining line for non-completers means students are dropping out\n", - "# of activity — not just reducing their intensity.\n", + "# of activity, not just reducing their intensity.\n", "fig, ax = plt.subplots(figsize=FIG_SIZE)\n", "for outcome in [LABEL_COMPLETED, LABEL_NOT_COMPLETED]:\n", " subset = df_trajectory[df_trajectory['outcome'] == outcome]\n", @@ -842,7 +842,7 @@ "id": "23", "metadata": {}, "source": [ - "## 6. Engagement Typology — Binge vs. Steady\n", + "## 6. Engagement Typology: Binge vs. Steady\n", "\n", "Not all engagement is created equal. Two students with 200 total clicks can have very different patterns:\n", "- **Binge**: 200 clicks in 2 days (concentrated bursts, cramming behavior)\n", @@ -850,7 +850,7 @@ "\n", "The `active_days_first_28` and `total_clicks_first_28` dimensions capture this distinction. A scatter plot of these two variables, colored by outcome, reveals whether *consistency* or *volume* matters more for completion.\n", "\n", - "**SaaS parallel:** This maps to the difference between DAU (daily active users) and session depth. A user who logs in briefly every day is behaviorally different from one who has one marathon session per week — even if their total usage time is similar." + "**SaaS parallel:** This maps to the difference between DAU (daily active users) and session depth. A user who logs in briefly every day is behaviorally different from one who has one marathon session per week, even if their total usage time is similar." ] }, { @@ -987,9 +987,9 @@ "source": [ "> **Interpretation:**\n", "> - The scatter plot shows that **completers cluster in the upper-right** (many active days, high total clicks), while non-completers dominate the lower-left (few days, few clicks).\n", - "> - The \"binge\" quadrant (few days, many clicks) is relatively sparse — most students with high click volumes also spread them across multiple days.\n", + "> - The \"binge\" quadrant (few days, many clicks) is relatively sparse: most students with high click volumes also spread them across multiple days.\n", "> - The 2×2 heatmap quantifies this: **high frequency is the stronger predictor**. Moving from low to high frequency (more active days) increases completion rate more than moving from low to high intensity (more clicks per session). Consistency beats cramming.\n", - "> - The \"minimal\" quadrant (low frequency, low intensity) has the lowest completion rate — these are students who barely engaged even when they did show up.\n", + "> - The \"minimal\" quadrant (low frequency, low intensity) has the lowest completion rate: these are students who barely engaged even when they did show up.\n", ">\n", "> **Implication for BQ5:** Interventions should prioritize *consistency* over *volume*. Encouraging students to log in regularly (even briefly) may be more effective than encouraging longer sessions. This is analogous to the \"daily streak\" mechanic used in consumer apps to build habit formation." ] @@ -1005,7 +1005,7 @@ "\n", "The OULAD VLE contains several activity types: content pages (`oucontent`), forums (`forumng`), quizzes, resources, homepage, subpages, and more. The `vle` table maps each resource (`id_site`) to its `activity_type`.\n", "\n", - "**Diversity hypothesis:** Students who interact with a wider variety of resource types may be more deeply engaged with the course — accessing not just core content but also forums, quizzes, and supplementary materials. If diversity correlates with completion, it suggests that *breadth* of engagement matters alongside *volume*." + "**Diversity hypothesis:** Students who interact with a wider variety of resource types may be more deeply engaged with the course, accessing not just core content but also forums, quizzes, and supplementary materials. If diversity correlates with completion, it suggests that *breadth* of engagement matters alongside *volume*." ] }, { @@ -1046,7 +1046,7 @@ "source": [ "# --- Activity type usage by outcome ---\n", "# Which activity types show the largest gap between completers and non-completers?\n", - "# CTE computes per-enrollment click totals first, then averages — this avoids\n", + "# CTE computes per-enrollment click totals first, then averages; this avoids\n", "# COUNT(DISTINCT concat(...)), a pattern that is not portable across SQL engines.\n", "df_activity_outcome = execute_query('''\n", " WITH enrollment_activity_clicks AS (\n", @@ -1159,11 +1159,11 @@ "metadata": {}, "source": [ "> **Interpretation:**\n", - "> - A few activity types dominate the click volume — `oucontent` (course content pages) and `homepage`/`subpage` likely account for the majority of clicks.\n", + "> - A few activity types dominate the click volume: `oucontent` (course content pages) and `homepage`/`subpage` likely account for the majority of clicks.\n", "> - The **gap between completers and non-completers** varies by activity type. Some resource types (e.g., forums, quizzes) may show a proportionally larger gap, suggesting they are more \"discriminating\" engagement signals.\n", "> - Activity type **diversity** shows a positive association with completion: students who use more resource types tend to complete at higher rates. This is another dose-response pattern, complementing the decile analysis in Section 4.\n", ">\n", - "> **Caveat:** Activity type diversity is partly confounded with total engagement volume — students who click more are more likely to encounter different resource types. The diversity signal may not be independent of the volume signal. BQ2 will test this more formally.\n", + "> **Caveat:** Activity type diversity is partly confounded with total engagement volume: students who click more are more likely to encounter different resource types. The diversity signal may not be independent of the volume signal. BQ2 will test this more formally.\n", ">\n", "> **SaaS parallel:** Feature breadth (number of distinct features used) is a common retention predictor in product analytics. Users who explore beyond the core feature set tend to be more engaged and less likely to churn." ] @@ -1175,7 +1175,7 @@ "source": [ "## 8. Course-Level Engagement Profiles\n", "\n", - "Notebook 01 (Section 7) examined course *design* characteristics — assessment density, VLE resource count, course length. Here we add the **engagement dimension**: how much do students actually click in each course?\n", + "Notebook 01 (Section 7) examined course *design* characteristics: assessment density, VLE resource count, course length. Here we add the **engagement dimension**: how much do students actually click in each course?\n", "\n", "This matters because courses with more resources may naturally generate more clicks. An engagement level that is \"low\" in a resource-heavy course might be \"normal\" in a lighter one. BQ4 (Notebook 06) will formalize this analysis; here we establish the baseline." ] @@ -1262,7 +1262,7 @@ "source": [ "> **Interpretation:**\n", "> - Engagement levels vary **substantially** across modules. Some courses generate much higher click volumes than others, likely reflecting differences in course design (number of VLE resources, assessment structure, content format).\n", - "> - Within each course, there is also considerable variation — the box plot whiskers and outliers show that some students click many times more than their peers in the same course.\n", + "> - Within each course, there is also considerable variation: the box plot whiskers and outliers show that some students click many times more than their peers in the same course.\n", "> - The mean-median gap in the summary table confirms right-skewed distributions within most courses: a few highly active students pull the mean above the median.\n", ">\n", "> **Implication:** Engagement metrics should be interpreted **relative to the course**, not in absolute terms. This is why the engagement decile (Section 4) normalizes within each course-presentation. A student with 100 clicks may be in the top decile of one course and the bottom decile of another. BQ4 (Notebook 06) will analyze how course design features relate to engagement levels and retention." @@ -1277,7 +1277,7 @@ "\n", "We now synthesize all early engagement metrics into a single view: which metrics have the **strongest linear association** with completion?\n", "\n", - "We compute Pearson correlations between all engagement variables and the binary `completed` variable. When one variable is binary (0/1) and the other is continuous, the Pearson correlation equals the point-biserial correlation — a standard measure of association between a continuous and a dichotomous variable.\n", + "We compute Pearson correlations between all engagement variables and the binary `completed` variable. When one variable is binary (0/1) and the other is continuous, the Pearson correlation equals the point-biserial correlation, a standard measure of association between a continuous and a dichotomous variable.\n", "\n", "**Important:** This analysis uses a `LEFT JOIN` to include ghost students (with engagement values set to 0). This gives the full-population perspective: ghost students are the extreme low end of engagement, and including them strengthens correlations because they almost all fail to complete." ] @@ -1362,14 +1362,14 @@ "metadata": {}, "source": [ "> **Interpretation:**\n", - "> - The engagement metrics are **inter-correlated** — active days, total clicks, and engagement decile tend to move together. This is expected: students who click more also tend to click on more days.\n", - "> - The strongest correlates with completion are likely **active days** and **total clicks** — the simplest engagement measures are also the most predictive.\n", + "> - The engagement metrics are **inter-correlated**: active days, total clicks, and engagement decile tend to move together. This is expected: students who click more also tend to click on more days.\n", + "> - The strongest correlates with completion are likely **active days** and **total clicks**: the simplest engagement measures are also the most predictive.\n", "> - **Clicks per active day** (intensity) may have a weaker correlation with completion than active days (frequency), reinforcing the Section 6 finding that consistency matters more than intensity.\n", "> - Including ghost students (as zeros) strengthens all correlations because they represent the extreme case: zero engagement, near-zero completion.\n", ">\n", "> **Caveat:** Pearson correlations capture **linear** relationships. Non-linear patterns (e.g., diminishing returns at high engagement levels, or a threshold effect) would be underestimated. The decile analysis in Section 4 partially addresses this by looking at the shape of the dose-response curve.\n", ">\n", - "> **Looking ahead:** BQ2 (Notebook 04) will formalize these associations with t-tests, effect sizes (Cohen's d), and confidence intervals — moving from correlation to actionable signal assessment." + "> **Looking ahead:** BQ2 (Notebook 04) will formalize these associations with t-tests, effect sizes (Cohen's d), and confidence intervals, moving from correlation to actionable signal assessment." ] }, { @@ -1381,13 +1381,13 @@ "\n", "### What we learned\n", "\n", - "1. **Ghost students are the clearest signal.** A measurable share of enrollments have zero VLE activity in the first 28 days. Their completion rate is near zero. They are the lowest-hanging fruit for intervention — they need *activation*, not academic support.\n", + "1. **Ghost students are the clearest signal.** A measurable share of enrollments have zero VLE activity in the first 28 days. Their completion rate is near zero. They are the lowest-hanging fruit for intervention: they need *activation*, not academic support.\n", "\n", "2. **Engagement separates outcomes across all metrics.** Completers show substantially higher active days, total clicks, and later last-active-day in the 28-day window. The separation is visible in every distribution plot.\n", "\n", "3. **Dose-response relationship.** Completion rate increases monotonically with engagement decile. The bottom 2–3 deciles are extreme-risk segments with far-below-average completion rates.\n", "\n", - "4. **Early divergence.** The daily trajectory shows that the engagement gap between completers and non-completers opens within the first 7–10 days. Non-completers both click less *and* stop showing up earlier — a dual signal of intensity and frequency.\n", + "4. **Early divergence.** The daily trajectory shows that the engagement gap between completers and non-completers opens within the first 7–10 days. Non-completers both click less *and* stop showing up earlier: a dual signal of intensity and frequency.\n", "\n", "5. **Consistency over intensity.** The typology analysis reveals that *frequency* (number of active days) predicts completion more strongly than *intensity* (clicks per active day). Steady engagement outperforms binge engagement.\n", "\n", @@ -1403,7 +1403,7 @@ "|----------|------------------|-------|\n", "| **03** | BQ1 | Where and when do students drop out? |\n", "| **04** | BQ2 | Which early behavioral signals predict drop-out? |\n", - "| **05** | BQ3 | Demographics vs. behavior — what predicts outcome more? |\n", + "| **05** | BQ3 | Demographics vs. behavior: what predicts outcome more? |\n", "| **06** | BQ4 | How do course characteristics affect retention? |\n", "| **07** | BQ5 | Top 3 actionable interventions |\n", "\n", diff --git a/notebooks/03_bq1_dropout_timing.ipynb b/notebooks/03_bq1_dropout_timing.ipynb index 434d480..372629f 100644 --- a/notebooks/03_bq1_dropout_timing.ipynb +++ b/notebooks/03_bq1_dropout_timing.ipynb @@ -5,7 +5,7 @@ "id": "0", "metadata": {}, "source": [ - "# 03 — BQ1: Where and When Do Students Drop Out?\n", + "# 03. BQ1: Where and When Do Students Drop Out?\n", "\n", "> **Notebook 03 of 7** | Learning Retention Analytics \n", "> First business question analysis: temporal patterns of student withdrawal across courses." @@ -18,14 +18,14 @@ "source": [ "## Purpose and Scope\n", "\n", - "This notebook answers **BQ1: Where and when do students drop out?** — the first of five business questions driving this project.\n", + "This notebook answers **BQ1: Where and when do students drop out?**, the first of five business questions driving this project.\n", "\n", "**What this notebook covers:**\n", "- Scale of the dropout problem: how many students withdraw, and from which courses\n", - "- Cumulative dropout curves — survival-style analysis showing when students leave\n", - "- Cliff detection — identifying critical moments of mass withdrawal\n", - "- Normalized timeline — comparing dropout timing across courses of different lengths\n", - "- Pre-course withdrawals — students who leave before the course even starts\n", + "- Cumulative dropout curves: survival-style analysis showing when students leave\n", + "- Cliff detection: identifying critical moments of mass withdrawal\n", + "- Normalized timeline: comparing dropout timing across courses of different lengths\n", + "- Pre-course withdrawals: students who leave before the course even starts\n", "- Demographic segmentation of dropout timing\n", "- Course design characteristics and their relationship to dropout patterns\n", "\n", @@ -46,10 +46,10 @@ "## Table of Contents\n", "\n", "1. [Environment Setup](#1.-Environment-Setup)\n", - "2. [Dropout Overview — Scale and Rate](#2.-Dropout-Overview-—-Scale-and-Rate)\n", + "2. [Dropout Overview: Scale and Rate](#2.-Dropout-Overview:-Scale-and-Rate)\n", "3. [Cumulative Dropout Curves](#3.-Cumulative-Dropout-Curves)\n", "4. [Identifying Dropout Cliffs](#4.-Identifying-Dropout-Cliffs)\n", - "5. [Normalized Timeline — Percentage Through Course](#5.-Normalized-Timeline-—-Percentage-Through-Course)\n", + "5. [Normalized Timeline: Percentage Through Course](#5.-Normalized-Timeline:-Percentage-Through-Course)\n", "6. [Pre-Course Withdrawals](#6.-Pre-Course-Withdrawals)\n", "7. [Dropout Timing by Demographics](#7.-Dropout-Timing-by-Demographics)\n", "8. [Course Design and Dropout Patterns](#8.-Course-Design-and-Dropout-Patterns)\n", @@ -61,7 +61,7 @@ "- The ETL pipeline must have been run: `python -m run_pipeline`\n", "- The DuckDB database at `data/db/oulad.duckdb` must contain all 5 analytical views\n", "\n", - "**Dataset:** Open University Learning Analytics Dataset (OULAD) — ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." + "**Dataset:** Open University Learning Analytics Dataset (OULAD), ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." ] }, { @@ -75,7 +75,7 @@ "\n", "**Technical notes for readers:**\n", "- Notebooks live in `notebooks/` but project modules are in `src/` at the project root. We add the project root to `sys.path` so that `from src.config import ...` works.\n", - "- All database queries go through `src.db.connection.execute_query()` — the project's DB abstraction layer (ADR-003).\n", + "- All database queries go through `src.db.connection.execute_query()`, the project's DB abstraction layer (ADR-003).\n", "- BQ1's primary SQL query lives in `sql/queries/q_bq1_dropout_curves.sql` and is loaded at runtime from disk. Additional analysis queries in this notebook may be defined inline, but they still execute through the same DB abstraction layer (ADR-003).\n", "- Figures are saved to `reports/figures/` at 150 DPI. Since `nbstripout` removes notebook outputs before commit, the saved PNGs are the persistent visual record." ] @@ -133,12 +133,12 @@ "\n", "# Semantic color palette: one color per outcome category\n", "PALETTE_OUTCOME = {\n", - " 'Pass': '#4C72B0', # blue — neutral positive\n", - " 'Distinction': '#55A868', # green — strong positive\n", - " 'Fail': '#C44E52', # red — negative\n", - " 'Withdrawn': '#8172B3', # purple — departed (distinct from failed)\n", + " 'Pass': '#4C72B0', # blue: neutral positive\n", + " 'Distinction': '#55A868', # green: strong positive\n", + " 'Fail': '#C44E52', # red: negative\n", + " 'Withdrawn': '#8172B3', # purple: departed (distinct from failed)\n", "}\n", - "# Outcome label constants — single source of truth for binary labels\n", + "# Outcome label constants: single source of truth for binary labels\n", "LABEL_COMPLETED = 'Completed'\n", "LABEL_NOT_COMPLETED = 'Not completed'\n", "PALETTE_BINARY = {1: '#55A868', 0: '#C44E52'}\n", @@ -153,7 +153,7 @@ "_TAB10 = plt.cm.tab10.colors\n", "PALETTE_COURSE = {m: _TAB10[i] for i, m in enumerate(_MODULE_ORDER)}\n", "\n", - "# Shared axis labels — avoids cross-cell string literal duplication\n", + "# Shared axis labels: avoids cross-cell string literal duplication\n", "LABEL_DROPOUT_DAY = 'Dropout day (relative to course start)'\n", "LABEL_CUMULATIVE_DROPOUT = 'Cumulative dropout rate (%)'\n", "LABEL_COMPLETION_RATE = 'Completion rate (%)'\n", @@ -205,11 +205,11 @@ "id": "5", "metadata": {}, "source": [ - "## 2. Dropout Overview — Scale and Rate\n", + "## 2. Dropout Overview: Scale and Rate\n", "\n", "Before examining *when* students drop out, we establish **how many** drop out and from which courses. This section provides the baseline numbers that contextualize all subsequent analyses.\n", "\n", - "**Key distinction:** In the OULAD dataset, \"withdrawn\" means the student explicitly unregistered from the course (`date_unregistration IS NOT NULL`). Students who stayed enrolled but failed are classified as \"Fail\", not \"Withdrawn\". This notebook focuses on **explicit withdrawals** — the students who actively chose to leave." + "**Key distinction:** In the OULAD dataset, \"withdrawn\" means the student explicitly unregistered from the course (`date_unregistration IS NOT NULL`). Students who stayed enrolled but failed are classified as \"Fail\", not \"Withdrawn\". This notebook focuses on **explicit withdrawals**: the students who actively chose to leave." ] }, { @@ -292,7 +292,7 @@ "id": "7", "metadata": {}, "source": [ - "> **Key finding:** Roughly one-third of all enrollments end in explicit withdrawal. The withdrawal rate varies across modules — some courses lose a significantly larger share of students than others.\n", + "> **Key finding:** Roughly one-third of all enrollments end in explicit withdrawal. The withdrawal rate varies across modules: some courses lose a significantly larger share of students than others.\n", ">\n", "> This variation is important: BQ4 (Notebook 06) will investigate whether course design features (assessment density, resource count, duration) explain these differences. For now, we note that the problem is substantial and module-dependent." ] @@ -304,11 +304,11 @@ "source": [ "## 3. Cumulative Dropout Curves\n", "\n", - "This is the **core BQ1 visualization** — a survival-style analysis showing how dropout accumulates over time.\n", + "This is the **core BQ1 visualization**: a survival-style analysis showing how dropout accumulates over time.\n", "\n", "The query `q_bq1_dropout_curves.sql` computes cumulative dropout counts and percentages per course-presentation using window functions (`SUM OVER ... ROWS BETWEEN UNBOUNDED PRECEDING`). Each point on the curve represents the percentage of the original cohort that has withdrawn by that day.\n", "\n", - "**Why step charts?** Dropout is a discrete event — students withdraw on specific days, not continuously. A step chart (rather than a smoothed line) preserves this reality and makes cliff events (sudden spikes) visible." + "**Why step charts?** Dropout is a discrete event: students withdraw on specific days, not continuously. A step chart (rather than a smoothed line) preserves this reality and makes cliff events (sudden spikes) visible." ] }, { @@ -346,7 +346,7 @@ "\n", "ax.set_xlabel(LABEL_DROPOUT_DAY)\n", "ax.set_ylabel(LABEL_CUMULATIVE_DROPOUT)\n", - "ax.set_title('Cumulative Dropout Curves — All Course-Presentations')\n", + "ax.set_title('Cumulative Dropout Curves: All Course-Presentations')\n", "ax.legend(title='Module', bbox_to_anchor=(1.02, 1), loc='upper left')\n", "ax.set_ylim(0, None)\n", "sns.despine()\n", @@ -422,7 +422,7 @@ "source": [ "## 4. Identifying Dropout Cliffs\n", "\n", - "Not all days are equal. Some days see **disproportionately large numbers of withdrawals** — \"dropout cliffs\" that may correspond to specific course events (assessment deadlines, feedback release, registration cutoffs).\n", + "Not all days are equal. Some days see **disproportionately large numbers of withdrawals**: \"dropout cliffs\" that may correspond to specific course events (assessment deadlines, feedback release, registration cutoffs).\n", "\n", "We identify cliffs by looking at the daily dropout count from the BQ1 query results and flagging days where the count exceeds the 95th percentile for that course. These are days when something triggered an unusual number of departures." ] @@ -437,7 +437,7 @@ "# --- Cliff detection: days with unusually high dropout counts ---\n", "# For each course-presentation, flag days where daily dropout count\n", "# exceeds the 95th percentile. These outlier days represent sudden\n", - "# mass departures — likely tied to course milestones.\n", + "# mass departures, likely tied to course milestones.\n", "df_daily_dropouts = df_curves[[\n", " 'code_module', 'code_presentation', 'dropout_day', 'n_dropouts'\n", "]].copy()\n", @@ -481,7 +481,7 @@ "\n", "# Label each bar with module + day for identification\n", "labels = [\n", - " f\"{row['code_module']} — day {int(row['dropout_day'])}\"\n", + " f\"{row['code_module']} - day {int(row['dropout_day'])}\"\n", " for _, row in df_cliff_top.iterrows()\n", "]\n", "ax.set_yticks(range(len(df_cliff_top)))\n", @@ -505,7 +505,7 @@ "> - Some modules show cliffs earlier in the timeline (suggesting onboarding-phase attrition), while others show cliffs mid-course (suggesting assessment-driven attrition).\n", "> - The consistency of cliff timing across presentations of the same module strengthens the hypothesis that these events are tied to **course design** rather than student-specific factors.\n", ">\n", - "> **Actionable insight:** If cliff days align with known course events, the platform operator can deploy targeted interventions *before* those dates — e.g., reminder emails, support resources, or simplified re-enrollment pathways for students who missed assessment deadlines." + "> **Actionable insight:** If cliff days align with known course events, the platform operator can deploy targeted interventions *before* those dates, e.g., reminder emails, support resources, or simplified re-enrollment pathways for students who missed assessment deadlines." ] }, { @@ -513,7 +513,7 @@ "id": "15", "metadata": {}, "source": [ - "## 5. Normalized Timeline — Percentage Through Course\n", + "## 5. Normalized Timeline: Percentage Through Course\n", "\n", "OULAD courses have different lengths (ranging from ~240 to ~270 days). Comparing dropout days in absolute terms mixes apples and oranges. To enable cross-course comparison, `v_dropout_timing` includes a `dropout_pct` field: the percentage of course duration elapsed when the student withdrew.\n", "\n", @@ -607,7 +607,7 @@ "\n", "A surprising subset of students withdraw **before the course even starts** (`dropout_day < 0`). These are enrollments where the student registered and then unregistered before day 0.\n", "\n", - "Pre-course withdrawals represent a distinct phenomenon: the student never experienced any course content. This is pure **registration churn** — analogous to SaaS users who sign up during a trial and cancel before ever using the product." + "Pre-course withdrawals represent a distinct phenomenon: the student never experienced any course content. This is pure **registration churn**, analogous to SaaS users who sign up during a trial and cancel before ever using the product." ] }, { @@ -678,7 +678,7 @@ "> - Pre-course withdrawal is a **registration quality** signal: the enrollment funnel converts users who are not yet committed.\n", "> - The distribution across modules suggests that some courses may have better registration-to-start conversion than others.\n", ">\n", - "> **Intervention opportunity:** Pre-course withdrawals could be reduced with better expectation-setting at registration time, welcome emails with concrete \"first steps\", or a simplified late-registration pathway. In SaaS terms, this is the \"activation email\" problem — the user signed up but needs a nudge to actually start." + "> **Intervention opportunity:** Pre-course withdrawals could be reduced with better expectation-setting at registration time, welcome emails with concrete \"first steps\", or a simplified late-registration pathway. In SaaS terms, this is the \"activation email\" problem: the user signed up but needs a nudge to actually start." ] }, { @@ -690,7 +690,7 @@ "\n", "Does *when* students drop out vary by demographic group? `v_dropout_timing` includes demographic columns from `v_student_enriched`, allowing us to segment dropout timing by education level, age band, and deprivation index.\n", "\n", - "**Important distinction:** This section is **descriptive only** — we observe differences in median dropout day but do not perform statistical tests. Formal hypothesis testing on demographic associations belongs in Notebook 05 (BQ3: demographics vs. behavior)." + "**Important distinction:** This section is **descriptive only**: we observe differences in median dropout day but do not perform statistical tests. Formal hypothesis testing on demographic associations belongs in Notebook 05 (BQ3: demographics vs. behavior)." ] }, { @@ -805,7 +805,7 @@ "> - **Education level** shows some differentiation: students with different educational backgrounds may drop out at different stages of the course lifecycle.\n", "> - **Age band** and **IMD band** (socioeconomic deprivation index) show patterns that suggest demographic context influences not just *whether* students drop out, but *when*.\n", ">\n", - "> **Caveat:** These are descriptive observations — no causal claims. The differences could be confounded by course choice, prior experience, or other factors. BQ3 (Notebook 05) will perform formal statistical tests comparing demographic and behavioral predictors." + "> **Caveat:** These are descriptive observations: no causal claims. The differences could be confounded by course choice, prior experience, or other factors. BQ3 (Notebook 05) will perform formal statistical tests comparing demographic and behavioral predictors." ] }, { @@ -910,8 +910,8 @@ "source": [ "> **Interpretation:**\n", "> - **Assessment density** and **withdrawal rate** may show a relationship: courses with more frequent assessments could have different retention profiles. The direction of this relationship (if any) is an empirical question that BQ4 will explore formally.\n", - "> - **Course length** and **median dropout day** show that longer courses tend to have later median dropout days in absolute terms, which is expected — but the normalized analysis in Section 5 already showed that the *relative* timing (percentage through course) may be more consistent across modules.\n", - "> - The scatter patterns suggest that course design features contribute to — but do not fully explain — the variation in dropout behavior. Student-level factors (demographics, engagement) also play a role, as BQ2 and BQ3 will explore.\n", + "> - **Course length** and **median dropout day** show that longer courses tend to have later median dropout days in absolute terms, which is expected, but the normalized analysis in Section 5 already showed that the *relative* timing (percentage through course) may be more consistent across modules.\n", + "> - The scatter patterns suggest that course design features contribute to, but do not fully explain, the variation in dropout behavior. Student-level factors (demographics, engagement) also play a role, as BQ2 and BQ3 will explore.\n", ">\n", "> **Preview of BQ4:** Notebook 06 will formalize this analysis with per-course retention metrics, assessment-interaction analysis, and resource diversity effects." ] @@ -933,7 +933,7 @@ "\n", "4. **Normalized timing enables cross-course comparison.** Converting dropout day to percentage-through-course reveals that the *phase* of dropout (early, mid, late) varies by module, suggesting different underlying causes.\n", "\n", - "5. **Pre-course withdrawals are a distinct phenomenon.** A measurable share of students withdraw before day 0 — pure registration churn. These students need activation, not academic support.\n", + "5. **Pre-course withdrawals are a distinct phenomenon.** A measurable share of students withdraw before day 0: pure registration churn. These students need activation, not academic support.\n", "\n", "6. **Demographic segmentation shows moderate variation.** Dropout timing differs somewhat by education level, age, and socioeconomic index, but the differences are modest compared to overall variation. Formal testing in BQ3 will determine statistical significance.\n", "\n", @@ -944,7 +944,7 @@ "| Notebook | Business Question | Focus |\n", "|----------|------------------|-------|\n", "| **04** | BQ2 | Which early behavioral signals predict drop-out? |\n", - "| **05** | BQ3 | Demographics vs. behavior — what predicts outcome more? |\n", + "| **05** | BQ3 | Demographics vs. behavior: what predicts outcome more? |\n", "| **06** | BQ4 | How do course characteristics affect retention? |\n", "| **07** | BQ5 | Top 3 actionable interventions |\n", "\n", @@ -958,9 +958,9 @@ "id": "28", "metadata": {}, "source": [ - "> **From timing to signals:** This notebook established *when* students leave — the temporal landscape of dropout. But timing alone does not enable prevention. The next step is to identify *what early behaviors predict departure*.\n", + "> **From timing to signals:** This notebook established *when* students leave: the temporal landscape of dropout. But timing alone does not enable prevention. The next step is to identify *what early behaviors predict departure*.\n", ">\n", - "> Continue to **Notebook 04** (`04_bq2_early_signals.ipynb`) for BQ2: which early behavioral signals predict drop-out? Notebook 04 introduces formal statistical testing with t-tests, effect sizes, and multiple comparison correction — the first use of `src/stats/tests.py` in analysis." + "> Continue to **Notebook 04** (`04_bq2_early_signals.ipynb`) for BQ2: which early behavioral signals predict drop-out? Notebook 04 introduces formal statistical testing with t-tests, effect sizes, and multiple comparison correction: the first use of `src/stats/tests.py` in analysis." ] } ], diff --git a/notebooks/04_bq2_early_signals.ipynb b/notebooks/04_bq2_early_signals.ipynb index 1683431..094d933 100644 --- a/notebooks/04_bq2_early_signals.ipynb +++ b/notebooks/04_bq2_early_signals.ipynb @@ -5,7 +5,7 @@ "id": "0", "metadata": {}, "source": [ - "# 04 — BQ2: Which Early Behavioral Signals Predict Drop-Out?\n", + "# 04. BQ2: Which Early Behavioral Signals Predict Drop-Out?\n", "\n", "> **Notebook 04 of 7** | Learning Retention Analytics \n", "> Second business question analysis: identifying early engagement signals that differentiate completers from non-completers, with formal statistical testing." @@ -18,7 +18,7 @@ "source": [ "## Purpose and Scope\n", "\n", - "This notebook answers **BQ2: Which early behavioral signals predict drop-out?** — the second of five business questions driving this project.\n", + "This notebook answers **BQ2: Which early behavioral signals predict drop-out?**, the second of five business questions driving this project.\n", "\n", "Notebook 02 (EDA) showed that engagement metrics *differ* between completers and non-completers. Notebook 03 (BQ1) established *when* students leave. This notebook moves from observation to **formal statistical testing**: which early signals (first 28 days) are statistically significant predictors of eventual dropout, and how strong are the effects?\n", "\n", @@ -27,7 +27,7 @@ "- Descriptive comparison of 8 early signals by completion outcome\n", "- Independent t-tests with Cohen's d effect sizes on each signal\n", "- Multiple comparison correction (Bonferroni and Benjamini-Hochberg)\n", - "- Forest plot of ranked effect sizes — the key deliverable figure\n", + "- Forest plot of ranked effect sizes: the key deliverable figure\n", "- Signal deep-dives: dose-response for the strongest predictors\n", "- Assessment-based signals: first score and submission timing\n", "- Ghost student signal quantification with bootstrap confidence intervals\n", @@ -37,9 +37,9 @@ "- No causal claims. We identify *associations*, not causes.\n", "\n", "**What comes next:**\n", - "- **Notebook 05** (`05_bq3_demographics_vs_behavior.ipynb`): demographics vs. behavior — what predicts outcome more? (BQ3)\n", + "- **Notebook 05** (`05_bq3_demographics_vs_behavior.ipynb`): demographics vs. behavior: what predicts outcome more? (BQ3)\n", "\n", - "> **Methodological transferability:** The signal-ranking approach used here — systematic t-tests with effect sizes, multiple comparison correction, and forest plots — is the standard toolkit for identifying predictive features in product analytics. Replace \"engagement signals\" with \"feature usage metrics\" and \"completion\" with \"retention\" to apply this framework to SaaS churn analysis, subscription renewal prediction, or fitness app engagement scoring." + "> **Methodological transferability:** The signal-ranking approach used here (systematic t-tests with effect sizes, multiple comparison correction, and forest plots) is the standard toolkit for identifying predictive features in product analytics. Replace \"engagement signals\" with \"feature usage metrics\" and \"completion\" with \"retention\" to apply this framework to SaaS churn analysis, subscription renewal prediction, or fitness app engagement scoring." ] }, { @@ -52,9 +52,9 @@ "1. [Environment Setup](#1.-Environment-Setup)\n", "2. [The Analysis Dataset](#2.-The-Analysis-Dataset)\n", "3. [Descriptive Comparison](#3.-Descriptive-Comparison)\n", - "4. [Statistical Testing — T-Tests](#4.-Statistical-Testing-—-T-Tests)\n", + "4. [Statistical Testing: T-Tests](#4.-Statistical-Testing:-T-Tests)\n", "5. [Multiple Comparison Correction](#5.-Multiple-Comparison-Correction)\n", - "6. [Effect Size Ranking — Forest Plot](#6.-Effect-Size-Ranking-—-Forest-Plot)\n", + "6. [Effect Size Ranking: Forest Plot](#6.-Effect-Size-Ranking:-Forest-Plot)\n", "7. [Signal Deep-Dives](#7.-Signal-Deep-Dives)\n", "8. [Assessment-Based Signals](#8.-Assessment-Based-Signals)\n", "9. [Ghost Student Signal](#9.-Ghost-Student-Signal)\n", @@ -66,7 +66,7 @@ "- The ETL pipeline must have been run: `python -m run_pipeline`\n", "- The DuckDB database at `data/db/oulad.duckdb` must contain all 5 analytical views\n", "\n", - "**Dataset:** Open University Learning Analytics Dataset (OULAD) — ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." + "**Dataset:** Open University Learning Analytics Dataset (OULAD), ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." ] }, { @@ -79,9 +79,9 @@ "We configure imports, visualization defaults, and reusable helper functions.\n", "\n", "**Technical notes for readers:**\n", - "- All database queries go through `src.db.connection.execute_query()` — the project's DB abstraction layer (ADR-003).\n", + "- All database queries go through `src.db.connection.execute_query()`, the project's DB abstraction layer (ADR-003).\n", "- BQ2's primary SQL query lives in `sql/queries/q_bq2_early_signals.sql` and is loaded at runtime from disk.\n", - "- Statistical tests use `src.stats.tests` — project wrappers around scipy that standardize output format (test statistic, p-value, effect size, confidence interval).\n", + "- Statistical tests use `src.stats.tests`, project wrappers around scipy that standardize output format (test statistic, p-value, effect size, confidence interval).\n", "- Figures are saved to `reports/figures/` at 150 DPI." ] }, @@ -159,7 +159,7 @@ "LABEL_NUM_ENROLLMENTS = 'Number of enrollments'\n", "LABEL_EFFECT_SIZE = \"Cohen's d\"\n", "\n", - "# Significance threshold — used for annotation and interpretation\n", + "# Significance threshold: used for annotation and interpretation\n", "ALPHA = 0.05\n", "\n", "FIG_DPI = 150\n", @@ -220,7 +220,7 @@ "| `date_registration` | Registration day (negative = early registration) | `v_student_enriched` |\n", "| `completed` | Binary outcome: 1 = completed, 0 = not completed | `v_student_enriched` |\n", "\n", - "Students with zero VLE activity get `COALESCE` values of 0 for engagement metrics — they are included as the extreme low end of the engagement spectrum." + "Students with zero VLE activity get `COALESCE` values of 0 for engagement metrics: they are included as the extreme low end of the engagement spectrum." ] }, { @@ -283,7 +283,7 @@ "\n", "Before statistical testing, we visually compare the distribution of each signal between the two outcome groups. This builds intuition about *where* the differences lie and whether the distributions overlap substantially.\n", "\n", - "Violin plots show the full distribution shape — more informative than simple bar charts of means, especially for detecting bimodality or heavy tails common in clickstream data." + "Violin plots show the full distribution shape: more informative than simple bar charts of means, especially for detecting bimodality or heavy tails common in clickstream data." ] }, { @@ -370,7 +370,7 @@ "id": "10", "metadata": {}, "source": [ - "> **Visual impression:** All six engagement signals show visible separation between completers and non-completers. The distributions overlap substantially — this is expected with behavioral data — but the central tendencies (means, medians) differ consistently.\n", + "> **Visual impression:** All six engagement signals show visible separation between completers and non-completers. The distributions overlap substantially (this is expected with behavioral data), but the central tendencies (means, medians) differ consistently.\n", ">\n", "> The question is: are these differences **statistically significant**, and how **large** are the effects? Sections 4–6 answer this formally." ] @@ -380,7 +380,7 @@ "id": "11", "metadata": {}, "source": [ - "## 4. Statistical Testing — T-Tests\n", + "## 4. Statistical Testing: T-Tests\n", "\n", "We use **Welch's independent t-test** (unequal variances) to compare each signal between the completed and not-completed groups. For each test, we report:\n", "\n", @@ -528,7 +528,7 @@ "id": "16", "metadata": {}, "source": [ - "> **Interpretation:** With a large dataset (~32K enrollments), all 8 signals remain significant even after the conservative Bonferroni correction (8/8 under both Bonferroni and BH). This confirms that the differences are real — but statistical significance alone does not tell us which signals are *practically meaningful*. That is what the effect size ranking (next section) addresses." + "> **Interpretation:** With a large dataset (~32K enrollments), all 8 signals remain significant even after the conservative Bonferroni correction (8/8 under both Bonferroni and BH). This confirms that the differences are real, but statistical significance alone does not tell us which signals are *practically meaningful*. That is what the effect size ranking (next section) addresses." ] }, { @@ -536,7 +536,7 @@ "id": "17", "metadata": {}, "source": [ - "## 6. Effect Size Ranking — Forest Plot\n", + "## 6. Effect Size Ranking: Forest Plot\n", "\n", "This is the **key deliverable figure** for BQ2. The forest plot ranks all signals by Cohen's d (absolute effect size), showing the standardized mean difference for each signal.\n", "\n", @@ -693,7 +693,7 @@ "id": "22", "metadata": {}, "source": [ - "> **Interpretation:** The dose-response plots confirm that the relationship between top signals and completion is **monotonic** (or near-monotonic): higher engagement consistently predicts higher completion rates. This is important — it means the signal is useful across its entire range, not just at extremes.\n", + "> **Interpretation:** The dose-response plots confirm that the relationship between top signals and completion is **monotonic** (or near-monotonic): higher engagement consistently predicts higher completion rates. This is important: it means the signal is useful across its entire range, not just at extremes.\n", ">\n", "> The steep gradient between the lowest and highest quartile quantifies the practical impact: students in the bottom quartile of key engagement signals have substantially lower completion rates than those in the top quartile." ] @@ -705,7 +705,7 @@ "source": [ "## 8. Assessment-Based Signals\n", "\n", - "Two of our signals come from early assessment behavior rather than VLE clickstream: `first_score` (score on the first assessment) and `first_submit_day` (when they submitted it). These signals have substantial missingness — students who never submitted any assessment in the first 28 days have NULL values.\n", + "Two of our signals come from early assessment behavior rather than VLE clickstream: `first_score` (score on the first assessment) and `first_submit_day` (when they submitted it). These signals have substantial missingness: students who never submitted any assessment in the first 28 days have NULL values.\n", "\n", "The missingness itself is informative: students who did not submit any early assessment are likely at higher dropout risk than those who did." ] @@ -775,7 +775,7 @@ "source": [ "> **Interpretation:**\n", "> - **Submission vs non-submission** is itself a powerful signal: students who submitted at least one assessment in the first 28 days have a substantially higher completion rate.\n", - "> - Among submitters, **first assessment score** shows separation — completers tend to score higher on their first assessment.\n", + "> - Among submitters, **first assessment score** shows separation: completers tend to score higher on their first assessment.\n", "> - **Submission timing** provides additional signal: earlier submission may indicate better time management and engagement.\n", ">\n", "> **Caveat:** Assessment signals have high missingness. The t-test results for `first_score` and `first_submit_day` reflect only the submitter subpopulation. The *non-submission* signal is captured by engagement metrics (ghost/low-activity students)." @@ -788,9 +788,9 @@ "source": [ "## 9. Ghost Student Signal\n", "\n", - "Notebook 02 identified \"ghost students\" — enrollments with zero VLE activity in the first 28 days. Here we quantify this signal formally using **bootstrap confidence intervals** for the completion rate difference between ghost and active students.\n", + "Notebook 02 identified \"ghost students\": enrollments with zero VLE activity in the first 28 days. Here we quantify this signal formally using **bootstrap confidence intervals** for the completion rate difference between ghost and active students.\n", "\n", - "The bootstrap approach is used because the ghost/active split produces groups of very different sizes, and the completion rate for ghost students is near zero — parametric CI assumptions may not hold well at these extremes." + "The bootstrap approach is used because the ghost/active split produces groups of very different sizes, and the completion rate for ghost students is near zero; parametric CI assumptions may not hold well at these extremes." ] }, { @@ -873,13 +873,13 @@ "id": "28", "metadata": {}, "source": [ - "> **Key finding:** The gap between ghost and active students is enormous and the 95% bootstrap confidence intervals do not overlap: 4.8% completion for ghosts (CI 4.2-5.4%) versus 54.3% for active students (CI 53.7-54.9%). Ghost students — those with zero VLE activity in the first 28 days — have a near-zero completion rate.\n", + "> **Key finding:** The gap between ghost and active students is enormous and the 95% bootstrap confidence intervals do not overlap: 4.8% completion for ghosts (CI 4.2-5.4%) versus 54.3% for active students (CI 53.7-54.9%). Ghost students, those with zero VLE activity in the first 28 days, have a near-zero completion rate.\n", ">\n", "> This is the **strongest single signal** in the dataset: if a student has not clicked on any VLE resource within the first 4 weeks, their probability of completing the course is negligible.\n", ">\n", "> **Definition note:** in this notebook \"ghost\" means zero VLE activity in the first 28 days (n = 4685). The BQ5 segment sizing uses a broader operational threshold instead (at most 1 active day and fewer than 10 clicks), which yields a larger segment with a 7.7% completion rate: the strict definition isolates the pure signal, the broad one sizes the intervention target.\n", ">\n", - "> **Intervention priority:** Ghost students are the lowest-hanging fruit. They do not need better content — they need to be *activated*. In SaaS terms, this is the onboarding gap: users who signed up but never experienced the core product value." + "> **Intervention priority:** Ghost students are the lowest-hanging fruit. They do not need better content: they need to be *activated*. In SaaS terms, this is the onboarding gap: users who signed up but never experienced the core product value." ] }, { @@ -891,13 +891,13 @@ "\n", "### What we learned\n", "\n", - "1. **All 8 early signals show statistically significant differences** between completers and non-completers. With ~32K enrollments, even modest differences reach significance — which is why effect size (Cohen's d) is the primary ranking criterion.\n", + "1. **All 8 early signals show statistically significant differences** between completers and non-completers. With ~32K enrollments, even modest differences reach significance, which is why effect size (Cohen's d) is the primary ranking criterion.\n", "\n", "2. **The strongest behavioral predictors** are engagement-volume metrics: engagement decile (d = 0.97), active days (d = 0.90), and total clicks (d = 0.63). These signals capture both frequency and intensity of platform interaction in the first 28 days.\n", "\n", "3. **Multiple comparison correction confirms robustness.** All 8 signals remain significant after both Bonferroni and Benjamini-Hochberg correction, indicating that the associations are real, not artifacts of multiple testing.\n", "\n", - "4. **Dose-response relationships are monotonic.** More engagement consistently predicts higher completion rates across all signal quartiles. There are no obvious thresholds or diminishing returns — the relationship is graded.\n", + "4. **Dose-response relationships are monotonic.** More engagement consistently predicts higher completion rates across all signal quartiles. There are no obvious thresholds or diminishing returns: the relationship is graded.\n", "\n", "5. **Assessment submission is a binary predictor.** Whether or not a student submitted any assessment in the first 28 days is itself a powerful signal, independent of the score achieved.\n", "\n", @@ -905,13 +905,13 @@ "\n", "7. **Effect sizes span Cohen's full scale.** The two leading volume signals reach the \"large\" threshold (d = 0.90 and 0.97), most others sit in the medium band (|d| ≈ 0.5-0.6), and the timing signals remain small (|d| ≈ 0.1-0.2). Individual signals still explain only part of the outcome variance: the practical value comes from combining multiple signals into an engagement score (BQ5).\n", "\n", - "8. **No causal claims.** All findings are associations. Motivated students may both engage more and complete more — engagement could be a proxy for motivation, not a cause of success.\n", + "8. **No causal claims.** All findings are associations. Motivated students may both engage more and complete more: engagement could be a proxy for motivation, not a cause of success.\n", "\n", "### What comes next\n", "\n", "| Notebook | Business Question | Focus |\n", "|----------|------------------|-------|\n", - "| **05** | BQ3 | Demographics vs. behavior — what predicts outcome more? |\n", + "| **05** | BQ3 | Demographics vs. behavior: what predicts outcome more? |\n", "| **06** | BQ4 | How do course characteristics affect retention? |\n", "| **07** | BQ5 | Top 3 actionable interventions |\n", "\n", @@ -927,7 +927,7 @@ "source": [ "> **From signals to comparison:** This notebook identified *which* early behaviors predict dropout and ranked them by effect size. The next question is: do these behavioral signals outperform demographic factors? Or does a student's background matter more than what they actually do on the platform?\n", ">\n", - "> Continue to **Notebook 05** (`05_bq3_demographics_vs_behavior.ipynb`) for BQ3: demographics vs. behavior — what predicts outcome more?" + "> Continue to **Notebook 05** (`05_bq3_demographics_vs_behavior.ipynb`) for BQ3: demographics vs. behavior: what predicts outcome more?" ] } ], diff --git a/notebooks/05_bq3_demographics_vs_behavior.ipynb b/notebooks/05_bq3_demographics_vs_behavior.ipynb index 204c332..4ecc11c 100644 --- a/notebooks/05_bq3_demographics_vs_behavior.ipynb +++ b/notebooks/05_bq3_demographics_vs_behavior.ipynb @@ -5,7 +5,7 @@ "id": "0", "metadata": {}, "source": [ - "# 05 — BQ3: Demographics vs Behavior — What Predicts Outcome More?\n", + "# 05. BQ3: Demographics vs Behavior, What Predicts Outcome More?\n", "\n", "> **Notebook 05 of 7** | Learning Retention Analytics \n", "> Third business question analysis: comparing the predictive strength of demographic features versus early behavioral signals." @@ -18,7 +18,7 @@ "source": [ "## Purpose and Scope\n", "\n", - "This notebook answers **BQ3: Do demographics or behavior predict outcome more strongly?** — the third of five business questions driving this project.\n", + "This notebook answers **BQ3: Do demographics or behavior predict outcome more strongly?**, the third of five business questions driving this project.\n", "\n", "Notebook 04 (BQ2) ranked early behavioral signals by their association with completion. This notebook adds the **demographic dimension**: if we know a student's age, education level, and socioeconomic background, does that predict their outcome better or worse than knowing how they interacted with the platform in the first 28 days?\n", "\n", @@ -38,7 +38,7 @@ "**What comes next:**\n", "- **Notebook 06** (`06_bq4_course_comparison.ipynb`): how do course characteristics affect retention? (BQ4)\n", "\n", - "> **Methodological transferability:** The demographics-vs-behavior comparison is a core question in product analytics. In SaaS: do user demographics (company size, industry, role) predict churn better than product usage patterns? In health tech: do patient demographics predict adherence better than app engagement? The framework used here — parallel effect-size testing and comparison — transfers directly." + "> **Methodological transferability:** The demographics-vs-behavior comparison is a core question in product analytics. In SaaS: do user demographics (company size, industry, role) predict churn better than product usage patterns? In health tech: do patient demographics predict adherence better than app engagement? The framework used here (parallel effect-size testing and comparison) transfers directly." ] }, { @@ -50,10 +50,10 @@ "\n", "1. [Environment Setup](#1.-Environment-Setup)\n", "2. [The Analysis Dataset](#2.-The-Analysis-Dataset)\n", - "3. [Part A — Demographic Associations](#3.-Part-A-—-Demographic-Associations)\n", - "4. [Part B — Behavioral Associations](#4.-Part-B-—-Behavioral-Associations)\n", - "5. [Part C — The Verdict: Demographics vs Behavior](#5.-Part-C-—-The-Verdict:-Demographics-vs-Behavior)\n", - "6. [Deep Dive — Education Level × Engagement](#6.-Deep-Dive-—-Education-Level-×-Engagement)\n", + "3. [Part A: Demographic Associations](#3.-Part-A:-Demographic-Associations)\n", + "4. [Part B: Behavioral Associations](#4.-Part-B:-Behavioral-Associations)\n", + "5. [Part C: The Verdict on Demographics vs Behavior](#5.-Part-C:-The-Verdict-on-Demographics-vs-Behavior)\n", + "6. [Deep Dive: Education Level × Engagement](#6.-Deep-Dive:-Education-Level-×-Engagement)\n", "7. [Ethical Framing](#7.-Ethical-Framing)\n", "8. [Key Takeaways and Next Steps](#8.-Key-Takeaways-and-Next-Steps)\n", "\n", @@ -63,7 +63,7 @@ "- The ETL pipeline must have been run: `python -m run_pipeline`\n", "- The DuckDB database at `data/db/oulad.duckdb` must contain all 5 analytical views\n", "\n", - "**Dataset:** Open University Learning Analytics Dataset (OULAD) — ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." + "**Dataset:** Open University Learning Analytics Dataset (OULAD), ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." ] }, { @@ -76,9 +76,9 @@ "We configure imports, visualization defaults, and reusable helper functions.\n", "\n", "**Technical notes for readers:**\n", - "- All database queries go through `src.db.connection.execute_query()` — the project's DB abstraction layer (ADR-003).\n", + "- All database queries go through `src.db.connection.execute_query()`, the project's DB abstraction layer (ADR-003).\n", "- BQ3's primary SQL query lives in `sql/queries/q_bq3_demographics_vs_behavior.sql` and is loaded at runtime from disk.\n", - "- Statistical tests use `src.stats.tests` — project wrappers around scipy. This notebook introduces `chi_square_test()` (Cramér's V) alongside the `independent_t_test()` (Cohen's d) used in NB04.\n", + "- Statistical tests use `src.stats.tests`, project wrappers around scipy. This notebook introduces `chi_square_test()` (Cramér's V) alongside the `independent_t_test()` (Cohen's d) used in NB04.\n", "- Figures are saved to `reports/figures/` at 150 DPI." ] }, @@ -151,14 +151,14 @@ "PALETTE_BINARY_LABELS = {LABEL_COMPLETED: '#55A868', LABEL_NOT_COMPLETED: '#C44E52'}\n", "PALETTE_SEQUENTIAL = 'YlOrRd'\n", "\n", - "# Shared axis labels — defined as constants to avoid\n", + "# Shared axis labels: defined as constants to avoid\n", "# duplicated string literals flagged by static analysis\n", "LABEL_COMPLETION_RATE = 'Completion rate (%)'\n", "LABEL_NUM_ENROLLMENTS = 'Number of enrollments'\n", "LABEL_EFFECT_SIZE = \"Cohen's d\"\n", "LABEL_CRAMERS_V = \"Cramér's V\"\n", "\n", - "# Significance threshold — used for annotation and interpretation\n", + "# Significance threshold: used for annotation and interpretation\n", "ALPHA = 0.05\n", "\n", "FIG_DPI = 150\n", @@ -213,7 +213,7 @@ "| **Demographic (numeric)** | num_of_prev_attempts, studied_credits | T-test + Cohen's d |\n", "| **Behavioral (continuous)** | active_days_first_28, total_clicks_first_28, avg_clicks_per_active_day, engagement_decile_in_course, first_score | T-test + Cohen's d |\n", "| **Behavioral (binary)** | submitted_first_assessment | T-test + Cohen's d |\n", - "| **Outcome** | completed (0/1) | — |\n", + "| **Outcome** | completed (0/1) | - |\n", "\n", "The key design: demographic features are known *before* the course starts. Behavioral features emerge during the first 28 days. If behavior is a stronger signal, the platform has a window to intervene.\n", "\n", @@ -268,12 +268,12 @@ " 'active_days_first_28': 'Active days (first 28)',\n", " 'total_clicks_first_28': 'Total clicks (first 28)',\n", " 'avg_clicks_per_active_day': 'Avg clicks per active day',\n", - " # NULL for students with no VLE activity — they have no position\n", + " # NULL for students with no VLE activity: they have no position\n", " # in the engagement ranking. Effect size is conditional on having\n", " # at least some platform activity (dropna excludes inactive students).\n", " 'engagement_decile_in_course': 'Engagement decile (active only)',\n", " 'submitted_first_assessment': 'Submitted first assessment',\n", - " # NULL for non-submitters — effect size is conditional on having\n", + " # NULL for non-submitters: effect size is conditional on having\n", " # submitted at least one early assessment.\n", " 'first_score': 'First score (submitters only)',\n", "}\n", @@ -301,7 +301,7 @@ "id": "7", "metadata": {}, "source": [ - "## 3. Part A — Demographic Associations\n", + "## 3. Part A: Demographic Associations\n", "\n", "We test each demographic feature for association with completion using the appropriate statistical test:\n", "\n", @@ -399,9 +399,9 @@ "id": "9", "metadata": {}, "source": [ - "> **Interpretation:** Most demographic features show statistically significant associations with completion — but significance alone is not the point. With ~32K enrollments, even trivial differences reach significance. The critical question is **effect size**: are these associations *meaningful*?\n", + "> **Interpretation:** Most demographic features show statistically significant associations with completion, but significance alone is not the point. With ~32K enrollments, even trivial differences reach significance. The critical question is **effect size**: are these associations *meaningful*?\n", ">\n", - "> Look at the Cramér's V values. By Cohen's conventions, values below 0.1 represent negligible associations. The fact that demographic features have small V values suggests they are weak predictors of completion — they tell us *something*, but not much." + "> Look at the Cramér's V values. By Cohen's conventions, values below 0.1 represent negligible associations. The fact that demographic features have small V values suggests they are weak predictors of completion: they tell us *something*, but not much." ] }, { @@ -474,9 +474,9 @@ "id": "11", "metadata": {}, "source": [ - "> **Visual impression:** While completion rates vary across demographic categories — especially for education level and IMD band — the differences are modest. No single demographic group has a near-zero or near-100% completion rate. Demographics shift the probability slightly, but they do not determine the outcome.\n", + "> **Visual impression:** While completion rates vary across demographic categories (especially for education level and IMD band), the differences are modest. No single demographic group has a near-zero or near-100% completion rate. Demographics shift the probability slightly, but they do not determine the outcome.\n", ">\n", - "> This contrasts sharply with behavioral signals from NB04, where ghost students had near-zero completion rates — a much starker separation." + "> This contrasts sharply with behavioral signals from NB04, where ghost students had near-zero completion rates, a much starker separation." ] }, { @@ -488,7 +488,7 @@ "source": [ "# --- Demographic effect sizes bar chart ---\n", "# Combine Cramér's V (categorical) and |Cohen's d| (numeric) in one plot.\n", - "# Both measure association strength, though on different scales — the\n", + "# Both measure association strength, though on different scales; the\n", "# metric type is noted in each bar's annotation for transparency.\n", "df_demo_effects = pd.concat([\n", " df_demo_chi[['label', 'effect_size', 'effect_metric', 'significant']],\n", @@ -544,7 +544,7 @@ "id": "14", "metadata": {}, "source": [ - "## 4. Part B — Behavioral Associations\n", + "## 4. Part B: Behavioral Associations\n", "\n", "We now test the 6 behavioral features (early engagement metrics) for association with completion using **Welch's t-test** with **Cohen's d** as effect size. These are the same metrics ranked in NB04, but here we frame them as the counterpart to demographics for head-to-head comparison.\n", "\n", @@ -616,11 +616,11 @@ "source": [ "> **Interpretation:** All 6 behavioral features show statistically significant associations with completion. More importantly, the **effect sizes are substantially larger** than the demographic ones. Every behavioral feature reaches at least a medium effect size (|d| between 0.52 and 0.90), compared to the small-to-modest demographic effects (|d| up to 0.28, V up to 0.15).\n", ">\n", - "> The strongest behavioral signals — engagement volume, activity frequency, and first assessment submission — provide far more discriminative information about eventual completion than any demographic variable.\n", + "> The strongest behavioral signals (engagement volume, activity frequency, and first assessment submission) provide far more discriminative information about eventual completion than any demographic variable.\n", ">\n", "> **Caveats on conditional features:** Two features are tested on reduced populations (see sample sizes in the table):\n", - "> - `engagement_decile` — excludes students with no VLE activity (NULL in the data). The effect size measures relative engagement *among active students*.\n", - "> - `first_score` — excludes non-submitters. The effect size measures score differences *among those who submitted*.\n", + "> - `engagement_decile`: excludes students with no VLE activity (NULL in the data). The effect size measures relative engagement *among active students*.\n", + "> - `first_score`: excludes non-submitters. The effect size measures score differences *among those who submitted*.\n", ">\n", "> Both are excluded from the head-to-head comparison in Part C to ensure a fair same-population comparison." ] @@ -680,14 +680,14 @@ "id": "19", "metadata": {}, "source": [ - "## 5. Part C — The Verdict: Demographics vs Behavior\n", + "## 5. Part C: The Verdict on Demographics vs Behavior\n", "\n", "This is the core analysis of BQ3: a head-to-head comparison of effect sizes.\n", "\n", - "**Methodology note:** Cramér's V and Cohen's d are measured on different scales, so direct numerical comparison requires caution. However, both have established thresholds for \"small,\" \"medium,\" and \"large\" effects, and the **relative ordering within and across groups** tells a clear story. Additionally, the numeric demographics (previous attempts, studied credits) use Cohen's d — the same metric as behavioral features — enabling direct comparison.\n", + "**Methodology note:** Cramér's V and Cohen's d are measured on different scales, so direct numerical comparison requires caution. However, both have established thresholds for \"small,\" \"medium,\" and \"large\" effects, and the **relative ordering within and across groups** tells a clear story. Additionally, the numeric demographics (previous attempts, studied credits) use Cohen's d (the same metric as behavioral features), enabling direct comparison.\n", "\n", "The figure below shows all-enrollment features in a unified view:\n", - "- **Left panel**: |Cohen's d| for all-enrollment continuous features (2 demographic + 4 behavioral), colored by type. Conditional features (engagement decile, first score) are excluded because they are measured on different populations — their values are reported separately in the quantitative summary.\n", + "- **Left panel**: |Cohen's d| for all-enrollment continuous features (2 demographic + 4 behavioral), colored by type. Conditional features (engagement decile, first score) are excluded because they are measured on different populations; their values are reported separately in the quantitative summary.\n", "- **Right panel**: Cramér's V for categorical demographics, for context" ] }, @@ -777,7 +777,7 @@ "sns.despine(ax=ax2)\n", "\n", "fig.suptitle(\n", - " 'BQ3: Demographics vs Behavior — Effect Size Comparison\\n'\n", + " 'BQ3: Demographics vs Behavior - Effect Size Comparison\\n'\n", " '(all-enrollment features only; conditional features reported below)',\n", " fontsize=14, y=1.02,\n", ")\n", @@ -789,7 +789,7 @@ "# Compare d-vs-d only on comparable samples (all-enrollment metrics).\n", "# Conditional features (engagement_decile, first_score) are excluded from\n", "# the main summary and reported separately to avoid skewing the mean/ratio.\n", - "# Cramér's V reported separately — different scale.\n", + "# Cramér's V reported separately: different scale.\n", "avg_demo_d = df_demo_ttest['effect_size'].mean()\n", "avg_behav_d = df_behav_comparable['abs_cohens_d'].mean()\n", "max_demo_d = df_demo_ttest['effect_size'].max()\n", @@ -837,7 +837,7 @@ "id": "22", "metadata": {}, "source": [ - "## 6. Deep Dive — Education Level × Engagement\n", + "## 6. Deep Dive: Education Level × Engagement\n", "\n", "Even though demographics are weak predictors overall, education level (`highest_education`) is typically the strongest demographic signal. The critical question is: does education level *still matter* once we account for engagement?\n", "\n", @@ -930,7 +930,7 @@ "source": [ "> **Key finding:** Within every education level, high-engagement students dramatically outperform low-engagement students. The within-education-level gap (engagement effect) is consistently **larger** than the between-level gap (education effect at the same engagement level).\n", ">\n", - "> This means a student with lower formal education but high engagement has a *better* chance of completing than a highly educated student who does not engage with the platform. Engagement is the dominant factor — education level merely shifts the baseline." + "> This means a student with lower formal education but high engagement has a *better* chance of completing than a highly educated student who does not engage with the platform. Engagement is the dominant factor; education level merely shifts the baseline." ] }, { @@ -940,7 +940,7 @@ "source": [ "## 7. Ethical Framing\n", "\n", - "The BQ3 finding — that **behavior predicts outcome more strongly than demographics** — has important implications for how a learning platform should be designed and operated:\n", + "The BQ3 finding, that **behavior predicts outcome more strongly than demographics**, has important implications for how a learning platform should be designed and operated:\n", "\n", "**1. Interventions should target behavior, not demographics.**\n", "Sending additional support messages only to students from lower education backgrounds would be both less effective and potentially discriminatory. Instead, monitoring engagement metrics and intervening when activity drops provides a fairer and more effective approach.\n", @@ -963,13 +963,13 @@ "\n", "### What we learned\n", "\n", - "1. **All demographic features show statistically significant but weak associations** with completion. The largest Cramér's V values peak at about 0.15 (education 0.150, IMD band 0.134), and numeric demographic Cohen's d values stay below 0.3. With ~32K enrollments, significance is easy to achieve — effect size is what matters.\n", + "1. **All demographic features show statistically significant but weak associations** with completion. The largest Cramér's V values peak at about 0.15 (education 0.150, IMD band 0.134), and numeric demographic Cohen's d values stay below 0.3. With ~32K enrollments, significance is easy to achieve: effect size is what matters.\n", "\n", "2. **Behavioral features have 2–4× larger effect sizes** than demographic features. Engagement volume, activity frequency, and first assessment submission are far more informative about eventual completion.\n", "\n", "3. **Within every education level, engagement is the swing factor.** High-engagement students outperform low-engagement students regardless of their educational background. The within-group behavioral gap exceeds the between-group demographic gap.\n", "\n", - "4. **Demographics cannot be changed; behavior can be influenced.** This makes behavioral signals not only *stronger* predictors but also *actionable* ones — the platform can influence engagement through design and intervention.\n", + "4. **Demographics cannot be changed; behavior can be influenced.** This makes behavioral signals not only *stronger* predictors but also *actionable* ones: the platform can influence engagement through design and intervention.\n", "\n", "5. **Ethical alignment:** Using behavioral signals for risk identification avoids the fairness concerns inherent in demographic profiling, while providing better predictive accuracy.\n", "\n", diff --git a/notebooks/06_bq4_course_comparison.ipynb b/notebooks/06_bq4_course_comparison.ipynb index 6ea67df..bf81465 100644 --- a/notebooks/06_bq4_course_comparison.ipynb +++ b/notebooks/06_bq4_course_comparison.ipynb @@ -5,7 +5,7 @@ "id": "0", "metadata": {}, "source": [ - "# 06 — BQ4: How Do Course Characteristics Affect Retention?\n", + "# 06. BQ4: How Do Course Characteristics Affect Retention?\n", "\n", "> **Notebook 06 of 7** | Learning Retention Analytics \n", "> Fourth business question analysis: comparing course design features and their relationship with student retention." @@ -18,7 +18,7 @@ "source": [ "## Purpose and Scope\n", "\n", - "This notebook answers **BQ4: How do course characteristics affect retention?** — the fourth of five business questions driving this project.\n", + "This notebook answers **BQ4: How do course characteristics affect retention?**, the fourth of five business questions driving this project.\n", "\n", "Previous notebooks focused on student-level analysis: when students drop out (BQ1), which early behaviors predict dropout (BQ2), and whether demographics or behavior predicts outcome more strongly (BQ3). This notebook **shifts the unit of analysis from students to courses**.\n", "\n", @@ -31,12 +31,12 @@ "\n", "**What this notebook does NOT do:**\n", "- No confirmatory inference. With only 7 modules, any correlation summaries or thresholds are descriptive heuristics, not formal significance tests. All analysis is exploratory.\n", - "- No causal claims. Differences in retention could be driven by student selection, subject difficulty, or unmeasured factors — not course design alone.\n", + "- No causal claims. Differences in retention could be driven by student selection, subject difficulty, or unmeasured factors, not course design alone.\n", "\n", "**What comes next:**\n", "- **Notebook 07** (`07_bq5_recommendations_synthesis.ipynb`): synthesizing BQ1–BQ4 findings into the top 3 actionable interventions (BQ5).\n", "\n", - "> **Methodological transferability:** Comparing course-level retention metrics parallels product-level or plan-level churn analysis in SaaS: which product tier retains best? Which plan characteristics (trial length, feature set, onboarding flow) correlate with lower churn? The approach — descriptive ranking, feature profiling, heatmap comparison — transfers directly." + "> **Methodological transferability:** Comparing course-level retention metrics parallels product-level or plan-level churn analysis in SaaS: which product tier retains best? Which plan characteristics (trial length, feature set, onboarding flow) correlate with lower churn? The approach (descriptive ranking, feature profiling, heatmap comparison) transfers directly." ] }, { @@ -61,7 +61,7 @@ "- The ETL pipeline must have been run: `python -m run_pipeline`\n", "- The DuckDB database at `data/db/oulad.duckdb` must contain all 5 analytical views\n", "\n", - "**Dataset:** Open University Learning Analytics Dataset (OULAD) — ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." + "**Dataset:** Open University Learning Analytics Dataset (OULAD), ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." ] }, { @@ -74,9 +74,9 @@ "We configure imports, visualization defaults, and reusable helper functions.\n", "\n", "**Technical notes for readers:**\n", - "- All database queries go through `src.db.connection.execute_query()` — the project's DB abstraction layer (ADR-003).\n", + "- All database queries go through `src.db.connection.execute_query()`, the project's DB abstraction layer (ADR-003).\n", "- BQ4's primary SQL query lives in `sql/queries/q_bq4_course_comparison.sql` and is loaded at runtime from disk. The query aggregates course-level metrics from `v_course_profile`, `v_engagement_daily`, and `v_engagement_early`.\n", - "- No statistical test functions are imported — with only 7 data points, all analysis is descriptive.\n", + "- No statistical test functions are imported: with only 7 data points, all analysis is descriptive.\n", "- Figures are saved to `reports/figures/` at 150 DPI." ] }, @@ -136,7 +136,7 @@ "_TAB10 = plt.cm.tab10.colors\n", "PALETTE_COURSE = {m: _TAB10[i] for i, m in enumerate(_MODULE_ORDER)}\n", "\n", - "# Shared axis labels — defined as constants to avoid\n", + "# Shared axis labels: defined as constants to avoid\n", "# duplicated string literals flagged by static analysis\n", "LABEL_COMPLETION_RATE = 'Completion rate (%)'\n", "LABEL_WITHDRAWAL_RATE = 'Withdrawal rate (%)'\n", @@ -253,7 +253,7 @@ " print(n_nulls[n_nulls > 0].to_string())\n", "else:\n", " print()\n", - " print('No null values — all 7 modules have complete data.')\n", + " print('No null values: all 7 modules have complete data.')\n", "\n", "# --- Full dataset overview ---\n", "print()\n", @@ -269,7 +269,7 @@ "source": [ "> **First look:** The 7 OULAD modules span a wide range of completion rates and course designs. Some modules are shorter with fewer assessments; others are longer with more resources and higher assessment density. The question is whether these design differences relate to retention performance.\n", ">\n", - "> The dataset is deliberately small — one row per module — which means every observation matters and no outlier can be dismissed. All patterns we identify are hypotheses that would need larger-scale validation." + "> The dataset is deliberately small (one row per module), which means every observation matters and no outlier can be dismissed. All patterns we identify are hypotheses that would need larger-scale validation." ] }, { @@ -346,7 +346,7 @@ "worst_module = worst['code_module']\n", "print()\n", "print(f'Completion rate range: {worst_rate:.1f}% ({worst_module}) '\n", - " f'to {best_rate:.1f}% ({best_module}) — gap of {gap:.1f} pp')" + " f'to {best_rate:.1f}% ({best_module}), gap of {gap:.1f} pp')" ] }, { @@ -354,9 +354,9 @@ "id": "10", "metadata": {}, "source": [ - "> **Key finding:** The completion rate gap between the best- and worst-performing modules is substantial. This variation is not random — it persists across multiple presentations of the same module, suggesting that something about the course itself (design, subject difficulty, student selection) drives the difference.\n", + "> **Key finding:** The completion rate gap between the best- and worst-performing modules is substantial. This variation is not random: it persists across multiple presentations of the same module, suggesting that something about the course itself (design, subject difficulty, student selection) drives the difference.\n", ">\n", - "> **Note on enrollment volume:** The annotation shows total enrollment for each module. Larger enrollment doesn't necessarily correlate with higher or lower completion — the pattern is more nuanced, and we'll explore it through design features next." + "> **Note on enrollment volume:** The annotation shows total enrollment for each module. Larger enrollment doesn't necessarily correlate with higher or lower completion; the pattern is more nuanced, and we'll explore it through design features next." ] }, { @@ -444,7 +444,7 @@ "metadata": {}, "source": [ "> **Visual patterns:**\n", - "> - **Course length:** The scatter may suggest a relationship between duration and retention — shorter or longer courses may perform differently. However, with 7 points, any apparent trend could be driven by one or two outliers.\n", + "> - **Course length:** The scatter may suggest a relationship between duration and retention: shorter or longer courses may perform differently. However, with 7 points, any apparent trend could be driven by one or two outliers.\n", "> - **Assessment density:** Courses with different assessment pacing may show different retention profiles. Whether more frequent assessment helps (through regular engagement checkpoints) or hurts (through assessment fatigue) is an open question.\n", ">\n", "> **Confounding factors:** These associations are confounded by subject difficulty, student self-selection, and module content quality. A course with high assessment density that also happens to be in a popular, well-taught subject will retain students regardless of assessment frequency." @@ -483,7 +483,7 @@ " elif abs(rho) >= 0.4:\n", " direction = '~ moderate'\n", " else:\n", - " direction = '— weak'\n", + " direction = '- weak'\n", " print(f' {label:35s} ρ = {rho:+.3f} ({direction})')" ] }, @@ -494,7 +494,7 @@ "source": [ "> **Correlation summary:** The Spearman rank correlations provide a compact summary of which features *move in the same direction* as completion rate. Features marked as notable (|ρ| ≥ 0.79) would pass the significance threshold for n=7, but even these should be treated as hypotheses.\n", ">\n", - "> **What correlations cannot tell us:** Rank correlations measure monotonic association, not causation. A strong positive ρ between assessment density and completion rate could mean that assessments help retention — or that popular, well-designed courses happen to have more assessments. The data cannot distinguish these explanations with 7 observations." + "> **What correlations cannot tell us:** Rank correlations measure monotonic association, not causation. A strong positive ρ between assessment density and completion rate could mean that assessments help retention, or that popular, well-designed courses happen to have more assessments. The data cannot distinguish these explanations with 7 observations." ] }, { @@ -647,7 +647,11 @@ "annot_array = np.array(annot_values)\n", "\n", "# --- Heatmap ---\n", - "fig, ax = plt.subplots(figsize=(14, 5))\n", + "# Explicit geometry: at 5 in of height, tight_layout reserves almost all the\n", + "# vertical space for the long column labels and collapses the 7 matrix rows\n", + "# into unreadable slivers (observed on headless nbconvert re-execution).\n", + "# Extra height plus angled labels keep the layout stable on any backend.\n", + "fig, ax = plt.subplots(figsize=(14, 7))\n", "sns.heatmap(\n", " df_z, annot=annot_array, fmt='',\n", " cmap='RdYlGn', center=0, linewidths=0.5,\n", @@ -662,8 +666,11 @@ " rate_labels.append(f'{m} ({rate:.1f}%)')\n", "ax.set_yticklabels(rate_labels, rotation=0)\n", "\n", + "# Angled column labels shorten the label block tight_layout must accommodate\n", + "ax.set_xticklabels(ax.get_xticklabels(), rotation=35, ha='right')\n", + "\n", "ax.set_title(\n", - " 'Course Design Heatmap — All Features Standardized\\n'\n", + " 'Course Design Heatmap: All Features Standardized\\n'\n", " '(z-score coloring, actual values annotated; sorted by completion rate)'\n", ")\n", "ax.set_ylabel('')\n", @@ -682,9 +689,9 @@ "> **What to look for:**\n", "> - Do high-completion courses (top rows) tend to be greener on specific feature columns? If so, those features may contribute to retention.\n", "> - Do low-completion courses (bottom rows) share red cells on the same features? That would strengthen the pattern.\n", - "> - Are there modules that are outliers — green on most features but low on completion, or vice versa? These break the pattern and suggest other factors at play.\n", + "> - Are there modules that are outliers: green on most features but low on completion, or vice versa? These break the pattern and suggest other factors at play.\n", ">\n", - "> **Caveat:** The z-score standardization amplifies visual contrast — a difference of 0.5 standard deviations looks dramatic in the color scale but may not be practically meaningful with only 7 observations." + "> **Caveat:** The z-score standardization amplifies visual contrast: a difference of 0.5 standard deviations looks dramatic in the color scale but may not be practically meaningful with only 7 observations." ] }, { @@ -698,7 +705,7 @@ "\n", "Based on the descriptive patterns observed:\n", "\n", - "1. **Assessment structure may influence retention.** Courses with a certain assessment pacing might provide regular engagement checkpoints that help students stay on track — or might create pressure points that drive departures. The direction of this effect is an empirical question.\n", + "1. **Assessment structure may influence retention.** Courses with a certain assessment pacing might provide regular engagement checkpoints that help students stay on track, or might create pressure points that drive departures. The direction of this effect is an empirical question.\n", "\n", "2. **Resource diversity may support engagement.** Courses offering a wider variety of VLE activity types may keep students engaged through varied learning experiences. However, resource diversity could also correlate with institutional investment in the course, which is a confound.\n", "\n", @@ -708,7 +715,7 @@ "\n", "This analysis has fundamental constraints that must be acknowledged:\n", "\n", - "- **n=7.** Seven data points cannot support inferential statistics. All correlations, patterns, and hypotheses are exploratory. A significant Spearman ρ at n=7 requires |ρ| > 0.79 — extremely strong associations.\n", + "- **n=7.** Seven data points cannot support inferential statistics. All correlations, patterns, and hypotheses are exploratory. A significant Spearman ρ at n=7 requires |ρ| > 0.79, extremely strong associations.\n", "\n", "- **Ecological fallacy.** Course-level averages can mask student-level variation. A course with high average engagement might have a bimodal distribution: many highly engaged students and many who never logged in.\n", "\n", @@ -728,13 +735,13 @@ "\n", "### What we learned\n", "\n", - "1. **Completion rates vary substantially across modules** — the gap between best and worst performers is meaningful and consistent across presentations. This variation is not noise.\n", + "1. **Completion rates vary substantially across modules**: the gap between best and worst performers is meaningful and consistent across presentations. This variation is not noise.\n", "\n", "2. **Course design features show suggestive patterns** with retention, but with only 7 observations, these are hypotheses rather than findings. Assessment density and resource diversity emerge as candidates for further investigation.\n", "\n", "3. **Engagement intensity differs by course.** Some modules generate more VLE interaction than others. Since early engagement is the strongest student-level predictor of completion (BQ2), course designs that promote higher initial engagement may indirectly support retention.\n", "\n", - "4. **The heatmap reveals course profiles** — modules differ not just in completion rate but in their overall design and engagement characteristics. High-retention courses may share certain design patterns.\n", + "4. **The heatmap reveals course profiles**: modules differ not just in completion rate but in their overall design and engagement characteristics. High-retention courses may share certain design patterns.\n", "\n", "5. **No causal claims are possible** at this sample size. The patterns observed here are inputs to the recommendations in BQ5, not standalone evidence.\n", "\n", @@ -744,7 +751,7 @@ "|----------|-------------------|-------------------------------------------------|\n", "| **07** | BQ5 | Top 3 actionable interventions for retention |\n", "\n", - "NB07 synthesizes findings from all five business questions (BQ1–BQ4) into concrete, prioritized recommendations for a platform operator — the final analytical output of this project.\n", + "NB07 synthesizes findings from all five business questions (BQ1–BQ4) into concrete, prioritized recommendations for a platform operator, the final analytical output of this project.\n", "\n", "---\n", "\n", diff --git a/notebooks/07_bq5_recommendations_synthesis.ipynb b/notebooks/07_bq5_recommendations_synthesis.ipynb index 1916fc5..6af5f04 100644 --- a/notebooks/07_bq5_recommendations_synthesis.ipynb +++ b/notebooks/07_bq5_recommendations_synthesis.ipynb @@ -5,7 +5,7 @@ "id": "0", "metadata": {}, "source": [ - "# 07 — BQ5: Top 3 Actionable Interventions for Student Retention\n", + "# 07. BQ5: Top 3 Actionable Interventions for Student Retention\n", "\n", "> **Notebook 07 of 7** | Learning Retention Analytics \n", "> Fifth business question: synthesizing BQ1–BQ4 findings into concrete, prioritized recommendations for a platform operator." @@ -18,7 +18,7 @@ "source": [ "## Purpose and Scope\n", "\n", - "This notebook answers **BQ5: What are the top 3 actionable interventions for a platform operator?** — the final business question and the capstone of this project.\n", + "This notebook answers **BQ5: What are the top 3 actionable interventions for a platform operator?**, the final business question and the capstone of this project.\n", "\n", "Notebooks 03–06 established *what happens* (BQ1: dropout timing), *what predicts it* (BQ2: early signals), *whether demographics or behavior matters more* (BQ3: behavior wins), and *how courses differ* (BQ4: course design vs retention). This notebook **transforms those findings into action**.\n", "\n", @@ -30,17 +30,17 @@ "- Proposing a phased implementation roadmap\n", "\n", "**What this notebook does NOT do:**\n", - "- No new statistical testing — all evidence comes from BQ1–BQ4\n", - "- No causal claims — impact estimates are projections, not measured effects\n", - "- No A/B test design — that is the natural next step after deploying interventions\n", + "- No new statistical testing: all evidence comes from BQ1–BQ4\n", + "- No causal claims: impact estimates are projections, not measured effects\n", + "- No A/B test design: that is the natural next step after deploying interventions\n", "\n", "**What came before:**\n", - "- **NB03** (BQ1): where and when students drop out — dropout curves, cliff detection\n", - "- **NB04** (BQ2): early behavioral signals that predict dropout — effect sizes, dose-response\n", - "- **NB05** (BQ3): demographics vs behavior — behavior predicts outcome 2–5× more strongly\n", - "- **NB06** (BQ4): course design vs retention — descriptive course profiles, exploratory correlations\n", + "- **NB03** (BQ1): where and when students drop out: dropout curves, cliff detection\n", + "- **NB04** (BQ2): early behavioral signals that predict dropout: effect sizes, dose-response\n", + "- **NB05** (BQ3): demographics vs behavior: behavior predicts outcome 2–5× more strongly\n", + "- **NB06** (BQ4): course design vs retention: descriptive course profiles, exploratory correlations\n", "\n", - "> **Methodological transferability:** This synthesis pattern — segment sizing → intervention design → impact estimation → prioritization — is the standard \"churn intervention playbook\" in SaaS product analytics. The three segments (ghost users, feature non-adopters, early disengagers) map directly to subscription churn, fitness app retention, and freemium conversion contexts." + "> **Methodological transferability:** This synthesis pattern (segment sizing → intervention design → impact estimation → prioritization) is the standard \"churn intervention playbook\" in SaaS product analytics. The three segments (ghost users, feature non-adopters, early disengagers) map directly to subscription churn, fitness app retention, and freemium conversion contexts." ] }, { @@ -51,11 +51,11 @@ "## Table of Contents\n", "\n", "1. [Environment Setup](#1.-Environment-Setup)\n", - "2. [Segment Sizing — The Three Target Populations](#2.-Segment-Sizing-—-The-Three-Target-Populations)\n", + "2. [Segment Sizing: The Three Target Populations](#2.-Segment-Sizing:-The-Three-Target-Populations)\n", "3. [Segment Overlap](#3.-Segment-Overlap)\n", - "4. [Recommendation 1 — Ghost Student Activation](#4.-Recommendation-1-—-Ghost-Student-Activation)\n", - "5. [Recommendation 2 — First Assessment Checkpoint](#5.-Recommendation-2-—-First-Assessment-Checkpoint)\n", - "6. [Recommendation 3 — Week 3 Re-engagement Campaign](#6.-Recommendation-3-—-Week-3-Re-engagement-Campaign)\n", + "4. [Recommendation 1: Ghost Student Activation](#4.-Recommendation-1:-Ghost-Student-Activation)\n", + "5. [Recommendation 2: First Assessment Checkpoint](#5.-Recommendation-2:-First-Assessment-Checkpoint)\n", + "6. [Recommendation 3: Week 3 Re-engagement Campaign](#6.-Recommendation-3:-Week-3-Re-engagement-Campaign)\n", "7. [Priority Matrix](#7.-Priority-Matrix)\n", "8. [Implementation Roadmap](#8.-Implementation-Roadmap)\n", "9. [Limitations and Caveats](#9.-Limitations-and-Caveats)\n", @@ -67,7 +67,7 @@ "- The ETL pipeline must have been run: `python -m run_pipeline`\n", "- The DuckDB database at `data/db/oulad.duckdb` must contain all 5 analytical views\n", "\n", - "**Dataset:** Open University Learning Analytics Dataset (OULAD) — ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." + "**Dataset:** Open University Learning Analytics Dataset (OULAD), ~32K students, 7 courses, complete behavioral clickstream. License: CC-BY 4.0." ] }, { @@ -80,10 +80,10 @@ "We configure imports, visualization defaults, and reusable helper functions.\n", "\n", "**Technical notes for readers:**\n", - "- All database queries go through `src.db.connection.execute_query()` — the project's DB abstraction layer (ADR-003).\n", + "- All database queries go through `src.db.connection.execute_query()`, the project's DB abstraction layer (ADR-003).\n", "- BQ5's primary SQL query lives in `sql/queries/q_bq5_segment_sizing.sql` and is loaded at runtime from disk. It sizes three intervention segments from `v_student_enriched` and `v_engagement_early`.\n", "- Additional inline SQL queries compute impact estimates for each recommendation. These are specific to this notebook's synthesis narrative and not reusable as standalone queries (consistent with the inline query pattern in NB03).\n", - "- No statistical test imports — this is a synthesis notebook, all evidence comes from NB03–NB06.\n", + "- No statistical test imports: this is a synthesis notebook, all evidence comes from NB03–NB06.\n", "- Figures are saved to `reports/figures/` at 150 DPI." ] }, @@ -148,7 +148,7 @@ "}\n", "SEGMENT_ORDER = list(PALETTE_SEGMENT.keys())\n", "\n", - "# Shared axis labels and section headers — defined as constants to avoid\n", + "# Shared axis labels and section headers: defined as constants to avoid\n", "# duplicated string literals flagged by static analysis\n", "LABEL_COMPLETION_RATE = 'Completion rate (%)'\n", "LABEL_NON_COMPLETION_RATE = 'Non-completion rate (%)'\n", @@ -199,9 +199,9 @@ "id": "5", "metadata": {}, "source": [ - "## 2. Segment Sizing — The Three Target Populations\n", + "## 2. Segment Sizing: The Three Target Populations\n", "\n", - "The BQ5 query (`q_bq5_segment_sizing.sql`) sizes three student segments defined by **observable, actionable criteria** — not demographics. Each segment represents a group where a targeted intervention could reduce non-completion:\n", + "The BQ5 query (`q_bq5_segment_sizing.sql`) sizes three student segments defined by **observable, actionable criteria**, not demographics. Each segment represents a group where a targeted intervention could reduce non-completion:\n", "\n", "| Segment | Definition | Rationale |\n", "|---------|-----------|-----------|\n", @@ -316,7 +316,7 @@ "sns.despine(ax=ax2)\n", "\n", "fig.suptitle(\n", - " 'BQ5 Segment Sizing — Who Should We Target?\\n'\n", + " 'BQ5 Segment Sizing: Who Should We Target?\\n'\n", " f'(total enrollments: {total_students:,})',\n", " fontsize=14, y=1.02,\n", ")\n", @@ -330,7 +330,7 @@ "id": "8", "metadata": {}, "source": [ - "> **Segment profile:** All three segments show non-completion rates substantially above the overall baseline. The gap between each segment's rate and the platform average quantifies the \"excess non-completion\" — the portion potentially addressable through targeted intervention.\n", + "> **Segment profile:** All three segments show non-completion rates substantially above the overall baseline. The gap between each segment's rate and the platform average quantifies the \"excess non-completion\": the portion potentially addressable through targeted intervention.\n", ">\n", "> **Note:** These segments are not mutually exclusive. A ghost student who never accessed the VLE almost certainly did not submit an assessment either. The next section quantifies this overlap to avoid double-counting when estimating aggregate impact." ] @@ -343,7 +343,7 @@ "## 3. Segment Overlap\n", "\n", "Students can belong to multiple segments simultaneously. Understanding the overlap is critical for two reasons:\n", - "1. **Impact estimation:** If most ghost students are also non-submitters, the two interventions target largely the same people — their impact should not be summed naively.\n", + "1. **Impact estimation:** If most ghost students are also non-submitters, the two interventions target largely the same people; their impact should not be summed naively.\n", "2. **Intervention sequencing:** A student in multiple segments would receive multiple interventions; we should prioritize which one they get first.\n", "\n", "The query below reproduces the BQ5 segment classification at the student level (instead of aggregating to a single row) and counts pairwise and three-way intersections." @@ -525,7 +525,7 @@ "ax.set_yticklabels(overlap_data['category'].values)\n", "ax.set_xlabel(LABEL_NUM_STUDENTS)\n", "ax.set_title(\n", - " 'Segment Overlap — Exclusive and Shared Membership\\n'\n", + " 'Segment Overlap: Exclusive and Shared Membership\\n'\n", " '(gray bars = students in multiple segments)'\n", ")\n", "sns.despine()\n", @@ -550,7 +550,7 @@ "id": "12", "metadata": {}, "source": [ - "> **Overlap insight:** Ghost students and assessment non-submitters overlap heavily — a student who never accesses the VLE cannot submit an assessment. This means **Recommendations 1 and 2 largely target the same population** from different angles. Early disengagers, by definition, had *some* initial activity, so they overlap less with ghost students. This makes Recommendation 3 an independent intervention targeting a different failure mode.\n", + "> **Overlap insight:** Ghost students and assessment non-submitters overlap heavily: a student who never accesses the VLE cannot submit an assessment. This means **Recommendations 1 and 2 largely target the same population** from different angles. Early disengagers, by definition, had *some* initial activity, so they overlap less with ghost students. This makes Recommendation 3 an independent intervention targeting a different failure mode.\n", ">\n", "> **Implication for impact estimation:** Because of the ghost–non-submitter overlap, the combined impact of all three interventions is **less than the sum of individual impacts**. Each recommendation section below estimates impact independently; the Priority Matrix (Section 7) presents those independent estimates and includes overlap as an interpretive consideration when reading combined totals." ] @@ -560,7 +560,7 @@ "id": "13", "metadata": {}, "source": [ - "## 4. Recommendation 1 — Ghost Student Activation\n", + "## 4. Recommendation 1: Ghost Student Activation\n", "\n", "### The Problem\n", "\n", @@ -570,7 +570,7 @@ "\n", "- **BQ2 (NB04):** Early engagement (active days, total clicks in first 28 days) has the largest effect size among all behavioral predictors. Students with zero early activity have near-zero completion rates.\n", "- **BQ3 (NB05):** Behavior predicts outcome 2–5× more strongly than demographics. This means we should target *what students do* (or fail to do), not *who they are*.\n", - "- **BQ1 (NB03):** A significant fraction of withdrawals happen in the first two weeks — before most students have even established a study routine.\n", + "- **BQ1 (NB03):** A significant fraction of withdrawals happen in the first two weeks, before most students have even established a study routine.\n", "\n", "### Proposed Intervention\n", "\n", @@ -579,7 +579,7 @@ "| **Trigger** | Student has zero VLE activity by day 3 of the course |\n", "| **Action** | Automated activation sequence: day-3 welcome email with \"first step\" link to a single easy resource, day-7 follow-up if still inactive |\n", "| **Channel** | Email + in-platform notification |\n", - "| **Cost** | **Low** — email automation only, no platform changes required |\n", + "| **Cost** | **Low**: email automation only, no platform changes required |\n", "\n", "### Impact Estimation" ] @@ -657,7 +657,7 @@ "source": [ "> **Interpretation:** The completion rate gap between ghost and active students is substantial. Even a modest activation rate (20%) would produce meaningful additional completions because the segment is large and the gap is wide.\n", ">\n", - "> **Assumption transparency:** The scenario assumes converted students achieve the *platform average* completion rate — a conservative estimate. In practice, late-activating students may perform worse than those who started on time. This would reduce the estimated impact." + "> **Assumption transparency:** The scenario assumes converted students achieve the *platform average* completion rate, a conservative estimate. In practice, late-activating students may perform worse than those who started on time. This would reduce the estimated impact." ] }, { @@ -665,11 +665,11 @@ "id": "16", "metadata": {}, "source": [ - "## 5. Recommendation 2 — First Assessment Checkpoint\n", + "## 5. Recommendation 2: First Assessment Checkpoint\n", "\n", "### The Problem\n", "\n", - "Students who miss the first assessment deadline are signaling disengagement. Missing this early milestone breaks the feedback loop that keeps students connected to the course — they lose the sense of progress that early assessment provides.\n", + "Students who miss the first assessment deadline are signaling disengagement. Missing this early milestone breaks the feedback loop that keeps students connected to the course: they lose the sense of progress that early assessment provides.\n", "\n", "### Evidence from BQ1–BQ4\n", "\n", @@ -682,9 +682,9 @@ "| Element | Details |\n", "|---------|---------|\n", "| **Trigger** | First assessment deadline is 3 days away and student has not submitted |\n", - "| **Action** | Automated reminder with assessment preview: \"Here's what to expect — the first assessment covers X and takes approximately Y minutes\" |\n", + "| **Action** | Automated reminder with assessment preview: \"Here's what to expect: the first assessment covers X and takes approximately Y minutes\" |\n", "| **Channel** | Email + in-platform notification + optional SMS |\n", - "| **Cost** | **Medium** — requires deadline-aware notification system integrated with course calendar |\n", + "| **Cost** | **Medium**: requires deadline-aware notification system integrated with course calendar |\n", "\n", "### Impact Estimation" ] @@ -758,7 +758,7 @@ "source": [ "> **Interpretation:** The gap between submitters and non-submitters is stark. Assessment submission is both a *signal* (it reveals commitment) and a *mechanism* (it creates accountability). This dual nature makes it an ideal intervention point.\n", ">\n", - "> **Caution on causality:** Submitting the assessment may not *cause* completion — both may be driven by an underlying motivation factor. The intervention works only if the reminder nudges students who *would have submitted* with a small push, not if it forces reluctant students through a gate. This is why the conversion rate assumptions are conservative (10–25%)." + "> **Caution on causality:** Submitting the assessment may not *cause* completion: both may be driven by an underlying motivation factor. The intervention works only if the reminder nudges students who *would have submitted* with a small push, not if it forces reluctant students through a gate. This is why the conversion rate assumptions are conservative (10–25%)." ] }, { @@ -766,16 +766,16 @@ "id": "19", "metadata": {}, "source": [ - "## 6. Recommendation 3 — Week 3 Re-engagement Campaign\n", + "## 6. Recommendation 3: Week 3 Re-engagement Campaign\n", "\n", "### The Problem\n", "\n", - "Early disengagers are students who *started* the course — they had VLE activity in the first two weeks — but then stopped entirely in weeks 3–4. Unlike ghost students who never began, these students demonstrated initial interest but lost momentum.\n", + "Early disengagers are students who *started* the course (they had VLE activity in the first two weeks) but then stopped entirely in weeks 3–4. Unlike ghost students who never began, these students demonstrated initial interest but lost momentum.\n", "\n", "### Evidence from BQ1–BQ4\n", "\n", "- **BQ1 (NB03):** Dropout curves show mid-course cliffs that often align with assessment deadlines or the transition from introductory to core content. Weeks 3–4 are a critical inflection point.\n", - "- **BQ2 (NB04):** `last_active_day_in_window` is a significant predictor — students whose last activity is early in the window are unlikely to complete.\n", + "- **BQ2 (NB04):** `last_active_day_in_window` is a significant predictor: students whose last activity is early in the window are unlikely to complete.\n", "- **BQ3 (NB05):** The behavioral signal (activity drop) is more predictive than any demographic factor. The intervention should target the behavior, not the student's profile.\n", "\n", "### Proposed Intervention\n", @@ -785,7 +785,7 @@ "| **Trigger** | Student had ≥1 active day in days 0–14 but zero activity in days 15–17 (3-day inactivity after initial engagement) |\n", "| **Action** | \"We miss you\" email at day 18: personalized progress summary (\"You completed X% of week 1–2 activities\"), social proof (\"Students like you who re-engaged at this point completed the course Y% of the time\"), and a direct link to the next resource |\n", "| **Channel** | Email + in-platform notification |\n", - "| **Cost** | **Medium-High** — requires real-time activity tracking pipeline and personalized message generation |\n", + "| **Cost** | **Medium-High**: requires real-time activity tracking pipeline and personalized message generation |\n", "\n", "### Impact Estimation" ] @@ -876,9 +876,9 @@ "id": "21", "metadata": {}, "source": [ - "> **Interpretation:** Early disengagers have already demonstrated willingness to engage — they are not ghost students. This means a re-engagement nudge has a plausible mechanism: reminding someone who *was* active to come back. The impact estimate uses a conservative target (halfway between disengaged and sustained rates) because re-engagement after a gap is harder than sustained momentum.\n", + "> **Interpretation:** Early disengagers have already demonstrated willingness to engage: they are not ghost students. This means a re-engagement nudge has a plausible mechanism: reminding someone who *was* active to come back. The impact estimate uses a conservative target (halfway between disengaged and sustained rates) because re-engagement after a gap is harder than sustained momentum.\n", ">\n", - "> **Why this is the most complex intervention:** Unlike simple email automation (Rec 1) or deadline-aware reminders (Rec 2), this intervention requires tracking individual activity patterns in near-real-time and generating personalized messages. The higher implementation cost is justified only if the segment is large enough — which the BQ5 sizing confirms." + "> **Why this is the most complex intervention:** Unlike simple email automation (Rec 1) or deadline-aware reminders (Rec 2), this intervention requires tracking individual activity patterns in near-real-time and generating personalized messages. The higher implementation cost is justified only if the segment is large enough, which the BQ5 sizing confirms." ] }, { @@ -1004,7 +1004,7 @@ "ax.set_xlabel('Implementation Cost')\n", "ax.set_ylabel('Estimated Additional Completions\\n(middle scenario)')\n", "ax.set_title(\n", - " 'Priority Matrix — Impact vs Cost\\n'\n", + " 'Priority Matrix: Impact vs Cost\\n'\n", " '(bubble size = segment size)'\n", ")\n", "ax.set_xlim(0.4, 3.6)\n", @@ -1021,9 +1021,9 @@ "id": "25", "metadata": {}, "source": [ - "> **Reading the matrix:** The ideal intervention sits in the top-left corner — high impact, low cost. Ghost Student Activation is the clear \"quick win\": it targets the largest segment with the highest non-completion rate, and requires only email automation. The First Assessment Checkpoint offers solid impact at moderate cost. The Week 3 Re-engagement Campaign has the highest implementation complexity but targets a distinct population (not overlapping with ghosts), making it a valuable addition.\n", + "> **Reading the matrix:** The ideal intervention sits in the top-left corner: high impact, low cost. Ghost Student Activation is the clear \"quick win\": it targets the largest segment with the highest non-completion rate, and requires only email automation. The First Assessment Checkpoint offers solid impact at moderate cost. The Week 3 Re-engagement Campaign has the highest implementation complexity but targets a distinct population (not overlapping with ghosts), making it a valuable addition.\n", ">\n", - "> **Priority ranking:** (1) Ghost Activation — start here, (2) Assessment Checkpoint — build next, (3) Re-engagement Campaign — invest when infrastructure is ready." + "> **Priority ranking:** (1) Ghost Activation (start here), (2) Assessment Checkpoint (build next), (3) Re-engagement Campaign (invest when infrastructure is ready)." ] }, { @@ -1035,7 +1035,7 @@ "\n", "A phased rollout minimizes risk and allows each intervention to be validated before investing in the next.\n", "\n", - "### Phase 1: Weeks 1–2 — Ghost Student Activation (Quick Win)\n", + "### Phase 1 (Weeks 1–2): Ghost Student Activation (Quick Win)\n", "\n", "| Step | Action | Owner |\n", "|------|--------|-------|\n", @@ -1047,7 +1047,7 @@ "\n", "**Success metric:** ≥15% of targeted ghost students access the VLE within 7 days of receiving the email.\n", "\n", - "### Phase 2: Weeks 3–4 — First Assessment Checkpoint\n", + "### Phase 2 (Weeks 3–4): First Assessment Checkpoint\n", "\n", "| Step | Action | Owner |\n", "|------|--------|-------|\n", @@ -1059,7 +1059,7 @@ "\n", "**Success metric:** ≥10% increase in first assessment submission rate among targeted students.\n", "\n", - "### Phase 3: Weeks 5–8 — Week 3 Re-engagement Campaign\n", + "### Phase 3 (Weeks 5–8): Week 3 Re-engagement Campaign\n", "\n", "| Step | Action | Owner |\n", "|------|--------|-------|\n", @@ -1073,7 +1073,7 @@ "\n", "### Cross-cutting Principle\n", "\n", - "All three interventions target **behavior, not demographics** — consistent with BQ3's finding that behavioral signals are stronger predictors. This is both more effective (targeting the actionable variable) and more ethical (avoiding demographic profiling)." + "All three interventions target **behavior, not demographics**, consistent with BQ3's finding that behavioral signals are stronger predictors. This is both more effective (targeting the actionable variable) and more ethical (avoiding demographic profiling)." ] }, { @@ -1099,7 +1099,7 @@ "\n", "### Ethical considerations\n", "\n", - "- All interventions target behavior, not demographics — no student is profiled by gender, age, disability, or socioeconomic status.\n", + "- All interventions target behavior, not demographics: no student is profiled by gender, age, disability, or socioeconomic status.\n", "- Automated interventions should include an opt-out mechanism to respect student autonomy.\n", "- \"Ghost students\" may have valid reasons for not engaging (changed circumstances, enrolled in error). The intervention should inform, not pressure." ] @@ -1123,11 +1123,11 @@ "\n", "| NB | BQ | Key Finding |\n", "|----|-----|-------------|\n", - "| 03 | BQ1 | Dropout is not uniform — it concentrates in cliffs at specific course milestones. Pre-course withdrawal is a significant fraction. |\n", + "| 03 | BQ1 | Dropout is not uniform: it concentrates in cliffs at specific course milestones. Pre-course withdrawal is a significant fraction. |\n", "| 04 | BQ2 | Early behavioral signals (active days, clicks, assessment submission) predict dropout with large effect sizes. The first 28 days are the critical window. |\n", "| 05 | BQ3 | Behavior predicts outcome 2–5× more strongly than demographics. Interventions should target what students *do*, not who they *are*. |\n", "| 06 | BQ4 | Completion rates vary substantially across courses. Course design features (assessment density, resource diversity) show suggestive associations with retention. |\n", - "| 07 | BQ5 | Three actionable interventions — ghost activation, assessment checkpoint, re-engagement campaign — target the largest at-risk segments with data-estimated impact. |\n", + "| 07 | BQ5 | Three actionable interventions (ghost activation, assessment checkpoint, re-engagement campaign) target the largest at-risk segments with data-estimated impact. |\n", "\n", "### The Analytical Pipeline Is Complete\n", "\n", @@ -1143,7 +1143,7 @@ "id": "29", "metadata": {}, "source": [ - "> **From analysis to action:** This project started with a dataset and five questions. Seven notebooks later, we have a complete picture: when students leave, what predicts it, what matters more (behavior), how courses differ, and — most importantly — what a platform operator can do about it. The three interventions proposed here are not speculative: they are sized by real data, supported by statistical evidence, and ranked by feasibility.\n", + "> **From analysis to action:** This project started with a dataset and five questions. Seven notebooks later, we have a complete picture: when students leave, what predicts it, what matters more (behavior), how courses differ, and, most importantly, what a platform operator can do about it. The three interventions proposed here are not speculative: they are sized by real data, supported by statistical evidence, and ranked by feasibility.\n", ">\n", "> The gap between analysis and impact is an A/B test. That is the next step." ] diff --git a/reports/REPORT.md b/reports/REPORT.md index b03873e..f30cad2 100644 --- a/reports/REPORT.md +++ b/reports/REPORT.md @@ -10,7 +10,8 @@ This report synthesizes findings from a SQL-driven analytical pipeline applied to the OULAD dataset: 32,593 student-course enrollments across 7 modules, with complete -behavioral clickstream, assessment records, and demographic profiles. +behavioral clickstream from the university's Virtual Learning Environment (VLE), +assessment records, and demographic profiles. **Outcome definition:** Each enrollment is classified as *Completed* (Pass or Distinction) or *Not completed* (Fail or Withdrawn). This binary split is consistent with the OULAD @@ -138,8 +139,8 @@ These behavioral signals are strong. But are they merely proxies for demographic > than demographic effect sizes. Within every education level, high engagement > dramatically outperforms low engagement. -We tested 6 categorical demographic features (gender, age band, education level, IMD -band, disability, region) and 2 numeric demographic features (previous attempts, studied +We tested 6 categorical demographic features (gender, age band, education level, +Index of Multiple Deprivation (IMD) band, disability, region) and 2 numeric demographic features (previous attempts, studied credits) against completion outcome. All 8 are statistically significant after Benjamini-Hochberg correction, but their effect sizes are uniformly weak. The strongest demographic predictor (highest education) reaches a Cramer's V of approximately **0.15**; diff --git a/reports/figures/02_course_engagement_boxplot.png b/reports/figures/02_course_engagement_boxplot.png index 6b98f94..9c9e15e 100644 Binary files a/reports/figures/02_course_engagement_boxplot.png and b/reports/figures/02_course_engagement_boxplot.png differ diff --git a/reports/figures/02_decile_completion_rate.png b/reports/figures/02_decile_completion_rate.png index 9e55c97..a86aa22 100644 Binary files a/reports/figures/02_decile_completion_rate.png and b/reports/figures/02_decile_completion_rate.png differ diff --git a/reports/figures/02_decile_outcome_stacked.png b/reports/figures/02_decile_outcome_stacked.png index d2acda6..959a552 100644 Binary files a/reports/figures/02_decile_outcome_stacked.png and b/reports/figures/02_decile_outcome_stacked.png differ diff --git a/reports/figures/02_engagement_typology_scatter.png b/reports/figures/02_engagement_typology_scatter.png index cb1e8ee..2b6f46b 100644 Binary files a/reports/figures/02_engagement_typology_scatter.png and b/reports/figures/02_engagement_typology_scatter.png differ diff --git a/reports/figures/02_ghost_outcome_distribution.png b/reports/figures/02_ghost_outcome_distribution.png index 4078f56..aad6b13 100644 Binary files a/reports/figures/02_ghost_outcome_distribution.png and b/reports/figures/02_ghost_outcome_distribution.png differ diff --git a/reports/figures/03_course_design_vs_dropout.png b/reports/figures/03_course_design_vs_dropout.png index 92989d7..9b7ad5d 100644 Binary files a/reports/figures/03_course_design_vs_dropout.png and b/reports/figures/03_course_design_vs_dropout.png differ diff --git a/reports/figures/03_dropout_by_demographics.png b/reports/figures/03_dropout_by_demographics.png index abefb43..bad2ca9 100644 Binary files a/reports/figures/03_dropout_by_demographics.png and b/reports/figures/03_dropout_by_demographics.png differ diff --git a/reports/figures/03_dropout_cliffs.png b/reports/figures/03_dropout_cliffs.png index cf2485e..4dc0850 100644 Binary files a/reports/figures/03_dropout_cliffs.png and b/reports/figures/03_dropout_cliffs.png differ diff --git a/reports/figures/03_dropout_curves_overlaid.png b/reports/figures/03_dropout_curves_overlaid.png index 91aabec..397eeb9 100644 Binary files a/reports/figures/03_dropout_curves_overlaid.png and b/reports/figures/03_dropout_curves_overlaid.png differ diff --git a/reports/figures/03_dropout_normalized.png b/reports/figures/03_dropout_normalized.png index 4915f0e..c162c0a 100644 Binary files a/reports/figures/03_dropout_normalized.png and b/reports/figures/03_dropout_normalized.png differ diff --git a/reports/figures/05_behavior_effect_sizes.png b/reports/figures/05_behavior_effect_sizes.png index a237ffe..897526e 100644 Binary files a/reports/figures/05_behavior_effect_sizes.png and b/reports/figures/05_behavior_effect_sizes.png differ diff --git a/reports/figures/05_demographic_completion_rates.png b/reports/figures/05_demographic_completion_rates.png index 9ac4506..4f6cdad 100644 Binary files a/reports/figures/05_demographic_completion_rates.png and b/reports/figures/05_demographic_completion_rates.png differ diff --git a/reports/figures/05_demographics_vs_behavior_comparison.png b/reports/figures/05_demographics_vs_behavior_comparison.png index d5fec17..18872ad 100644 Binary files a/reports/figures/05_demographics_vs_behavior_comparison.png and b/reports/figures/05_demographics_vs_behavior_comparison.png differ diff --git a/reports/figures/06_course_completion_ranking.png b/reports/figures/06_course_completion_ranking.png index 1a23142..6c66018 100644 Binary files a/reports/figures/06_course_completion_ranking.png and b/reports/figures/06_course_completion_ranking.png differ diff --git a/reports/figures/06_course_design_heatmap.png b/reports/figures/06_course_design_heatmap.png index 0c7beb0..24c600e 100644 Binary files a/reports/figures/06_course_design_heatmap.png and b/reports/figures/06_course_design_heatmap.png differ diff --git a/reports/figures/06_engagement_by_course.png b/reports/figures/06_engagement_by_course.png index 0743287..6fe3549 100644 Binary files a/reports/figures/06_engagement_by_course.png and b/reports/figures/06_engagement_by_course.png differ diff --git a/reports/figures/07_priority_matrix.png b/reports/figures/07_priority_matrix.png index 8687e29..0a32de2 100644 Binary files a/reports/figures/07_priority_matrix.png and b/reports/figures/07_priority_matrix.png differ diff --git a/reports/figures/07_segment_overlap.png b/reports/figures/07_segment_overlap.png index 0210b84..d7e0470 100644 Binary files a/reports/figures/07_segment_overlap.png and b/reports/figures/07_segment_overlap.png differ diff --git a/reports/figures/07_segment_sizing_overview.png b/reports/figures/07_segment_sizing_overview.png index 5234a2f..bedc9ee 100644 Binary files a/reports/figures/07_segment_sizing_overview.png and b/reports/figures/07_segment_sizing_overview.png differ diff --git a/reports/it/REPORT.md b/reports/it/REPORT.md index 983216c..3ca9032 100644 --- a/reports/it/REPORT.md +++ b/reports/it/REPORT.md @@ -10,7 +10,8 @@ Questo report sintetizza i risultati di una pipeline analitica SQL-driven applicata al dataset OULAD: 32.593 iscrizioni studente-corso distribuite su 7 moduli, con clickstream -comportamentale completo, record delle valutazioni e profili demografici. +comportamentale completo dal Virtual Learning Environment (VLE) dell'università, record +delle valutazioni e profili demografici. **Definizione dell'outcome:** Ogni iscrizione è classificata come *Completato* (Pass o Distinction) o *Non completato* (Fail o Withdrawn). Questa suddivisione binaria è @@ -145,7 +146,8 @@ Questi segnali comportamentali sono forti. Ma sono semplicemente proxy della dem > l'engagement alto supera drammaticamente l'engagement basso. Abbiamo testato 6 variabili demografiche categoriche (genere, fascia d'età, livello di -istruzione, fascia IMD, disabilità, regione) e 2 variabili demografiche numeriche +istruzione, fascia IMD (Index of Multiple Deprivation), disabilità, regione) e 2 +variabili demografiche numeriche (tentativi precedenti, crediti studiati) contro l'esito di completamento. Tutte le 8 sono statisticamente significative dopo correzione Benjamini-Hochberg, ma i loro effect size sono uniformemente deboli. Il predittore demografico più forte (livello di istruzione diff --git a/run_pipeline.py b/run_pipeline.py index d99a2a9..d5b6ebb 100644 --- a/run_pipeline.py +++ b/run_pipeline.py @@ -1,4 +1,4 @@ -"""Pipeline orchestrator — runs all ETL steps, or a single one via --step. +"""Pipeline orchestrator: runs all ETL steps, or a single one via --step. Entry point for the entire analytical pipeline: python -m run_pipeline # all steps, full dataset @@ -29,7 +29,7 @@ def main() -> None: """Parse arguments and run the full pipeline.""" parser = argparse.ArgumentParser( - description="Learning Retention Analytics — ETL Pipeline" + description="Learning Retention Analytics: ETL Pipeline" ) parser.add_argument( "--step", @@ -62,21 +62,21 @@ def main() -> None: run_step: str | None = args.step logger.info( - "Pipeline starting — source: %s, step: %s", + "Pipeline starting (source: %s, step: %s)", "data_sample" if args.sample else "data/raw", run_step or "all", ) if run_step in (None, "ingest"): - with step_timer("Step 01 — Ingest"): + with step_timer("Step 01: Ingest"): ingest(use_sample=args.sample) if run_step in (None, "transform"): - with step_timer("Step 02 — Transform"): + with step_timer("Step 02: Transform"): transform() if run_step in (None, "export"): - with step_timer("Step 03 — Export"): + with step_timer("Step 03: Export"): export() if run_step in (None, "stats"): diff --git a/sql/queries/q_bq1_dropout_curves.sql b/sql/queries/q_bq1_dropout_curves.sql index ca252a0..d3e9861 100644 --- a/sql/queries/q_bq1_dropout_curves.sql +++ b/sql/queries/q_bq1_dropout_curves.sql @@ -1,4 +1,4 @@ --- q_bq1_dropout_curves — BQ1: Where and when do students drop out? +-- q_bq1_dropout_curves (BQ1): Where and when do students drop out? -- -- Computes cumulative dropout curves per course-presentation. -- Each row represents a day when at least one dropout occurred, diff --git a/sql/queries/q_bq2_early_signals.sql b/sql/queries/q_bq2_early_signals.sql index 988f3a3..8daf981 100644 --- a/sql/queries/q_bq2_early_signals.sql +++ b/sql/queries/q_bq2_early_signals.sql @@ -1,4 +1,4 @@ --- q_bq2_early_signals — BQ2: Which early behavioral signals predict drop-out? +-- q_bq2_early_signals (BQ2): Which early behavioral signals predict drop-out? -- -- Joins early engagement metrics (first 28 days) with student outcomes -- to create a dataset ready for statistical testing in Python. @@ -8,7 +8,7 @@ -- then runs t-tests comparing completed=1 vs completed=0 groups on each -- engagement metric, with effect sizes and multiple comparison correction. -- --- This query intentionally does NOT perform the statistical tests in SQL — +-- This query intentionally does NOT perform the statistical tests in SQL; -- those require scipy/statsmodels and are better handled in Python. -- SQL's role here is to prepare the clean, joined dataset. @@ -37,7 +37,7 @@ SELECT FROM v_student_enriched se -- LEFT JOIN because some students may have zero VLE activity in the first 28 days --- (they enrolled but never clicked anything — a strong dropout signal itself) +-- (they enrolled but never clicked anything, a strong dropout signal itself) LEFT JOIN v_engagement_early ee ON se.id_student = ee.id_student AND se.code_module = ee.code_module diff --git a/sql/queries/q_bq3_demographics_vs_behavior.sql b/sql/queries/q_bq3_demographics_vs_behavior.sql index 2ce7dbd..4f9602f 100644 --- a/sql/queries/q_bq3_demographics_vs_behavior.sql +++ b/sql/queries/q_bq3_demographics_vs_behavior.sql @@ -1,4 +1,4 @@ --- q_bq3_demographics_vs_behavior — BQ3: Demographics vs behavior as outcome predictors +-- q_bq3_demographics_vs_behavior (BQ3): Demographics vs behavior as outcome predictors -- -- Combines demographic features from studentInfo with behavioral features -- from early engagement to create a dataset for comparative analysis. diff --git a/sql/queries/q_bq4_course_comparison.sql b/sql/queries/q_bq4_course_comparison.sql index a7126eb..d4fa3bf 100644 --- a/sql/queries/q_bq4_course_comparison.sql +++ b/sql/queries/q_bq4_course_comparison.sql @@ -1,4 +1,4 @@ --- q_bq4_course_comparison — BQ4: How do course characteristics affect retention? +-- q_bq4_course_comparison (BQ4): How do course characteristics affect retention? -- -- Aggregates course-level metrics to compare retention across the 7 OULAD modules. -- Each row represents one module (aggregated across all presentations) with: diff --git a/sql/queries/q_bq5_segment_sizing.sql b/sql/queries/q_bq5_segment_sizing.sql index e37912a..b3634a4 100644 --- a/sql/queries/q_bq5_segment_sizing.sql +++ b/sql/queries/q_bq5_segment_sizing.sql @@ -1,13 +1,13 @@ --- q_bq5_segment_sizing — BQ5: Sizing the target segments for interventions +-- q_bq5_segment_sizing (BQ5): Sizing the target segments for interventions -- -- Quantifies the student segments that the top 3 interventions would target. -- Each segment is defined by observable, actionable criteria (not demographics) -- so that a platform operator can implement automated triggers. -- -- The three segments correspond to the expected BQ5 recommendations: --- 1. "Ghost students" — enrolled but zero/minimal VLE activity in week 1-4 --- 2. "Assessment non-submitters" — didn't submit the first assessment --- 3. "Early disengagers" — active initially but activity drops to zero by week 3-4 +-- 1. "Ghost students": enrolled but zero/minimal VLE activity in week 1-4 +-- 2. "Assessment non-submitters": didn't submit the first assessment +-- 3. "Early disengagers": active initially but activity drops to zero by week 3-4 -- -- This query sizes each segment and computes their dropout rates to -- estimate the impact of targeted interventions. diff --git a/sql/schema.sql b/sql/schema.sql index 1244a18..68d3984 100644 --- a/sql/schema.sql +++ b/sql/schema.sql @@ -1,4 +1,4 @@ --- schema.sql — DDL for OULAD raw tables +-- schema.sql: DDL for OULAD raw tables -- -- Defines the 7 raw tables that mirror the OULAD CSV structure. -- All types are kept generic (VARCHAR, INTEGER, DOUBLE) for ANSI compliance. diff --git a/sql/views/v_course_profile.sql b/sql/views/v_course_profile.sql index 429d0b8..3076c02 100644 --- a/sql/views/v_course_profile.sql +++ b/sql/views/v_course_profile.sql @@ -1,4 +1,4 @@ --- v_course_profile — Course-level characteristics and retention metrics +-- v_course_profile: Course-level characteristics and retention metrics -- -- One row per course-presentation with aggregated metrics that characterize -- the course design and student outcomes. Used in BQ4 ("how do course diff --git a/sql/views/v_dropout_timing.sql b/sql/views/v_dropout_timing.sql index 22c5877..d7d6fd0 100644 --- a/sql/views/v_dropout_timing.sql +++ b/sql/views/v_dropout_timing.sql @@ -1,11 +1,11 @@ --- v_dropout_timing — When students drop out relative to course timeline +-- v_dropout_timing: When students drop out relative to course timeline -- -- Enriches each withdrawal event with course context (total duration, -- percentage through the course) to answer BQ1: "where and when do -- students drop out?" -- -- Only includes students who explicitly withdrew (date_unregistration IS NOT NULL). --- Students who failed but never withdrew are excluded — they represent +-- Students who failed but never withdrew are excluded: they represent -- a different phenomenon (academic failure vs. active departure). -- -- The dropout_pct field normalizes timing across courses of different lengths, diff --git a/sql/views/v_engagement_daily.sql b/sql/views/v_engagement_daily.sql index 521323c..113af60 100644 --- a/sql/views/v_engagement_daily.sql +++ b/sql/views/v_engagement_daily.sql @@ -1,4 +1,4 @@ --- v_engagement_daily — Daily engagement aggregates per student +-- v_engagement_daily: Daily engagement aggregates per student -- -- Aggregates the raw clickstream (studentVle) to one row per student per day, -- summing clicks across all VLE resources and counting distinct resource types. diff --git a/sql/views/v_engagement_early.sql b/sql/views/v_engagement_early.sql index 0d91c76..763804f 100644 --- a/sql/views/v_engagement_early.sql +++ b/sql/views/v_engagement_early.sql @@ -1,4 +1,4 @@ --- v_engagement_early — Engagement metrics for the first 28 days (0-28) +-- v_engagement_early: Engagement metrics for the first 28 days (0-28) -- -- Aggregates clickstream data from the first 4 weeks of each course -- to create early behavioral signals per student. This is the key view diff --git a/sql/views/v_student_enriched.sql b/sql/views/v_student_enriched.sql index 74feb94..75197d5 100644 --- a/sql/views/v_student_enriched.sql +++ b/sql/views/v_student_enriched.sql @@ -1,4 +1,4 @@ --- v_student_enriched — Core student-level view with enriched outcome data +-- v_student_enriched: Core student-level view with enriched outcome data -- -- Combines studentInfo (demographics + final result) with studentRegistration -- (enrollment/withdrawal dates) to create a single row per student-module diff --git a/src/config.py b/src/config.py index 9601ce3..a880ce1 100644 --- a/src/config.py +++ b/src/config.py @@ -1,4 +1,4 @@ -"""Centralized configuration — paths, constants, and env vars. +"""Centralized configuration: paths, constants, and env vars. All paths are relative to PROJECT_ROOT. Single env var: PUSH_TO_SHEETS (default false). diff --git a/src/db/connection.py b/src/db/connection.py index acb51de..286a336 100644 --- a/src/db/connection.py +++ b/src/db/connection.py @@ -1,4 +1,4 @@ -"""Database abstraction layer — DuckDB now, BigQuery later. +"""Database abstraction layer: DuckDB now, BigQuery later. All database access in the project MUST go through this module. No direct duckdb.connect() calls elsewhere in the codebase. @@ -38,7 +38,7 @@ def get_connection( duckdb.DuckDBPyConnection """ if db_path is None: - # DuckDB does not support read_only on :memory: connections — + # DuckDB does not support read_only on :memory: connections; # the parameter is silently ignored. This is expected: in-memory # DBs are ephemeral test fixtures, not shared resources. logger.debug("Opening in-memory DuckDB connection") diff --git a/src/pipeline/step_01_ingest.py b/src/pipeline/step_01_ingest.py index 16e2f42..082ceda 100644 --- a/src/pipeline/step_01_ingest.py +++ b/src/pipeline/step_01_ingest.py @@ -1,11 +1,11 @@ -"""Step 01 — Ingest OULAD CSV files into raw DuckDB tables. +"""Step 01: Ingest OULAD CSV files into raw DuckDB tables. Reads the 7 OULAD CSV files from data/raw/ (or data_sample/) and loads them into DuckDB tables defined by sql/schema.sql. This step is idempotent: running it again will DROP and re-CREATE all tables, ensuring a clean state. This is acceptable because raw data is always -available on disk as CSVs — the DuckDB tables are a derived artifact. +available on disk as CSVs; the DuckDB tables are a derived artifact. """ import logging diff --git a/src/pipeline/step_02_transform.py b/src/pipeline/step_02_transform.py index 83b5f85..2ed0004 100644 --- a/src/pipeline/step_02_transform.py +++ b/src/pipeline/step_02_transform.py @@ -1,4 +1,4 @@ -"""Step 02 — Transform raw tables into analytical views. +"""Step 02: Transform raw tables into analytical views. Executes all SQL view definitions from sql/views/ against the DuckDB database. Views are the analytical backbone of the project: they encapsulate the business diff --git a/src/pipeline/step_03_export.py b/src/pipeline/step_03_export.py index 89724b6..d64fed1 100644 --- a/src/pipeline/step_03_export.py +++ b/src/pipeline/step_03_export.py @@ -1,4 +1,4 @@ -"""Step 03 — Export analytical views to CSV and optionally to Google Sheets. +"""Step 03: Export analytical views to CSV and optionally to Google Sheets. Materializes each analytical view into a CSV file under data/analysis/. When PUSH_TO_SHEETS is enabled, the same data is also pushed to Google Sheets diff --git a/src/pipeline/step_04_stats.py b/src/pipeline/step_04_stats.py index cdb05e5..54e9894 100644 --- a/src/pipeline/step_04_stats.py +++ b/src/pipeline/step_04_stats.py @@ -38,7 +38,7 @@ # Significance threshold for the significant_* flags in the exported CSVs. # 0.05 matches the notebooks; the flags are a convenience for dashboard -# consumers — effect size remains the primary ranking criterion. +# consumers; effect size remains the primary ranking criterion. ALPHA: float = 0.05 # Bootstrap resamples for the ghost/active completion-rate CIs. diff --git a/src/sheets/push.py b/src/sheets/push.py index 814b4e8..8743a0c 100644 --- a/src/sheets/push.py +++ b/src/sheets/push.py @@ -49,7 +49,7 @@ def _get_credentials_from_keychain() -> dict: SHEETS_KEYCHAIN_SERVICE, SHEETS_KEYCHAIN_ACCOUNT ) if not raw: - # Generic error message — never reveal what we expected to find + # Generic error message: never reveal what we expected to find raise RuntimeError("Credentials not found in macOS Keychain.") try: diff --git a/src/stats/tests.py b/src/stats/tests.py index 1642e23..8a23d07 100644 --- a/src/stats/tests.py +++ b/src/stats/tests.py @@ -1,4 +1,4 @@ -"""Statistical test wrappers — t-test, chi-square, effect sizes, confidence intervals. +"""Statistical test wrappers: t-test, chi-square, effect sizes, confidence intervals. These wrappers standardize the interface for all statistical tests used in the project, ensuring consistent output format (test statistic, p-value, @@ -71,7 +71,7 @@ def independent_t_test( Includes t-statistic, p-value, Cohen's d, and 95% CI for the difference in means. """ - # Drop NaN values — missing data should not influence the test + # Drop NaN values: missing data should not influence the test g1: np.ndarray = np.asarray(group1, dtype=float) g2: np.ndarray = np.asarray(group2, dtype=float) g1 = g1[~np.isnan(g1)] @@ -103,8 +103,8 @@ def independent_t_test( ) # When pooled_std is zero both groups have zero variance (all values # identical within each group). If the means also match, Cohen's d is - # genuinely 0. If the means differ, the effect is theoretically infinite - # — returning 0.0 would silently hide a real difference. + # genuinely 0. If the means differ, the effect is theoretically + # infinite; returning 0.0 would silently hide a real difference. if pooled_std == 0: mean_g1: float = float(np.mean(g1)) mean_g2: float = float(np.mean(g2)) @@ -116,7 +116,7 @@ def independent_t_test( cohens_d = float((np.mean(g1) - np.mean(g2)) / pooled_std) # 95% CI for the difference in means using Welch-Satterthwaite - # degrees of freedom — matches the Welch t-test above instead of + # degrees of freedom, which matches the Welch t-test above instead of # the normal approximation (z=1.96), which undercovers for small samples mean_diff: float = float(np.mean(g1) - np.mean(g2)) s1_sq_n: float = np.var(g1, ddof=1) / len(g1) @@ -187,7 +187,7 @@ def chi_square_test( if observed.ndim != 2 or observed.shape[0] < 2 or observed.shape[1] < 2: raise ValueError( f"Contingency table must be at least 2×2, got {observed.shape}. " - "A degenerate table means one variable has a single category — " + "A degenerate table means one variable has a single category; " "chi-square test is not applicable." ) @@ -200,7 +200,7 @@ def chi_square_test( # Cramér's V: effect size for chi-square # Ranges from 0 (no association) to 1 (perfect association) - # k = min(rows, cols) — the smaller dimension of the contingency table + # k = min(rows, cols), the smaller dimension of the contingency table n: int = int(observed.sum()) k: int = min(observed.shape) - 1 cramers_v: float = np.sqrt(chi2 / (n * k)) if (n * k) > 0 else 0.0 @@ -319,7 +319,7 @@ def bootstrap_ci( tuple[float, float] (lower_bound, upper_bound) of the confidence interval. """ - # Validate parameters at the boundary — invalid values would + # Validate parameters at the boundary: invalid values would # produce confusing numpy errors deeper in the computation if not (0 < confidence < 1): raise ValueError( @@ -334,7 +334,7 @@ def bootstrap_ci( # Guard: bootstrap requires at least one finite value to resample from. # An empty array (all NaN or empty input) would produce a (nan, nan) - # interval silently — better to fail explicitly. + # interval silently; better to fail explicitly. if len(arr) == 0: raise ValueError( "Cannot compute bootstrap CI: no finite values remain after " diff --git a/src/utils/runtime.py b/src/utils/runtime.py index 6a89482..af14e8e 100644 --- a/src/utils/runtime.py +++ b/src/utils/runtime.py @@ -1,4 +1,4 @@ -"""Runtime utilities — step timing and environment info. +"""Runtime utilities: step timing and environment info. Provides a context manager for timing pipeline steps and a function to log the current runtime environment (Python version, key library versions). @@ -21,9 +21,9 @@ def step_timer(step_name: str) -> Generator[None, None, None]: Usage ----- - >>> with step_timer("Step 01 — Ingest"): + >>> with step_timer("Step 01: Ingest"): ... ingest() - # logs: "Step 01 — Ingest completed in 3.42s" + # logs: "Step 01: Ingest completed in 3.42s" """ start: float = time.perf_counter() logger.info("Starting: %s", step_name) @@ -48,7 +48,7 @@ def log_environment() -> None: for lib_name in ["duckdb", "pandas", "numpy", "scipy"]: try: lib = __import__(lib_name) - # Not all modules expose __version__ — guard to avoid + # Not all modules expose __version__, so guard to avoid # AttributeError breaking the entire startup log version: str = getattr(lib, "__version__", "unknown") logger.info("%s %s", lib_name, version) diff --git a/tests/conftest.py b/tests/conftest.py index 9e26927..83be04a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,4 +1,4 @@ -"""Shared test fixtures — DuckDB in-memory database with sample data. +"""Shared test fixtures: DuckDB in-memory database with sample data. All tests use an in-memory DuckDB connection loaded with data_sample/ CSVs. This avoids touching the real database and makes tests fast and isolated. diff --git a/tests/test_config_stress.py b/tests/test_config_stress.py index edafb21..e374dee 100644 --- a/tests/test_config_stress.py +++ b/tests/test_config_stress.py @@ -1,4 +1,4 @@ -"""Stress tests for src/config.py — env var edge cases and path validation. +"""Stress tests for src/config.py: env var edge cases and path validation. The config module is loaded once at import time. Since PUSH_TO_SHEETS is computed at module level from os.environ, testing different env var values diff --git a/tests/test_connection_stress.py b/tests/test_connection_stress.py index a29ec90..c7845c4 100644 --- a/tests/test_connection_stress.py +++ b/tests/test_connection_stress.py @@ -115,7 +115,7 @@ def test_sql_injection_attempt_in_value(self) -> None: injection_payload: str = "Robert'; DROP TABLE users;--" conn.execute("INSERT INTO users VALUES (?)", [injection_payload]) - # Retrieve via parameterized query — binding must treat it as literal + # Retrieve via parameterized query: binding must treat it as literal df: pd.DataFrame = execute_query( "SELECT * FROM users WHERE name = $name", conn=conn, diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py index 6a412c1..f1330f7 100644 --- a/tests/test_pipeline.py +++ b/tests/test_pipeline.py @@ -1,4 +1,4 @@ -"""Tests for pipeline steps — ingest, transform, export. +"""Tests for pipeline steps: ingest, transform, export. Most tests use the session-scoped db_conn fixture from conftest.py, which provides an in-memory DuckDB pre-loaded with sample data. @@ -17,7 +17,7 @@ class TestIngest: - """Test step 01 — CSV to DuckDB raw tables.""" + """Test step 01: CSV to DuckDB raw tables.""" def test_ingest_creates_all_tables( self, db_conn: duckdb.DuckDBPyConnection @@ -76,7 +76,7 @@ def test_ingest_is_idempotent(self) -> None: class TestTransform: - """Test step 02 — raw tables to analytical views.""" + """Test step 02: raw tables to analytical views.""" def test_views_exist(self, db_conn: duckdb.DuckDBPyConnection) -> None: """All 5 analytical views should be queryable.""" @@ -144,7 +144,7 @@ def test_course_profile_has_all_courses( class TestExport: - """Test step 03 — views to CSV files.""" + """Test step 03: views to CSV files.""" def test_export_creates_csv_files(self, db_conn: duckdb.DuckDBPyConnection) -> None: """Export should create CSV files in the output directory.""" diff --git a/tests/test_pipeline_stress.py b/tests/test_pipeline_stress.py index 2eeff51..6fe4469 100644 --- a/tests/test_pipeline_stress.py +++ b/tests/test_pipeline_stress.py @@ -83,7 +83,7 @@ class TestTransformStress: def test_transform_on_empty_tables(self) -> None: """Transform on empty raw tables should create views without error. - Views may return 0 rows but should not fail — this tests + Views may return 0 rows but should not fail; this tests resilience of SQL views to empty input. """ conn: duckdb.DuckDBPyConnection = get_connection(db_path=None) @@ -133,7 +133,7 @@ def test_transform_with_missing_view_file(self, tmp_path: Path) -> None: conn: duckdb.DuckDBPyConnection = get_connection(db_path=None) ingest(conn=conn, use_sample=True) - # Patch VIEWS_DIR to an empty directory — all files will be "missing" + # Patch VIEWS_DIR to an empty directory: all files will be "missing" empty_views: Path = tmp_path / "views" empty_views.mkdir() with patch("src.pipeline.step_02_transform.VIEWS_DIR", empty_views): @@ -206,7 +206,7 @@ def test_export_empty_views(self) -> None: df: pd.DataFrame = pd.read_csv(path) # All exported files should be valid CSVs with a header. # Views produce 0 data rows, but aggregate queries (e.g. - # BQ5 segment_sizing) return 1 row of zeros — both are valid. + # BQ5 segment_sizing) return 1 row of zeros; both are valid. assert len(df.columns) > 0 conn.close() diff --git a/tests/test_sheets_push.py b/tests/test_sheets_push.py index 86936b8..a237642 100644 --- a/tests/test_sheets_push.py +++ b/tests/test_sheets_push.py @@ -1,4 +1,4 @@ -"""Tests for src/sheets/push.py — Google Sheets push via gspread + Keychain. +"""Tests for src/sheets/push.py: Google Sheets push via gspread + Keychain. All external dependencies (keyring, gspread, google.oauth2) are mocked so these tests run without credentials or network access. @@ -72,13 +72,13 @@ def test_valid_credentials_returns_dict(self, mock_kp: MagicMock) -> None: @patch("src.sheets.push.keyring.get_password", return_value="") def test_empty_string_treated_as_missing(self, mock_kp: MagicMock) -> None: - """Empty string from Keychain is falsy — should raise like None.""" + """Empty string from Keychain is falsy; should raise like None.""" with pytest.raises(RuntimeError, match="Credentials not found"): _get_credentials_from_keychain() class TestAuthorize: - """Tests for _authorize — gspread client creation.""" + """Tests for _authorize: gspread client creation.""" @patch("src.sheets.push.gspread.authorize") @patch("src.sheets.push.Credentials.from_service_account_info") @@ -126,7 +126,7 @@ def test_authorize_gspread_failure_propagates( class TestPushCsvsToSheets: - """Tests for push_csvs_to_sheets — worksheet create/update logic.""" + """Tests for push_csvs_to_sheets: worksheet create/update logic.""" def test_no_spreadsheet_id_skips_push( self, caplog: pytest.LogCaptureFixture diff --git a/tests/test_smoke.py b/tests/test_smoke.py index b0c9e72..4c76251 100644 --- a/tests/test_smoke.py +++ b/tests/test_smoke.py @@ -1,4 +1,4 @@ -"""Smoke tests — quick sanity checks that imports and connections work. +"""Smoke tests: quick sanity checks that imports and connections work. These tests run first and fast. If they fail, something fundamental is broken (missing dependency, bad config, import error). diff --git a/tests/test_sql_queries.py b/tests/test_sql_queries.py index 4a3018e..d841fbb 100644 --- a/tests/test_sql_queries.py +++ b/tests/test_sql_queries.py @@ -1,4 +1,4 @@ -"""Tests for BQ SQL queries — validate business logic invariants on sample data. +"""Tests for BQ SQL queries: validate business logic invariants on sample data. These tests verify that the BQ1–BQ5 queries produce structurally correct results. The queries are loaded from sql/queries/ and executed against the @@ -12,7 +12,7 @@ from src.db.connection import execute_query # --------------------------------------------------------------------------- -# Helper — load a query file and execute it against the test DB +# Helper: load a query file and execute it against the test DB # --------------------------------------------------------------------------- @@ -23,7 +23,7 @@ def _run_query_file(filename: str, conn: duckdb.DuckDBPyConnection) -> pd.DataFr # =================================================================== -# BQ1 — q_bq1_dropout_curves +# BQ1: q_bq1_dropout_curves # =================================================================== @@ -82,7 +82,7 @@ def test_n_enrolled_consistent_with_student_enriched( ) # n_enrolled from the BQ1 query (must be constant within each - # course-presentation — verify before comparing) + # course-presentation; verify before comparing) df_bq1: pd.DataFrame = _run_query_file("q_bq1_dropout_curves.sql", db_conn) n_unique_per_group: pd.Series = df_bq1.groupby( ["code_module", "code_presentation"] @@ -96,7 +96,7 @@ def test_n_enrolled_consistent_with_student_enriched( .reset_index() ) - # Merge and compare — only for courses that appear in the dropout query + # Merge and compare: only for courses that appear in the dropout query # (courses with zero dropouts won't have rows in BQ1) merged: pd.DataFrame = df_bq1_enrolled.merge( df_se, @@ -112,7 +112,7 @@ def test_n_enrolled_consistent_with_student_enriched( # =================================================================== -# BQ2 — q_bq2_early_signals +# BQ2: q_bq2_early_signals # =================================================================== @@ -183,7 +183,7 @@ def test_first_assessment_populated( n_with_score: int = df["first_score"].notna().sum() assert n_with_score > 0, ( - "No students have first_score — first-assessment subquery " + "No students have first_score: first-assessment subquery " "returned no rows (check sample data has assessments with date <= 28)" ) @@ -198,7 +198,7 @@ def test_first_score_bounded(self, db_conn: duckdb.DuckDBPyConnection) -> None: # =================================================================== -# BQ3 — q_bq3_demographics_vs_behavior +# BQ3: q_bq3_demographics_vs_behavior # =================================================================== @@ -269,7 +269,7 @@ def test_engagement_decile_bounded( """When not NULL, engagement_decile_in_course must be between 1 and 10. Unlike BQ2 (which COALESCEs NULL to 1), BQ3 preserves NULL for - ghost students — so we filter NULLs before checking the range. + ghost students, so we filter NULLs before checking the range. """ df: pd.DataFrame = _run_query_file( "q_bq3_demographics_vs_behavior.sql", db_conn @@ -295,7 +295,7 @@ def test_decile_null_iff_zero_active_days( "q_bq3_demographics_vs_behavior.sql", db_conn ) - # NULL decile but positive active_days — should never happen + # NULL decile but positive active_days: should never happen null_decile_positive_days: pd.DataFrame = df[ df["engagement_decile_in_course"].isna() & (df["active_days_first_28"] > 0) ] @@ -304,7 +304,7 @@ def test_decile_null_iff_zero_active_days( f"but active_days_first_28 > 0" ) - # Non-NULL decile but zero active_days — should never happen + # Non-NULL decile but zero active_days: should never happen has_decile_zero_days: pd.DataFrame = df[ df["engagement_decile_in_course"].notna() & (df["active_days_first_28"] == 0) @@ -329,7 +329,7 @@ def test_first_assessment_populated( n_submitted: int = (df["submitted_first_assessment"] == 1).sum() assert n_submitted > 0, ( - "No students have submitted_first_assessment=1 — " + "No students have submitted_first_assessment=1: " "first-assessment subquery returned no rows " "(check sample data has assessments with date <= 28)" ) @@ -346,7 +346,7 @@ def test_first_score_consistent_with_submitted_flag( "q_bq3_demographics_vs_behavior.sql", db_conn ) - # Flag=1 but score is NULL — should never happen + # Flag=1 but score is NULL: should never happen flag_but_no_score: pd.DataFrame = df[ (df["submitted_first_assessment"] == 1) & df["first_score"].isna() ] @@ -355,7 +355,7 @@ def test_first_score_consistent_with_submitted_flag( f"but NULL first_score" ) - # Score exists but flag=0 — should never happen + # Score exists but flag=0: should never happen score_but_no_flag: pd.DataFrame = df[ (df["submitted_first_assessment"] == 0) & df["first_score"].notna() ] @@ -383,7 +383,7 @@ def test_row_count_matches_student_enriched( # =================================================================== -# BQ4 — q_bq4_course_comparison +# BQ4: q_bq4_course_comparison # =================================================================== @@ -433,7 +433,7 @@ def test_design_features_positive(self, db_conn: duckdb.DuckDBPyConnection) -> N """Course design metrics must be strictly positive. Every course has a duration, at least one assessment, and at - least one VLE resource — zero would indicate missing data. + least one VLE resource; zero would indicate missing data. """ df: pd.DataFrame = _run_query_file("q_bq4_course_comparison.sql", db_conn) @@ -447,7 +447,7 @@ def test_design_features_positive(self, db_conn: duckdb.DuckDBPyConnection) -> N # =================================================================== -# BQ5 — q_bq5_segment_sizing +# BQ5: q_bq5_segment_sizing # =================================================================== diff --git a/tests/test_sql_views.py b/tests/test_sql_views.py index 2c83b17..21e4bd2 100644 --- a/tests/test_sql_views.py +++ b/tests/test_sql_views.py @@ -1,4 +1,4 @@ -"""Tests for SQL views — validate analytical logic on sample data. +"""Tests for SQL views: validate analytical logic on sample data. These tests verify that the SQL views produce correct results by checking business logic invariants on the sample data. diff --git a/tests/test_stats_stress.py b/tests/test_stats_stress.py index 55a86ff..4515973 100644 --- a/tests/test_stats_stress.py +++ b/tests/test_stats_stress.py @@ -20,7 +20,7 @@ ) # =================================================================== -# independent_t_test — stress & edge cases +# independent_t_test: stress & edge cases # =================================================================== @@ -32,7 +32,7 @@ class TestTTestZeroVariance: scipy.stats.ttest_ind emits a RuntimeWarning when both groups are constant (catastrophic cancellation in moment calculation). This is - expected — our code handles the degenerate case in the pooled_std + expected: our code handles the degenerate case in the pooled_std and se_diff guards before using scipy's result. """ @@ -109,7 +109,7 @@ def test_mixed_positive_negative(self) -> None: assert result.ci_lower < result.ci_upper def test_single_outlier_group(self) -> None: - """One group with a massive outlier — should not crash.""" + """One group with a massive outlier; should not crash.""" g1: np.ndarray = np.array([1.0, 2.0, 3.0, 4.0, 1e10]) g2: np.ndarray = np.array([1.0, 2.0, 3.0, 4.0, 5.0]) result: TestResult = independent_t_test(g1, g2, "outlier") @@ -147,7 +147,7 @@ class TestTTestAsymmetricGroups: """Very unbalanced group sizes.""" def test_highly_unbalanced_groups(self) -> None: - """One group much larger — Welch's t-test should handle this.""" + """One group much larger; Welch's t-test should handle this.""" rng = np.random.default_rng(42) g1: np.ndarray = rng.normal(10, 2, size=5) g2: np.ndarray = rng.normal(10, 2, size=500) @@ -237,7 +237,7 @@ def test_effect_size_sign_matches_mean_diff(self, seed: int) -> None: # =================================================================== -# chi_square_test — stress & edge cases +# chi_square_test: stress & edge cases # =================================================================== @@ -254,7 +254,7 @@ def test_large_contingency_table(self) -> None: assert 0 <= result.effect_size <= 1 def test_very_sparse_table(self) -> None: - """Table with small counts — should still compute. + """Table with small counts; should still compute. Fully-zero expected cells make scipy raise, so we use a table that is sparse (low counts, strong association) but has no @@ -277,7 +277,7 @@ def test_single_column_raises(self) -> None: chi_square_test(observed, "single_col") def test_1d_array_raises(self) -> None: - """1D array is not a contingency table — should raise.""" + """1D array is not a contingency table; should raise.""" with pytest.raises(ValueError, match="at least 2×2"): chi_square_test(np.array([10, 20, 30]), "1d") @@ -313,7 +313,7 @@ def test_cramers_v_between_0_and_1(self, seed: int) -> None: # =================================================================== -# apply_multiple_comparison_correction — stress +# apply_multiple_comparison_correction: stress # =================================================================== @@ -389,7 +389,7 @@ def test_corrected_p_values_always_valid(self, seed: int) -> None: # =================================================================== -# bootstrap_ci — stress & edge cases +# bootstrap_ci: stress & edge cases # =================================================================== @@ -489,7 +489,7 @@ def test_pandas_series_input(self) -> None: # =================================================================== -# independent_t_test — additional edge cases +# independent_t_test: additional edge cases # =================================================================== @@ -498,7 +498,7 @@ class TestTTestInfValues: """Test t-test behavior when input data contains inf.""" def test_inf_in_group_produces_finite_or_nan_result(self) -> None: - """Groups containing inf should not crash — result may be nan/inf + """Groups containing inf should not crash; result may be nan/inf but the wrapper should not raise.""" g1: np.ndarray = np.array([1.0, 2.0, 3.0, float("inf"), 5.0]) g2: np.ndarray = np.array([10.0, 11.0, 12.0, 13.0, 14.0]) @@ -558,7 +558,7 @@ def test_one_element_raises(self) -> None: # =================================================================== -# chi_square_test — additional edge cases +# chi_square_test: additional edge cases # =================================================================== @@ -581,7 +581,7 @@ def test_table_with_multiple_zero_cells(self) -> None: # =================================================================== -# bootstrap_ci — additional edge cases +# bootstrap_ci: additional edge cases # =================================================================== @@ -598,7 +598,7 @@ def sometimes_nan(arr: np.ndarray) -> float: return float("nan") if arr[0] < 0 else float(np.mean(arr)) data: np.ndarray = np.array([-1.0, 2.0, 3.0, 4.0, 5.0]) - # Should not crash — numpy percentile handles NaN + # Should not crash: numpy percentile handles NaN lower, upper = bootstrap_ci(data, statistic_fn=sometimes_nan) # Result may be NaN but should not raise assert isinstance(lower, float) diff --git a/tests/test_utils_stress.py b/tests/test_utils_stress.py index 741b041..23c70d3 100644 --- a/tests/test_utils_stress.py +++ b/tests/test_utils_stress.py @@ -1,4 +1,4 @@ -"""Stress tests for src/utils/ — logging and runtime utilities. +"""Stress tests for src/utils/: logging and runtime utilities. Tests edge cases: exception handling in step_timer, missing libraries in log_environment, setup_logging idempotency, and timer accuracy. @@ -14,7 +14,7 @@ from src.utils.runtime import log_environment, step_timer # =================================================================== -# step_timer — stress tests +# step_timer: stress tests # =================================================================== @@ -53,7 +53,7 @@ def test_timer_measures_actual_time(self, caplog: pytest.LogCaptureFixture) -> N assert "Timed Step completed in" in caplog.text def test_timer_with_empty_step_name(self, caplog: pytest.LogCaptureFixture) -> None: - """Empty step name should not crash — just produce odd log messages.""" + """Empty step name should not crash; just produce odd log messages.""" with caplog.at_level(logging.INFO, logger="src.utils.runtime"): with step_timer(""): pass @@ -64,7 +64,7 @@ def test_timer_with_unicode_step_name( ) -> None: """Unicode characters in step name should be logged correctly.""" with caplog.at_level(logging.INFO, logger="src.utils.runtime"): - with step_timer("Étape 1 — Ingest données"): + with step_timer("Étape 1 · Ingest données"): pass assert "Étape 1" in caplog.text @@ -82,7 +82,7 @@ def test_nested_timers(self, caplog: pytest.LogCaptureFixture) -> None: # =================================================================== -# log_environment — stress tests +# log_environment: stress tests # =================================================================== @@ -164,7 +164,7 @@ def test_log_environment_is_idempotent( # =================================================================== -# setup_logging — stress tests +# setup_logging: stress tests # ===================================================================