From c0cd77c8f0c13761059c42e77f60883a2253e277 Mon Sep 17 00:00:00 2001 From: iabaako Date: Sun, 4 Oct 2026 20:37:39 +0000 Subject: [PATCH 1/4] feat(backchecks): track progress against the backcheck target Backcheck Coverage is now the share of eligible unique survey IDs with a matching backcheck, shown against the target %, and no longer needs comparison columns. A Targets row shows backchecks done against ceil(survey_target x target% / 100) when the survey target is set. The target % resolves from the settings panel, then the page config, then 10% with a warning; clearing the panel falls back to the page config. An optional eligibility filter restricts the base. The enumerator view of the error statistics table shows per-enumerator coverage, including enumerators with no backchecks; the backchecker view shows backchecks done. Closes #318 Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 16 + docs/USER_GUIDE.md | 26 +- src/datasure/checks/backchecks/compute.py | 9 +- src/datasure/checks/backchecks/coverage.py | 213 +++++++++++++ src/datasure/checks/backchecks/models.py | 16 +- src/datasure/checks/backchecks/report_ui.py | 197 ++++++++---- src/datasure/checks/backchecks/settings_ui.py | 123 +++++++- src/datasure/utils/onboarding_utils.py | 5 +- src/datasure/views/output_view_template.py | 1 + tests/checks/backchecks/conftest.py | 2 +- tests/checks/backchecks/test_compute.py | 2 - tests/checks/backchecks/test_coverage.py | 295 ++++++++++++++++++ tests/checks/backchecks/test_models.py | 2 +- tests/checks/backchecks/test_report_ui.py | 22 +- 14 files changed, 833 insertions(+), 96 deletions(-) create mode 100644 src/datasure/checks/backchecks/coverage.py create mode 100644 tests/checks/backchecks/test_coverage.py diff --git a/CHANGELOG.md b/CHANGELOG.md index c40b4a6b..7a8f7863 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -82,6 +82,22 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 `CorrectionEntry.severity` sets it and is rejected on non-accept actions, on acceptances of other checks, and with any value other than `hard`. Hard acceptances are highlighted in the Correction Log — #298 +- **Backcheck targets**: The Backchecks Summary tracks progress against the + backcheck target. "Backcheck Coverage" is now the share of eligible unique + survey IDs (after duplicate handling) with a matching backcheck, shown + against the target, and is calculated before any comparison columns are + configured. A new Targets row shows backchecks done against + `ceil(survey_target × target% / 100)` when the page configuration sets + `survey_target`. The target % resolves from the settings panel, then the + page configuration, then 10% (`BackcheckSettings.backcheck_target_percent` + is now `float | None`; a page-config target of 0 counts as not set). A new + optional eligibility filter (column plus values) restricts the base. In the + Enumerator Backchecker Error Statistics table, the enumerator view's + "Surveys" and "Backchecks" are now eligible unique submissions and how many + were backchecked, with new "Coverage %" and "vs target" columns; the + backchecker view shows backchecks done. The table renders before comparison + columns are configured. Calculations live in the new + `checks/backchecks/coverage.py` — #318 ### Fixed diff --git a/docs/USER_GUIDE.md b/docs/USER_GUIDE.md index 0ffff2fe..edef7a0d 100644 --- a/docs/USER_GUIDE.md +++ b/docs/USER_GUIDE.md @@ -994,9 +994,26 @@ Configure validation: - **Enumerator**: Original data collector - **Back Checker**: QC validator - **Date**: Back check date -- **Target %**: Target back check rate (e.g., 10%) +- **Backcheck target (%)**: Share of surveys to back check (e.g., 10%). + Pre-filled from the page configuration. A value you enter here is saved and + overrides the page configuration; clear it to fall back. If neither is set, + 10% is used and a warning is shown. +- **Eligibility Filter**: Optional survey column and values that mark a survey + eligible for back checks (e.g., `consent` in `1`). Only eligible surveys + count towards coverage. - **Handle Duplicates**: Include or exclude duplicates +##### Backchecks Summary + +- **Backcheck Coverage**: Share of eligible unique survey IDs (after duplicate + handling) with at least one matching back check, and how many points it is + above or below the target. It is calculated before any back check columns + are configured. +- **Targets**: When the page configuration sets the target number of survey + responses, back checks done against the back checks expected + (survey target × target %, rounded up), with a progress bar. Values over + 100% are shown as is. + **Add Back Check Columns**: Click "Add a back check column" (+ button): @@ -1039,7 +1056,10 @@ Detailed column-level validation: Performance by original enumerator: - Enumerator ID -- \# surveys back checked +- Surveys: eligible unique submissions +- Backchecks: how many of those were back checked +- Coverage % and points vs target (coverage below target is highlighted; + enumerators with no back checks show 0%) - \# values compared - \# different values - Error rate (%) @@ -1049,7 +1069,7 @@ Performance by original enumerator: Performance by validator: - Back Checker ID -- \# surveys validated +- Backchecks: unique surveys back checked - \# values compared - \# discrepancies - Error rate (%) diff --git a/src/datasure/checks/backchecks/compute.py b/src/datasure/checks/backchecks/compute.py index fcde2265..670a4a30 100644 --- a/src/datasure/checks/backchecks/compute.py +++ b/src/datasure/checks/backchecks/compute.py @@ -29,9 +29,8 @@ def load_default_backchecks_settings( Loads previously saved backcheck report settings from the settings file and merges them with the provided default configuration. Saved settings - take precedence over defaults. - - Cached for 60 seconds to reduce file I/O operations. + take precedence over defaults, except a cleared (None) backcheck target, + which falls back to the configured one. Parameters ---------- @@ -46,6 +45,8 @@ def load_default_backchecks_settings( Merged settings combining saved and default configurations. """ saved_settings = load_check_settings(settings_file, TAB_NAME) + if saved_settings.get("backcheck_target_percent") is None: + saved_settings.pop("backcheck_target_percent", None) default_settings: dict = dict(config) default_settings.update(saved_settings) @@ -919,8 +920,6 @@ def _calculate_staff_statistics( # Initialize stats dict staff_stats = { staff_col: staff_name, - "Surveys": staff_data[survey_key].n_unique(), - "Backchecks": staff_data[survey_key].n_unique(), "Avg Days": _calculate_average_days(staff_data, survey_date, backcheck_date), } diff --git a/src/datasure/checks/backchecks/coverage.py b/src/datasure/checks/backchecks/coverage.py new file mode 100644 index 00000000..85f14068 --- /dev/null +++ b/src/datasure/checks/backchecks/coverage.py @@ -0,0 +1,213 @@ +"""Backcheck coverage against the backcheck target. + +The target % resolves from the settings panel, then the page config, then +`DEFAULT_TARGET_PERCENT`. Coverage counts unique survey IDs, after duplicate +handling and the optional eligibility filter, that have a matching backcheck. +""" + +import math +from dataclasses import dataclass + +import polars as pl + +from datasure.checks.backchecks.compute import _prepare_data_for_merge +from datasure.checks.backchecks.models import BackcheckSettings + +DEFAULT_TARGET_PERCENT: float = 10.0 + +# Column `backchecked_surveys` adds to flag surveys with a matching backcheck. +BACKCHECKED: str = "_backchecked" + + +def settings_from_page_config(config: dict) -> BackcheckSettings: + """Build backcheck settings from the page configuration. + + The page configuration stores 0 for a target left blank, so 0 means + "not set" for both the backcheck target % and the survey target. + """ + return BackcheckSettings( + **{ + **config, + "backcheck_target_percent": config.get("backcheck_target_percent") or None, + "survey_target": config.get("survey_target") or None, + } + ) + + +def effective_target_percent(settings: BackcheckSettings) -> float: + """Return the backcheck target %, or the default when none is set.""" + if settings.backcheck_target_percent is None: + return DEFAULT_TARGET_PERCENT + return settings.backcheck_target_percent + + +@dataclass(frozen=True) +class BackcheckCoverage: + """Backcheck progress against the target %. + + `on_track_percent` and `points_vs_target` are None when there are no + eligible surveys. The expected-total fields are None when the survey + target is not set. + """ + + eligible: int + backchecked: int + target_percent: float + on_track_percent: float | None + points_vs_target: float | None + expected_backchecks: int | None + expected_progress_percent: float | None + + +def backchecked_surveys( + survey_data: pl.DataFrame, + backcheck_data: pl.DataFrame, + settings: BackcheckSettings, +) -> pl.DataFrame | None: + """Return one row per eligible survey ID, flagged if it was backchecked. + + Both datasets go through the duplicate-handling option first, as in the + backcheck comparison. Matching is by survey ID only. + + Returns + ------- + pl.DataFrame | None + The deduplicated eligible survey rows plus a boolean `BACKCHECKED` + column, or None if either dataset lacks the survey ID column. + """ + survey_id = settings.survey_id + if ( + not survey_id + or survey_id not in survey_data.columns + or survey_id not in backcheck_data.columns + ): + return None + + option = settings.drop_duplicates_option + surveys = _prepare_data_for_merge(survey_data, survey_id, option).unique( + subset=[survey_id], keep="first", maintain_order=True + ) + backchecked_ids = _prepare_data_for_merge(backcheck_data, survey_id, option)[ + survey_id + ].drop_nulls() + + surveys = surveys.filter(pl.col(survey_id).is_not_null()) + eligibility_column = settings.eligibility_column + if ( + eligibility_column + and eligibility_column in surveys.columns + and settings.eligibility_values + ): + surveys = surveys.filter( + pl.col(eligibility_column).cast(pl.Utf8).is_in(settings.eligibility_values) + ) + + return surveys.with_columns( + pl.col(survey_id).is_in(backchecked_ids.implode()).alias(BACKCHECKED) + ) + + +def compute_backcheck_coverage( + survey_data: pl.DataFrame, + backcheck_data: pl.DataFrame, + settings: BackcheckSettings, +) -> BackcheckCoverage | None: + """Compute overall backcheck coverage against the target. + + Returns None if either dataset lacks the survey ID column. + """ + surveys = backchecked_surveys(survey_data, backcheck_data, settings) + if surveys is None: + return None + + eligible = surveys.height + backchecked = int(surveys[BACKCHECKED].sum()) + target_percent = effective_target_percent(settings) + on_track_percent = backchecked / eligible * 100 if eligible else None + + expected = None + if settings.survey_target: + # Round away float noise first so an exact product doesn't round up. + expected = math.ceil(round(settings.survey_target * target_percent / 100, 9)) + + return BackcheckCoverage( + eligible=eligible, + backchecked=backchecked, + target_percent=target_percent, + on_track_percent=on_track_percent, + points_vs_target=( + on_track_percent - target_percent if on_track_percent is not None else None + ), + expected_backchecks=expected, + expected_progress_percent=backchecked / expected * 100 if expected else None, + ) + + +def compute_staff_coverage( + survey_data: pl.DataFrame, + backcheck_data: pl.DataFrame, + settings: BackcheckSettings, + staff_type: str = "enumerator", +) -> pl.DataFrame: + """Compute backcheck coverage per enumerator, or backchecks per backchecker. + + Parameters + ---------- + survey_data : pl.DataFrame + Survey dataset. + backcheck_data : pl.DataFrame + Backcheck dataset. + settings : BackcheckSettings + Backcheck settings. + staff_type : str + Either "enumerator" or "backchecker". + + Returns + ------- + pl.DataFrame + For enumerators: the enumerator column, "Surveys" (eligible unique + submissions), "Backchecks" (how many of those were backchecked), + "Coverage %" and "vs target" (points). Enumerators with no backchecks + appear at 0%. For backcheckers: the backchecker column and + "Backchecks" (unique backchecked survey IDs). Empty if the survey ID + or staff column is missing. + """ + surveys = backchecked_surveys(survey_data, backcheck_data, settings) + if surveys is None: + return pl.DataFrame() + + if staff_type == "enumerator": + staff_col = settings.enumerator + if not staff_col or staff_col not in surveys.columns: + return pl.DataFrame() + target_percent = effective_target_percent(settings) + return ( + surveys.filter(pl.col(staff_col).is_not_null()) + .group_by(staff_col, maintain_order=True) + .agg( + pl.len().alias("Surveys"), + pl.col(BACKCHECKED).sum().cast(pl.Int64).alias("Backchecks"), + ) + .with_columns( + (pl.col("Backchecks") / pl.col("Surveys") * 100).alias("Coverage %") + ) + .with_columns((pl.col("Coverage %") - target_percent).alias("vs target")) + .with_columns(pl.col("Surveys").cast(pl.Int64)) + ) + + staff_col = settings.backchecker + if not staff_col or staff_col not in backcheck_data.columns: + return pl.DataFrame() + survey_id = settings.survey_id + backchecked_ids = surveys.filter(pl.col(BACKCHECKED))[survey_id] + return ( + _prepare_data_for_merge( + backcheck_data, survey_id, settings.drop_duplicates_option + ) + .filter( + pl.col(survey_id).is_in(backchecked_ids.implode()) + & pl.col(staff_col).is_not_null() + ) + .group_by(staff_col, maintain_order=True) + .agg(pl.col(survey_id).n_unique().cast(pl.Int64).alias("Backchecks")) + ) diff --git a/src/datasure/checks/backchecks/models.py b/src/datasure/checks/backchecks/models.py index dbc7e9de..d75e5b3a 100644 --- a/src/datasure/checks/backchecks/models.py +++ b/src/datasure/checks/backchecks/models.py @@ -76,8 +76,20 @@ class BackcheckSettings(BaseModel): ) enumerator: str | None = Field(None, description="Column containing enumerator") backchecker: str | None = Field(None, description="Column containing back checker") - backcheck_target_percent: int = Field( - 10, description="Target percentage of backchecks" + backcheck_target_percent: float | None = Field( + None, + ge=0, + le=100, + description="Target percentage of surveys to backcheck; None if not set", + ) + survey_target: int | None = Field( + None, ge=0, description="Target number of survey responses" + ) + eligibility_column: str | None = Field( + None, description="Survey column that marks a survey eligible" + ) + eligibility_values: list[str] | None = Field( + None, description="Values of eligibility_column that mark a survey eligible" ) drop_duplicates_option: str = Field( "drop", description="How to handle duplicate entries" diff --git a/src/datasure/checks/backchecks/report_ui.py b/src/datasure/checks/backchecks/report_ui.py index a653fd69..f2f2b0ae 100644 --- a/src/datasure/checks/backchecks/report_ui.py +++ b/src/datasure/checks/backchecks/report_ui.py @@ -12,6 +12,13 @@ compute_enumerator_backchecker_stats, expand_col_names, ) +from datasure.checks.backchecks.coverage import ( + BackcheckCoverage, + compute_backcheck_coverage, + compute_staff_coverage, + effective_target_percent, + settings_from_page_config, +) from datasure.checks.backchecks.models import ( TAB_NAME, WEEKDAY_NAMES, @@ -33,6 +40,7 @@ save_check_settings, trigger_save, ) +from datasure.utils.ui_utils import styled_dataframe # ============================================================================== # COLUMN CONFIGURATION FUNCTIONS @@ -567,10 +575,9 @@ def _render_backcheck_test_options(backcheck_category: int) -> BackcheckTestOpti def _render_backcheck_summary( survey_data: pl.DataFrame, backcheck_data: pl.DataFrame, - backcheck_analysis: pl.DataFrame, backcheck_settings: BackcheckSettings, ) -> None: - """Render summary metrics for backcheck analysis. + """Render summary metrics and progress against the backcheck target. Parameters ---------- @@ -578,25 +585,12 @@ def _render_backcheck_summary( Survey dataset. backcheck_data : pl.DataFrame Backcheck dataset. - backcheck_analysis : pl.DataFrame - Results from compute_backcheck_analysis. backcheck_settings : BackcheckSettings - Backcheck settings including enumerator and backchecker columns. + Backcheck settings including the staff columns and targets. """ - # Calculate basic metrics - n_survey_obs = len(survey_data) - n_backcheck_obs = len(backcheck_data) - - # Calculate percentage of surveys with backcheck responses - # Get unique survey keys that have backchecks - survey_key = backcheck_settings.survey_key - if survey_key and not backcheck_analysis.is_empty(): - unique_backchecked_surveys = backcheck_analysis[survey_key].n_unique() - backcheck_coverage_pct = ( - (unique_backchecked_surveys / n_survey_obs * 100) if n_survey_obs > 0 else 0 - ) - else: - backcheck_coverage_pct = 0 + coverage = compute_backcheck_coverage( + survey_data, backcheck_data, backcheck_settings + ) # Count unique enumerators and back checkers enumerator_col = backcheck_settings.enumerator @@ -617,16 +611,13 @@ def _render_backcheck_summary( lc1, lc2, _, _ = st.columns(4) with uc1, st.container(border=True): - st.metric("Survey Observations", f"{n_survey_obs:,}") + st.metric("Survey Observations", f"{len(survey_data):,}") with uc2, st.container(border=True): - st.metric("Backcheck Observations", f"{n_backcheck_obs:,}") + st.metric("Backcheck Observations", f"{len(backcheck_data):,}") with uc3, st.container(border=True): - st.metric( - "Backcheck Coverage", - f"{backcheck_coverage_pct:.1f}%", - ) + _render_coverage_metric(coverage, backcheck_settings) with lc1, st.container(border=True): st.metric( @@ -639,6 +630,57 @@ def _render_backcheck_summary( f"{n_backcheckers:,}" if n_backcheckers > 0 else "N/A", ) + if coverage is not None: + _render_expected_backchecks(coverage) + + +def _render_coverage_metric( + coverage: BackcheckCoverage | None, backcheck_settings: BackcheckSettings +) -> None: + """Render the on-track coverage card with its delta against the target.""" + target_percent = effective_target_percent(backcheck_settings) + if coverage is None or coverage.on_track_percent is None: + st.metric( + "Backcheck Coverage", + "N/A", + help="Needs the survey ID column in both the survey and backcheck " + "data, and at least one eligible survey.", + ) + return + + st.metric( + "Backcheck Coverage", + f"{coverage.on_track_percent:.1f}%", + delta=f"{coverage.points_vs_target:+.1f} pts vs {target_percent:g}%", + help=f"{coverage.backchecked:,} of {coverage.eligible:,} eligible unique " + f"surveys have been backchecked. Target: {target_percent:g}%.", + ) + + +def _render_expected_backchecks(coverage: BackcheckCoverage) -> None: + """Render the Targets row: backchecks done against backchecks expected.""" + st.markdown("##### Targets") + if coverage.expected_backchecks is None: + st.info( + "Set the target number of responses for the survey in the page " + "configuration to track progress towards the expected total " + "number of backchecks." + ) + return + + progress = coverage.expected_progress_percent or 0.0 + tc1, tc2 = st.columns([1, 3], vertical_alignment="center") + with tc1, st.container(border=True): + st.metric( + "Backchecks vs Expected", + f"{coverage.backchecked:,} / {coverage.expected_backchecks:,} " + f"({progress:.0f}%)", + help=f"Expected backchecks: {coverage.target_percent:g}% of the " + "survey target, rounded up.", + ) + with tc2: + st.progress(min(progress / 100, 1.0)) + def _render_backchecker_productivity( data: pl.DataFrame, @@ -827,10 +869,11 @@ def _render_enum_bcer_stats( backcheck_settings: BackcheckSettings, settings_file: str, ) -> None: - """Render enumerator and backchecker error rate statistics. + """Render enumerator and backchecker coverage and error rate statistics. - Displays statistics tables showing error rates by category for either - enumerators or backcheckers, with a pills selector to switch between views. + Displays per-staff backcheck coverage, plus error rates by category once + backcheck columns are configured, with a pills selector to switch + between enumerators and backcheckers. Parameters ---------- @@ -845,12 +888,6 @@ def _render_enum_bcer_stats( settings_file : str Path to settings file for saving/loading configurations. """ - if backcheck_analysis.is_empty(): - st.info( - "No backcheck analysis results available. Configure backcheck columns in the settings section above." - ) - return - # Check if required columns are configured enumerator_col = backcheck_settings.enumerator backchecker_col = backcheck_settings.backchecker @@ -871,6 +908,17 @@ def _render_enum_bcer_stats( ) +def _highlight_below_target(target_percent: float): + """Return a Styler cell function that flags coverage below the target.""" + + def style(value: object) -> str: + if isinstance(value, int | float) and value < target_percent: + return "background-color: #f8d7da; color: #842029" + return "" + + return style + + @st.fragment def _render_enum_bcer_stats_table( survey_data: pl.DataFrame, @@ -920,8 +968,8 @@ def _render_enum_bcer_stats_table( # Compute and display statistics staff_type = "enumerator" if view_selection == "Enumerator" else "backchecker" - stats_df = compute_enumerator_backchecker_stats( - survey_data, backcheck_data, backcheck_analysis, backcheck_settings, staff_type + stats_df = compute_staff_coverage( + survey_data, backcheck_data, backcheck_settings, staff_type ) if stats_df.is_empty(): @@ -931,11 +979,35 @@ def _render_enum_bcer_stats_table( # Get staff column name for display staff_col = enumerator_col if staff_type == "enumerator" else backchecker_col + if backcheck_analysis.is_empty(): + st.info( + "Error rates appear here once backcheck columns are configured in " + "the Backchecks Columns Configuration section above." + ) + else: + error_stats = compute_enumerator_backchecker_stats( + survey_data, + backcheck_data, + backcheck_analysis, + backcheck_settings, + staff_type, + ) + if not error_stats.is_empty(): + stats_df = stats_df.join(error_stats, on=staff_col, how="left") + # Configure columns for wide format column_config = { staff_col: st.column_config.TextColumn(view_selection, pinned=True), - "Surveys": st.column_config.NumberColumn("Surveys", format="%d"), - "Backchecks": st.column_config.NumberColumn("Backchecks", format="%d"), + "Surveys": st.column_config.NumberColumn( + "Surveys", format="%d", help="Eligible unique submissions" + ), + "Backchecks": st.column_config.NumberColumn( + "Backchecks", format="%d", help="Unique submissions backchecked" + ), + "Coverage %": st.column_config.NumberColumn("Coverage %"), + "vs target": st.column_config.NumberColumn( + "vs target", help="Coverage minus the target, in percentage points" + ), "Avg Days": st.column_config.NumberColumn("Avg Days", format="%.1f"), } @@ -978,8 +1050,24 @@ def _render_enum_bcer_stats_table( "Error % (Total)", format="%.2f" ) - st.dataframe( - stats_df, hide_index=True, width="stretch", column_config=column_config + if staff_type == "backchecker": + st.dataframe( + stats_df, hide_index=True, width="stretch", column_config=column_config + ) + return + + # st.dataframe shows a Styler's formatted text, so format every cell here: + # plain text by default, blank error rates for unbackchecked enumerators. + target_percent = effective_target_percent(backcheck_settings) + formatters = {col: str for col in stats_df.columns} + formatters.update({"Coverage %": "{:.1f}%", "vs target": "{:+.1f}"}) + styler = ( + stats_df.to_pandas(use_pyarrow_extension_array=True) + .style.map(_highlight_below_target(target_percent), subset=["Coverage %"]) + .format(formatters, na_rep="") + ) + styled_dataframe( + styler, hide_index=True, width="stretch", column_config=column_config ) @@ -1574,10 +1662,6 @@ def backchecks_report( """ ) - # Convert Polars DataFrames to Pandas for compatibility - survey_data_pd = survey_data.to_pandas() - backcheck_data_pd = backcheck_data.to_pandas() - # Get column information for settings UI survey_categorical_columns = survey_columns.categorical_columns survey_datetime_columns = survey_columns.datetime_columns @@ -1586,12 +1670,12 @@ def backchecks_report( backcheck_datetime_columns = backcheck_columns.datetime_columns # Configure settings - config_settings = BackcheckSettings(**config) + config_settings = settings_from_page_config(config) backcheck_settings = backchecks_report_settings( project_id, setting_file, - survey_data_pd, - backcheck_data_pd, + survey_data, + backcheck_data, config_settings, survey_categorical_columns, survey_datetime_columns, @@ -1651,7 +1735,12 @@ def backchecks_report( """ ##### Backchecks Summary Five metrics appear here: Survey Observations, Backcheck Observations, - Backcheck Coverage %, Total Enumerators, and Total Back Checkers. + Backcheck Coverage, Total Enumerators, and Total Back Checkers. + Backcheck Coverage is the share of eligible unique surveys that have been + backchecked, with how far it is above or below the backcheck target. + When the survey's target number of responses is set in the page + configuration, a **Targets** row shows backchecks done against the + total number of backchecks expected. Below the metrics, a **Backchecker Productivity** table shows submission counts per backchecker over time. Use the **Daily / Weekly / Monthly** pills @@ -1659,9 +1748,7 @@ def backchecks_report( """ ) - _render_backcheck_summary( - survey_data, backcheck_data, _backcheck_analysis, backcheck_settings - ) + _render_backcheck_summary(survey_data, backcheck_data, backcheck_settings) _render_backchecker_productivity( backcheck_data, @@ -1676,9 +1763,11 @@ def backchecks_report( """ ##### Enumerator Backchecker Error Statistics Use the **Enumerator / Backchecker** pills to switch between two views. - Each view shows a table with submission counts, values compared, number of - mismatches, and error rate — broken down by category — for either the - original enumerator or the backchecker. + The enumerator view shows each enumerator's eligible surveys, how many + were backchecked, and their coverage against the target, with coverage + below target highlighted. The backchecker view shows backchecks done. + Once backcheck columns are configured, both views also show values + compared, number of mismatches, and error rate, broken down by category. """ ) diff --git a/src/datasure/checks/backchecks/settings_ui.py b/src/datasure/checks/backchecks/settings_ui.py index 92901eee..3c494a78 100644 --- a/src/datasure/checks/backchecks/settings_ui.py +++ b/src/datasure/checks/backchecks/settings_ui.py @@ -1,10 +1,10 @@ """Settings UI for the backchecks report.""" -import pandas as pd import polars as pl import streamlit as st from datasure.checks.backchecks.compute import load_default_backchecks_settings +from datasure.checks.backchecks.coverage import DEFAULT_TARGET_PERCENT from datasure.checks.backchecks.models import ( TAB_NAME, BackcheckSettings, @@ -247,9 +247,13 @@ def _render_staff_identifiers( def _render_tracking_options( settings_file: str, default_settings: BackcheckSettings -) -> int: +) -> float | None: """Render tracking options section. + The target input is pre-filled with the saved panel value, else the page + config target. A value the user changes is saved and wins over the page + config; clearing it saves None, which falls back to the page config. + Parameters ---------- settings_file : str @@ -259,20 +263,25 @@ def _render_tracking_options( Returns ------- - int - Backcheck target percent. + float | None + Backcheck target percent, or None if not set anywhere. """ with st.container(border=True): st.subheader("Tracking Options") to1, _, _ = st.columns(3) with to1: + default_target = default_settings.backcheck_target_percent backcheck_target_percent = st.number_input( - "Target number of backchecks", - min_value=0, - help="Total number of backchecks expected", + "Backcheck target (%)", + min_value=0.0, + max_value=100.0, + step=1.0, + format="%.1f", + help="Percentage of survey submissions to backcheck. Leave blank " + "to use the target from the page configuration.", key="backcheck_goal_backchecks", - value=default_settings.backcheck_target_percent, + value=float(default_target) if default_target is not None else None, on_change=trigger_save, kwargs={"state_name": TAB_NAME + "_backcheck_target_percent"}, ) @@ -282,9 +291,89 @@ def _render_tracking_options( {"backcheck_target_percent": backcheck_target_percent}, ) + if backcheck_target_percent is None: + st.warning( + "No backcheck target is set here or in the page configuration, " + f"so the default of {DEFAULT_TARGET_PERCENT:g}% is used." + ) + return backcheck_target_percent +def _render_eligibility_filter( + settings_file: str, + default_settings: BackcheckSettings, + survey_data: pl.DataFrame, +) -> tuple[str | None, list[str]]: + """Render the eligibility filter for the backcheck coverage base. + + Parameters + ---------- + settings_file : str + Path to settings file. + default_settings : BackcheckSettings + Default settings. + survey_data : pl.DataFrame + Survey dataset, used for the column and value options. + + Returns + ------- + tuple[str | None, list[str]] + Eligibility column and the values that mark a survey eligible. + """ + with st.container(border=True): + st.markdown("##### Eligibility Filter (Optional)") + st.write( + "Only count surveys whose selected column has one of the selected " + "values towards backcheck coverage, for example `consent` in `1`. " + "With no values selected, every survey counts." + ) + ef1, ef2, _ = st.columns(3) + + with ef1: + eligibility_column = _render_selectbox_with_save( + "Eligibility column", + list(survey_data.columns), + "eligibility_column_backchecks", + settings_file, + "eligibility_column", + default_settings.eligibility_column, + "Select the survey column that marks a survey eligible", + ) + + value_options = [] + if eligibility_column and eligibility_column in survey_data.columns: + # Cast as coverage.backchecked_surveys does, so the values match. + value_options = ( + survey_data[eligibility_column] + .cast(pl.Utf8) + .drop_nulls() + .unique() + .sort() + .to_list() + ) + saved_values = default_settings.eligibility_values or [] + + with ef2: + eligibility_values = st.multiselect( + "Eligible values", + options=value_options, + default=[v for v in saved_values if v in value_options], + key="eligibility_values_backchecks", + help="Select the values that mark a survey eligible", + disabled=not eligibility_column, + on_change=trigger_save, + kwargs={"state_name": TAB_NAME + "_eligibility_values"}, + ) + save_check_settings( + settings_file, + TAB_NAME, + {"eligibility_values": eligibility_values}, + ) + + return eligibility_column, eligibility_values + + def _render_duplicate_handling( settings_file: str, default_settings: BackcheckSettings ) -> str: @@ -431,8 +520,8 @@ def _render_additional_options( def backchecks_report_settings( project_id: str, settings_file: str, - survey_data: pd.DataFrame, - backcheck_data: pd.DataFrame, + survey_data: pl.DataFrame, + backcheck_data: pl.DataFrame, config: BackcheckSettings, survey_categorical_columns: list[str], survey_datetime_columns: list[str], @@ -446,7 +535,8 @@ def backchecks_report_settings( - Survey identifiers (key and ID columns) - Survey date column selection - Enumerator and backchecker columns - - Tracking options (backcheck goal and duplicate handling) + - Tracking options (backcheck target % and eligibility filter) + - Additional options (duplicate handling and value comparison) Settings are automatically saved to the settings file when changed and loaded from previous sessions if available. @@ -457,9 +547,9 @@ def backchecks_report_settings( Unique project identifier for database operations. settings_file : str Path to settings file for saving/loading configurations. - survey_data : pd.DataFrame + survey_data : pl.DataFrame Survey dataset. - backcheck_data : pd.DataFrame + backcheck_data : pl.DataFrame Backcheck dataset. config : BackcheckSettings Default configuration used as fallback values. @@ -506,6 +596,10 @@ def backchecks_report_settings( settings_file, default_settings ) + eligibility_column, eligibility_values = _render_eligibility_filter( + settings_file, default_settings, survey_data + ) + ( drop_duplicates_option, no_diff_values, @@ -521,6 +615,9 @@ def backchecks_report_settings( enumerator=enumerator, backchecker=backchecker, backcheck_target_percent=backcheck_target_percent, + survey_target=default_settings.survey_target, + eligibility_column=eligibility_column, + eligibility_values=eligibility_values, drop_duplicates_option=drop_duplicates_option, no_differences_list=no_diff_values, exclude_values_list=exclude_values, diff --git a/src/datasure/utils/onboarding_utils.py b/src/datasure/utils/onboarding_utils.py index cba4bccb..6966a838 100644 --- a/src/datasure/utils/onboarding_utils.py +++ b/src/datasure/utils/onboarding_utils.py @@ -855,7 +855,10 @@ class OutputOnboardingInfo: - **Enumerator**: Column identifying the original data collector (e.g., enum_name). - **Backchecker**: Column in the backcheck dataset identifying who conducted the back check (e.g., backchecker_name). - - **Target number of backchecks**: Expected total number of back checks. + - **Backcheck target (%)**: Percentage of surveys to backcheck. Defaults to + the page configuration target, or 10% if neither is set. + - **Eligibility Filter**: Optional column and values that mark a survey + eligible for backchecks (e.g., consent = 1). - **Additional Options**: Duplicate handling (Drop All / Keep First / Keep Last), No Differences Values, Exclude Values, and String Comparison Options (case sensitivity, trim spaces, remove symbols). diff --git a/src/datasure/views/output_view_template.py b/src/datasure/views/output_view_template.py index 1e628b36..6c8b4a2e 100644 --- a/src/datasure/views/output_view_template.py +++ b/src/datasure/views/output_view_template.py @@ -419,6 +419,7 @@ def render_check_tabs(project_id: str, config: PageConfig, data: CheckData) -> N "backchecker": config.backchecker, "backchecker_team": config.backchecker_team, "backcheck_target_percent": config.backcheck_target_percent, + "survey_target": config.survey_target, } backchecks_report( project_id, diff --git a/tests/checks/backchecks/conftest.py b/tests/checks/backchecks/conftest.py index 1e97dd7c..4be405c8 100644 --- a/tests/checks/backchecks/conftest.py +++ b/tests/checks/backchecks/conftest.py @@ -24,7 +24,7 @@ def make_col(): col.text_input.return_value = "" return col - def mock_columns(n_or_spec): + def mock_columns(n_or_spec, **_kwargs): if isinstance(n_or_spec, int): n = n_or_spec elif isinstance(n_or_spec, list | tuple): diff --git a/tests/checks/backchecks/test_compute.py b/tests/checks/backchecks/test_compute.py index ebf0fc51..1e4ed135 100644 --- a/tests/checks/backchecks/test_compute.py +++ b/tests/checks/backchecks/test_compute.py @@ -619,8 +619,6 @@ def test_compute_enumerator_backchecker_stats_enumerator( assert not result.is_empty() assert "enumerator" in result.columns - assert "Surveys" in result.columns - assert "Backchecks" in result.columns assert "Error Rate % (Total)" in result.columns diff --git a/tests/checks/backchecks/test_coverage.py b/tests/checks/backchecks/test_coverage.py new file mode 100644 index 00000000..0695af65 --- /dev/null +++ b/tests/checks/backchecks/test_coverage.py @@ -0,0 +1,295 @@ +"""Tests for backcheck target resolution and coverage against the target.""" + +import json + +import polars as pl +import pytest + +from datasure.checks.backchecks.compute import load_default_backchecks_settings +from datasure.checks.backchecks.coverage import ( + compute_backcheck_coverage, + compute_staff_coverage, + effective_target_percent, + settings_from_page_config, +) +from datasure.checks.backchecks.models import BackcheckSettings + + +def _settings_file(tmp_path, saved: dict) -> str: + file_path = tmp_path / "settings.json" + file_path.write_text(json.dumps({"backchecks": saved})) + return str(file_path) + + +# ============================================ +# TARGET RESOLUTION: panel, then page config, then default +# ============================================ + + +def test_cleared_panel_target_falls_back_to_page_config(tmp_path): + """A cleared panel value does not hide the page config target.""" + settings_file = _settings_file(tmp_path, {"backcheck_target_percent": None}) + page_config = BackcheckSettings(survey_key="key", backcheck_target_percent=15) + + result = load_default_backchecks_settings(settings_file, page_config) + + assert result.backcheck_target_percent == 15 + + +def test_saved_panel_target_wins_over_page_config(tmp_path): + """A target saved in the panel overrides the page config target.""" + settings_file = _settings_file(tmp_path, {"backcheck_target_percent": 25}) + page_config = BackcheckSettings(survey_key="key", backcheck_target_percent=15) + + result = load_default_backchecks_settings(settings_file, page_config) + + assert result.backcheck_target_percent == 25 + + +def test_page_config_without_targets_builds_settings(): + """A page config with no targets builds settings with targets unset.""" + settings = settings_from_page_config( + { + "survey_key": "key", + "backcheck_target_percent": None, + "survey_target": None, + } + ) + + assert settings.backcheck_target_percent is None + assert settings.survey_target is None + + +def test_page_config_zero_targets_mean_not_set(): + """The page config stores 0 when a target is left blank.""" + settings = settings_from_page_config( + {"survey_key": "key", "backcheck_target_percent": 0.0, "survey_target": 0} + ) + + assert settings.backcheck_target_percent is None + assert settings.survey_target is None + + +def test_page_config_targets_are_kept(): + """Page config targets carry through to the settings.""" + settings = settings_from_page_config( + {"survey_key": "key", "backcheck_target_percent": 12.0, "survey_target": 500} + ) + + assert settings.backcheck_target_percent == 12 + assert settings.survey_target == 500 + + +def test_effective_target_defaults_to_ten_percent(): + """With no target set anywhere, 10% is used.""" + settings = BackcheckSettings(survey_key="key") + + assert effective_target_percent(settings) == 10 + + +def test_effective_target_uses_set_target(): + """A set target, including 0, is used as is.""" + assert ( + effective_target_percent( + BackcheckSettings(survey_key="key", backcheck_target_percent=0) + ) + == 0 + ) + + +# ============================================ +# OVERALL COVERAGE +# ============================================ + + +def _coverage_settings(**overrides) -> BackcheckSettings: + return BackcheckSettings( + survey_key="KEY", + survey_id="hhid", + enumerator="enum", + backchecker="bcer", + drop_duplicates_option="drop", + **overrides, + ) + + +def test_on_track_counts_unique_ids_after_duplicate_handling(): + """Duplicated survey IDs are dropped from the base before counting.""" + survey = pl.DataFrame( + { + "KEY": ["k1", "k2", "k3", "k4", "k5"], + "hhid": ["H1", "H1", "H2", "H3", "H4"], + "enum": ["E1", "E1", "E1", "E2", "E2"], + } + ) + backcheck = pl.DataFrame({"KEY": ["b1", "b2"], "hhid": ["H1", "H2"]}) + + coverage = compute_backcheck_coverage(survey, backcheck, _coverage_settings()) + + # H1 is duplicated and dropped: base H2-H4, of which H2 is backchecked. + assert coverage.eligible == 3 + assert coverage.backchecked == 1 + assert coverage.on_track_percent == pytest.approx(100 / 3) + + +def test_eligibility_filter_restricts_the_base(): + """Only surveys whose eligibility column is in the chosen values count.""" + survey = pl.DataFrame( + { + "KEY": ["k1", "k2", "k3", "k4"], + "hhid": ["H1", "H2", "H3", "H4"], + "consent": [1, 0, 1, 1], + } + ) + backcheck = pl.DataFrame({"KEY": ["b1", "b2"], "hhid": ["H1", "H2"]}) + settings = _coverage_settings( + eligibility_column="consent", eligibility_values=["1"] + ) + + coverage = compute_backcheck_coverage(survey, backcheck, settings) + + # H2 did not consent, so its backcheck does not count either. + assert coverage.eligible == 3 + assert coverage.backchecked == 1 + + +def _surveys_with_backchecks(n_surveys: int, n_backchecked: int): + ids = [f"H{i}" for i in range(n_surveys)] + survey = pl.DataFrame({"KEY": [f"k{i}" for i in ids], "hhid": ids}) + backcheck = pl.DataFrame( + {"KEY": [f"b{i}" for i in ids[:n_backchecked]], "hhid": ids[:n_backchecked]} + ) + return survey, backcheck + + +def test_points_vs_target_is_negative_when_behind(): + """1 of 20 backchecked is 5%, 5 points behind a 10% target.""" + survey, backcheck = _surveys_with_backchecks(20, 1) + + coverage = compute_backcheck_coverage( + survey, backcheck, _coverage_settings(backcheck_target_percent=10) + ) + + assert coverage.on_track_percent == pytest.approx(5) + assert coverage.points_vs_target == pytest.approx(-5) + + +def test_expected_backchecks_rounds_up(): + """10% of a 95-response target is 9.5, so 10 backchecks are expected.""" + survey, backcheck = _surveys_with_backchecks(20, 4) + + coverage = compute_backcheck_coverage( + survey, + backcheck, + _coverage_settings(backcheck_target_percent=10, survey_target=95), + ) + + assert coverage.expected_backchecks == 10 + assert coverage.expected_progress_percent == pytest.approx(40) + + +def test_expected_progress_can_exceed_100_percent(): + """Backchecks past the expected total show the real percentage.""" + survey, backcheck = _surveys_with_backchecks(20, 3) + + coverage = compute_backcheck_coverage( + survey, + backcheck, + _coverage_settings(backcheck_target_percent=10, survey_target=20), + ) + + assert coverage.expected_backchecks == 2 + assert coverage.expected_progress_percent == pytest.approx(150) + + +def test_expected_backchecks_unset_without_survey_target(): + """Without a survey target there is no expected total.""" + survey, backcheck = _surveys_with_backchecks(20, 3) + + coverage = compute_backcheck_coverage(survey, backcheck, _coverage_settings()) + + assert coverage.expected_backchecks is None + assert coverage.expected_progress_percent is None + + +def test_coverage_unavailable_without_survey_id_in_backcheck_data(): + """Coverage needs the survey ID in both datasets.""" + survey, _ = _surveys_with_backchecks(5, 0) + backcheck = pl.DataFrame({"KEY": ["b1"]}) + + assert compute_backcheck_coverage(survey, backcheck, _coverage_settings()) is None + + +# ============================================ +# PER-STAFF COVERAGE +# ============================================ + + +def _staff_data(): + survey = pl.DataFrame( + { + "KEY": ["k1", "k2", "k3", "k4"], + "hhid": ["H1", "H2", "H3", "H4"], + "enum": ["E1", "E1", "E1", "E2"], + } + ) + backcheck = pl.DataFrame( + { + "KEY": ["b1", "b2", "b3"], + "hhid": ["H1", "H2", "H9"], + "bcer": ["B1", "B2", "B2"], + } + ) + return survey, backcheck + + +def test_enumerator_coverage_includes_enumerators_without_backchecks(): + """Each enumerator's eligible surveys and how many were backchecked.""" + survey, backcheck = _staff_data() + + result = compute_staff_coverage( + survey, backcheck, _coverage_settings(backcheck_target_percent=10), "enumerator" + ).sort("enum") + + assert result.to_dicts() == [ + { + "enum": "E1", + "Surveys": 3, + "Backchecks": 2, + "Coverage %": pytest.approx(200 / 3), + "vs target": pytest.approx(200 / 3 - 10), + }, + { + "enum": "E2", + "Surveys": 1, + "Backchecks": 0, + "Coverage %": 0.0, + "vs target": -10.0, + }, + ] + + +def test_backchecker_view_counts_matched_backchecks_only(): + """Backcheckers get unique backchecked survey IDs and no coverage columns.""" + survey, backcheck = _staff_data() + + result = compute_staff_coverage( + survey, backcheck, _coverage_settings(), "backchecker" + ).sort("bcer") + + # B2's backcheck of H9 matches no survey, so it does not count. + assert result.to_dicts() == [ + {"bcer": "B1", "Backchecks": 1}, + {"bcer": "B2", "Backchecks": 1}, + ] + + +def test_staff_coverage_empty_without_staff_column(): + """No table when the enumerator column is missing from the survey data.""" + survey, backcheck = _staff_data() + + result = compute_staff_coverage( + survey.drop("enum"), backcheck, _coverage_settings(), "enumerator" + ) + + assert result.is_empty() diff --git a/tests/checks/backchecks/test_models.py b/tests/checks/backchecks/test_models.py index d579dec4..e8a6151e 100644 --- a/tests/checks/backchecks/test_models.py +++ b/tests/checks/backchecks/test_models.py @@ -68,7 +68,7 @@ def test_backcheck_settings_model_valid(): def test_backcheck_settings_model_defaults(): """Test BackcheckSettings model with default values.""" settings = BackcheckSettings(survey_key="survey_id") - assert settings.backcheck_target_percent == 10 + assert settings.backcheck_target_percent is None assert settings.drop_duplicates_option == "drop" assert settings.no_differences_list is None assert settings.exclude_values_list is None diff --git a/tests/checks/backchecks/test_report_ui.py b/tests/checks/backchecks/test_report_ui.py index fb7a1db5..bea3a8d4 100644 --- a/tests/checks/backchecks/test_report_ui.py +++ b/tests/checks/backchecks/test_report_ui.py @@ -458,21 +458,16 @@ def test_render_backcheck_summary(patched_bc): """_render_backcheck_summary renders metrics without errors.""" survey_data = pl.DataFrame({"key": [1, 2], "enum": ["a", "b"]}) backcheck_data = pl.DataFrame({"key": [1], "bcer": ["c"]}) - analysis = pl.DataFrame( - { - "key": [1], - "column_name": ["age"], - "survey_value": ["25"], - "backcheck_value": ["26"], - "match_status": ["mismatch"], - "category": [1], - } - ) settings = BackcheckSettings( - survey_key="key", enumerator="enum", backchecker="bcer" + survey_key="key", + survey_id="key", + enumerator="enum", + backchecker="bcer", + survey_target=10, ) - _render_backcheck_summary(survey_data, backcheck_data, analysis, settings) + _render_backcheck_summary(survey_data, backcheck_data, settings) patched_bc.metric.assert_called() + patched_bc.progress.assert_called_once_with(1.0) def test_render_time_period_selector_backchecks(patched_bc): @@ -773,9 +768,8 @@ def test_render_backcheck_summary_no_key_no_enum_no_bcer(patched_bc): """_render_backcheck_summary hits else branches when key/enum/bcer are None.""" survey_data = pl.DataFrame({"col1": [1, 2]}) backcheck_data = pl.DataFrame({"col1": [1]}) - analysis = pl.DataFrame() settings = BackcheckSettings(survey_key=None, enumerator=None, backchecker=None) - _render_backcheck_summary(survey_data, backcheck_data, analysis, settings) + _render_backcheck_summary(survey_data, backcheck_data, settings) patched_bc.metric.assert_called() From a09055fe274a868e7ec36a03f14792aa3e122dab Mon Sep 17 00:00:00 2001 From: iabaako Date: Sun, 4 Oct 2026 20:41:35 +0000 Subject: [PATCH 2/4] fix(backchecks): fall back to the page config target when the panel is cleared Clearing the backcheck target input left the widget at None for the rest of the session, so the 10% default applied and the "no target set" warning showed even when the page config had a target. The input's change callback now saves the cleared value and restores the page config target in the input. A saved target outside 0-100 (from the old count-based input) is ignored instead of failing validation. Also reads the target from BackcheckCoverage in the coverage card, and moves the breaking BackcheckSettings and error-statistics changes under Changed in the changelog. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 42 ++++++++++++------- src/datasure/checks/backchecks/compute.py | 9 ++-- src/datasure/checks/backchecks/report_ui.py | 8 ++-- src/datasure/checks/backchecks/settings_ui.py | 38 ++++++++++++++--- tests/checks/backchecks/test_coverage.py | 10 +++++ tests/checks/backchecks/test_settings_ui.py | 4 +- 6 files changed, 79 insertions(+), 32 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7a8f7863..0bd9d86d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -82,22 +82,32 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 `CorrectionEntry.severity` sets it and is rejected on non-accept actions, on acceptances of other checks, and with any value other than `hard`. Hard acceptances are highlighted in the Correction Log — #298 -- **Backcheck targets**: The Backchecks Summary tracks progress against the - backcheck target. "Backcheck Coverage" is now the share of eligible unique - survey IDs (after duplicate handling) with a matching backcheck, shown - against the target, and is calculated before any comparison columns are - configured. A new Targets row shows backchecks done against - `ceil(survey_target × target% / 100)` when the page configuration sets - `survey_target`. The target % resolves from the settings panel, then the - page configuration, then 10% (`BackcheckSettings.backcheck_target_percent` - is now `float | None`; a page-config target of 0 counts as not set). A new - optional eligibility filter (column plus values) restricts the base. In the - Enumerator Backchecker Error Statistics table, the enumerator view's - "Surveys" and "Backchecks" are now eligible unique submissions and how many - were backchecked, with new "Coverage %" and "vs target" columns; the - backchecker view shows backchecks done. The table renders before comparison - columns are configured. Calculations live in the new - `checks/backchecks/coverage.py` — #318 +- **Backcheck targets**: New `checks/backchecks/coverage.py`. + `compute_backcheck_coverage` returns `BackcheckCoverage`: eligible unique + survey IDs (after duplicate handling and the optional eligibility filter), + how many have a matching backcheck, the on-track %, points vs the target, and + expected backchecks, `ceil(survey_target × target% / 100)`, when + `survey_target` is set. `compute_staff_coverage` returns per-enumerator + coverage, including enumerators with no backchecks, or backchecks done per + backchecker. Neither needs comparison columns. The Backchecks Summary shows + coverage against the target and a new Targets row, and the settings panel + gains an eligibility filter (`eligibility_column`, `eligibility_values`). + `settings_from_page_config` builds `BackcheckSettings` from the page config, + where a target of 0 means not set — #318 + +### Changed + +- **Breaking**: `BackcheckSettings.backcheck_target_percent` is now + `float | None` (0–100), defaulting to None instead of 10; + `effective_target_percent` applies the 10% default. The target resolves from + the settings panel, then the page config, then 10%. A cleared panel value, or + a saved value outside 0–100, falls back to the page config. + `BackcheckSettings` gains `survey_target` — #318 +- **Breaking**: `compute_enumerator_backchecker_stats` no longer returns the + "Surveys" and "Backchecks" columns (both were the count of compared survey + KEYs). The Enumerator Backchecker Error Statistics table now takes them from + `compute_staff_coverage` and renders before comparison columns are + configured — #318 ### Fixed diff --git a/src/datasure/checks/backchecks/compute.py b/src/datasure/checks/backchecks/compute.py index 670a4a30..8c693eef 100644 --- a/src/datasure/checks/backchecks/compute.py +++ b/src/datasure/checks/backchecks/compute.py @@ -29,8 +29,8 @@ def load_default_backchecks_settings( Loads previously saved backcheck report settings from the settings file and merges them with the provided default configuration. Saved settings - take precedence over defaults, except a cleared (None) backcheck target, - which falls back to the configured one. + take precedence over defaults, except a cleared or invalid backcheck + target, which falls back to the configured one. Parameters ---------- @@ -45,7 +45,10 @@ def load_default_backchecks_settings( Merged settings combining saved and default configurations. """ saved_settings = load_check_settings(settings_file, TAB_NAME) - if saved_settings.get("backcheck_target_percent") is None: + # A cleared target falls back to the configured one, as does a value saved + # by the old count-based input that is not a valid percentage. + saved_target = saved_settings.get("backcheck_target_percent") + if not isinstance(saved_target, int | float) or not 0 <= saved_target <= 100: saved_settings.pop("backcheck_target_percent", None) default_settings: dict = dict(config) diff --git a/src/datasure/checks/backchecks/report_ui.py b/src/datasure/checks/backchecks/report_ui.py index f2f2b0ae..a12a3d57 100644 --- a/src/datasure/checks/backchecks/report_ui.py +++ b/src/datasure/checks/backchecks/report_ui.py @@ -617,7 +617,7 @@ def _render_backcheck_summary( st.metric("Backcheck Observations", f"{len(backcheck_data):,}") with uc3, st.container(border=True): - _render_coverage_metric(coverage, backcheck_settings) + _render_coverage_metric(coverage) with lc1, st.container(border=True): st.metric( @@ -634,11 +634,8 @@ def _render_backcheck_summary( _render_expected_backchecks(coverage) -def _render_coverage_metric( - coverage: BackcheckCoverage | None, backcheck_settings: BackcheckSettings -) -> None: +def _render_coverage_metric(coverage: BackcheckCoverage | None) -> None: """Render the on-track coverage card with its delta against the target.""" - target_percent = effective_target_percent(backcheck_settings) if coverage is None or coverage.on_track_percent is None: st.metric( "Backcheck Coverage", @@ -648,6 +645,7 @@ def _render_coverage_metric( ) return + target_percent = coverage.target_percent st.metric( "Backcheck Coverage", f"{coverage.on_track_percent:.1f}%", diff --git a/src/datasure/checks/backchecks/settings_ui.py b/src/datasure/checks/backchecks/settings_ui.py index 3c494a78..67ea4201 100644 --- a/src/datasure/checks/backchecks/settings_ui.py +++ b/src/datasure/checks/backchecks/settings_ui.py @@ -245,14 +245,38 @@ def _render_staff_identifiers( return enumerator, backchecker +TARGET_INPUT_KEY: str = "backcheck_goal_backchecks" +TARGET_STATE_NAME: str = TAB_NAME + "_backcheck_target_percent" + + +def _on_target_change(settings_file: str, page_config_target: float | None) -> None: + """Flag a changed target for saving; on clear, fall back to the page config. + + Clearing saves None (so the page config target applies in later sessions) + and shows the page config target in the input straight away. The save + clears the flag, so the input's restored value is not saved as a panel + value. + """ + trigger_save(state_name=TARGET_STATE_NAME) + if ( + TARGET_INPUT_KEY in st.session_state + and st.session_state[TARGET_INPUT_KEY] is None + ): + save_check_settings(settings_file, TAB_NAME, {"backcheck_target_percent": None}) + if page_config_target is not None: + st.session_state[TARGET_INPUT_KEY] = float(page_config_target) + + def _render_tracking_options( - settings_file: str, default_settings: BackcheckSettings + settings_file: str, + default_settings: BackcheckSettings, + page_config_target: float | None = None, ) -> float | None: """Render tracking options section. The target input is pre-filled with the saved panel value, else the page config target. A value the user changes is saved and wins over the page - config; clearing it saves None, which falls back to the page config. + config; clearing it saves None and falls back to the page config. Parameters ---------- @@ -260,6 +284,8 @@ def _render_tracking_options( Path to settings file. default_settings : BackcheckSettings Default settings. + page_config_target : float | None + Backcheck target % from the page configuration, if set. Returns ------- @@ -280,10 +306,10 @@ def _render_tracking_options( format="%.1f", help="Percentage of survey submissions to backcheck. Leave blank " "to use the target from the page configuration.", - key="backcheck_goal_backchecks", + key=TARGET_INPUT_KEY, value=float(default_target) if default_target is not None else None, - on_change=trigger_save, - kwargs={"state_name": TAB_NAME + "_backcheck_target_percent"}, + on_change=_on_target_change, + args=(settings_file, page_config_target), ) save_check_settings( settings_file, @@ -593,7 +619,7 @@ def backchecks_report_settings( ) backcheck_target_percent = _render_tracking_options( - settings_file, default_settings + settings_file, default_settings, config.backcheck_target_percent ) eligibility_column, eligibility_values = _render_eligibility_filter( diff --git a/tests/checks/backchecks/test_coverage.py b/tests/checks/backchecks/test_coverage.py index 0695af65..c827574e 100644 --- a/tests/checks/backchecks/test_coverage.py +++ b/tests/checks/backchecks/test_coverage.py @@ -46,6 +46,16 @@ def test_saved_panel_target_wins_over_page_config(tmp_path): assert result.backcheck_target_percent == 25 +def test_saved_target_outside_percent_range_is_ignored(tmp_path): + """A saved value from the old count-based input does not break loading.""" + settings_file = _settings_file(tmp_path, {"backcheck_target_percent": 250}) + page_config = BackcheckSettings(survey_key="key", backcheck_target_percent=15) + + result = load_default_backchecks_settings(settings_file, page_config) + + assert result.backcheck_target_percent == 15 + + def test_page_config_without_targets_builds_settings(): """A page config with no targets builds settings with targets unset.""" settings = settings_from_page_config( diff --git a/tests/checks/backchecks/test_settings_ui.py b/tests/checks/backchecks/test_settings_ui.py index 97f53262..fc9df107 100644 --- a/tests/checks/backchecks/test_settings_ui.py +++ b/tests/checks/backchecks/test_settings_ui.py @@ -289,8 +289,8 @@ def test_render_tracking_options_persists_changed_target(tmp_path): mock_st = make_mock_st() mock_st.session_state = session_state - def change_target(*_args, on_change, kwargs, **_widget_kwargs): - on_change(**kwargs) + def change_target(*_args, on_change, args, **_widget_kwargs): + on_change(*args) return 35 mock_st.number_input.side_effect = change_target From 5280dcabdbea8b4a5a4bf1728a1fdd54a2901598 Mon Sep 17 00:00:00 2001 From: iabaako Date: Sun, 4 Oct 2026 21:05:04 +0000 Subject: [PATCH 3/4] feat(backchecks): show coverage and expected backchecks side by side in Targets The summary's count cards now share one row, and a Targets section shows Backcheck Coverage next to Backchecks vs Expected. The progress bar is gone; Backchecks vs Expected shows how many backchecks above or below the expected total as its delta, like coverage's points vs target. The wider card also stops the "done / expected (%)" value from being truncated. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 8 ++-- docs/USER_GUIDE.md | 11 +++-- src/datasure/checks/backchecks/report_ui.py | 50 ++++++++++----------- tests/checks/backchecks/test_report_ui.py | 13 ++++-- 4 files changed, 47 insertions(+), 35 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0bd9d86d..7e9b4392 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -89,9 +89,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 expected backchecks, `ceil(survey_target × target% / 100)`, when `survey_target` is set. `compute_staff_coverage` returns per-enumerator coverage, including enumerators with no backchecks, or backchecks done per - backchecker. Neither needs comparison columns. The Backchecks Summary shows - coverage against the target and a new Targets row, and the settings panel - gains an eligibility filter (`eligibility_column`, `eligibility_values`). + backchecker. Neither needs comparison columns. The Backchecks Summary has a + new Targets section showing coverage against the target % and backchecks + done against expected, each with its deviation as a delta. The settings + panel gains an eligibility filter (`eligibility_column`, + `eligibility_values`). `settings_from_page_config` builds `BackcheckSettings` from the page config, where a target of 0 means not set — #318 diff --git a/docs/USER_GUIDE.md b/docs/USER_GUIDE.md index edef7a0d..72343026 100644 --- a/docs/USER_GUIDE.md +++ b/docs/USER_GUIDE.md @@ -1005,14 +1005,17 @@ Configure validation: ##### Backchecks Summary +A row of counts (survey observations, back check observations, enumerators and +back checkers), then a **Targets** section with two metrics side by side: + - **Backcheck Coverage**: Share of eligible unique survey IDs (after duplicate handling) with at least one matching back check, and how many points it is above or below the target. It is calculated before any back check columns are configured. -- **Targets**: When the page configuration sets the target number of survey - responses, back checks done against the back checks expected - (survey target × target %, rounded up), with a progress bar. Values over - 100% are shown as is. +- **Backchecks vs Expected**: When the page configuration sets the target + number of survey responses, back checks done against the back checks + expected (survey target × target %, rounded up), and how many back checks + above or below that it is. Values over 100% are shown as is. **Add Back Check Columns**: Click "Add a back check column" (+ button): diff --git a/src/datasure/checks/backchecks/report_ui.py b/src/datasure/checks/backchecks/report_ui.py index a12a3d57..48f5b063 100644 --- a/src/datasure/checks/backchecks/report_ui.py +++ b/src/datasure/checks/backchecks/report_ui.py @@ -607,31 +607,34 @@ def _render_backcheck_summary( n_backcheckers = 0 # Display metrics in columns - uc1, uc2, uc3, _ = st.columns(4) - lc1, lc2, _, _ = st.columns(4) + c1, c2, c3, c4 = st.columns(4) - with uc1, st.container(border=True): + with c1, st.container(border=True): st.metric("Survey Observations", f"{len(survey_data):,}") - with uc2, st.container(border=True): + with c2, st.container(border=True): st.metric("Backcheck Observations", f"{len(backcheck_data):,}") - with uc3, st.container(border=True): - _render_coverage_metric(coverage) - - with lc1, st.container(border=True): + with c3, st.container(border=True): st.metric( "Total Enumerators", f"{n_enumerators:,}" if n_enumerators > 0 else "N/A" ) - with lc2, st.container(border=True): + with c4, st.container(border=True): st.metric( "Total Back Checkers", f"{n_backcheckers:,}" if n_backcheckers > 0 else "N/A", ) + st.markdown("##### Targets") + tc1, tc2 = st.columns(2) + + with tc1, st.container(border=True): + _render_coverage_metric(coverage) + if coverage is not None: - _render_expected_backchecks(coverage) + with tc2: + _render_expected_backchecks(coverage) def _render_coverage_metric(coverage: BackcheckCoverage | None) -> None: @@ -656,8 +659,7 @@ def _render_coverage_metric(coverage: BackcheckCoverage | None) -> None: def _render_expected_backchecks(coverage: BackcheckCoverage) -> None: - """Render the Targets row: backchecks done against backchecks expected.""" - st.markdown("##### Targets") + """Render backchecks done against expected, with the deviation as a delta.""" if coverage.expected_backchecks is None: st.info( "Set the target number of responses for the survey in the page " @@ -666,18 +668,16 @@ def _render_expected_backchecks(coverage: BackcheckCoverage) -> None: ) return - progress = coverage.expected_progress_percent or 0.0 - tc1, tc2 = st.columns([1, 3], vertical_alignment="center") - with tc1, st.container(border=True): + deviation = coverage.backchecked - coverage.expected_backchecks + with st.container(border=True): st.metric( "Backchecks vs Expected", f"{coverage.backchecked:,} / {coverage.expected_backchecks:,} " - f"({progress:.0f}%)", + f"({coverage.expected_progress_percent:.0f}%)", + delta=f"{deviation:+,} backchecks vs target", help=f"Expected backchecks: {coverage.target_percent:g}% of the " "survey target, rounded up.", ) - with tc2: - st.progress(min(progress / 100, 1.0)) def _render_backchecker_productivity( @@ -1732,13 +1732,13 @@ def backchecks_report( demo_callout( """ ##### Backchecks Summary - Five metrics appear here: Survey Observations, Backcheck Observations, - Backcheck Coverage, Total Enumerators, and Total Back Checkers. - Backcheck Coverage is the share of eligible unique surveys that have been - backchecked, with how far it is above or below the backcheck target. - When the survey's target number of responses is set in the page - configuration, a **Targets** row shows backchecks done against the - total number of backchecks expected. + Four metrics appear here: Survey Observations, Backcheck Observations, + Total Enumerators, and Total Back Checkers. Below them, the **Targets** + section shows Backcheck Coverage, the share of eligible unique surveys + that have been backchecked, with how far it is above or below the + backcheck target. When the survey's target number of responses is set + in the page configuration, it also shows backchecks done against the + total number of backchecks expected, with how many above or below. Below the metrics, a **Backchecker Productivity** table shows submission counts per backchecker over time. Use the **Daily / Weekly / Monthly** pills diff --git a/tests/checks/backchecks/test_report_ui.py b/tests/checks/backchecks/test_report_ui.py index bea3a8d4..68dbe533 100644 --- a/tests/checks/backchecks/test_report_ui.py +++ b/tests/checks/backchecks/test_report_ui.py @@ -463,11 +463,18 @@ def test_render_backcheck_summary(patched_bc): survey_id="key", enumerator="enum", backchecker="bcer", - survey_target=10, + survey_target=30, ) _render_backcheck_summary(survey_data, backcheck_data, settings) - patched_bc.metric.assert_called() - patched_bc.progress.assert_called_once_with(1.0) + # 10% of 30 expects 3 backchecks; 1 is done, 2 short of the target. + expected_card = next( + c + for c in patched_bc.metric.call_args_list + if c.args[0] == "Backchecks vs Expected" + ) + assert expected_card.args[1] == "1 / 3 (33%)" + assert expected_card.kwargs["delta"] == "-2 backchecks vs target" + patched_bc.progress.assert_not_called() def test_render_time_period_selector_backchecks(patched_bc): From dfcd46310413df153e1c2a5809c7de2dfb3775de Mon Sep 17 00:00:00 2001 From: iabaako Date: Mon, 5 Oct 2026 13:57:23 +0000 Subject: [PATCH 4/4] fix(backchecks): render Backchecks vs Expected with a 0% target A 0% target with a survey target set expects 0 backchecks, which leaves the progress percentage unset, and formatting it crashed the summary. The card now shows "done / 0" without a percentage, and keeps the deviation delta. Co-Authored-By: Claude Opus 5.5 --- src/datasure/checks/backchecks/report_ui.py | 7 +++++-- tests/checks/backchecks/test_report_ui.py | 20 ++++++++++++++++++++ 2 files changed, 25 insertions(+), 2 deletions(-) diff --git a/src/datasure/checks/backchecks/report_ui.py b/src/datasure/checks/backchecks/report_ui.py index 48f5b063..dfccfaeb 100644 --- a/src/datasure/checks/backchecks/report_ui.py +++ b/src/datasure/checks/backchecks/report_ui.py @@ -669,11 +669,14 @@ def _render_expected_backchecks(coverage: BackcheckCoverage) -> None: return deviation = coverage.backchecked - coverage.expected_backchecks + value = f"{coverage.backchecked:,} / {coverage.expected_backchecks:,}" + # A 0% target expects no backchecks, so there is no percentage to show. + if coverage.expected_progress_percent is not None: + value += f" ({coverage.expected_progress_percent:.0f}%)" with st.container(border=True): st.metric( "Backchecks vs Expected", - f"{coverage.backchecked:,} / {coverage.expected_backchecks:,} " - f"({coverage.expected_progress_percent:.0f}%)", + value, delta=f"{deviation:+,} backchecks vs target", help=f"Expected backchecks: {coverage.target_percent:g}% of the " "survey target, rounded up.", diff --git a/tests/checks/backchecks/test_report_ui.py b/tests/checks/backchecks/test_report_ui.py index 68dbe533..0bd90075 100644 --- a/tests/checks/backchecks/test_report_ui.py +++ b/tests/checks/backchecks/test_report_ui.py @@ -477,6 +477,26 @@ def test_render_backcheck_summary(patched_bc): patched_bc.progress.assert_not_called() +def test_render_backcheck_summary_zero_target(patched_bc): + """A 0% target expects no backchecks and renders without a percentage.""" + survey_data = pl.DataFrame({"key": [1, 2]}) + backcheck_data = pl.DataFrame({"key": [1]}) + settings = BackcheckSettings( + survey_key="key", + survey_id="key", + backcheck_target_percent=0, + survey_target=30, + ) + _render_backcheck_summary(survey_data, backcheck_data, settings) + expected_card = next( + c + for c in patched_bc.metric.call_args_list + if c.args[0] == "Backchecks vs Expected" + ) + assert expected_card.args[1] == "1 / 0" + assert expected_card.kwargs["delta"] == "+1 backchecks vs target" + + def test_render_time_period_selector_backchecks(patched_bc): """_render_time_period_selector_backchecks returns selected time period.""" patched_bc.pills.return_value = "Week"