Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
38 changes: 38 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -156,6 +156,43 @@ to a JSON file, a JSON string, a pre-instantiated config object, or `None`
(see [How to implement new metrics](#how-to-implement-new-metrics)). If no
`writer_config_path` is given, results are printed to the console.

### Coverage-gap diversity from MUPs

`diversity_coverageGap` calculates the exact finite-domain DNF union induced
by a supplied set of Maximal Uncovered Patterns (MUPs). The paper-defined
coverage gap and exact pattern counts are included in `DQexplanation`.
`DQvalue` is the complementary coverage-space score (`1 - coverage_gap`) so
that it follows the METIS convention that higher scores indicate better data
quality.

```python
from metis.metric.diversity.diversity_coverageGap_config import (
diversity_coverageGap_config,
)

config = diversity_coverageGap_config(
mups_path="path/to/mups_bluenile.csv_mincov_19000.txt",
attributes=[
"shape", "color", "cut", "clarity",
"polish", "symmetry", "florescence",
],
mincov=19000,
)

orchestrator.assess(
metrics=["diversity_coverageGap"],
metric_configs=[config],
)
```

MUP fields are mapped positionally to `attributes`. Use the same attribute
order as during MUP discovery. The wildcard defaults to `x`; trailing fields
are interpreted exactly as in FLAPS: the final value is parsed as the MUP's
actual coverage and validated against `mincov`. The coverage value verifies the
frontier but does not change the geometric DNF volume. In the GUI, select
**Diversity: Coverage Gap** and upload the MUP file directly in its metric
configuration.

### Data loader configs

Datasets are described by small JSON configs (see `data/*.json`). File paths
Expand Down Expand Up @@ -214,6 +251,7 @@ Writer config details are also covered in
| Consistency | `consistency_ruleBasedHinrichs` | Rule-based consistency score after Hinrichs (attribute and tuple rules) |
| Consistency | `consistency_ruleBasedPipino` | Rule-based consistency score after Pipino (boolean rules) |
| Correctness | `correctness_heinrich` | Cell-wise correctness against a reference dataset after Heinrich |
| Diversity | `diversity_coverageGap` | Exact MUP-induced DNF coverage space; details include the coverage gap |
| Minimality | `minimality_duplicateCount` | Duplicate rows in the dataset |
| Timeliness | `timeliness_heinrich` | Decay-based timeliness of date columns after Heinrich |
| Validity | `validity_outOfVocabulary` | Share of values outside a known vocabulary |
Expand Down
5 changes: 4 additions & 1 deletion docs/GUI.md
Original file line number Diff line number Diff line change
Expand Up @@ -75,14 +75,17 @@ metric is selected and no blockers remain.
availability warnings. Metrics whose native dependencies are missing (for
example FAHES for `completeness_nullAndDMVRatio`) are disabled with a
warning.
- Metrics are configured inline through one of three editors, chosen by the
- Metrics are configured inline through the appropriate editor, chosen by the
metric's metadata (see the config conventions in the
[README](../README.md#config-conventions)):
- a form editor for plain dataclass configs
- a Python editor for callable rule configs (`consistency_ruleBased*`)
- an inline rule editor for functional dependencies
(`consistency_countFDViolations`)
- `timeliness_heinrich` gets a dedicated per-column editor
- `diversity_coverageGap` gets a dedicated MUP-file uploader and positional
dataset-attribute mapping; its `mincov` is inferred from filenames such as
`*_mincov_19000.txt` when available
- Select all and deselect buttons exist per dimension. The page lists
blockers (missing required configs, missing reference dataset) before
letting you continue.
Expand Down
174 changes: 174 additions & 0 deletions gui/ui/components/config_editors/mups_editor.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,174 @@
"""Upload and configure a FLAPS MUP file for coverage-gap assessment."""

from __future__ import annotations

import csv
import io
import re

import streamlit as st

_MINCOV_RE = re.compile(r"mincov[_=-]?(\d+)", re.IGNORECASE)
_IDENTIFIER_NAMES = {"id", "row_id", "rowid", "index"}


def render(config_class, key_prefix: str, df_columns: list[str]):
"""Render the MUP upload and positional attribute mapping controls."""
st.caption(
"Upload the MUP output for this dataset, then select the dataset "
"attributes used during MUP discovery. Attributes are interpreted in "
"dataset-column order. The final field of each row is read as that "
"MUP's actual coverage."
)

delimiter = st.text_input(
"MUP delimiter",
value=",",
max_chars=1,
key=f"{key_prefix}__delimiter",
)
wildcard = st.text_input(
"Wildcard token",
value="x",
key=f"{key_prefix}__wildcard",
help="Token used by FLAPS for an unspecified attribute.",
)
uploaded = st.file_uploader(
"MUP file",
type=["txt", "csv"],
key=f"{key_prefix}__mups_upload",
help="FLAPS rows contain the positional pattern followed by its actual coverage.",
)

if uploaded is None:
st.caption("Upload a MUP file to complete this metric configuration.")
return None

raw = uploaded.getvalue()
try:
content = raw.decode("utf-8-sig")
except UnicodeDecodeError:
content = raw.decode("latin-1")

if len(delimiter) != 1:
st.error("The MUP delimiter must be exactly one character.")
return None

first_fields = _first_mup_fields(content, delimiter)
file_id = f"{uploaded.name}::{uploaded.size}"
file_state_key = f"{key_prefix}__mups_file_id"
attributes_key = f"{key_prefix}__attributes"
mincov_key = f"{key_prefix}__mincov"
if st.session_state.get(file_state_key) != file_id:
st.session_state[file_state_key] = file_id
st.session_state[attributes_key] = _default_attributes(
df_columns, len(first_fields)
)
inferred_mincov = _infer_mincov(uploaded.name)
st.session_state[mincov_key] = (
str(inferred_mincov) if inferred_mincov is not None else ""
)

selected = st.multiselect(
"Diversity attributes",
options=df_columns,
key=attributes_key,
help=(
"Select exactly the attributes used to discover these MUPs. Their "
"dataset order must match the positional fields in the MUP file."
),
)
attributes = [column for column in df_columns if column in set(selected)]

mincov_raw = st.text_input(
"Minimum coverage threshold (mincov, optional)",
key=mincov_key,
placeholder="e.g. 19000",
help=(
"Checks that every MUP's final coverage value is below the "
"threshold. The supplied MUP frontier determines the DNF count."
),
)

if not attributes:
st.caption("Select at least one diversity attribute.")
return None
if first_fields and len(first_fields) < len(attributes) + 1:
st.error(
f"The first MUP row has {len(first_fields)} fields, but "
f"{len(attributes)} pattern fields plus one final coverage field "
"are required."
)
return None

intermediate = len(first_fields) - len(attributes) - 1 if first_fields else 0
mapping = ", ".join(
f"{index + 1}: `{attribute}`" for index, attribute in enumerate(attributes)
)
st.caption(f"Positional mapping — {mapping}")
if not first_fields:
st.warning(
"The uploaded file contains no MUP rows. If this is intentional, "
"the coverage gap is 0 and the coverage-space score is 1."
)
else:
st.caption(
"The final field of every MUP row is parsed as its actual coverage; "
"it validates the frontier but is not a DNF dimension."
)
if intermediate > 0:
st.caption(
f"The {intermediate} field{'s' if intermediate != 1 else ''} between "
"the pattern and final coverage will be ignored as metadata."
)

mincov = None
if mincov_raw.strip():
try:
mincov = int(mincov_raw)
except ValueError:
st.error("mincov must be a positive integer.")
return None

try:
config = config_class(
mups_content=content,
mups_filename=uploaded.name,
attributes=attributes,
mincov=mincov,
wildcard=wildcard,
delimiter=delimiter,
)
config.validate()
return config
except (TypeError, ValueError) as exc:
st.error(f"Config error: {exc}")
return None


def _first_mup_fields(content: str, delimiter: str) -> list[str]:
for fields in csv.reader(io.StringIO(content), delimiter=delimiter):
if not fields or all(not field.strip() for field in fields):
continue
if fields[0].lstrip().startswith("#"):
continue
return fields
return []


def _default_attributes(df_columns: list[str], mup_field_count: int) -> list[str]:
non_identifiers = [
column for column in df_columns if column.lower() not in _IDENTIFIER_NAMES
]
# FLAPS output commonly appends one coverage value after the pattern.
likely_pattern_width = max(1, mup_field_count - 1)
if len(non_identifiers) == likely_pattern_width:
return non_identifiers
if len(df_columns) == likely_pattern_width:
return list(df_columns)
return non_identifiers or list(df_columns)


def _infer_mincov(filename: str) -> int | None:
match = _MINCOV_RE.search(filename)
return int(match.group(1)) if match else None
1 change: 1 addition & 0 deletions gui/ui/icons.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
"Completeness": ":material/water_drop:",
"Consistency": ":material/link:",
"Correctness": ":material/check_circle:",
"Diversity": ":material/diversity_3:",
"Minimality": ":material/compress:",
"Timeliness": ":material/schedule:",
"Validity": ":material/fact_check:",
Expand Down
11 changes: 11 additions & 0 deletions gui/ui/pages/metrics_page.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@
from metis.utils.dq_granularity import DQGranularity
from ui.components.config_editors import (
callable_editor,
mups_editor,
simple_editor,
timeliness_editor,
)
Expand Down Expand Up @@ -718,6 +719,16 @@ def _render_inline_config(info: MetricInfo, df) -> None:
AppState.set_metric_config(info.name, cfg)
return

if info.name == "diversity_coverageGap":
cfg = mups_editor.render(
info.config_class,
key_prefix=info.name,
df_columns=list(df.columns),
)
if cfg is not None:
AppState.set_metric_config(info.name, cfg)
return

if info.config_class:
# If cell-level output isn't recommended for this metric, pre-select
# column-axis aggregation so the user doesn't have to know that the
Expand Down
1 change: 1 addition & 0 deletions metis/metric/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
from .consistency.consistency_ruleBasedHinrichs import consistency_ruleBasedHinrichs
from .consistency.consistency_ruleBasedPipino import consistency_ruleBasedPipino
from .correctness.correctness_heinrich import correctness_heinrich
from .diversity.diversity_coverageGap import diversity_coverageGap
from .metric import Metric
from .minimality.minimality_duplicateCount import minimality_duplicateCount
from .minimality.minimality_clustering import minimality_clustering
Expand Down
5 changes: 5 additions & 0 deletions metis/metric/diversity/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
"""Coverage-based diversity metrics."""

from .diversity_coverageGap import diversity_coverageGap

__all__ = ["diversity_coverageGap"]
Loading