feat(P7b): explainable aggregate scores — expose denominator + formula

Matter readiness, the completeness grade and command/status coverage each now
carry a `math` block: numerator, denominator, the exact formula that reproduces
the percentage, and (for Matter) the functions excluded from the denominator
because their category has no Matter cluster — previously a silent skip and the
biggest "why is this number what it is?" gap. Completeness also states its band
thresholds. Report-only, additive; scores unchanged, only now auditable.
Raised by external review. tests/test_explainable_aggregates.py (12/12 green).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Nikolay1
2026-07-15 10:15:40 +02:00
co-authored by Claude Opus 4.8
parent f68c85bb48
commit 1a60264c78
4 changed files with 138 additions and 5 deletions
+8
View File
@@ -22,6 +22,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added
- **Explainable aggregate scores** (`advanced.py`, `handover.py`). Every headline percentage now ships
the numbers behind it instead of a bare figure: Matter readiness, the completeness grade and
command/status coverage each carry a `math` block with the numerator, denominator, the exact formula
(that reproduces the percentage) and, for Matter, the functions **excluded** from the denominator
because their category has no Matter cluster (previously a silent skip — the biggest source of "why is
this number what it is?"). The completeness grade also states its band thresholds. A reviewer can now
audit or reproduce any score. Report-only, additive. Raised by external review. `tests/test_explainable_aggregates.py`.
- **Provenance / confidence** (`explain.py`, new MCP tool `explain_ga`). The enriched model mixes ETS
facts, DPT-derived structure and name heuristics; downstream tools then treat the result almost like a
fact. `explain_ga(address)` makes the reasoning auditable for one GA: per decision (category / kind /
+44 -3
View File
@@ -26,6 +26,30 @@ def _functional_commands(project: LoadedProject) -> list:
and g.dpt_main is not None]
def _ratio_explain(numerator: int, denominator: int, unit: str,
excluded: dict[str, Any] | None = None,
bands: dict[str, str] | None = None) -> dict[str, Any]:
"""A percentage a reader can audit: the numbers behind it, the exact formula,
what was left OUT of the denominator, and (optionally) the grade bands.
Every aggregate score this tool reports carries one of these so the percentage
is never a bare number — a reviewer can see the denominator and reproduce it.
"""
pct = round(100 * numerator / denominator) if denominator else 0
out: dict[str, Any] = {
"pct": pct,
"numerator": numerator,
"denominator": denominator,
"formula": f"{numerator} / {denominator} {unit} = {pct}%"
if denominator else f"0 {unit} — ratio undefined (denominator 0)",
}
if excluded:
out["excluded_from_denominator"] = excluded
if bands:
out["bands"] = bands
return out
def _status_gas(project: LoadedProject) -> list:
from .analyze import _is_status_ga
return [g for g in project.gas.values() if _is_status_ga(g)]
@@ -48,9 +72,11 @@ def matter_readiness(project: LoadedProject) -> dict[str, Any]:
"""Which controllable functions round-trip to a Matter cluster, and what's missing."""
stats = _status_gas(project)
ready, not_ready = [], []
no_cluster: Counter = Counter()
for ga in _functional_commands(project):
cluster = _MATTER.get(ga.category)
if cluster is None:
no_cluster[ga.category or "unknown"] += 1 # excluded from the ratio — count it
continue
has_status = find_status(ga, [s for s in stats if s.main == ga.main]) is not None \
or find_status(ga, stats) is not None
@@ -58,14 +84,25 @@ def matter_readiness(project: LoadedProject) -> dict[str, Any]:
"matter": cluster[0], "has_status": has_status}
(ready if has_status else not_ready).append(row)
total = len(ready) + len(not_ready)
excluded_n = sum(no_cluster.values())
math = _ratio_explain(
len(ready), total, "Matter-mappable controllable functions with a status GA",
excluded={
"controllable_functions_without_a_matter_cluster": excluded_n,
"by_category": dict(no_cluster.most_common()),
"why": "categories with no Matter cluster (e.g. scenes, diagnostics) can't "
"round-trip, so they are not counted in the readiness denominator.",
} if excluded_n else None)
return {
"controllable_functions": total,
"matter_ready": len(ready),
"ready_pct": (100 * len(ready) // total) if total else 0,
"ready_pct": math["pct"],
"math": math,
"not_ready": not_ready[:200],
"note": "Ready = has a status GA + decodable DPT, so the Matter cluster can report "
"state. A Matter bridge (e.g. HA Matter server) exposes these; functions "
"without a status GA won't round-trip. Static readiness only — no bridging here.",
"without a status GA won't round-trip. Static readiness only — no bridging here. "
"See `math` for the exact denominator and what was excluded.",
}
@@ -113,12 +150,16 @@ def completeness_grade(project: LoadedProject) -> dict[str, Any]:
if n == 0:
missing.append(label)
hit = sum(1 for v in present.values() if v)
score = round(100 * hit / len(_PATTERNS))
bands = {"as-built grade": ">=75", "near-complete": "55-74",
"functional skeleton": "30-54", "bare skeleton": "<30"}
math = _ratio_explain(hit, len(_PATTERNS), "as-built patterns present", bands=bands)
score = math["pct"]
grade = ("as-built grade" if score >= 75 else
"near-complete" if score >= 55 else
"functional skeleton" if score >= 30 else "bare skeleton")
return {
"grade": grade, "score": score,
"math": math,
"patterns_present": {k: v for k, v in present.items() if v},
"patterns_missing": missing,
"note": "Completeness = presence of the as-built patterns a professional adds beyond "
+9 -2
View File
@@ -86,7 +86,14 @@ def _feedback_coverage(project: LoadedProject,
and ga.dpt_main is not None]
gaps = sum(1 for f in missing if f["code"] == "missing_status_address")
total = len(commands)
return {"commands": total, "with_status": max(total - gaps, 0), "missing": gaps}
with_status = max(total - gaps, 0)
pct = round(100 * with_status / total) if total else 0
return {
"commands": total, "with_status": with_status, "missing": gaps,
"pct": pct,
"formula": f"{with_status} / {total} functional command GAs have a status = {pct}%"
if total else "no functional command GAs — coverage undefined",
}
# --------------------------------------------------------------------------- #
@@ -211,7 +218,7 @@ def build_handover(project: LoadedProject,
# 4. Feedback coverage
md.append("\n## 4. Command / status coverage\n")
pct = (100 * cover["with_status"] // cover["commands"]) if cover["commands"] else 0
pct = cover["pct"]
md.append(
f"- Functional command GAs: **{cover['commands']}**\n"
f"- With a status/feedback GA: **{cover['with_status']}** ({pct}%)\n"
+77
View File
@@ -0,0 +1,77 @@
"""P7b explainable aggregates: every headline percentage (Matter readiness,
completeness grade, feedback coverage) must ship the numbers behind it — the
denominator, the exact formula, and what was excluded — so a reviewer can
reproduce it instead of trusting a bare number. The Matter denominator in
particular silently dropped functions with no Matter cluster; that exclusion is
now stated.
"""
from nickol_knx_mcp.project import build_loaded_from_raw
from nickol_knx_mcp.advanced import matter_readiness, completeness_grade
from nickol_knx_mcp.handover import _feedback_coverage
from nickol_knx_mcp.analyze import detect_missing_status
def _ga(addr, name, dmain, dsub):
return {"name": name, "identifier": f"GA-{addr}", "raw_address": 0, "address": addr,
"project_uid": None, "dpt": {"main": dmain, "sub": dsub}, "data_secure": False,
"communication_object_ids": [], "description": "", "comment": ""}
def _proj(gas):
raw = {"info": {"group_address_style": "ThreeLevel", "schema_version": "21"},
"group_addresses": gas, "communication_objects": {}, "devices": {},
"functions": {}, "topology": {}, "group_ranges": {}}
return build_loaded_from_raw(raw, "t.knxproj")
def main():
# two lighting commands (one with status), plus a SCENE command that has no
# Matter cluster -> must be excluded from the Matter denominator, not silently.
gas = {
"1/0/1": _ga("1/0/1", "Kitchen light switch", 1, 1),
"1/4/1": _ga("1/4/1", "Kitchen light status", 1, 11),
"1/0/2": _ga("1/0/2", "Hall light switch", 1, 1), # no status
"0/0/1": _ga("0/0/1", "Evening scene", 17, 1), # scene -> no Matter cluster
}
p = _proj(gas)
# --- Matter readiness ---
m = matter_readiness(p)
assert "math" in m, "matter_readiness must expose a math block"
math = m["math"]
# denominator must be the 2 lighting commands, numerator the 1 with status
assert math["denominator"] == 2, math
assert math["numerator"] == 1, math
assert math["pct"] == m["ready_pct"] == 50, (math, m["ready_pct"])
assert "50%" in math["formula"], math["formula"]
# the scene must be visibly excluded, not dropped
exc = math.get("excluded_from_denominator")
assert exc and exc["controllable_functions_without_a_matter_cluster"] >= 1, exc
assert "scene" in {k for k in exc["by_category"]}, exc
# --- Completeness grade ---
g = completeness_grade(p)
gm = g["math"]
assert gm["denominator"] == 8 and gm["pct"] == g["score"], gm
assert "bands" in gm and "as-built grade" in gm["bands"], gm
assert gm["formula"].endswith(f"{g['score']}%"), gm["formula"]
# --- Feedback coverage (handover) ---
missing = detect_missing_status(p)
cov = _feedback_coverage(p, missing)
# 2 lighting commands + the scene command is functional too -> commands counts
# all functional command GAs; formula must reproduce the pct exactly
assert cov["with_status"] + cov["missing"] == cov["commands"], cov
assert f"{cov['pct']}%" in cov["formula"], cov
# --- empty project: ratio undefined, no crash, formula says so ---
empty = matter_readiness(_proj({}))
assert empty["ready_pct"] == 0 and "undefined" in empty["math"]["formula"], empty["math"]
print("test_explainable_aggregates: OK — Matter/completeness/coverage each ship a "
"reproducible formula + denominator; Matter states its excluded (no-cluster) "
"functions; empty project is undefined, not a crash.")
if __name__ == "__main__":
main()