diff --git a/CHANGELOG.md b/CHANGELOG.md index 9e863cf..d620c65 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -22,6 +22,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- **Explainable aggregate scores** (`advanced.py`, `handover.py`). Every headline percentage now ships + the numbers behind it instead of a bare figure: Matter readiness, the completeness grade and + command/status coverage each carry a `math` block with the numerator, denominator, the exact formula + (that reproduces the percentage) and, for Matter, the functions **excluded** from the denominator + because their category has no Matter cluster (previously a silent skip — the biggest source of "why is + this number what it is?"). The completeness grade also states its band thresholds. A reviewer can now + audit or reproduce any score. Report-only, additive. Raised by external review. `tests/test_explainable_aggregates.py`. + - **Provenance / confidence** (`explain.py`, new MCP tool `explain_ga`). The enriched model mixes ETS facts, DPT-derived structure and name heuristics; downstream tools then treat the result almost like a fact. `explain_ga(address)` makes the reasoning auditable for one GA: per decision (category / kind / diff --git a/nickol_knx_mcp/advanced.py b/nickol_knx_mcp/advanced.py index 5f2e57e..af2af77 100644 --- a/nickol_knx_mcp/advanced.py +++ b/nickol_knx_mcp/advanced.py @@ -26,6 +26,30 @@ def _functional_commands(project: LoadedProject) -> list: and g.dpt_main is not None] +def _ratio_explain(numerator: int, denominator: int, unit: str, + excluded: dict[str, Any] | None = None, + bands: dict[str, str] | None = None) -> dict[str, Any]: + """A percentage a reader can audit: the numbers behind it, the exact formula, + what was left OUT of the denominator, and (optionally) the grade bands. + + Every aggregate score this tool reports carries one of these so the percentage + is never a bare number — a reviewer can see the denominator and reproduce it. + """ + pct = round(100 * numerator / denominator) if denominator else 0 + out: dict[str, Any] = { + "pct": pct, + "numerator": numerator, + "denominator": denominator, + "formula": f"{numerator} / {denominator} {unit} = {pct}%" + if denominator else f"0 {unit} — ratio undefined (denominator 0)", + } + if excluded: + out["excluded_from_denominator"] = excluded + if bands: + out["bands"] = bands + return out + + def _status_gas(project: LoadedProject) -> list: from .analyze import _is_status_ga return [g for g in project.gas.values() if _is_status_ga(g)] @@ -48,9 +72,11 @@ def matter_readiness(project: LoadedProject) -> dict[str, Any]: """Which controllable functions round-trip to a Matter cluster, and what's missing.""" stats = _status_gas(project) ready, not_ready = [], [] + no_cluster: Counter = Counter() for ga in _functional_commands(project): cluster = _MATTER.get(ga.category) if cluster is None: + no_cluster[ga.category or "unknown"] += 1 # excluded from the ratio — count it continue has_status = find_status(ga, [s for s in stats if s.main == ga.main]) is not None \ or find_status(ga, stats) is not None @@ -58,14 +84,25 @@ def matter_readiness(project: LoadedProject) -> dict[str, Any]: "matter": cluster[0], "has_status": has_status} (ready if has_status else not_ready).append(row) total = len(ready) + len(not_ready) + excluded_n = sum(no_cluster.values()) + math = _ratio_explain( + len(ready), total, "Matter-mappable controllable functions with a status GA", + excluded={ + "controllable_functions_without_a_matter_cluster": excluded_n, + "by_category": dict(no_cluster.most_common()), + "why": "categories with no Matter cluster (e.g. scenes, diagnostics) can't " + "round-trip, so they are not counted in the readiness denominator.", + } if excluded_n else None) return { "controllable_functions": total, "matter_ready": len(ready), - "ready_pct": (100 * len(ready) // total) if total else 0, + "ready_pct": math["pct"], + "math": math, "not_ready": not_ready[:200], "note": "Ready = has a status GA + decodable DPT, so the Matter cluster can report " "state. A Matter bridge (e.g. HA Matter server) exposes these; functions " - "without a status GA won't round-trip. Static readiness only — no bridging here.", + "without a status GA won't round-trip. Static readiness only — no bridging here. " + "See `math` for the exact denominator and what was excluded.", } @@ -113,12 +150,16 @@ def completeness_grade(project: LoadedProject) -> dict[str, Any]: if n == 0: missing.append(label) hit = sum(1 for v in present.values() if v) - score = round(100 * hit / len(_PATTERNS)) + bands = {"as-built grade": ">=75", "near-complete": "55-74", + "functional skeleton": "30-54", "bare skeleton": "<30"} + math = _ratio_explain(hit, len(_PATTERNS), "as-built patterns present", bands=bands) + score = math["pct"] grade = ("as-built grade" if score >= 75 else "near-complete" if score >= 55 else "functional skeleton" if score >= 30 else "bare skeleton") return { "grade": grade, "score": score, + "math": math, "patterns_present": {k: v for k, v in present.items() if v}, "patterns_missing": missing, "note": "Completeness = presence of the as-built patterns a professional adds beyond " diff --git a/nickol_knx_mcp/handover.py b/nickol_knx_mcp/handover.py index a590db6..5af61c1 100644 --- a/nickol_knx_mcp/handover.py +++ b/nickol_knx_mcp/handover.py @@ -86,7 +86,14 @@ def _feedback_coverage(project: LoadedProject, and ga.dpt_main is not None] gaps = sum(1 for f in missing if f["code"] == "missing_status_address") total = len(commands) - return {"commands": total, "with_status": max(total - gaps, 0), "missing": gaps} + with_status = max(total - gaps, 0) + pct = round(100 * with_status / total) if total else 0 + return { + "commands": total, "with_status": with_status, "missing": gaps, + "pct": pct, + "formula": f"{with_status} / {total} functional command GAs have a status = {pct}%" + if total else "no functional command GAs — coverage undefined", + } # --------------------------------------------------------------------------- # @@ -211,7 +218,7 @@ def build_handover(project: LoadedProject, # 4. Feedback coverage md.append("\n## 4. Command / status coverage\n") - pct = (100 * cover["with_status"] // cover["commands"]) if cover["commands"] else 0 + pct = cover["pct"] md.append( f"- Functional command GAs: **{cover['commands']}**\n" f"- With a status/feedback GA: **{cover['with_status']}** ({pct}%)\n" diff --git a/tests/test_explainable_aggregates.py b/tests/test_explainable_aggregates.py new file mode 100644 index 0000000..9bd54a2 --- /dev/null +++ b/tests/test_explainable_aggregates.py @@ -0,0 +1,77 @@ +"""P7b explainable aggregates: every headline percentage (Matter readiness, +completeness grade, feedback coverage) must ship the numbers behind it — the +denominator, the exact formula, and what was excluded — so a reviewer can +reproduce it instead of trusting a bare number. The Matter denominator in +particular silently dropped functions with no Matter cluster; that exclusion is +now stated. +""" +from nickol_knx_mcp.project import build_loaded_from_raw +from nickol_knx_mcp.advanced import matter_readiness, completeness_grade +from nickol_knx_mcp.handover import _feedback_coverage +from nickol_knx_mcp.analyze import detect_missing_status + + +def _ga(addr, name, dmain, dsub): + return {"name": name, "identifier": f"GA-{addr}", "raw_address": 0, "address": addr, + "project_uid": None, "dpt": {"main": dmain, "sub": dsub}, "data_secure": False, + "communication_object_ids": [], "description": "", "comment": ""} + + +def _proj(gas): + raw = {"info": {"group_address_style": "ThreeLevel", "schema_version": "21"}, + "group_addresses": gas, "communication_objects": {}, "devices": {}, + "functions": {}, "topology": {}, "group_ranges": {}} + return build_loaded_from_raw(raw, "t.knxproj") + + +def main(): + # two lighting commands (one with status), plus a SCENE command that has no + # Matter cluster -> must be excluded from the Matter denominator, not silently. + gas = { + "1/0/1": _ga("1/0/1", "Kitchen light switch", 1, 1), + "1/4/1": _ga("1/4/1", "Kitchen light status", 1, 11), + "1/0/2": _ga("1/0/2", "Hall light switch", 1, 1), # no status + "0/0/1": _ga("0/0/1", "Evening scene", 17, 1), # scene -> no Matter cluster + } + p = _proj(gas) + + # --- Matter readiness --- + m = matter_readiness(p) + assert "math" in m, "matter_readiness must expose a math block" + math = m["math"] + # denominator must be the 2 lighting commands, numerator the 1 with status + assert math["denominator"] == 2, math + assert math["numerator"] == 1, math + assert math["pct"] == m["ready_pct"] == 50, (math, m["ready_pct"]) + assert "50%" in math["formula"], math["formula"] + # the scene must be visibly excluded, not dropped + exc = math.get("excluded_from_denominator") + assert exc and exc["controllable_functions_without_a_matter_cluster"] >= 1, exc + assert "scene" in {k for k in exc["by_category"]}, exc + + # --- Completeness grade --- + g = completeness_grade(p) + gm = g["math"] + assert gm["denominator"] == 8 and gm["pct"] == g["score"], gm + assert "bands" in gm and "as-built grade" in gm["bands"], gm + assert gm["formula"].endswith(f"{g['score']}%"), gm["formula"] + + # --- Feedback coverage (handover) --- + missing = detect_missing_status(p) + cov = _feedback_coverage(p, missing) + # 2 lighting commands + the scene command is functional too -> commands counts + # all functional command GAs; formula must reproduce the pct exactly + assert cov["with_status"] + cov["missing"] == cov["commands"], cov + assert f"{cov['pct']}%" in cov["formula"], cov + + # --- empty project: ratio undefined, no crash, formula says so --- + empty = matter_readiness(_proj({})) + assert empty["ready_pct"] == 0 and "undefined" in empty["math"]["formula"], empty["math"] + + print("test_explainable_aggregates: OK — Matter/completeness/coverage each ship a " + "reproducible formula + denominator; Matter states its excluded (no-cluster) " + "functions; empty project is undefined, not a crash.") + + +if __name__ == "__main__": + main()