mirror of
https://github.com/NickoScope/nickol-knx-mcp.git
synced 2026-09-29 19:31:12 +02:00
feat(P7b): explainable aggregate scores — expose denominator + formula
Matter readiness, the completeness grade and command/status coverage each now carry a `math` block: numerator, denominator, the exact formula that reproduces the percentage, and (for Matter) the functions excluded from the denominator because their category has no Matter cluster — previously a silent skip and the biggest "why is this number what it is?" gap. Completeness also states its band thresholds. Report-only, additive; scores unchanged, only now auditable. Raised by external review. tests/test_explainable_aggregates.py (12/12 green). Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
f68c85bb48
commit
1a60264c78
@@ -22,6 +22,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
### Added
|
||||
|
||||
- **Explainable aggregate scores** (`advanced.py`, `handover.py`). Every headline percentage now ships
|
||||
the numbers behind it instead of a bare figure: Matter readiness, the completeness grade and
|
||||
command/status coverage each carry a `math` block with the numerator, denominator, the exact formula
|
||||
(that reproduces the percentage) and, for Matter, the functions **excluded** from the denominator
|
||||
because their category has no Matter cluster (previously a silent skip — the biggest source of "why is
|
||||
this number what it is?"). The completeness grade also states its band thresholds. A reviewer can now
|
||||
audit or reproduce any score. Report-only, additive. Raised by external review. `tests/test_explainable_aggregates.py`.
|
||||
|
||||
- **Provenance / confidence** (`explain.py`, new MCP tool `explain_ga`). The enriched model mixes ETS
|
||||
facts, DPT-derived structure and name heuristics; downstream tools then treat the result almost like a
|
||||
fact. `explain_ga(address)` makes the reasoning auditable for one GA: per decision (category / kind /
|
||||
|
||||
@@ -26,6 +26,30 @@ def _functional_commands(project: LoadedProject) -> list:
|
||||
and g.dpt_main is not None]
|
||||
|
||||
|
||||
def _ratio_explain(numerator: int, denominator: int, unit: str,
|
||||
excluded: dict[str, Any] | None = None,
|
||||
bands: dict[str, str] | None = None) -> dict[str, Any]:
|
||||
"""A percentage a reader can audit: the numbers behind it, the exact formula,
|
||||
what was left OUT of the denominator, and (optionally) the grade bands.
|
||||
|
||||
Every aggregate score this tool reports carries one of these so the percentage
|
||||
is never a bare number — a reviewer can see the denominator and reproduce it.
|
||||
"""
|
||||
pct = round(100 * numerator / denominator) if denominator else 0
|
||||
out: dict[str, Any] = {
|
||||
"pct": pct,
|
||||
"numerator": numerator,
|
||||
"denominator": denominator,
|
||||
"formula": f"{numerator} / {denominator} {unit} = {pct}%"
|
||||
if denominator else f"0 {unit} — ratio undefined (denominator 0)",
|
||||
}
|
||||
if excluded:
|
||||
out["excluded_from_denominator"] = excluded
|
||||
if bands:
|
||||
out["bands"] = bands
|
||||
return out
|
||||
|
||||
|
||||
def _status_gas(project: LoadedProject) -> list:
|
||||
from .analyze import _is_status_ga
|
||||
return [g for g in project.gas.values() if _is_status_ga(g)]
|
||||
@@ -48,9 +72,11 @@ def matter_readiness(project: LoadedProject) -> dict[str, Any]:
|
||||
"""Which controllable functions round-trip to a Matter cluster, and what's missing."""
|
||||
stats = _status_gas(project)
|
||||
ready, not_ready = [], []
|
||||
no_cluster: Counter = Counter()
|
||||
for ga in _functional_commands(project):
|
||||
cluster = _MATTER.get(ga.category)
|
||||
if cluster is None:
|
||||
no_cluster[ga.category or "unknown"] += 1 # excluded from the ratio — count it
|
||||
continue
|
||||
has_status = find_status(ga, [s for s in stats if s.main == ga.main]) is not None \
|
||||
or find_status(ga, stats) is not None
|
||||
@@ -58,14 +84,25 @@ def matter_readiness(project: LoadedProject) -> dict[str, Any]:
|
||||
"matter": cluster[0], "has_status": has_status}
|
||||
(ready if has_status else not_ready).append(row)
|
||||
total = len(ready) + len(not_ready)
|
||||
excluded_n = sum(no_cluster.values())
|
||||
math = _ratio_explain(
|
||||
len(ready), total, "Matter-mappable controllable functions with a status GA",
|
||||
excluded={
|
||||
"controllable_functions_without_a_matter_cluster": excluded_n,
|
||||
"by_category": dict(no_cluster.most_common()),
|
||||
"why": "categories with no Matter cluster (e.g. scenes, diagnostics) can't "
|
||||
"round-trip, so they are not counted in the readiness denominator.",
|
||||
} if excluded_n else None)
|
||||
return {
|
||||
"controllable_functions": total,
|
||||
"matter_ready": len(ready),
|
||||
"ready_pct": (100 * len(ready) // total) if total else 0,
|
||||
"ready_pct": math["pct"],
|
||||
"math": math,
|
||||
"not_ready": not_ready[:200],
|
||||
"note": "Ready = has a status GA + decodable DPT, so the Matter cluster can report "
|
||||
"state. A Matter bridge (e.g. HA Matter server) exposes these; functions "
|
||||
"without a status GA won't round-trip. Static readiness only — no bridging here.",
|
||||
"without a status GA won't round-trip. Static readiness only — no bridging here. "
|
||||
"See `math` for the exact denominator and what was excluded.",
|
||||
}
|
||||
|
||||
|
||||
@@ -113,12 +150,16 @@ def completeness_grade(project: LoadedProject) -> dict[str, Any]:
|
||||
if n == 0:
|
||||
missing.append(label)
|
||||
hit = sum(1 for v in present.values() if v)
|
||||
score = round(100 * hit / len(_PATTERNS))
|
||||
bands = {"as-built grade": ">=75", "near-complete": "55-74",
|
||||
"functional skeleton": "30-54", "bare skeleton": "<30"}
|
||||
math = _ratio_explain(hit, len(_PATTERNS), "as-built patterns present", bands=bands)
|
||||
score = math["pct"]
|
||||
grade = ("as-built grade" if score >= 75 else
|
||||
"near-complete" if score >= 55 else
|
||||
"functional skeleton" if score >= 30 else "bare skeleton")
|
||||
return {
|
||||
"grade": grade, "score": score,
|
||||
"math": math,
|
||||
"patterns_present": {k: v for k, v in present.items() if v},
|
||||
"patterns_missing": missing,
|
||||
"note": "Completeness = presence of the as-built patterns a professional adds beyond "
|
||||
|
||||
@@ -86,7 +86,14 @@ def _feedback_coverage(project: LoadedProject,
|
||||
and ga.dpt_main is not None]
|
||||
gaps = sum(1 for f in missing if f["code"] == "missing_status_address")
|
||||
total = len(commands)
|
||||
return {"commands": total, "with_status": max(total - gaps, 0), "missing": gaps}
|
||||
with_status = max(total - gaps, 0)
|
||||
pct = round(100 * with_status / total) if total else 0
|
||||
return {
|
||||
"commands": total, "with_status": with_status, "missing": gaps,
|
||||
"pct": pct,
|
||||
"formula": f"{with_status} / {total} functional command GAs have a status = {pct}%"
|
||||
if total else "no functional command GAs — coverage undefined",
|
||||
}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
@@ -211,7 +218,7 @@ def build_handover(project: LoadedProject,
|
||||
|
||||
# 4. Feedback coverage
|
||||
md.append("\n## 4. Command / status coverage\n")
|
||||
pct = (100 * cover["with_status"] // cover["commands"]) if cover["commands"] else 0
|
||||
pct = cover["pct"]
|
||||
md.append(
|
||||
f"- Functional command GAs: **{cover['commands']}**\n"
|
||||
f"- With a status/feedback GA: **{cover['with_status']}** ({pct}%)\n"
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
"""P7b explainable aggregates: every headline percentage (Matter readiness,
|
||||
completeness grade, feedback coverage) must ship the numbers behind it — the
|
||||
denominator, the exact formula, and what was excluded — so a reviewer can
|
||||
reproduce it instead of trusting a bare number. The Matter denominator in
|
||||
particular silently dropped functions with no Matter cluster; that exclusion is
|
||||
now stated.
|
||||
"""
|
||||
from nickol_knx_mcp.project import build_loaded_from_raw
|
||||
from nickol_knx_mcp.advanced import matter_readiness, completeness_grade
|
||||
from nickol_knx_mcp.handover import _feedback_coverage
|
||||
from nickol_knx_mcp.analyze import detect_missing_status
|
||||
|
||||
|
||||
def _ga(addr, name, dmain, dsub):
|
||||
return {"name": name, "identifier": f"GA-{addr}", "raw_address": 0, "address": addr,
|
||||
"project_uid": None, "dpt": {"main": dmain, "sub": dsub}, "data_secure": False,
|
||||
"communication_object_ids": [], "description": "", "comment": ""}
|
||||
|
||||
|
||||
def _proj(gas):
|
||||
raw = {"info": {"group_address_style": "ThreeLevel", "schema_version": "21"},
|
||||
"group_addresses": gas, "communication_objects": {}, "devices": {},
|
||||
"functions": {}, "topology": {}, "group_ranges": {}}
|
||||
return build_loaded_from_raw(raw, "t.knxproj")
|
||||
|
||||
|
||||
def main():
|
||||
# two lighting commands (one with status), plus a SCENE command that has no
|
||||
# Matter cluster -> must be excluded from the Matter denominator, not silently.
|
||||
gas = {
|
||||
"1/0/1": _ga("1/0/1", "Kitchen light switch", 1, 1),
|
||||
"1/4/1": _ga("1/4/1", "Kitchen light status", 1, 11),
|
||||
"1/0/2": _ga("1/0/2", "Hall light switch", 1, 1), # no status
|
||||
"0/0/1": _ga("0/0/1", "Evening scene", 17, 1), # scene -> no Matter cluster
|
||||
}
|
||||
p = _proj(gas)
|
||||
|
||||
# --- Matter readiness ---
|
||||
m = matter_readiness(p)
|
||||
assert "math" in m, "matter_readiness must expose a math block"
|
||||
math = m["math"]
|
||||
# denominator must be the 2 lighting commands, numerator the 1 with status
|
||||
assert math["denominator"] == 2, math
|
||||
assert math["numerator"] == 1, math
|
||||
assert math["pct"] == m["ready_pct"] == 50, (math, m["ready_pct"])
|
||||
assert "50%" in math["formula"], math["formula"]
|
||||
# the scene must be visibly excluded, not dropped
|
||||
exc = math.get("excluded_from_denominator")
|
||||
assert exc and exc["controllable_functions_without_a_matter_cluster"] >= 1, exc
|
||||
assert "scene" in {k for k in exc["by_category"]}, exc
|
||||
|
||||
# --- Completeness grade ---
|
||||
g = completeness_grade(p)
|
||||
gm = g["math"]
|
||||
assert gm["denominator"] == 8 and gm["pct"] == g["score"], gm
|
||||
assert "bands" in gm and "as-built grade" in gm["bands"], gm
|
||||
assert gm["formula"].endswith(f"{g['score']}%"), gm["formula"]
|
||||
|
||||
# --- Feedback coverage (handover) ---
|
||||
missing = detect_missing_status(p)
|
||||
cov = _feedback_coverage(p, missing)
|
||||
# 2 lighting commands + the scene command is functional too -> commands counts
|
||||
# all functional command GAs; formula must reproduce the pct exactly
|
||||
assert cov["with_status"] + cov["missing"] == cov["commands"], cov
|
||||
assert f"{cov['pct']}%" in cov["formula"], cov
|
||||
|
||||
# --- empty project: ratio undefined, no crash, formula says so ---
|
||||
empty = matter_readiness(_proj({}))
|
||||
assert empty["ready_pct"] == 0 and "undefined" in empty["math"]["formula"], empty["math"]
|
||||
|
||||
print("test_explainable_aggregates: OK — Matter/completeness/coverage each ship a "
|
||||
"reproducible formula + denominator; Matter states its excluded (no-cluster) "
|
||||
"functions; empty project is undefined, not a crash.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user