diff --git a/CHANGELOG.md b/CHANGELOG.md index 9cb3552..6b2881a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- **Provenance / confidence** (`explain.py`, new MCP tool `explain_ga`). The enriched model mixes ETS + facts, DPT-derived structure and name heuristics; downstream tools then treat the result almost like a + fact. `explain_ga(address)` makes the reasoning auditable for one GA: per decision (category / kind / + status pairing) it reports the signals that fired with a confidence tier — **authoritative** (an ETS + Function role) > **structural** (the KNX DPT) > **heuristic** (a name keyword) — and flags **conflicts** + (e.g. a GA the DPT calls `lighting` while the name says "AC" → `contested`), the hotspot for silent + misclassification. Additive and read-only (no change to the core model). Asked for by three independent + reviewers (two external councils + a field integrator). `tests/test_explain.py`. Tool count 27 → 28. + - **Role-aware feedback completeness** (`detect_role_completeness` in `analyze.py`, surfaced through `check_missing_status`). "Does the function have *a* status?" was not enough — a dimmer with an on/off status but no brightness status silently passed and inflated coverage/Matter scores. The new check diff --git a/README.md b/README.md index 9ad56c9..1d9f37a 100644 --- a/README.md +++ b/README.md @@ -250,7 +250,7 @@ keyring handling, and the recommended workflow). --- -## MCP tools (27) +## MCP tools (28) **Read** | Tool | Purpose | @@ -259,6 +259,7 @@ keyring handling, and the recommended workflow). | `list_group_addresses(category?, kind?)` | list GAs with classification and filters | | `get_devices()` | devices + their communication objects | | `get_topology()` | topology (areas / lines / devices) | +| `explain_ga(address)` | **provenance** for one GA: why it's classified this way — evidence per decision with a confidence tier (**authoritative** ETS Function > **structural** DPT > **heuristic** name), how its status was paired, and **conflicts** (name says "AC", DPT says lighting → `contested`) | **Validate** | Tool | Purpose | diff --git a/nickol_knx_mcp/explain.py b/nickol_knx_mcp/explain.py new file mode 100644 index 0000000..516d09e --- /dev/null +++ b/nickol_knx_mcp/explain.py @@ -0,0 +1,119 @@ +"""Provenance / confidence for a single group address — "why did the tool decide this?". + +External reviewers (twice) and a field integrator all asked for the same thing: the +enriched model mixes ETS facts, DPT-derived structure, and name heuristics, and the +downstream tools treat the result almost like a fact. This module makes the reasoning +**explicit and auditable** for one GA, without changing the core model: it replays the +same primitives and reports, per decision, the signals that fired and a confidence tier: + + * ``authoritative`` — an ETS Function role (the project author's own tagging); + * ``structural`` — the KNX DPT (a typed, deterministic signal); + * ``heuristic`` — a name keyword (language- and school-dependent, may mislead); + * ``none`` / ``low`` — insufficient signal. + +It also shows how a status was (or wasn't) paired and flags **conflicts** — e.g. a GA +classified ``lighting`` by DPT while its name says "AC" — which is exactly where silent +misclassification hides. +""" +from __future__ import annotations + +from typing import Any, Optional + +from .project import LoadedProject, _refine_category, _override_kind_by_name +from .dpt_map import classify_dpt +from .intent import classify_intent +from .pairing import (function_status_pairs, find_status, positional_status, + self_reporting, base_tokens) + + +def explain_ga(project: LoadedProject, address: str) -> dict[str, Any]: + """Return the evidence and confidence behind one GA's classification + pairing.""" + ga = project.gas.get(address) + if ga is None: + return {"error": f"no group address {address} in the loaded project"} + + base = classify_dpt(ga.dpt_main, ga.dpt_sub) + + # --- category: DPT (structural) vs name refine (heuristic) --- + refined_cat = _refine_category(ga.name, base["category"]) + cat_ev: list[dict[str, str]] = [] + if ga.dpt_main is not None: + cat_ev.append({"signal": f"DPT {ga.dpt or ga.dpt_main} → {base['category']}", + "tier": "structural"}) + if refined_cat != base["category"]: + cat_ev.append({"signal": f"name resolved ambiguous DPT → {refined_cat}", + "tier": "heuristic"}) + + # --- kind: DPT vs name override --- + refined_kind = _override_kind_by_name(ga.name, base["kind"]) + kind_ev: list[dict[str, str]] = [{"signal": f"DPT → {base['kind']}", "tier": "structural"}] + if refined_kind != base["kind"]: + kind_ev.append({"signal": f"name keyword → {refined_kind}", "tier": "heuristic"}) + + # --- ETS Function membership (authoritative) --- + fn_hits = [] + for fid, fn in (project.functions or {}).items(): + roles = fn.get("group_addresses", {}) or {} + for gaddr, ref in roles.items(): + if ref.get("address", gaddr) == address: + fn_hits.append({"function": fn.get("name", fid), + "type": fn.get("function_type"), + "role": ref.get("role")}) + + # --- conflict: DPT-domain vs name-domain (the silent-misclassification hotspot) --- + conflicts = [] + low = (ga.name or "").lower() + _NAME_DOMAIN = {"hvac": ("ac", " a/c", "climate", "heat", "cool", "hvac", "конд", "клима", + "отоплен", "температур"), + "shutter": ("blind", "shutter", "cover", "жалюзи", "штор", "ролл"), + "energy": ("meter", "energy", "power", "счётчик", "энерг", "мощност")} + for dom, toks in _NAME_DOMAIN.items(): + if any(t in low for t in toks) and refined_cat != dom and refined_cat != "unknown": + conflicts.append( + f"name suggests '{dom}' but classified '{refined_cat}' (DPT took precedence " + "over the name)") + + # --- status pairing: which strategy, if any --- + fpairs = function_status_pairs(project) + pairing: dict[str, Any] = {"paired": False, "method": "none"} + if address in fpairs: + pairing = {"paired": True, "method": "ets_function_role", "status": fpairs[address], + "tier": "authoritative"} + elif ga.kind == "command": + cand = [g for g in project.gas.values() + if (g.kind == "status" or "status" in (g.name or "").lower())] + m = find_status(ga, [c for c in cand if c.main == ga.main]) or find_status(ga, cand) + if m is not None: + pairing = {"paired": True, "method": "name_token", "status": m.address, + "tier": "heuristic"} + elif positional_status(ga, project) is not None: + pairing = {"paired": True, "method": "positional", "tier": "structural"} + elif self_reporting(ga, project): + pairing = {"paired": True, "method": "self_reporting_R+T", "tier": "structural"} + + # top confidence for the DOMAIN classification. A name/DPT conflict dominates: + # the classification is contested regardless of any ETS-Function pairing signal. + if conflicts: + confidence = "contested" + elif fn_hits: + confidence = "authoritative" + elif ga.dpt_main is not None and refined_cat != "unknown": + confidence = "heuristic" if refined_cat != base["category"] else "structural" + else: + confidence = "low" + + return { + "address": address, "name": ga.name, "dpt": ga.dpt or None, + "raw_facts": {"co_ids": ga.co_ids, "data_secure": ga.data_secure, + "identity_tokens": sorted(base_tokens(ga.name))}, + "category": {"value": refined_cat, "evidence": cat_ev}, + "kind": {"value": refined_kind, "evidence": kind_ev}, + "intent": {"value": classify_intent(ga.name), "tier": "heuristic (name-based)"}, + "ets_functions": fn_hits or "none — no ETS Function tags this GA (heuristics only)", + "status_pairing": pairing, + "conflicts": conflicts or "none", + "confidence": confidence, + "note": "Evidence tiers: authoritative (ETS Function role) > structural (DPT) > " + "heuristic (name keyword). A conflict means the name and the DPT disagree — " + "review before trusting the classification or generating an entity from it.", + } diff --git a/nickol_knx_mcp/server.py b/nickol_knx_mcp/server.py index fa8fe70..1981806 100644 --- a/nickol_knx_mcp/server.py +++ b/nickol_knx_mcp/server.py @@ -33,6 +33,7 @@ from .iot import generate_knx_iot_turtle from .param_check import check_device_parameters as _check_device_parameters from .policy import (check_policy as _check_policy, load_policy as _load_policy, example_policy_yaml as _example_policy_yaml) +from .explain import explain_ga as _explain_ga mcp = FastMCP("nickol-knx") @@ -380,6 +381,17 @@ def check_policy(profile_path: Optional[str] = None, return _check_policy(_project(), _load_policy(profile_path)) +@mcp.tool() +def explain_ga(address: str) -> dict[str, Any]: + """**Provenance** for one group address — why the tool classified it the way it did. + Replays the classification and shows, per decision (category / kind / status pairing), + the signals that fired with a **confidence tier**: authoritative (an ETS Function role) > + structural (the KNX DPT) > heuristic (a name keyword). Flags **conflicts** (e.g. a GA + the DPT calls `lighting` while its name says "AC") — the hotspot for silent + misclassification. Read-only; use before trusting a category or generating an entity.""" + return _explain_ga(_project(), address) + + @mcp.tool() def check_matter() -> dict[str, Any]: """Matter-readiness lint: which controllable functions round-trip to a Matter diff --git a/tests/test_explain.py b/tests/test_explain.py new file mode 100644 index 0000000..81253b1 --- /dev/null +++ b/tests/test_explain.py @@ -0,0 +1,51 @@ +"""explain_ga — provenance / confidence for one GA (evidence + conflict + pairing tier).""" +from nickol_knx_mcp.project import build_loaded_from_raw +from nickol_knx_mcp.explain import explain_ga + + +def _ga(addr, name, dmain, dsub): + return {"name": name, "identifier": f"GA-{addr}", "raw_address": 0, "address": addr, + "project_uid": None, "dpt": {"main": dmain, "sub": dsub}, "data_secure": False, + "communication_object_ids": [], "description": "", "comment": ""} + + +def _proj(gas): + raw = {"info": {"group_address_style": "ThreeLevel", "schema_version": "21"}, + "group_addresses": gas, "communication_objects": {}, "devices": {}, + "functions": {}, "topology": {}, "group_ranges": {}} + return build_loaded_from_raw(raw, "t.knxproj") + + +def main(): + gas = { + "3/2/1": _ga("3/2/1", "Living room AC on/off", 1, 1), # DPT lighting, name hvac -> conflict + "1/0/1": _ga("1/0/1", "Kitchen light switch", 1, 1), # clean lighting command + "1/4/1": _ga("1/4/1", "Kitchen light status", 1, 11), # its status + } + p = _proj(gas) + + # 1. conflict surfaced + confidence contested + ac = explain_ga(p, "3/2/1") + assert ac["category"]["value"] == "lighting" + assert ac["conflicts"] != "none" and any("hvac" in c for c in ac["conflicts"]), ac["conflicts"] + assert ac["confidence"] == "contested", ac["confidence"] + assert any(e["tier"] == "structural" for e in ac["category"]["evidence"]) + + # 2. clean lighting switch: no conflict, structural evidence, paired to its status by name token + sw = explain_ga(p, "1/0/1") + assert sw["conflicts"] == "none", sw["conflicts"] + assert sw["confidence"] == "structural", sw["confidence"] + assert sw["status_pairing"]["paired"] is True + assert sw["status_pairing"]["method"] == "name_token" + assert sw["status_pairing"]["status"] == "1/4/1" + assert sw["status_pairing"]["tier"] == "heuristic" + + # 3. missing address -> error + assert "error" in explain_ga(p, "9/9/9") + + print("test_explain: OK — AC on/off flagged as name/DPT conflict (contested); clean " + "switch is structural + name-token paired to its status; missing address errors.") + + +if __name__ == "__main__": + main()