feat: positional status pairing + self-reporting commands, app-program parser v2 (ComObjectRef merge, version pick), authoritative main names, grader range-name evidence (v0.8.0)

Closes #3, closes #4, closes #5, closes #6
This commit is contained in:
Nikolay Miroshnichenko
2026-07-07 12:24:24 +02:00
parent 4dae539d31
commit ff93d44166
9 changed files with 358 additions and 15 deletions
+3
View File
@@ -37,6 +37,9 @@ jobs:
- name: App-program parser (synthetic .knxprod round-trip)
run: python tests/test_appprog_parser.py
- name: v0.8 fixes (positional pairing, ref-merge, range names)
run: python tests/test_v08_fixes.py
- name: Console script is installed
run: |
python -c "import importlib.metadata as m; print('entry points:', [e.name for e in m.entry_points(group='console_scripts') if e.name == 'nickol-knx-mcp'])"
+33 -1
View File
@@ -6,8 +6,39 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
## [Unreleased]
## [0.8.0] — 2026-07-07
**Lessons from a second field project (a different integrator naming school):** positional
command/status pairing, ref-level application programs, and two detector blind spots. All four
changes were validated against a real 859-GA as-built project.
### Added
- **Positional command/status pairing** (#4, `pairing.py` / `analyze.py`). A widespread naming
school keeps feedback in a *parallel middle group* (command `1/0/s` ↔ status `1/1/s`) with the
GA name duplicated 1:1 and no status keyword at all — invisible to lexical pairing.
`detect_missing_status` now also pairs positionally (same main, same sub, different middle,
identical name, DPT-compatible) and recognises **self-reporting commands** (the actuator's
R+T communication object linked to the command GA itself). On the field project this cut
missing-status warnings from 339 to 89 — the ~90 that are actually real — and reports a
`status_pairing_summary` info with both counters.
- **Application-program parser v2** (#6, `appprog_parser.py`). Vendors like HDL, Ekinex and
Creatrol publish object names/DPTs/flags on `<ComObjectRef>` while the base `<ComObject>` is a
near-empty stub. The parser now merges ref-level data over the base (conservatively: refs only
fill empty fields; a DPT is taken only when every declaring ref agrees, disagreeing variants are
recorded as `dpt_variants`, never guessed). Unverified DPTs on the field project's 15 device
models dropped 4341 → 1167 (−73%). The app-program version is now picked by the **highest
ApplicationVersion** instead of list position — a stale v16 was being parsed where v18 existed.
### Fixed
- **`main_group_unnamed` read the wrong names** (#3, `analyze.py`): main-group names are now taken
from the authoritative `group_ranges` (a GA record's `main_name` can carry a middle range's
name), eliminating both false "unnamed" findings and masked truly-unnamed mains.
- **Completeness grader missed pattern-named ranges** (#5, `advanced.py`): a middle range literally
called «Сцены» full of plain-named scene GAs now counts as scene evidence — the grader reads
main/middle range names in addition to GA names (and the scene token matches Russian plural
forms).
- **Pinned `mcp>=1.10,<2`** — the MCP Python SDK v2 (in alpha) renames
`mcp.server.fastmcp.FastMCP` to `mcp.server.MCPServer`; without the upper bound a future
`pip install` would pull v2 and break the server import. Verified against the SDK docs.
@@ -299,7 +330,8 @@ Initial public beta.
- Tested end-to-end on a synthetic project only; real-world `.knxproj` testing is ongoing
(see the call for testers in the README).
[Unreleased]: https://github.com/NickoScope/nickol-knx-mcp/compare/v0.7.0...HEAD
[Unreleased]: https://github.com/NickoScope/nickol-knx-mcp/compare/v0.8.0...HEAD
[0.8.0]: https://github.com/NickoScope/nickol-knx-mcp/compare/v0.7.0...v0.8.0
[0.7.0]: https://github.com/NickoScope/nickol-knx-mcp/compare/v0.6.0...v0.7.0
[0.6.0]: https://github.com/NickoScope/nickol-knx-mcp/compare/v0.5.0...v0.6.0
[0.5.0]: https://github.com/NickoScope/nickol-knx-mcp/compare/v0.4.0...v0.5.0
+1 -1
View File
@@ -1,2 +1,2 @@
"""nickol-knx-mcp: design-time KNX/ETS project assistant MCP server."""
__version__ = "0.7.0"
__version__ = "0.8.0"
+20 -1
View File
@@ -78,18 +78,37 @@ _PATTERNS = {
"astro / meteo": ("солнц", "азимут", "восход", "закат", "метео", "sun ", "meteo"),
"monitoring": ("мониторинг", "статус работы", "неисправ", "monitoring", "fault"),
"deep metering": ("счётчик", "потреблен", "тариф", "meter", "consumption"),
"scenes": ("сцена", "сценар", "scene", "szene"),
"scenes": ("сцен", "scene", "szene"),
"reserves": ("резерв", "reserve", "spare"),
"debug main": ("отладк", "временно", "debug", "logic for"),
}
def _range_names_text(project: LoadedProject) -> str:
"""All main/middle GroupRange names, lowercased, as extra pattern evidence.
Integrators often name a whole middle range for a pattern ("Сцены",
"Мониторинг") while the GAs inside carry plain load names — grading by GA
names alone misses the pattern entirely, so range names count too.
"""
out: list[str] = []
for mrange in (project.raw.get("group_ranges") or {}).values():
out.append((mrange.get("name") or "").lower())
for srange in (mrange.get("group_ranges") or {}).values():
out.append((srange.get("name") or "").lower())
return " \n ".join(out)
def completeness_grade(project: LoadedProject) -> dict[str, Any]:
"""Grade a project: bare functional skeleton vs as-built grade (§ alex-skill patterns)."""
names = " \n ".join(g.name.lower() for g in project.gas.values())
range_names = _range_names_text(project)
present, missing = {}, []
for label, toks in _PATTERNS.items():
n = sum(names.count(t) for t in toks)
if n == 0:
# a main/middle RANGE named for the pattern is evidence too
n = sum(range_names.count(t) for t in toks)
present[label] = n
if n == 0:
missing.append(label)
+31 -5
View File
@@ -15,7 +15,8 @@ from collections import defaultdict
from typing import Any, Optional
from .project import LoadedProject, GARecord, STATUS_KEYWORDS
from .pairing import find_status, function_status_pairs, base_tokens
from .pairing import (find_status, function_status_pairs, base_tokens,
positional_status, self_reporting)
from .intent import INTENT_FUNCTIONAL, INTENT_RESERVE, INTENT_SCRATCH
SEVERITY_ERROR = "error"
@@ -95,11 +96,19 @@ def validate_naming(project: LoadedProject,
addresses=addrs,
))
# Main-group names present?
# Main-group names present? Read them from the project's GroupRanges — the
# authoritative source. GARecord.main_name from the parser can carry a
# MIDDLE range's name instead of the main's (first-seen shadowing), which
# both hid truly unnamed mains and mislabeled named ones.
main_named: dict[int, str] = {}
for ga in project.gas.values():
if ga.main is not None and ga.main not in main_named:
main_named[ga.main] = ga.main_name
for mkey, mrange in (project.raw.get("group_ranges") or {}).items():
head = str(mkey).split("/")[0]
if head.isdigit():
main_named[int(head)] = mrange.get("name") or ""
if not main_named: # fallback for synthetic projects without group_ranges
for ga in project.gas.values():
if ga.main is not None and ga.main not in main_named:
main_named[ga.main] = ga.main_name
for main, mname in sorted(main_named.items()):
if not mname:
findings.append(_finding(
@@ -225,6 +234,7 @@ def detect_missing_status(project: LoadedProject) -> list[dict[str, Any]]:
# All status-like GAs become pairing candidates (project-wide).
status_gas = [ga for ga in project.gas.values() if _is_status_ga(ga)]
n_positional = n_self_report = 0
for addr, ga in project.gas.items():
if addr in covered:
continue
@@ -238,6 +248,15 @@ def detect_missing_status(project: LoadedProject) -> list[dict[str, Any]]:
same_main = [s for s in status_gas if s.main == ga.main]
match = find_status(ga, same_main) or find_status(ga, status_gas)
if match is None:
# positional school: parallel status middle, identical name, same sub
pos = positional_status(ga, project)
if pos is not None:
n_positional += 1
continue
# actuator's R+T status object linked to the command GA itself
if self_reporting(ga, project):
n_self_report += 1
continue
if _is_central_macro(ga.name):
findings.append(_finding(
SEVERITY_INFO, "central_macro_no_status", addr,
@@ -253,6 +272,13 @@ def detect_missing_status(project: LoadedProject) -> list[dict[str, Any]]:
name=ga.name, dpt=ga.dpt, category=ga.category,
))
if n_positional or n_self_report:
findings.append(_finding(
SEVERITY_INFO, "status_pairing_summary", "-",
f"{n_positional} command(s) paired positionally (parallel status middle, "
f"identical name) and {n_self_report} self-reporting (actuator R+T object "
"on the command GA) — no separate status GA needed.",
))
return findings
+76 -6
View File
@@ -24,6 +24,7 @@ from typing import Any, Optional
_ATTR_RE = re.compile(r'(\w+)="([^"]*)"')
_COMOBJ_RE = re.compile(r"<ComObject\b([^>]*?)/?>")
_COMREF_RE = re.compile(r"<ComObjectRef\b([^>]*?)/?>")
_HW_SPLIT_RE = re.compile(r"<Hardware\b")
_APPREF_RE = re.compile(r'<ApplicationProgramRef\s+RefId="([^"]+)"')
_APPVER_RE = re.compile(r'ApplicationVersion="(\d+)"')
@@ -90,11 +91,52 @@ def _channel_token(text: str, name: str) -> Optional[str]:
def _parse_comobjects(xml_text: str) -> list[dict[str, Any]]:
objs: list[dict[str, Any]] = []
"""Parse the object model from base ``<ComObject>`` elements, then enrich
from ``<ComObjectRef>``.
Some vendors (Zennio, ABB, EOS…) publish the full model on the base
elements; others (HDL, Ekinex, Creatrol…) leave the base as a near-empty
stub and put names/DPTs/flags on the refs. Merging is conservative: a ref
value only FILLS a field the base left empty, and the DPT is taken only
when every ref of that object that declares one AGREES — refs are
parameter-dependent variants, and a disagreeing DPT stays honestly unset
rather than guessed.
"""
base: dict[str, dict[str, Any]] = {} # base Id -> raw attrs
for m in _COMOBJ_RE.finditer(xml_text):
a = _attrs(m.group(1))
if "Number" not in a:
continue
if "Number" in a and a.get("Id"):
base[a["Id"]] = a
elif "Number" in a:
base[f"_anon{len(base)}"] = a
refs: dict[str, list[dict[str, str]]] = {}
for m in _COMREF_RE.finditer(xml_text):
a = _attrs(m.group(1))
rid = a.get("RefId")
if rid:
refs.setdefault(rid, []).append(a)
def _merged(a: dict[str, str], rlist: list[dict[str, str]]) -> dict[str, str]:
out = dict(a)
for field in ("Text", "Name", "FunctionText", "ObjectSize",
"ReadFlag", "WriteFlag", "CommunicationFlag",
"TransmitFlag", "UpdateFlag"):
if not out.get(field):
vals = [r[field] for r in rlist if r.get(field)]
if vals:
out[field] = vals[0]
if not out.get("DatapointType"):
dpts = {r["DatapointType"] for r in rlist if r.get("DatapointType")}
if len(dpts) == 1:
out["DatapointType"] = dpts.pop()
elif len(dpts) > 1:
out["_dpt_variants"] = ", ".join(sorted(dpts))
return out
objs: list[dict[str, Any]] = []
for oid, a in base.items():
a = _merged(a, refs.get(oid, []))
w = a.get("WriteFlag", "").lower() == "enabled"
t = a.get("TransmitFlag", "").lower() == "enabled"
r = a.get("ReadFlag", "").lower() == "enabled"
@@ -103,7 +145,7 @@ def _parse_comobjects(xml_text: str) -> list[dict[str, Any]]:
number = int(a["Number"])
except ValueError:
continue
objs.append({
rec = {
"number": number,
"name": text,
"internal_name": a.get("Name"),
@@ -114,7 +156,11 @@ def _parse_comobjects(xml_text: str) -> list[dict[str, Any]]:
"R": r, "W": w, "T": t,
"U": a.get("UpdateFlag", "").lower() == "enabled"},
"role": _role(w, t, r, text),
})
}
if a.get("_dpt_variants"):
rec["dpt_variants"] = ", ".join(
sorted(_dpt(x) or x for x in a["_dpt_variants"].split(", ")))
objs.append(rec)
objs.sort(key=lambda o: o["number"])
return objs
@@ -152,6 +198,30 @@ def _detect_blocks(objs: list[dict[str, Any]]) -> dict[str, Any]:
return {"blocks": blocks, "general_objects": general, "logic_function_objects": lf}
def _pick_app_ref(refs: list[str]) -> Optional[str]:
"""Pick the NEWEST application program by its version segment.
``ApplicationProgramRef`` ids look like ``M-0073_A-1105-12-ABCD`` — the third
dash-segment is the application version in hex. Hardware entries can list
several app versions, and the newest is NOT always the last listed (seen in
the field: an HDL DALI gateway listing v18 before v16), so positional
``[-1]`` picked a stale program. Fall back to the last entry when the id
doesn't match the expected shape.
"""
if not refs:
return None
def ver(ref: str) -> int:
m = re.match(r"M-[0-9A-Fa-f]+_A-[0-9A-Fa-f]+-([0-9A-Fa-f]+)-", ref)
try:
return int(m.group(1), 16) if m else -1
except ValueError:
return -1
best = max(refs, key=ver)
return best if ver(best) >= 0 else refs[-1]
def _hardware_map(xml_text: str) -> list[dict[str, Any]]:
"""From a Hardware.xml, list {order_number, name, app_refs[]} (latest app last).
@@ -198,7 +268,7 @@ def parse_project(path: str, password: Optional[str] = None) -> dict[str, Any]:
continue
for entry in _hardware_map(hw_xml):
hw_count += 1
app_ref = entry["app_refs"][-1] if entry["app_refs"] else None
app_ref = _pick_app_ref(entry["app_refs"])
app_path = f"{mcode}/{app_ref}.xml" if app_ref else None
if not app_path or app_path not in appfiles:
devices.append({
+60
View File
@@ -62,6 +62,66 @@ def find_status(command: GARecord, candidates: Iterable[GARecord]) -> Optional[G
return best
# --------------------------------------------------------------------------- #
# Positional pairing school (parallel status middles, identical names)
# --------------------------------------------------------------------------- #
def _norm_name(s: str) -> str:
return " ".join((s or "").lower().split())
def positional_status(command: GARecord, project) -> Optional[GARecord]:
"""Find a status GA by POSITION instead of by name marker.
A widespread naming school (common around HDL-style projects) keeps statuses
in a *parallel middle group*: command in ``main/m/s``, feedback in
``main/m'/s`` — same main, same sub, different middle — and duplicates the
GA name 1:1 with no status suffix at all. A lexical marker search can never
pair these, so `detect_missing_status` would cry wolf on every command.
Deliberately conservative to avoid false pairs: requires exact sub match and
an identical normalised name (or identical-plus-status-keyword), and a
DPT-compatible candidate per STATUS_COMPAT.
"""
want = _norm_name(command.name)
if not want or command.main is None or command.sub is None:
return None
compat = STATUS_COMPAT.get(command.dpt_main or -1, {command.dpt_main})
for c in project.gas.values():
if c.address == command.address:
continue
if c.main != command.main or c.sub != command.sub:
continue
if c.middle == command.middle:
continue
n = _norm_name(c.name)
if n != want:
# allow the "same name + status word" variant of the same school
if not (n.startswith(want)
and any(k in n[len(want):] for k in STATUS_KEYWORDS)):
continue
if c.dpt_main is not None and c.dpt_main not in compat:
continue
return c
return None
def self_reporting(ga: GARecord, project) -> bool:
"""True when the actuator reports state on the command GA itself.
Some integrators link the actuator's *status* communication object to the
same group address as the command (the CO carries Read+Transmit flags), so
the GA is its own feedback. There is then no separate status GA to find —
and none is missing.
"""
cos = (project.raw or {}).get("communication_objects", {}) or {}
for cid in (ga.co_ids or []):
co = cos.get(cid) or {}
fl = co.get("flags") or {}
if fl.get("read") and fl.get("transmit"):
return True
return False
# --------------------------------------------------------------------------- #
# Authoritative pairing from ETS Functions (roles)
# --------------------------------------------------------------------------- #
+1 -1
View File
@@ -1,6 +1,6 @@
[project]
name = "nickol-knx-mcp"
version = "0.7.0"
version = "0.8.0"
description = "Design-time KNX/ETS6 project assistant as an MCP server (parse .knxproj, validate, generate HA YAML + ETS CSV/XML). No live bus access."
readme = "README.md"
requires-python = ">=3.10"
+133
View File
@@ -0,0 +1,133 @@
"""v0.8 fixes, synthetic and self-contained (issues #3-#6).
1. positional command/status pairing (parallel middles, identical names)
2. self-reporting commands (actuator R+T object on the command GA)
3. main_group_unnamed reads authoritative GroupRange names
4. completeness grader counts pattern-named ranges
5. app-program parser: ComObjectRef merge + newest-version pick
"""
import sys, os, zipfile, tempfile
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from nickol_knx_mcp.project import build_loaded_from_raw
from nickol_knx_mcp.analyze import detect_missing_status, validate_naming
from nickol_knx_mcp.advanced import completeness_grade
from nickol_knx_mcp.appprog_parser import parse_project
def ga(addr, name, main, sub, co_ids=None):
return {
"name": name, "identifier": f"GA-{addr}", "raw_address": 0,
"address": addr, "project_uid": None,
"dpt": ({"main": main, "sub": sub} if main is not None else None),
"data_secure": False, "communication_object_ids": co_ids or [],
"description": "", "comment": "",
}
def make(gas, cos=None, ranges=None):
raw = {
"info": {"project_id": "P-T", "name": "SynthV08", "last_modified": "",
"group_address_style": "ThreeLevel", "guid": "x", "created_by": "t",
"schema_version": "21", "tool_version": "t", "xknxproject_version": "t",
"language_code": "ru-RU"},
"communication_objects": cos or {},
"devices": {}, "topology": {},
"group_addresses": {g["address"]: g for g in gas},
"group_ranges": ranges or {},
"functions": {},
}
return build_loaded_from_raw(raw, "synthetic-v08")
# 1) POSITIONAL PAIRING: command 1/0/7 and status 1/1/7 share the exact name —
# no lexical marker anywhere; must NOT be flagged. An unpaired command must stay flagged.
proj = make([
ga("1/0/7", "Light-Кухня-Люстра-onoff", 1, 1),
ga("1/1/7", "Light-Кухня-Люстра-onoff", 1, 1), # parallel middle, same sub, plain 1.001 — invisible to the lexical matcher
ga("2/0/9", "Штора Кабинет вверх/вниз", 1, 8), # no pair anywhere -> warning
])
f = detect_missing_status(proj)
warn = [x for x in f if x["code"] == "missing_status_address"]
assert len(warn) == 1 and warn[0]["address"] == "2/0/9", warn
summary = [x for x in f if x["code"] == "status_pairing_summary"]
# both directions count (the parallel-middle twin is itself command-classified),
# which is fine: neither GA lacks feedback
assert summary and "2 command(s) paired positionally" in summary[0]["message"], summary
# 2) SELF-REPORTING: the actuator's R+T object sits on the command GA itself.
cos = {
"CO-W": {"name": "cmd", "number": 1, "flags": {"read": False, "write": True,
"communication": True, "transmit": False, "update": False},
"dpts": [{"main": 1, "sub": 1}], "group_address_links": ["1/0/1"]},
"CO-RT": {"name": "state", "number": 2, "flags": {"read": True, "write": False,
"communication": True, "transmit": True, "update": False},
"dpts": [{"main": 1, "sub": 1}], "group_address_links": ["1/0/1"]},
}
proj = make([ga("1/0/1", "Розетка Кабинет", 1, 1, co_ids=["CO-W", "CO-RT"])], cos=cos)
f = detect_missing_status(proj)
assert not [x for x in f if x["code"] == "missing_status_address"], f
assert any("self-reporting" in x["message"] for x in f if x["code"] == "status_pairing_summary")
# 3) MAIN NAMES from GroupRanges: main 1 IS named in ranges (GA-level main_name may
# lie); main 2 truly unnamed -> exactly one finding, for main 2.
ranges = {
"1": {"name": "Освещение", "group_ranges": {"1/0": {"name": "Команды"}}},
"2": {"name": "", "group_ranges": {}},
}
proj = make([ga("1/0/1", "Свет Кухня статус", 1, 11), ga("2/0/1", "Штора Спальня вверх/вниз", 1, 8)],
ranges=ranges)
un = [x for x in validate_naming(proj) if x["code"] == "main_group_unnamed"]
assert len(un) == 1 and un[0]["address"].startswith("2/"), un
# 4) GRADER counts pattern-named ranges: GA names are plain, the middle is called «Сцены».
ranges = {"1": {"name": "Свет", "group_ranges": {"1/5": {"name": "Сцены"}}}}
proj = make([ga("1/5/1", "Гостиная вечер", 18, 1)], ranges=ranges)
g = completeness_grade(proj)
assert g["patterns_present"].get("scenes"), g
# 5) PARSER v2: ref-level merge + version pick.
HW = """<?xml version="1.0" encoding="utf-8"?>
<KNX><ManufacturerData><Manufacturer RefId="M-0073"><Hardware>
<Hardware Id="M-0073_H-T-1" Name="RefVendor Dev" SerialNumber="RV-1">
<Products><Product Id="M-0073_H-T-1_P-1" OrderNumber="RV-1" /></Products>
<Hardware2Programs><Hardware2Program>
<ApplicationProgramRef RefId="M-0073_A-0001-12-AAAA" />
<ApplicationProgramRef RefId="M-0073_A-0001-10-BBBB" />
</Hardware2Program></Hardware2Programs>
</Hardware>
</Hardware></Manufacturer></ManufacturerData></KNX>"""
APP_V18 = """<?xml version="1.0" encoding="utf-8"?>
<KNX><ManufacturerData><Manufacturer><ApplicationPrograms>
<ApplicationProgram Id="M-0073_A-0001-12-AAAA" ApplicationVersion="18"><Static><ComObjectTable>
<ComObject Id="O-1" Name="Object 1" Number="1" ObjectSize="1 Bit" CommunicationFlag="Enabled" />
<ComObject Id="O-2" Name="Object 2" Number="2" ObjectSize="1 Byte" CommunicationFlag="Enabled" />
</ComObjectTable>
<ComObjectRefs>
<ComObjectRef Id="O-1_R-1" RefId="O-1" Text="Switch A" DatapointType="DPST-1-1" WriteFlag="Enabled" />
<ComObjectRef Id="O-2_R-1" RefId="O-2" Text="Value A" DatapointType="DPST-5-1" />
<ComObjectRef Id="O-2_R-2" RefId="O-2" Text="Value B" DatapointType="DPST-5-10" />
</ComObjectRefs>
</Static></ApplicationProgram>
</ApplicationPrograms></Manufacturer></ManufacturerData></KNX>"""
APP_V16 = APP_V18.replace("A-0001-12-AAAA", "A-0001-10-BBBB").replace(
'ApplicationVersion="18"', 'ApplicationVersion="16"').replace("Switch A", "OLD Switch")
with tempfile.TemporaryDirectory() as d:
arc = os.path.join(d, "refvendor.knxprod")
with zipfile.ZipFile(arc, "w") as z:
z.writestr("M-0073/Hardware.xml", HW)
z.writestr("M-0073/M-0073_A-0001-12-AAAA.xml", APP_V18)
z.writestr("M-0073/M-0073_A-0001-10-BBBB.xml", APP_V16)
res = parse_project(arc)
dev = res["devices"][0]
# newest app (v18, hex 12) picked even though listed FIRST is fine — but the
# stale one (hex 10) was listed LAST, which the old [-1] logic would pick.
assert dev["application_program"]["app_id"].endswith("12-AAAA"), dev["application_program"]
o1 = [o for o in dev["comm_objects"] if o["number"] == 1][0]
assert o1["dpt"] == "1.001" and o1["name"] == "Switch A", o1 # merged from ref
assert o1["flags"]["W"] is True, o1 # flag from ref
o2 = [o for o in dev["comm_objects"] if o["number"] == 2][0]
assert o2["dpt"] is None and "5.001" in o2.get("dpt_variants", ""), o2 # refs disagree -> honest
print("OK — v0.8: positional pairing, self-reporting, range names (naming+grader), ref-merge + version pick")