ContractViolation takes (rule, detail, offending_output=""). The guard passed a single string, so every retired name raised TypeError: ContractViolation.__init__() missing 1 required positional argument: 'detail' instead of the clean error it was written to produce - a guard that crashes the request it was meant to explain. It was not caught before deployment because this machine has neither fastapi nor psycopg, so the guard could not be imported. It was tested by lifting its source out of the file and exercising it against a stubbed ContractViolation - a plain Exception, which accepts one argument. The stub is what made the test pass. Running it in the built image found it in one command, which is the argument for running the suite where the dependencies are rather than reasoning about it where they are not. Also records what the error actually does: main.py logs the rule, returns 422, and deliberately keeps the detail out of the operator-facing message, so the explanation lands in the log and in Langfuse rather than on screen. That is the right trade - an operator asking in plain English never types one of these names, so the audience is a developer or a stale integration. Verified in the running container: the retired names and their lowercase forms are rejected with rule "retired_tag_name"; the poll groups, the item names, the instrument tags and plain English all pass. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
209 lines
8.2 KiB
Python
209 lines
8.2 KiB
Python
"""Alias resolution: what the operator said -> what the plant calls it.
|
|
|
|
"Pump 02" is equipment. Its data lives on tags. No historian point is called
|
|
"Pump 02", so without this module every equipment-level question fails.
|
|
|
|
Resolution is a database lookup against the GIN-indexed alias arrays, not a
|
|
model call. It is deterministic, it is cheap, and when it is wrong the fix goes
|
|
in db/seed/tags.csv - not in a prompt.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import re
|
|
from dataclasses import dataclass
|
|
|
|
import psycopg
|
|
from psycopg.rows import dict_row
|
|
|
|
from config import settings
|
|
from contracts import ContractViolation
|
|
|
|
log = logging.getLogger("tools.equipment")
|
|
|
|
|
|
@dataclass
|
|
class Resolved:
|
|
canonical_id: str
|
|
display_name: str
|
|
kind: str # "equipment" or "tag"
|
|
matched_alias: str
|
|
confidence: float # 1.0 exact id, 0.9 exact alias, lower for fuzzy
|
|
description: str = ""
|
|
|
|
|
|
def _normalise(text: str) -> str:
|
|
"""Fold the ways an operator types the same thing.
|
|
|
|
'Pump 02', 'pump 2', 'PUMP-2' and 'pump2' all become 'pump 2'. Leading
|
|
zeros go, because 'P-002' and 'P2' are the same unit to everyone except a
|
|
string comparison.
|
|
"""
|
|
t = text.strip().lower()
|
|
t = re.sub(r"[-_/]+", " ", t)
|
|
t = re.sub(r"\s+", " ", t)
|
|
t = re.sub(r"\b0+(\d)", r"\1", t)
|
|
return t
|
|
|
|
|
|
def _connect() -> psycopg.Connection:
|
|
cfg = settings()
|
|
return psycopg.connect(
|
|
cfg.dsn(),
|
|
row_factory=dict_row,
|
|
application_name="ai-api", # so DBAs can see who is connecting
|
|
connect_timeout=5,
|
|
)
|
|
|
|
|
|
# PS_* IS NOT A NAMESPACE, AND ASKING FOR ONE IS AN ERROR - NOT AN EMPTY RESULT.
|
|
#
|
|
# The point list delivered in August carried names like PS_STN_WET_WELL_LEVEL
|
|
# where the CI Server ITEM name belongs (AID.WRPS.STN.LEVEL). CI Server never
|
|
# used them; they were a proposal derived from the PLC register map. A corrected
|
|
# delivery arrived 2026-09-01 and the seed carries no PS_ key any more.
|
|
#
|
|
# Without this guard the old names simply match nothing, and the assistant says
|
|
# "no records found" - which an operator reads as "the plant recorded nothing",
|
|
# not as "you asked with a retired name". That is the exact confusion the four
|
|
# namespaces exist to prevent, so it fails loudly instead.
|
|
#
|
|
# The four Modbus POLL GROUPS keep the prefix legitimately. They are groups, not
|
|
# point or tag names, and nothing resolves an operator term onto one.
|
|
_POLL_GROUPS = {"PS_STATUS_BITS", "PS_PUBLISHED", "PS_SETPOINTS", "PS_SIM_CONTROL"}
|
|
_RETIRED_TAG = re.compile(r"^PS_[A-Z][A-Z0-9_]*$")
|
|
|
|
|
|
def reject_retired_tag(term: str) -> None:
|
|
"""Raise if `term` is one of the retired PS_* names. See the note above."""
|
|
candidate = term.strip().upper()
|
|
if _RETIRED_TAG.match(candidate) and candidate not in _POLL_GROUPS:
|
|
# ContractViolation takes (rule, detail). main.py logs the rule, returns
|
|
# 422, and deliberately does NOT put the detail in the operator-facing
|
|
# message - an error is not a side channel for content that failed a
|
|
# check. So the explanation below lands in the log and in Langfuse, not
|
|
# on the screen. That is the right trade here: an operator asking in
|
|
# plain English never types one of these names, so the audience for this
|
|
# error is a developer or a stale integration, and both read logs.
|
|
raise ContractViolation(
|
|
"retired_tag_name",
|
|
f"{candidate} is a retired name and is not a tag in this system. "
|
|
"It came from a superseded SCADA point list; CI Server never used it. "
|
|
"Use the CI Server item name instead, for example AID.WRPS.STN.LEVEL. "
|
|
"See db/seed/scada-source/README.md.",
|
|
)
|
|
|
|
|
|
def resolve(term: str, *, conn: psycopg.Connection | None = None) -> list[Resolved]:
|
|
"""Resolve one operator term to equipment and/or tags, best first.
|
|
|
|
Returns every plausible match rather than picking one. An ambiguous term is
|
|
a clarifying question, not a coin toss - agent.py surfaces the alternatives.
|
|
"""
|
|
reject_retired_tag(term)
|
|
owned = conn is None
|
|
conn = conn or _connect()
|
|
try:
|
|
needle = _normalise(term)
|
|
out: list[Resolved] = []
|
|
|
|
with conn.cursor() as cur:
|
|
cur.execute(
|
|
"""
|
|
SELECT equipment_id AS id, display_name, description,
|
|
aliases, 'equipment' AS kind
|
|
FROM equipment
|
|
WHERE lower(equipment_id) = %(raw)s
|
|
OR lower(display_name) = %(raw)s
|
|
OR EXISTS (SELECT 1 FROM unnest(aliases) a
|
|
WHERE lower(a) = %(raw)s)
|
|
UNION ALL
|
|
SELECT tag_id AS id, display_name, description, aliases, 'tag' AS kind
|
|
FROM tags
|
|
WHERE lower(tag_id) = %(raw)s
|
|
OR lower(display_name) = %(raw)s
|
|
OR EXISTS (SELECT 1 FROM unnest(aliases) a
|
|
WHERE lower(a) = %(raw)s)
|
|
""",
|
|
{"raw": term.strip().lower()},
|
|
)
|
|
for row in cur.fetchall():
|
|
out.append(
|
|
Resolved(
|
|
canonical_id=row["id"],
|
|
display_name=row["display_name"],
|
|
kind=row["kind"],
|
|
matched_alias=term.strip(),
|
|
confidence=1.0 if row["id"].lower() == term.strip().lower() else 0.9,
|
|
description=row["description"] or "",
|
|
)
|
|
)
|
|
|
|
if out:
|
|
return sorted(out, key=lambda r: -r.confidence)
|
|
|
|
# Nothing matched literally. Try the normalised forms, in Python, so
|
|
# the same folding applies to both sides.
|
|
with conn.cursor() as cur:
|
|
cur.execute(
|
|
"SELECT equipment_id AS id, display_name, description, aliases,"
|
|
" 'equipment' AS kind FROM equipment"
|
|
" UNION ALL "
|
|
"SELECT tag_id AS id, display_name, description, aliases,"
|
|
" 'tag' AS kind FROM tags"
|
|
)
|
|
for row in cur.fetchall():
|
|
candidates = [row["id"], row["display_name"], *(row["aliases"] or [])]
|
|
for candidate in candidates:
|
|
if candidate and _normalise(candidate) == needle:
|
|
out.append(
|
|
Resolved(
|
|
canonical_id=row["id"],
|
|
display_name=row["display_name"],
|
|
kind=row["kind"],
|
|
matched_alias=candidate,
|
|
confidence=0.8,
|
|
description=row["description"] or "",
|
|
)
|
|
)
|
|
break
|
|
|
|
return sorted(out, key=lambda r: -r.confidence)
|
|
finally:
|
|
if owned:
|
|
conn.close()
|
|
|
|
|
|
def tags_for_equipment(equipment_id: str, *, conn: psycopg.Connection | None = None) -> list[dict]:
|
|
"""Every tag belonging to a piece of equipment, with its metadata.
|
|
|
|
The description field says whether the tag is historised at all. Field
|
|
inputs to the PLC have no history; an answer that trends PU-301 vibration
|
|
is fabricating data. Check before querying Cube for it.
|
|
"""
|
|
owned = conn is None
|
|
conn = conn or _connect()
|
|
try:
|
|
with conn.cursor() as cur:
|
|
cur.execute(
|
|
"SELECT tag_id, display_name, signal_type, engineering_unit,"
|
|
" range_low, range_high, alarm_setpoint_hi, alarm_setpoint_lo,"
|
|
" trip_setpoint, description"
|
|
" FROM tags WHERE equipment_id = %s ORDER BY tag_id",
|
|
(equipment_id,),
|
|
)
|
|
return cur.fetchall()
|
|
finally:
|
|
if owned:
|
|
conn.close()
|
|
|
|
|
|
def is_historised(tag_row: dict) -> bool:
|
|
"""Whether a tag has history to query.
|
|
|
|
Encoded in the description because the CSV is the source of truth and a
|
|
boolean column would drift from it. HISTORISED and NOT HISTORISED are
|
|
written in capitals at the start of the description for exactly this.
|
|
"""
|
|
return not (tag_row.get("description") or "").upper().startswith("NOT HISTORISED")
|