Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 14 additions & 7 deletions ainode/api/systemone.py
Original file line number Diff line number Diff line change
Expand Up @@ -118,14 +118,16 @@ class Translated(NamedTuple):
``options`` is what the model reads, one line per option. ``names`` is what
each of those options answers to on the wire, in the same order: a choice
criteria key verbatim, ``true`` / ``false``, or a score level's position.
Keeping the pair here is what lets the answer name the caller's own key
rather than the letter the engine was constrained to.
``legend`` keeps an object-form score's display labels separate from those
keys. Keeping these values here lets the answer preserve both parts of the
caller's rubric rather than trying to recover labels from prompt text.
"""

kind: str
question: str
options: list[str]
names: list[str]
legend: Optional[list[str]] = None


# ----------------------------------------------------------------- translate in
Expand Down Expand Up @@ -212,7 +214,7 @@ def noul_options(key: str, criteria: Any) -> tuple[list[str], list[str]]:
list(NOUL_OPTIONS))


def score_options(key: str, criteria: Any) -> tuple[list[str], list[str]]:
def score_options(key: str, criteria: Any) -> tuple[list[str], list[str], list[str]]:
"""A score's levels in rubric order: a list by position, an object by insertion.

Both spellings are accepted because both are in the wild: the list form names
Expand All @@ -228,10 +230,13 @@ def score_options(key: str, criteria: Any) -> tuple[list[str], list[str]]:
f"a non-empty string (got {level!r})")
names.append(level.strip())
options = list(names)
legend = list(names)
elif isinstance(criteria, dict):
pairs = criteria_pairs(key, criteria, "score")
names = [name for name, _ in pairs]
options = [option_text(name, desc) for name, desc in pairs]
legend = [desc if isinstance(desc, str) and desc.strip() else name
for name, desc in pairs]
else:
raise DecideError(
f"question '{key}': a score question needs 'criteria', either an ordered "
Expand All @@ -243,7 +248,7 @@ def score_options(key: str, criteria: Any) -> tuple[list[str], list[str]]:
if len(set(names)) != len(names):
raise DecideError(f"question '{key}': 'criteria' repeats a level name. Every "
"level must be distinct so a score names one of them")
return options, names
return options, names, legend


def translate_one(key: str, spec: Any) -> Translated:
Expand All @@ -267,10 +272,11 @@ def translate_one(key: str, spec: Any) -> Translated:
if kind == NOUL:
options, names = noul_options(key, criteria)
elif kind == SCORE:
options, names = score_options(key, criteria)
options, names, legend = score_options(key, criteria)
else:
options, names = choice_options(key, criteria)
return Translated(kind, instructions.strip(), options, names)
return Translated(kind, instructions.strip(), options, names,
legend if kind == SCORE else None)


def translate_questions(raw: Any) -> dict[str, Translated]:
Expand Down Expand Up @@ -377,7 +383,8 @@ def answer_from_decision(item: Translated, entry: dict) -> Optional[dict]:
"noul": by_name["true"] if dist else float(picked == "true")}

if item.kind == SCORE:
legend = {str(index): name for index, name in enumerate(item.names)}
legend = {str(index): label for index, label in enumerate(
item.legend if item.legend is not None else item.names)}
if not dist:
return {"type": SCORE, "score": float(item.names.index(picked)),
"confidence": confidence, "legend": legend}
Expand Down
31 changes: 30 additions & 1 deletion tests/test_systemone.py
Original file line number Diff line number Diff line change
Expand Up @@ -338,6 +338,34 @@ def test_a_score_is_the_expected_level_with_a_legend_by_position():
assert math.isclose(sum(answer["probabilities"].values()), 1.0, abs_tol=1e-6)


def test_a_score_legend_uses_the_labels_from_object_criteria():
item = translate_questions({
"frustration": {
"type": "score",
"instructions": "How frustrated is the customer?",
"criteria": {
"0": "Calm",
"1": "Mildly annoyed",
"2": "Frustrated",
"3": "Angry",
},
},
})["frustration"]
entry = _entry("2: Frustrated", {
"0: Calm": 0.0,
"1: Mildly annoyed": 0.0,
"2: Frustrated": 1.0,
"3: Angry": 0.0,
}, 1.0)

assert answer_from_decision(item, entry)["legend"] == {
"0": "Calm",
"1": "Mildly annoyed",
"2": "Frustrated",
"3": "Angry",
}


def test_a_score_on_one_certain_level_is_that_level():
entry = _entry("bad: money at risk", {"none: cosmetic": 0.0,
"some: workaround": 0.0,
Expand Down Expand Up @@ -470,7 +498,8 @@ async def test_every_question_type_round_trips(client, engine_fake):
assert needs_human["type"] == "noul" and needs_human["noul"] > 0.9

severity = data["answers"]["severity"]
assert severity["legend"] == {"0": "none", "1": "some", "2": "bad"}
assert severity["legend"] == {"0": "cosmetic", "1": "a workaround exists",
"2": "money or data is at risk"}
assert set(severity["probabilities"]) == {"0", "1", "2"}
assert 0.0 <= severity["score"] < 0.1 # nearly all the mass on level 0

Expand Down
Loading