Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
72 changes: 72 additions & 0 deletions apps/skillnet-api/src/services/activity_authoring.py
Original file line number Diff line number Diff line change
Expand Up @@ -300,6 +300,7 @@ def assert_grounded_activity_draft(draft: ActivityAuthoringDraft) -> None:
"""

_reject_unsubstituted_example(draft.component_id, draft.definition)
_reject_unsubstituted_expected(draft.component_id, draft.definition)


def _reject_unsubstituted_example(component_id: str, definition: Mapping[str, Any]) -> None:
Expand Down Expand Up @@ -334,6 +335,77 @@ def _reject_unsubstituted_example(component_id: str, definition: Mapping[str, An
)


#: Components whose ``evaluation.expected`` holds free-text answer content (not
#: structural ids/refs like matching pairs, option values or step ids). For these,
#: the example's placeholder answers ("respuesta", "la respuesta con fundamento"...)
#: are graded text, not decoration — see ``_reject_unsubstituted_expected``.
_FREE_TEXT_EXPECTED_COMPONENTS = frozenset(
{
"didact.quiz.fill-in-the-blank",
"didact.quiz.short-answer",
"didact.completion-problem",
}
)


def _expected_texts(value: Any) -> list[str]:
"""Flatten a ``keyed_text``/``normalized_any`` ``expected`` value into text leaves."""

if isinstance(value, str):
text = value.strip()
return [text] if text else []
if isinstance(value, Mapping):
out: list[str] = []
for child in value.values():
out.extend(_expected_texts(child))
return out
if isinstance(value, (list, tuple)):
out = []
for item in value:
out.extend(_expected_texts(item))
return out
return []


def _reject_unsubstituted_expected(component_id: str, definition: Mapping[str, Any]) -> None:
"""Reject a free-text answer key that is still the contract's example placeholder.

``evaluation``/``expected`` is deliberately excluded from ``_content_texts`` above:
for most components it holds structural ids (matching pairs, option values, step
ids), never content worth grounding. But for the three ``keyed_text``/
``normalized_any`` components in ``_FREE_TEXT_EXPECTED_COMPONENTS`` the expected
values ARE the graded answer text. A model that leaves them as the example's
placeholder produces a fill-in-the-blank (or short-answer) whose question looks
grounded but whose correct answer is literally the word "respuesta" — the exact bug
reported by users. Unlike a stray content leaf, a single echoed answer is enough to
reject: it is exactly what grading compares against, so any overlap makes the
exercise unanswerable regardless of how much else was grounded correctly.
"""

if component_id not in _FREE_TEXT_EXPECTED_COMPONENTS:
return
try:
contract = authoring_definition_contract(component_id)
except Exception: # noqa: BLE001 - a missing contract is handled by shape validation
return
example_expected = {
text.casefold()
for text in _expected_texts(contract.get("evaluation", {}).get("expected"))
}
if not example_expected:
return
given_expected = {
text.casefold()
for text in _expected_texts(definition.get("evaluation", {}).get("expected"))
}
echoed = given_expected & example_expected
if echoed:
raise ValueError(
"activity content not grounded: contract example answer key was not "
f"substituted ({sorted(echoed)})"
)


def validate_authoring_draft(
draft: ActivityAuthoringDraft,
*,
Expand Down
28 changes: 28 additions & 0 deletions apps/skillnet-api/tests/test_activity_authoring.py
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,34 @@ def test_grounding_gate_accepts_source_substituted_content():
assert_grounded_activity_draft(draft) # does not raise


def test_grounding_gate_rejects_a_fill_blank_whose_answer_is_still_the_placeholder():
# The question itself is genuinely grounded (not an echoed example leaf), but the
# model left `evaluation.expected` as the contract's literal placeholder answer.
# This is the reported bug: a fill-in-the-blank whose correct answer is "respuesta".
draft = ActivityAuthoringDraft(
component_id="didact.quiz.fill-in-the-blank",
definition={
"question": "El proceso que convierte glucosa en energia se llama ___.",
"evaluation": {"mode": "normalized_any", "expected": ["respuesta"]},
},
source_refs=["atom-1"],
)
with pytest.raises(ValueError, match="answer key was not"):
assert_grounded_activity_draft(draft)


def test_grounding_gate_accepts_a_fill_blank_with_a_real_answer():
draft = ActivityAuthoringDraft(
component_id="didact.quiz.fill-in-the-blank",
definition={
"question": "El proceso que convierte glucosa en energia se llama ___.",
"evaluation": {"mode": "normalized_any", "expected": ["respiracion celular"]},
},
source_refs=["atom-1"],
)
assert_grounded_activity_draft(draft) # does not raise


def test_server_refs_replace_model_invented_reference_objects_before_validation():
draft = authoring_draft_with_server_refs(
_model_authoring_payload(
Expand Down
Loading