Files
Forge-Engine/tests/test_ai_chat.py
T
williamandClaude Sonnet 5 7db4803b93
Build and deploy / test-python (push) Successful in 12m10s
Build and deploy / test-js (push) Successful in 1m27s
Build and deploy / build-and-push (push) Skipped
Build and deploy / deploy (push) Skipped
Ajoute la fonctionnalite quiz autonome/plein ecran a la boite a quiz
Introduit la double categorie de modeles (boite de dialogue / page de
quiz plein ecran) avec plein ecran, minuteur, score integre et ecran de
resultat pour les modeles page ; ajoute les modeles "Manga" (boite et
page) et "Classique" (page), pilotables aussi par l'assistant IA Ruby.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-13 15:07:23 +02:00

218 lines
10 KiB
Python

"""ai/chat.py — la boucle tool-use (voir plan Phase 2, §5). Jamais un
vrai appel à l'API Claude ici : client.messages.create est monkeypatché
par une fausse classe qui rejoue une séquence de réponses programmée."""
import ai
import db
import screens
def _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup, name="pytest_ai_chat"):
resp = client.post("/games/new", data={"name": name}, follow_redirects=False)
slug = tmp_game_slug_cleanup(resp.headers["Location"].rstrip("/").split("/")[-1])
screen_id = screens.create_screen(slug, "Scène 1", kind="jeu_2d")
conversation_id = screens.create_ia_conversation(slug, screen_id)
return slug, screen_id, conversation_id
class _FakeBlock:
def __init__(self, type, text=None, id=None, name=None, input=None):
self.type = type
self.text = text
self.id = id
self.name = name
self.input = input or {}
class _FakeResponse:
def __init__(self, content, stop_reason):
self.content = content
self.stop_reason = stop_reason
class _FakeMessages:
def __init__(self, responses):
self._responses = list(responses)
self.call_count = 0
self.last_kwargs = None
def create(self, **kwargs):
self.call_count += 1
self.last_kwargs = kwargs
if len(self._responses) > 1:
return self._responses.pop(0)
return self._responses[0] # rejoue la dernière indéfiniment (voir test de la borne)
class _FakeClient:
def __init__(self, responses):
self.messages = _FakeMessages(responses)
def test_run_chat_turn_returns_text_directly_when_no_tool_is_used(client, tmp_game_slug_cleanup, monkeypatch):
slug, screen_id, conversation_id = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
fake_client = _FakeClient([_FakeResponse([_FakeBlock("text", text="Bonjour, que veux-tu créer ?")], "end_turn")])
monkeypatch.setattr("ai.chat.get_client", lambda: fake_client)
reply = ai.run_chat_turn(slug, screen_id, conversation_id, 1, "Salut")
assert reply == "Bonjour, que veux-tu créer ?"
assert fake_client.messages.call_count == 1
def test_run_chat_turn_dispatches_a_tool_call_then_returns_the_final_text(client, tmp_game_slug_cleanup, monkeypatch):
slug, screen_id, conversation_id = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
tool_use_response = _FakeResponse(
[_FakeBlock("tool_use", id="t1", name="create_global_variable", input={"name": "score_ia", "var_type": "nombre_entier"})],
"tool_use",
)
final_response = _FakeResponse([_FakeBlock("text", text="Variable créée.")], "end_turn")
fake_client = _FakeClient([tool_use_response, final_response])
monkeypatch.setattr("ai.chat.get_client", lambda: fake_client)
reply = ai.run_chat_turn(slug, screen_id, conversation_id, 1, "Crée une variable score_ia")
assert reply == "Variable créée."
names = [v["name"] for v in db.list_global_variables(slug)]
assert "score_ia" in names
def test_run_chat_turn_stops_after_the_max_iterations_even_if_claude_keeps_calling_tools(client, tmp_game_slug_cleanup, monkeypatch):
slug, screen_id, conversation_id = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
always_tool_use = _FakeResponse(
[_FakeBlock("tool_use", id="t1", name="create_global_variable", input={"name": "boucle_infinie"})],
"tool_use",
)
fake_client = _FakeClient([always_tool_use])
monkeypatch.setattr("ai.chat.get_client", lambda: fake_client)
reply = ai.run_chat_turn(slug, screen_id, conversation_id, 1, "Fais quelque chose")
assert fake_client.messages.call_count == ai.chat._MAX_TOOL_ITERATIONS
assert reply == "(pas de réponse textuelle)"
def test_run_chat_turn_explains_when_cut_short_by_max_tokens(client, tmp_game_slug_cleanup, monkeypatch):
"""Bug corrigé : max_tokens=4096 pouvait couper Claude EN PLEINE
RÉFLEXION sur une demande riche, avant le moindre appel d'outil —
symptôme observé : "(pas de réponse textuelle)" dès le premier tour,
aucune progression. Message désormais plus clair pour le créateur."""
slug, screen_id, conversation_id = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
cut_short = _FakeResponse([], "max_tokens") # aucun bloc texte, coupé en cours de réflexion
fake_client = _FakeClient([cut_short])
monkeypatch.setattr("ai.chat.get_client", lambda: fake_client)
reply = ai.run_chat_turn(slug, screen_id, conversation_id, 1, "Fais quelque chose de complexe")
assert "interrompue" in reply
assert fake_client.messages.call_count == 1 # stop_reason != "tool_use" -> sort dès le premier tour
def test_run_chat_turn_uses_a_generous_max_tokens_not_the_old_lowballed_value(client, tmp_game_slug_cleanup, monkeypatch):
slug, screen_id, conversation_id = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
fake_client = _FakeClient([_FakeResponse([_FakeBlock("text", text="ok")], "end_turn")])
monkeypatch.setattr("ai.chat.get_client", lambda: fake_client)
ai.run_chat_turn(slug, screen_id, conversation_id, 1, "Salut")
assert fake_client.messages.last_kwargs["max_tokens"] >= 16000
def test_run_chat_turn_raises_when_anthropic_is_not_configured(client, tmp_game_slug_cleanup):
slug, screen_id, conversation_id = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
try:
ai.run_chat_turn(slug, screen_id, conversation_id, 1, "Salut")
assert False, "devrait lever AnthropicNotConfiguredError"
except ai.AnthropicNotConfiguredError:
pass
# ---------- État de scène ré-injecté à chaque tour (bug corrigé : Ruby
# dupliquait des objets faute de voir ce qui existait déjà) ----------
def test_describe_scene_state_reports_dimensions_and_no_objects(client, tmp_game_slug_cleanup):
slug, screen_id, _ = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
description = ai.chat._describe_scene_state(slug, screen_id)
assert "0 à 960" in description and "0 à 540" in description
assert "Aucun objet" in description
def test_describe_scene_state_lists_existing_objects_with_role_and_trigger_flag(client, tmp_game_slug_cleanup):
slug, screen_id, _ = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
player_id = screens.add_scene_object(slug, screen_id, kind="personnage")
screens.set_scene_object_role(slug, player_id, "joueur")
pnj_id = screens.add_scene_object(slug, screen_id, kind="personnage")
screens.set_scene_object_collision_rules(slug, pnj_id, screens.sanitize_collision_rules([
{"trigger": "collision", "action": {"type": "dialogue", "dialogue": {"id": "d_1", "lines": []}}},
]))
description = ai.chat._describe_scene_state(slug, screen_id)
assert f"id={player_id}" in description
assert "rôle=joueur" in description
assert f"id={pnj_id}" in description
assert "rôle=pnj" in description
assert "[déclencheur déjà configuré]" in description
def test_describe_scene_state_reports_no_screen_trigger_by_default(client, tmp_game_slug_cleanup):
slug, screen_id, _ = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
description = ai.chat._describe_scene_state(slug, screen_id)
assert "Aucun déclencheur d'écran" in description
def test_describe_scene_state_reports_an_existing_screen_trigger(client, tmp_game_slug_cleanup):
slug, screen_id, _ = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
screens.set_screen_triggers(slug, screen_id, screens.sanitize_screen_triggers([
{"trigger": "affichage", "action": {"type": "dialogue"}},
]))
description = ai.chat._describe_scene_state(slug, screen_id)
assert "1 déclencheur(s) D'ÉCRAN" in description
def test_system_prompt_documents_the_new_triggers_and_actions():
"""Bug à éviter : ajouter un type de déclencheur/action côté moteur
(screens/rendering/collision_rules.py) sans jamais le documenter dans
le system prompt le rendrait invisible pour Ruby malgré l'outil qui
l'accepte techniquement."""
prompt = ai.chat._SYSTEM_PROMPT
for keyword in ("\"clic\"", "\"survol\"", "\"affichage\"", "surbrillance", "visibilite", "\"son\"", "\"video\"", "indication"):
assert keyword in prompt, keyword
def test_system_prompt_asks_ruby_to_clarify_ambiguous_trigger_choice():
assert "pose la question" in ai.chat._SYSTEM_PROMPT
def test_system_prompt_asks_ruby_to_clarify_interagir_vs_immediate_action():
assert "interagir" in ai.chat._SYSTEM_PROMPT
assert "APPUYER SUR UNE TOUCHE" in ai.chat._SYSTEM_PROMPT
def test_system_prompt_documents_the_quiz_box_config_tool():
"""Demande explicite : "Ruby dois pouvoir aussi piloter ces réglages"
(plein écran/minuteur/modèle d'un quiz autonome) — pas seulement
l'éditeur manuel."""
prompt = ai.chat._SYSTEM_PROMPT
assert "set_quiz_box_config" in prompt
for keyword in ("fullscreen", "timer_mode", "dialog_template", "page_template"):
assert keyword in prompt, keyword
def test_system_prompt_asks_ruby_to_clarify_the_timer_before_enabling_it():
assert "jamais imposé" in ai.chat._SYSTEM_PROMPT
def test_system_prompt_documents_the_two_quiz_template_categories():
"""Demande explicite : "Ruby doit également comprendre cette
distinction" (modèles "boîte de dialogue" hors plein écran vs modèles
"page de quiz" en plein écran, structures HTML totalement séparées)."""
prompt = ai.chat._SYSTEM_PROMPT
assert "page de quiz" in prompt
assert "manga" in prompt
def test_run_chat_turn_passes_the_scene_state_in_the_system_prompt(client, tmp_game_slug_cleanup, monkeypatch):
slug, screen_id, conversation_id = _create_jeu2d_game_with_conversation(client, tmp_game_slug_cleanup)
screens.add_scene_object(slug, screen_id, kind="personnage")
fake_client = _FakeClient([_FakeResponse([_FakeBlock("text", text="ok")], "end_turn")])
monkeypatch.setattr("ai.chat.get_client", lambda: fake_client)
ai.run_chat_turn(slug, screen_id, conversation_id, 1, "Salut")
system_prompt = fake_client.messages.last_kwargs["system"]
assert "0 à 960" in system_prompt and "0 à 540" in system_prompt
assert "1 objet(s)" in system_prompt