Major improvements to battleship game feedback accuracy and user experience: ## New Multi-Prompt Feedback System - Replaced single feedback with 3 specialized prompts: Shot Report, Ship Status, Game Over - Each prompt has individual metadata filtering to see only relevant data - Shot Report only sees hit/miss data, Ship Status only sees ship destruction data - Added STFU token system to suppress empty messages (filtered out automatically) ## Technical Implementation - Added per-prompt metadata_filter support in YAML structure - Updated app.py and guarded_ai.py to handle prompt-specific filtering - Legacy single-prompt system still works with transition-level filtering - Added comprehensive test suite for feedback system validation ## User Experience Fixes - Fixed TTS queue blocking JavaScript execution (async promises instead of await) - Ship Status now correctly reports who destroyed which ship (role confusion fixed) - Game Over only appears when game actually ends (no more random messages) - Maintained dramatic storytelling while ensuring factual accuracy ## Battleship-Specific Improvements - Ship destruction messages only appear when ships actually sink - Clear separation of concerns: hits/misses vs ship destruction vs game over - Eliminated false positive ship destruction reports - Fixed role reversal where wrong player got credit for destruction The battleship narrator now provides accurate, contextual feedback while preserving the dramatic naval warfare atmosphere.
542 lines
21 KiB
Python
542 lines
21 KiB
Python
import argparse
|
|
import yaml
|
|
import json
|
|
import random
|
|
import os
|
|
from openai import OpenAI
|
|
|
|
# Global model-client mapping
|
|
MODEL_CLIENT_MAP = {}
|
|
|
|
|
|
def get_client_for_endpoint(endpoint, api_key):
|
|
"""Create OpenAI client for any endpoint"""
|
|
return OpenAI(api_key=api_key, base_url=endpoint)
|
|
|
|
|
|
def initialize_model_map():
|
|
"""Initialize the model-client mapping from environment variables"""
|
|
global MODEL_CLIENT_MAP
|
|
|
|
# Load endpoints from environment variables
|
|
for i in range(1000): # Support up to 1000 endpoints
|
|
endpoint_key = f"MODEL_ENDPOINT_{i}"
|
|
api_key_key = f"MODEL_API_KEY_{i}"
|
|
|
|
endpoint = os.getenv(endpoint_key)
|
|
api_key = os.getenv(api_key_key)
|
|
|
|
if endpoint and api_key:
|
|
try:
|
|
client = get_client_for_endpoint(endpoint, api_key)
|
|
# Try to get models (simplified - just register endpoint)
|
|
MODEL_CLIENT_MAP[f"endpoint_{i}"] = (client, endpoint)
|
|
except Exception as e:
|
|
print(f"Warning: Failed to initialize endpoint {endpoint}: {e}")
|
|
|
|
|
|
def get_openai_client_and_model(model_name=None):
|
|
"""Get OpenAI client and model name"""
|
|
if not model_name:
|
|
model_name = "adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic"
|
|
|
|
# Try to find client for specific model
|
|
for stored_model, (client, base_url) in MODEL_CLIENT_MAP.items():
|
|
if model_name in stored_model or stored_model == model_name:
|
|
return client, model_name
|
|
|
|
# Fallback to first available client
|
|
if MODEL_CLIENT_MAP:
|
|
client, _ = next(iter(MODEL_CLIENT_MAP.values()))
|
|
return client, model_name
|
|
|
|
# Final fallback to environment or default OpenAI
|
|
api_key = os.getenv("OPENAI_API_KEY", "dummy-key")
|
|
endpoint = os.getenv("MODEL_ENDPOINT_0", "https://api.openai.com/v1")
|
|
|
|
client = get_client_for_endpoint(endpoint, api_key)
|
|
return client, model_name
|
|
|
|
|
|
# Initialize the model mapping on startup
|
|
initialize_model_map()
|
|
|
|
|
|
# Load the YAML activity file
|
|
def load_yaml_activity(file_path):
|
|
with open(file_path, "r") as file:
|
|
return yaml.safe_load(file)
|
|
|
|
|
|
# Categorize the user's response using gpt-4o-mini
|
|
def categorize_response(question, response, buckets, tokens_for_ai):
|
|
bucket_list = ", ".join([str(bucket) for bucket in buckets])
|
|
messages = [
|
|
{
|
|
"role": "system",
|
|
"content": f"{tokens_for_ai} Categorize the following response into one of the following buckets: {bucket_list}. Return ONLY a bucket label.",
|
|
},
|
|
{
|
|
"role": "user",
|
|
"content": f"Question: {question}\nResponse: {response}\n\nCategory:",
|
|
},
|
|
]
|
|
|
|
try:
|
|
client, model_name = get_openai_client_and_model()
|
|
completion = client.chat.completions.create(
|
|
model=model_name,
|
|
messages=messages,
|
|
max_tokens=5,
|
|
temperature=0,
|
|
)
|
|
category = (
|
|
completion.choices[0].message.content.strip().lower().replace(" ", "_")
|
|
)
|
|
return category
|
|
except Exception as e:
|
|
return f"Error: {e}"
|
|
|
|
|
|
# Generate AI feedback using gpt-4o-mini
|
|
def generate_ai_feedback(category, question, user_response, tokens_for_ai, metadata):
|
|
messages = [
|
|
{
|
|
"role": "system",
|
|
"content": f"{tokens_for_ai} Generate a human-readable feedback message based on the following:",
|
|
},
|
|
{
|
|
"role": "user",
|
|
"content": f"Question: {question}\nResponse: {user_response}\nCategory: {category},\nMetadata: {metadata}",
|
|
},
|
|
]
|
|
|
|
try:
|
|
client, model_name = get_openai_client_and_model()
|
|
completion = client.chat.completions.create(
|
|
model=model_name, messages=messages, max_tokens=250, temperature=0.7
|
|
)
|
|
feedback = completion.choices[0].message.content.strip()
|
|
return feedback
|
|
except Exception as e:
|
|
return f"Error: {e}"
|
|
|
|
|
|
# Provide feedback based on the category (legacy single feedback system)
|
|
def provide_feedback(
|
|
transition,
|
|
category,
|
|
question,
|
|
user_response,
|
|
user_language,
|
|
tokens_for_ai,
|
|
metadata,
|
|
):
|
|
feedback = ""
|
|
if "ai_feedback" in transition:
|
|
tokens_for_ai += f" Provide the feedback in {user_language}. {transition['ai_feedback'].get('tokens_for_ai', '')}."
|
|
|
|
# Filter metadata for feedback if metadata_feedback_filter is specified
|
|
feedback_metadata = metadata
|
|
if "metadata_feedback_filter" in transition:
|
|
filter_keys = transition["metadata_feedback_filter"]
|
|
feedback_metadata = {k: v for k, v in metadata.items() if k in filter_keys}
|
|
|
|
ai_feedback = generate_ai_feedback(
|
|
category, question, user_response, tokens_for_ai, feedback_metadata
|
|
)
|
|
feedback += f"\n\nAI Feedback: {ai_feedback}"
|
|
|
|
return feedback
|
|
|
|
|
|
# Provide feedback using multiple prompts (new system)
|
|
def provide_feedback_prompts(
|
|
transition,
|
|
category,
|
|
question,
|
|
feedback_prompts,
|
|
user_response,
|
|
user_language,
|
|
metadata,
|
|
legacy_tokens_for_ai="",
|
|
):
|
|
"""Generate feedback from multiple prompts"""
|
|
feedback_messages = []
|
|
|
|
for prompt in feedback_prompts:
|
|
prompt_name = prompt.get("name", "unnamed")
|
|
tokens_for_ai = prompt.get("tokens_for_ai", "")
|
|
|
|
# Apply per-prompt metadata filtering if specified
|
|
prompt_metadata = metadata
|
|
if "metadata_filter" in prompt:
|
|
filter_keys = prompt["metadata_filter"]
|
|
prompt_metadata = {k: v for k, v in metadata.items() if k in filter_keys}
|
|
|
|
# Combine legacy tokens with prompt-specific tokens
|
|
if legacy_tokens_for_ai:
|
|
tokens_for_ai = legacy_tokens_for_ai + " " + tokens_for_ai
|
|
|
|
# Add language instruction
|
|
tokens_for_ai += f" Provide the feedback in {user_language}."
|
|
|
|
# Add transition-specific AI feedback if present
|
|
if "ai_feedback" in transition:
|
|
tokens_for_ai += f" {transition['ai_feedback'].get('tokens_for_ai', '')}"
|
|
|
|
ai_feedback = generate_ai_feedback(
|
|
category, question, user_response, tokens_for_ai, prompt_metadata
|
|
)
|
|
|
|
# Only add feedback if it has content and isn't exactly the STFU token
|
|
if ai_feedback and ai_feedback.strip() and ai_feedback.strip() != "STFU":
|
|
feedback_messages.append({
|
|
"name": prompt_name,
|
|
"content": ai_feedback.strip()
|
|
})
|
|
|
|
return feedback_messages
|
|
|
|
|
|
def execute_processing_script(metadata, script):
|
|
# Prepare the local environment for the script
|
|
local_env = {"metadata": metadata, "script_result": None}
|
|
|
|
# Execute the script
|
|
exec(script, {}, local_env)
|
|
|
|
# Return the result from the script
|
|
return local_env["script_result"]
|
|
|
|
|
|
def get_next_section_and_step(activity_content, current_section_id, current_step_id):
|
|
for section in activity_content["sections"]:
|
|
if section["section_id"] == current_section_id:
|
|
for i, step in enumerate(section["steps"]):
|
|
if step["step_id"] == current_step_id:
|
|
if i + 1 < len(section["steps"]):
|
|
return section["section_id"], section["steps"][i + 1]["step_id"]
|
|
else:
|
|
# Move to the next section
|
|
next_section_index = (
|
|
activity_content["sections"].index(section) + 1
|
|
)
|
|
if next_section_index < len(activity_content["sections"]):
|
|
next_section = activity_content["sections"][
|
|
next_section_index
|
|
]
|
|
return (
|
|
next_section["section_id"],
|
|
next_section["steps"][0]["step_id"],
|
|
)
|
|
return None, None
|
|
|
|
|
|
def translate_text(text, target_language):
|
|
# Guard clause for default language
|
|
if target_language.lower() == "english":
|
|
return text
|
|
|
|
messages = [
|
|
{
|
|
"role": "system",
|
|
"content": f"Translate the following text to {target_language}:",
|
|
},
|
|
{
|
|
"role": "user",
|
|
"content": text,
|
|
},
|
|
]
|
|
|
|
try:
|
|
completion = client.chat.completions.create(
|
|
model="gpt-4o-mini", messages=messages, max_tokens=500, temperature=0.7
|
|
)
|
|
translation = completion.choices[0].message.content.strip()
|
|
return translation
|
|
except Exception as e:
|
|
return f"Error: {e}"
|
|
|
|
|
|
def simulate_activity(yaml_file_path):
|
|
yaml_content = load_yaml_activity(yaml_file_path)
|
|
max_attempts = yaml_content.get("default_max_attempts_per_step", 3)
|
|
|
|
current_section_id = yaml_content["sections"][0]["section_id"]
|
|
current_step_id = yaml_content["sections"][0]["steps"][0]["step_id"]
|
|
|
|
metadata = {"language": "English"} # Default language
|
|
|
|
while current_section_id and current_step_id:
|
|
print(
|
|
f"\n\nCurrent section: {current_section_id}, Current step: {current_step_id}\n\n"
|
|
)
|
|
section = next(
|
|
(
|
|
s
|
|
for s in yaml_content["sections"]
|
|
if s["section_id"] == current_section_id
|
|
),
|
|
None,
|
|
)
|
|
|
|
step = next(
|
|
(s for s in section["steps"] if s["step_id"] == current_step_id), None
|
|
)
|
|
|
|
# Get the user's language preference from metadata
|
|
user_language = metadata.get("language", "English")
|
|
|
|
# Translate and print all content blocks once per step
|
|
if "content_blocks" in step:
|
|
content = "\n\n".join(step["content_blocks"])
|
|
translated_content = translate_text(content, user_language)
|
|
print(translated_content)
|
|
|
|
# Skip classification and feedback if there's no question
|
|
if "question" not in step:
|
|
current_section_id, current_step_id = get_next_section_and_step(
|
|
yaml_content, current_section_id, current_step_id
|
|
)
|
|
continue
|
|
|
|
question = step["question"]
|
|
translated_question = translate_text(question, user_language)
|
|
print(f"\nQuestion: {translated_question}")
|
|
|
|
attempts = 0
|
|
while attempts < max_attempts:
|
|
user_response = input("\nYour Response: ")
|
|
|
|
# Execute pre-script if it exists (runs before categorization, with user_response available)
|
|
if "pre_script" in step:
|
|
print(f"DEBUG: Executing pre-script")
|
|
# Add user_response to a temporary copy of metadata for pre_script
|
|
temp_metadata = metadata.copy()
|
|
temp_metadata["user_response"] = user_response
|
|
pre_result = execute_processing_script(
|
|
temp_metadata, step["pre_script"]
|
|
)
|
|
|
|
# Update metadata with pre-script results
|
|
for key, value in pre_result.get("metadata", {}).items():
|
|
metadata[key] = value
|
|
print(f"DEBUG: Pre-script completed, updated metadata")
|
|
|
|
category = categorize_response(
|
|
question, user_response, step["buckets"], step["tokens_for_ai"]
|
|
)
|
|
print(f"\nCategory: {category}")
|
|
|
|
# Determine the transition based on the category (with integer/boolean matching)
|
|
transition = None
|
|
if category in step["transitions"]:
|
|
transition = step["transitions"][category]
|
|
elif category.isdigit() and int(category) in step["transitions"]:
|
|
transition = step["transitions"][int(category)]
|
|
else:
|
|
if category.lower() in ["yes", "true"]:
|
|
category = True
|
|
elif category.lower() in ["no", "false"]:
|
|
category = False
|
|
if category in step["transitions"]:
|
|
transition = step["transitions"][category]
|
|
|
|
if not transition:
|
|
print(
|
|
f"\nError: No valid transition found for category '{category}'. Please try again."
|
|
)
|
|
continue
|
|
|
|
# Check metadata conditions
|
|
if "metadata_conditions" in transition:
|
|
conditions_met = all(
|
|
metadata.get(key) == value
|
|
for key, value in transition["metadata_conditions"].items()
|
|
)
|
|
if not conditions_met:
|
|
print("\nYou do not meet the required conditions to proceed.")
|
|
print(f"Current Metadata: {json.dumps(metadata, indent=2)}")
|
|
continue
|
|
|
|
# Print transition content blocks if they exist
|
|
if "content_blocks" in transition:
|
|
transition_content = "\n\n".join(transition["content_blocks"])
|
|
translated_transition_content = translate_text(
|
|
transition_content, user_language
|
|
)
|
|
print(translated_transition_content)
|
|
|
|
# Track temporary metadata keys
|
|
metadata_tmp_keys = []
|
|
|
|
# Update metadata based on user actions
|
|
if "metadata_add" in transition:
|
|
for key, value in transition["metadata_add"].items():
|
|
if value == "the-users-response":
|
|
value = user_response
|
|
elif isinstance(value, str):
|
|
if value.startswith("n+random(") and value.endswith(")"):
|
|
# Extract the range and apply the random increment
|
|
range_values = value[9:-1].split(",")
|
|
if len(range_values) == 2:
|
|
x, y = map(int, range_values)
|
|
value = metadata.get(key, 0) + random.randint(x, y)
|
|
elif value.startswith("n+") or value.startswith("n-"):
|
|
# Extract the numeric part c and apply the operation +/-
|
|
c = int(value[1:])
|
|
if value.startswith("n+"):
|
|
value = metadata.get(key, 0) + c
|
|
elif value.startswith("n-"):
|
|
value = metadata.get(key, 0) - c
|
|
metadata[key] = value
|
|
|
|
if "metadata_tmp_add" in transition:
|
|
for key, value in transition["metadata_tmp_add"].items():
|
|
if value == "the-users-response":
|
|
value = user_response
|
|
elif isinstance(value, str):
|
|
if value.startswith("n+random(") and value.endswith(")"):
|
|
# Extract the range and apply the random increment
|
|
range_values = value[9:-1].split(",")
|
|
if len(range_values) == 2:
|
|
x, y = map(int, range_values)
|
|
value = random.randint(x, y)
|
|
elif value.startswith("n+") or value.startswith("n-"):
|
|
# Extract the numeric part c and apply the operation +/-
|
|
c = int(value[1:])
|
|
if value.startswith("n+"):
|
|
value = metadata.get(key, 0) + c
|
|
elif value.startswith("n-"):
|
|
value = metadata.get(key, 0) - c
|
|
metadata[key] = value
|
|
metadata_tmp_keys.append(key) # Track temporary keys
|
|
|
|
if "metadata_remove" in transition:
|
|
for key in transition["metadata_remove"]:
|
|
if key in metadata:
|
|
del metadata[key]
|
|
|
|
# Handle metadata_clear - clear all metadata if set to True
|
|
if "metadata_clear" in transition and transition["metadata_clear"] == True:
|
|
metadata.clear()
|
|
|
|
# Handle metadata_random
|
|
if "metadata_random" in transition:
|
|
random_key = random.choice(list(transition["metadata_random"].keys()))
|
|
random_value = transition["metadata_random"][random_key]
|
|
metadata[random_key] = random_value
|
|
|
|
if "metadata_tmp_random" in transition:
|
|
random_key = random.choice(
|
|
list(transition["metadata_tmp_random"].keys())
|
|
)
|
|
random_value = transition["metadata_tmp_random"][random_key]
|
|
metadata[random_key] = random_value
|
|
metadata_tmp_keys.append(random_key) # Track temporary keys
|
|
|
|
# Execute the processing script if it exists
|
|
if "processing_script" in step and transition.get(
|
|
"run_processing_script", False
|
|
):
|
|
# Add user_response to metadata temporarily for processing script
|
|
temp_metadata = metadata.copy()
|
|
temp_metadata["user_response"] = user_response
|
|
|
|
result = execute_processing_script(
|
|
temp_metadata, step["processing_script"]
|
|
)
|
|
|
|
# Copy any changes back to main metadata (except user_response)
|
|
for key, value in temp_metadata.items():
|
|
if key != "user_response":
|
|
metadata[key] = value
|
|
metadata["processing_script_result"] = result
|
|
metadata_tmp_keys.append("processing_script_result")
|
|
|
|
# Update metadata with results from the processing script
|
|
for key, value in result.get("metadata", {}).items():
|
|
metadata[key] = value
|
|
|
|
print(f"\nMetadata: {json.dumps(metadata, indent=2)}")
|
|
|
|
# Provide feedback based on the category
|
|
feedback_messages = []
|
|
|
|
if "feedback_prompts" in step:
|
|
# New multi-prompt system - legacy tokens get combined with each prompt
|
|
multi_feedback_messages = provide_feedback_prompts(
|
|
transition,
|
|
category,
|
|
question,
|
|
step["feedback_prompts"],
|
|
user_response,
|
|
user_language,
|
|
metadata,
|
|
step.get("feedback_tokens_for_ai", "") # Pass legacy tokens to be combined
|
|
)
|
|
feedback_messages.extend(multi_feedback_messages)
|
|
elif step.get("feedback_tokens_for_ai"):
|
|
# Legacy single feedback system - only if no feedback_prompts
|
|
feedback = provide_feedback(
|
|
transition,
|
|
category,
|
|
question,
|
|
user_response,
|
|
user_language,
|
|
step.get("feedback_tokens_for_ai", ""),
|
|
metadata,
|
|
)
|
|
if feedback and feedback.strip():
|
|
feedback_messages.append({
|
|
"name": "Feedback",
|
|
"content": feedback
|
|
})
|
|
|
|
# Display all feedback messages
|
|
for feedback_msg in feedback_messages:
|
|
print(f"\n{feedback_msg['name']}: {feedback_msg['content']}")
|
|
|
|
if category not in [
|
|
"partial_understanding",
|
|
"limited_effort",
|
|
"asking_clarifying_questions",
|
|
"set_language",
|
|
"off_topic",
|
|
]:
|
|
break
|
|
|
|
# Access counts_as_attempt directly from the transition
|
|
counts_as_attempt = transition.get("counts_as_attempt", True)
|
|
if counts_as_attempt:
|
|
attempts += 1
|
|
|
|
if attempts == max_attempts:
|
|
print("\nMaximum attempts reached. Moving to the next step.")
|
|
|
|
# Remove temporary metadata at the end of the step
|
|
for key in metadata_tmp_keys:
|
|
if key in metadata:
|
|
del metadata[key]
|
|
|
|
# Access next_section_and_step directly from the transition
|
|
next_section_and_step = transition.get("next_section_and_step", None)
|
|
if next_section_and_step:
|
|
current_section_id, current_step_id = next_section_and_step.split(":")
|
|
else:
|
|
current_section_id, current_step_id = get_next_section_and_step(
|
|
yaml_content, current_section_id, current_step_id
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
parser = argparse.ArgumentParser(description="Simulate an activity.")
|
|
parser.add_argument(
|
|
"yaml_file_path",
|
|
type=str,
|
|
help="Path to the activity YAML file",
|
|
default="activity0.yaml",
|
|
)
|
|
args = parser.parse_args()
|
|
simulate_activity(args.yaml_file_path)
|