89 lines
3.4 KiB
Python
89 lines
3.4 KiB
Python
import json
|
|
import sys
|
|
import tempfile
|
|
import unittest
|
|
from pathlib import Path
|
|
|
|
|
|
MODULE_DIR = Path(__file__).resolve().parents[1]
|
|
sys.path.insert(0, str(MODULE_DIR))
|
|
|
|
import label_nucleic_prompts as base
|
|
import label_swe_chat_prompts
|
|
from purpose_data import prompt_hash
|
|
|
|
|
|
class LabelSWEChatPromptsTests(unittest.TestCase):
|
|
def candidate(self):
|
|
value = {
|
|
"schemaVersion": 1,
|
|
"repoID": "repo",
|
|
"userID": "user",
|
|
"sessionID": "session",
|
|
"sourceTurnIDs": ["one", "two"],
|
|
"prompt": "What is making this test fail?",
|
|
"teacherResponse": "It fails only on CI.",
|
|
}
|
|
value["promptHash"] = prompt_hash(value["prompt"])
|
|
value["teacherResponseHash"] = prompt_hash(value["teacherResponse"])
|
|
line = base.SourceLine(1, json.dumps(value), "line-hash")
|
|
return label_swe_chat_prompts.candidates([line])[0]
|
|
|
|
def decision(self, candidate, recoverable):
|
|
return {
|
|
"items": [{
|
|
"id": candidate.id, "keep": True, "junkReason": None,
|
|
"purpose": "debugging", "secondary": None, "mixed": False,
|
|
"difficulty": 0.6, "slice": "boundary", "lang": "en",
|
|
"recoverableFromFirst": recoverable,
|
|
}]
|
|
}
|
|
|
|
def test_response_hash_is_checked_and_context_dependent_labels_become_vague_eval(self):
|
|
candidate = self.candidate()
|
|
decisions = label_swe_chat_prompts.validate_decisions(
|
|
[candidate], self.decision(candidate, False)
|
|
)
|
|
self.assertEqual("vague-eval", decisions[0][1]["slice"])
|
|
self.assertNotIn("It fails only on CI.", candidate.prompt)
|
|
|
|
def test_candidate_rejects_context_hash_mismatch(self):
|
|
candidate = self.candidate()
|
|
value = json.loads(candidate.line.raw)
|
|
value["teacherResponseHash"] = "bad"
|
|
with self.assertRaisesRegex(ValueError, "teacherResponseHash"):
|
|
label_swe_chat_prompts.candidates(
|
|
[base.SourceLine(1, json.dumps(value), "line-hash")]
|
|
)
|
|
|
|
def test_state_never_carries_later_text(self):
|
|
candidate = self.candidate()
|
|
state = label_swe_chat_prompts._state(
|
|
candidate, status="labeled", record=None, reason=None, recoverable=True
|
|
)
|
|
encoded = json.dumps(state)
|
|
self.assertNotIn("It fails only on CI.", encoded)
|
|
|
|
def test_response_sanitizer_preserves_all_prose_and_removes_code_payloads(self):
|
|
response = (
|
|
"I found the likely cause.\n\n"
|
|
"```swift\nlet secret = \"not teacher context\"\n```\n\n"
|
|
"The fix is to await the task.\n"
|
|
"<tool_use>{\"cmd\": \"rm -rf /\"}</tool_use>\n"
|
|
"Then rerun the focused test."
|
|
)
|
|
sanitized = label_swe_chat_prompts.response_for_labeling(response, 2_000)
|
|
self.assertIn("I found the likely cause.", sanitized)
|
|
self.assertIn("The fix is to await the task.", sanitized)
|
|
self.assertIn("Then rerun the focused test.", sanitized)
|
|
self.assertNotIn("let secret", sanitized)
|
|
self.assertNotIn("rm -rf", sanitized)
|
|
|
|
def test_response_sanitizer_refuses_to_truncate_prose(self):
|
|
with self.assertRaisesRegex(ValueError, "rather than truncating"):
|
|
label_swe_chat_prompts.response_for_labeling("x" * 1_001, 1_000)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|