Strengthen local model evaluation checks
This commit is contained in:
@@ -60,6 +60,17 @@ def run_check(check: dict, answer: str) -> bool:
|
||||
if check["type"] == "fenced_code":
|
||||
pattern = rf"```{re.escape(check['language'])}\b[\s\S]+?```"
|
||||
return re.search(pattern, answer, re.IGNORECASE) is not None
|
||||
if check["type"] == "bullet_count":
|
||||
count = sum(
|
||||
re.match(r"^\s*(?:[-*+•]\s+|\d{1,2}[.)]\s+)", line) is not None
|
||||
for line in answer.splitlines()
|
||||
)
|
||||
return count == int(check["value"])
|
||||
if check["type"] == "single_sentence":
|
||||
if "\n" in answer.strip():
|
||||
return False
|
||||
endings = re.findall(r"[.!?؟]+(?=\s|$)", answer.strip())
|
||||
return len(endings) == int(check.get("sentence_count", 1))
|
||||
raise ValueError(f"Unsupported automatic check: {check['type']}")
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user