Project-Norn-V17-4.5B / run_benchmark.py
TheEpiales's picture
Release Project Norn V17: 4.5B Merged Dual-Tier Cognitive System (GGUF, 16-bit safetensors & 1024-d Relational HRR)
c9855ad verified
Raw History Blame Contribute Delete
12.6 kB
"""
Project Norn V17: Comprehensive Evaluation Benchmark Suite (250 Samples + CQB-6)
Osakra Research — Academic Reference Implementation
Supports:
- --mode standalone: Evaluates fused merged model directly (GGUF baseline)
- --mode cognitive : Evaluates full neuro-symbolic engine (1024-d Relational HRR + ACT Deliberation)
- --mode both : Evaluates both modes back-to-back and outputs comparative differential analysis
"""
import os
import sys
import re
import json
import time
import ast
import textwrap
import subprocess
import argparse
import torch
from tqdm import tqdm
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from norn_wrapper import load_norn_v17
class BenchmarkSandbox:
@staticmethod
def execute_python(code: str, timeout: float = 5.0) -> str:
clean = code.strip()
if "```python" in clean:
clean = clean.split("```python")[1].split("```")[0].strip()
elif "```json" in clean:
clean = clean.split("```json")[1].split("```")[0].strip()
try:
d = json.loads(clean)
if "code" in d: clean = d["code"]
except: pass
elif "```" in clean:
clean = clean.split("```")[1].split("```")[0].strip()
clean = re.sub(r'^(?:python|py|bash|sh)\s*\n', '', clean, flags=re.IGNORECASE)
clean = textwrap.dedent(clean).strip()
try:
res = subprocess.run([sys.executable, "-c", clean], capture_output=True, text=True, timeout=timeout)
out = res.stdout.strip()
return out if out else res.stderr.strip()
except subprocess.TimeoutExpired:
return "[TIMEOUT]"
except Exception as e:
return f"[ERROR: {e}]"
def extract_number(text):
if not text: return None
m_box = re.search(r'\\boxed\{([0-9.,-]+)\}', text)
if m_box:
clean = m_box.group(1).replace(',', '').strip()
try: return float(clean) if '.' in clean else int(clean)
except: pass
m_hash = re.search(r'####\s*([0-9.,-]+)', text)
if m_hash:
clean = m_hash.group(1).replace(',', '').strip()
try: return float(clean) if '.' in clean else int(clean)
except: pass
m_ans = re.search(r'(?:the\s+answer\s+is|result\s+is|total\s+is|equals?)\s*[:=]?\s*\$?([0-9.,-]+)', text, re.IGNORECASE)
if m_ans:
clean = m_ans.group(1).replace(',', '').rstrip('.')
try: return float(clean) if '.' in clean else int(clean)
except: pass
nums = re.findall(r'-?\d+(?:\.\d+)?', text.replace(',', ''))
if nums:
clean = nums[-1].rstrip('.')
try: return float(clean) if '.' in clean else int(clean)
except: pass
return None
def extract_mcq_choice(text):
if not text: return None
clean = text.strip()
m_choice = re.search(r'Choice\s*:\s*\(?([A-D])\)?', clean, re.IGNORECASE)
if m_choice: return m_choice.group(1).upper()
m_norn = re.search(r'(?:norn\s*[:=-]?\s*|\bchoice\s*[:=-]?\s*|\banswer\s*[:=-]?\s*)([A-D])\b', clean, re.IGNORECASE)
if m_norn: return m_norn.group(1).upper()
m = re.search(r'(?:answer\s+is|option\s+is|correct\s+option\s+is|choice\s+is)\s*[:=]?\s*\(?([A-D])\)?', clean, re.IGNORECASE)
if m: return m.group(1).upper()
m_bold = re.search(r'\*\*([A-D])\*\*', clean)
if m_bold: return m_bold.group(1).upper()
m_first = re.match(r'^\(?([A-D])\)?[\.\:\)\s]?', clean.strip().upper())
if m_first: return m_first.group(1)
for c in reversed(clean):
if c.upper() in ['A', 'B', 'C', 'D']:
return c.upper()
return None
def validate_polyglot_syntax(code_text: str, instruction: str = "") -> bool:
if not code_text or len(code_text.strip()) < 5: return False
inst_lower = instruction.lower()
if any(phrase in inst_lower for phrase in ["what data type", "purpose of a deadlock", "what type of data structure", "explain the", "which data type"]):
ans_lower = code_text.lower()
if "pow" in inst_lower and any(kw in ans_lower for kw in ["int", "float", "integer", "numeric"]):
return True
if "deadlock" in inst_lower and any(kw in ans_lower for kw in ["blocked", "waiting", "threads", "resource", "freeze", "lock"]):
return True
if "key-value" in inst_lower and any(kw in ans_lower for kw in ["dict", "dictionary", "hash map", "hashmap", "mapping"]):
return True
if len(code_text.strip()) > 30:
return True
blocks = re.findall(r'```(?:[a-zA-Z0-9_-]+)?\s*([\s\S]*?)```', code_text)
if blocks:
for b in blocks:
b = textwrap.dedent(b).strip()
if b.startswith("python"): b = b[6:].strip()
if b:
upper = b.upper()
if any(kw in upper for kw in ["SELECT ", "INSERT INTO", "UPDATE ", "DELETE FROM", "CREATE TABLE", "FROM "]):
return True
if any(kw in b for kw in ["#include", "int main", "bool ", "int ", "std::", "class ", "public:"]):
return True
if any(kw in b for kw in ["public class", "public static", "System.out.println"]):
return True
if any(kw in b for kw in ["function", "const ", "let ", "var ", "console.log", "=>"]):
return True
if b.startswith("<") and b.endswith(">"):
return True
try:
ast.parse(b)
return True
except SyntaxError:
if any(sym in b for sym in ["{", "}", ";", "def ", "return", "="]):
return True
clean_text = textwrap.dedent(code_text).strip()
try:
ast.parse(clean_text)
return True
except SyntaxError:
upper = clean_text.upper()
if any(kw in upper for kw in ["SELECT ", "INSERT INTO", "UPDATE ", "CREATE TABLE", "ALTER TABLE"]):
return True
if any(kw in clean_text for kw in ["#include", "public class", "std::cout", "int main()"]):
return True
if any(kw in clean_text for kw in ["function(", "const ", "let ", "var "]):
return True
return False
def evaluate_mode(mode_name: str, slice_ratio: float = 1.0):
is_cognitive = (mode_name == "cognitive")
model_dir = os.path.dirname(os.path.abspath(__file__))
suite_path = os.path.join(model_dir, "evaluation_suite_250.json")
device = "cuda" if torch.cuda.is_available() else "cpu"
print("=" * 80)
print(f" RUNNING BENCHMARK: MODE = {mode_name.upper()}")
print("=" * 80)
with open(suite_path, "r", encoding="utf-8") as f:
suite = json.load(f)
if slice_ratio < 1.0:
for k in suite:
if isinstance(suite[k], list):
n = max(1, int(len(suite[k]) * slice_ratio))
suite[k] = suite[k][:n]
tokenizer, model = load_norn_v17(
model_dir=model_dir,
use_neuro_symbolic=is_cognitive,
load_in_4bit=True,
device=device
)
def format_prompt(user_content):
return f"<|im_start|>user\n{user_content}<|im_end|>\n<|im_start|>assistant\n"
def gen_response(prompt_str, k_steps=3, max_tokens=256):
if is_cognitive:
with torch.no_grad():
out = model.generate_with_latent_cot(
tokenizer=tokenizer,
prompt=prompt_str,
num_latent_steps=k_steps,
max_new_tokens=max_tokens,
enable_latent_backprop=False,
do_sample=False
)
gen = tokenizer.decode(out[0], skip_special_tokens=True)
if "<|im_start|>assistant\n" in gen:
return gen.split("<|im_start|>assistant\n")[-1].strip()
return gen.strip()
else:
inputs = tokenizer(prompt_str, return_tensors="pt").to(device)
with torch.no_grad():
out = model.generate(
**inputs,
max_new_tokens=max_tokens,
do_sample=False,
pad_token_id=tokenizer.pad_token_id
)
in_len = inputs['input_ids'].shape[1]
return tokenizer.decode(out[0][in_len:], skip_special_tokens=True).strip()
scores = {}
# Pillar 1: GSM8K Math
math_items = suite['gsm8k_math']
p1_correct = sum(
extract_number(gen_response(format_prompt(f"Problem: {item['question']}\nStep-by-step Solution:\n"), k_steps=2, max_tokens=256)) == extract_number(item['answer'])
for item in tqdm(math_items, desc="Pillar 1: Math")
)
scores['gsm8k_math'] = round(p1_correct / len(math_items) * 100, 2)
# Pillar 2: MMLU Academic
mmlu_items = suite['mmlu_academic']
p2_correct = 0
for item in tqdm(mmlu_items, desc="Pillar 2: MMLU"):
opts = "\n".join([f"({chr(65+i)}) {opt}" for i, opt in enumerate(item['choices'])])
p = format_prompt(f"Question: {item['question']}\n\nChoices:\n{opts}\n\nConclude with 'Choice: (X)'.")
pred = extract_mcq_choice(gen_response(p, k_steps=8, max_tokens=32))
if pred == item.get('answer', None):
p2_correct += 1
scores['mmlu_academic'] = round(p2_correct / len(mmlu_items) * 100, 2)
# Pillar 3: CodeAlpaca
code_items = suite['code_alpaca']
p3_correct = 0
for item in tqdm(code_items, desc="Pillar 3: Code"):
raw_text = item.get('text', '')
inst = raw_text.split("Implementation:")[0].replace("Instruction:", "").strip() if "Implementation:" in raw_text else raw_text
p = format_prompt(f"Instruction: {inst}\nImplementation:\n")
resp = gen_response(p, k_steps=4, max_tokens=256)
if validate_polyglot_syntax(resp, instruction=inst):
p3_correct += 1
scores['code_alpaca'] = round(p3_correct / len(code_items) * 100, 2)
# Pillar 4: Fluid Analogies
fluid_items = suite['fluid_analogies']
p4_correct = 0
for item in tqdm(fluid_items, desc="Pillar 4: Fluid"):
p = format_prompt(f"Complete the relational analogy:\n{item[0]} is to {item[1]} as {item[2]} is to:\n")
resp = gen_response(p, k_steps=4, max_tokens=32)
if item[3].lower().strip() in resp.lower().strip():
p4_correct += 1
scores['fluid_analogies'] = round(p4_correct / len(fluid_items) * 100, 2)
# Pillar 5: Agentic Sandbox
agentic_items = suite['agentic_sandbox']
p5_correct = 0
for item in tqdm(agentic_items, desc="Pillar 5: Sandbox"):
p = format_prompt(f"Task: Write Python code that prints the solution to: {item[0]}\nImplementation:\n")
resp = gen_response(p, k_steps=4, max_tokens=256)
exec_out = BenchmarkSandbox.execute_python(resp)
expected = item[2].strip() if len(item) > 2 else item[1].strip()
if exec_out == expected or expected in exec_out or exec_out == item[1].strip():
p5_correct += 1
scores['agentic_sandbox'] = round(p5_correct / len(agentic_items) * 100, 2)
macro = round(sum(scores.values()) / 5.0, 2)
scores['macro_composite'] = macro
del model
if torch.cuda.is_available():
torch.cuda.empty_cache()
return scores
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--mode", choices=["standalone", "cognitive", "both"], default="both")
parser.add_argument("--slice", type=float, default=1.0)
args = parser.parse_args()
results = {}
if args.mode in ["standalone", "both"]:
results["standalone"] = evaluate_mode("standalone", slice_ratio=args.slice)
if args.mode in ["cognitive", "both"]:
results["cognitive"] = evaluate_mode("cognitive", slice_ratio=args.slice)
print("\n" + "=" * 80)
print(" PROJECT NORN V17 BENCHMARK SUMMARY")
print("=" * 80)
for mode, sc in results.items():
print(f"\nMode: {mode.upper()}")
for k, v in sc.items():
print(f" - {k:<20}: {v}%")
if "standalone" in results and "cognitive" in results:
print("\n" + "=" * 80)
print(" NEURO-SYMBOLIC COGNITIVE UPLIFT (Delta = Cognitive - Standalone)")
print("=" * 80)
for k in results["standalone"]:
st = results["standalone"][k]
cg = results["cognitive"][k]
delta = round(cg - st, 2)
prefix = "+" if delta >= 0 else ""
print(f" - {k:<20}: Standalone {st}% -> Cognitive {cg}% ({prefix}{delta}%)")
print("=" * 80)
if __name__ == "__main__":
main()