Repository navigation
Expand file tree
/
Copy pathevaluate_enriched_prompts.py
More file actions
119 lines (96 loc) · 3.6 KB
/
Copy pathevaluate_enriched_prompts.py
File metadata and controls
119 lines (96 loc) · 3.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
import subprocess
import sys
import os
import pandas as pd
import re
# --- CONFIGURATION ---
INPUT_FILE = "test_mateus_enriched.xlsx"
OUTPUT_FILE = "test_mateus_results.xlsx"
MODEL_NAME = "gpt-oss:20b"
NUM_RUNS = 1
OUTPUT_DIR = "output"
# Ensure output directory exists
if not os.path.exists(OUTPUT_DIR):
os.makedirs(OUTPUT_DIR)
# --- HELPER: Sanitize Filename ---
def sanitize_filename(name):
return re.sub(r'[\\/*?:"<>|]', '_', name)
# --- HELPER: Execute Generation ---
def execute_generation(prompt_text, index, model_name):
"""
Writes the prompt to a file, runs generate.py, and reads the result.
"""
safe_model = sanitize_filename(model_name)
prompt_filename = os.path.join(OUTPUT_DIR, f"eval_prompt_{index}_{safe_model}.txt")
temp_csv_filename = os.path.join(OUTPUT_DIR, f"eval_temp_{index}_{safe_model}.csv")
# 1. Write prompt file
with open(prompt_filename, "w", encoding="utf-8") as f:
f.write(str(prompt_text))
# 2. Build command for generate.py
cmd = [
sys.executable, "generate.py",
prompt_filename,
"-n", str(NUM_RUNS),
"-m", model_name,
"-o", temp_csv_filename,
"--no-progress"
]
response_text = None
wall_time = None
try:
# Run generation
subprocess.run(cmd, check=True)
# 3. Read result
if os.path.exists(temp_csv_filename):
df_batch = pd.read_csv(temp_csv_filename)
if not df_batch.empty:
response_text = df_batch.iloc[0].get('response', None)
wall_time = df_batch.iloc[0].get('wall_time_sec', None)
# Cleanup CSV
os.remove(temp_csv_filename)
except subprocess.CalledProcessError:
print(f" ❌ Failed generation for row {index}")
except Exception as e:
print(f" ❌ Error processing row {index}: {e}")
finally:
# Cleanup Prompt File
if os.path.exists(prompt_filename):
os.remove(prompt_filename)
return response_text, wall_time
# --- MAIN LOGIC ---
def main():
print(f"📂 Loading {INPUT_FILE}...")
try:
df = pd.read_excel(INPUT_FILE)
except FileNotFoundError:
print(f"❌ Error: File {INPUT_FILE} not found.")
sys.exit(1)
if 'enriched_prompt' not in df.columns:
print("❌ Error: Column 'enriched_prompt' missing from input file.")
sys.exit(1)
print(f"🚀 Starting Evaluation | Model: {MODEL_NAME} | Rows: {len(df)}")
# Initialize columns if they don't exist
if 'response' not in df.columns:
df['response'] = None
if 'wall_time_sec' not in df.columns:
df['wall_time_sec'] = None
if 'model_used' not in df.columns:
df['model_used'] = MODEL_NAME
# Iterate rows
for index, row in df.iterrows():
prompt = row['enriched_prompt']
# Skip if empty prompt or already processed (optional check)
if pd.isna(prompt) or str(prompt).strip() == "":
print(f" ⚠️ Row {index}: Empty prompt. Skipping.")
continue
print(f" ▶️ Processing Row {index}...")
response, time_sec = execute_generation(prompt, index, MODEL_NAME)
# Update DataFrame immediately
df.at[index, 'response'] = response
df.at[index, 'wall_time_sec'] = time_sec
# Incremental save every row (safety)
df.to_excel(OUTPUT_FILE, index=False)
print("\n" + "="*60)
print(f"🎉 Evaluation complete. Results saved to {OUTPUT_FILE}")
if __name__ == "__main__":
main()