Repository navigation
Expand file tree
/
Copy path_polyglot_one.py
More file actions
214 lines (169 loc) · 6.91 KB
/
Copy path_polyglot_one.py
File metadata and controls
214 lines (169 loc) · 6.91 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
"""Stage 1 of the Aider polyglot benchmark wiring: one exercise, end to end.
Prove the loop works on a single Python exercise (`robot-name`). If this
is green, Stage 2 extends to all 34 Python exercises; Stage 3 adds the
other languages and wires it into ``forge eval polyglot``.
Mode A: invoke Forge's ``_run_opencode`` worker adapter directly with
an Aider-style single-file-edit prompt. No decomposer, no reviewer,
no merge -- just worker-writes-then-tests. Directly comparable to
Aider's per-model polyglot scoreboard.
Run:
.venv/bin/python scripts/_polyglot_one.py
"""
from __future__ import annotations
import json
import shutil
import subprocess
import sys
import tempfile
import time
from pathlib import Path
REPO = Path(__file__).resolve().parents[1]
BENCHMARK = REPO / ".forge" / "eval" / "polyglot-benchmark"
EXERCISE = BENCHMARK / "python" / "exercises" / "practice" / "robot-name"
def _stage_exercise(src: Path, dst: Path) -> dict:
"""Copy the exercise into a fresh worktree and return the file map.
The meta/config.json lists which files are solution (writable) vs
test (read-only gold). We copy everything EXCEPT ``.meta/example.*``
-- the reference solution must not leak to the worker.
"""
shutil.copytree(src, dst, dirs_exist_ok=True)
# Scrub the reference solution so the worker can't stumble on it.
for p in (dst / ".meta").glob("example.*"):
p.unlink()
meta = json.loads((src / ".meta" / "config.json").read_text())
return {
"solution_files": meta["files"]["solution"],
"test_files": meta["files"]["test"],
}
def _build_prompt(exercise_dir: Path, file_map: dict) -> str:
instructions = (exercise_dir / ".docs" / "instructions.md").read_text()
append_path = exercise_dir / ".docs" / "instructions.append.md"
if append_path.exists():
instructions += "\n\n" + append_path.read_text()
solution_files_block = "\n".join(
f" - `{p}`" for p in file_map["solution_files"]
)
test_files_block = "\n".join(
f" - `{p}`" for p in file_map["test_files"]
)
stub_dump = []
for sol in file_map["solution_files"]:
content = (exercise_dir / sol).read_text()
stub_dump.append(f"### `{sol}` (current contents)\n```python\n{content}\n```")
stub_block = "\n\n".join(stub_dump)
return f"""# Programming exercise
You are solving a self-contained programming exercise. Your job is to
edit the solution file(s) so that the test file(s) pass.
## Problem
{instructions}
## Files
Solution files (edit these):
{solution_files_block}
Test files (READ ONLY -- do not modify):
{test_files_block}
## Current stub
{stub_block}
## What to do
1. Read the problem above carefully.
2. Edit the solution file(s) so that `python -m pytest` passes all tests
in the test file(s).
3. Do NOT modify the test file(s). The tests define correctness; your
code must satisfy them as-written.
4. When you are done, stop. The tests will be run automatically after
your edits.
Keep the solution idiomatic and focused on the problem. Do not add
extra files, do not add print statements, do not add pip dependencies.
"""
def _run_tests(cwd: Path, test_files: list[str]) -> tuple[bool, str]:
"""Run pytest against the given test files. Return (passed, output)."""
cmd = [sys.executable, "-m", "pytest", "-x", "--tb=short", "-q"] + test_files
try:
r = subprocess.run(
cmd, cwd=cwd, capture_output=True, text=True, timeout=120,
)
except subprocess.TimeoutExpired:
return False, "TIMEOUT after 120s"
output = (r.stdout or "") + "\n" + (r.stderr or "")
return r.returncode == 0, output.strip()
def _git_init_and_commit(cwd: Path) -> None:
for cmd in (
["git", "init", "-q"],
["git", "config", "user.email", "polyglot@forge.eval"],
["git", "config", "user.name", "Polyglot Eval"],
["git", "add", "-A"],
["git", "commit", "-q", "-m", "initial: exercise stub"],
):
subprocess.run(cmd, cwd=cwd, check=True, capture_output=True)
def main() -> int:
if not EXERCISE.exists():
print(f"FATAL: exercise not found at {EXERCISE}", file=sys.stderr)
return 2
t0 = time.time()
with tempfile.TemporaryDirectory(prefix="polyglot-robot-name-") as tmp:
work = Path(tmp) / "exercise"
file_map = _stage_exercise(EXERCISE, work)
_git_init_and_commit(work)
prompt = _build_prompt(work, file_map)
print(f"[stage1] worktree: {work}")
print(f"[stage1] prompt len: {len(prompt)} chars")
# Lazy import to avoid pulling in the whole studio/ package if
# this script runs outside the venv.
sys.path.insert(0, str(REPO / "platform"))
sys.path.insert(0, str(REPO))
from studio.workflows.tdd import _run_opencode
attempt = 1
max_attempts = 2
prev_output = None
final_passed = False
final_output = ""
while attempt <= max_attempts:
turn_prompt = prompt
if prev_output:
turn_prompt += (
f"\n\n## Previous attempt failed\n\n"
f"The tests failed on the previous attempt with:\n\n"
f"```\n{prev_output[-3000:]}\n```\n\n"
f"Please review your solution and fix it."
)
print(f"[stage1] attempt {attempt}/{max_attempts} -- invoking worker...")
t_attempt = time.time()
result = _run_opencode(
prompt=turn_prompt,
cwd=str(work),
role="worker",
timeout=300,
)
print(
f"[stage1] attempt {attempt} done in "
f"{time.time() - t_attempt:.1f}s, exit={result.get('exit_code')}"
)
passed, output = _run_tests(work, file_map["test_files"])
final_passed = passed
final_output = output
if passed:
print(f"[stage1] PASSED on attempt {attempt}")
break
print(f"[stage1] tests failed on attempt {attempt}; will retry with feedback")
prev_output = output
attempt += 1
elapsed = time.time() - t0
print()
print("=" * 70)
print(f"Exercise: robot-name (python)")
print(f"Result: {'PASSED' if final_passed else 'FAILED'}")
print(f"Attempts: {attempt - (0 if final_passed else 1)}")
print(f"Wall time: {elapsed:.1f}s")
print("=" * 70)
if not final_passed:
print("\nFinal test output (tail):")
print(final_output[-2000:])
# Dump the final solution for inspection
print("\nFinal solution:")
for sol in file_map["solution_files"]:
p = work / sol
if p.exists():
print(f"--- {sol} ---")
print(p.read_text())
return 0 if final_passed else 1
if __name__ == "__main__":
sys.exit(main())