पाठ 3 / 29
Agent Loop कार्य में
छोटा coding agent चलाएँ जो bug ठीक करता है और हर चरण देखें।
देखें, तय करें, कार्य करें, जाँचें
Loop सरल है। (1) देखें: tests चलाएँ और देखें क्या विफल होता है। (2) तय करें: प्रासंगिक कोड पढ़ें और परिकल्पना बनाएँ। (3) कार्य करें: छोटा संपादन करें। (4) जाँचें: tests फिर चलाएँ। (5) जाँचें पास हों या सीमा आ जाए तब रुकें। नीचे का उदाहरण छोटा, असली agent loop है: throwaway project के apply_discount में bug है (वह प्रतिशत को 100 की जगह 10 से भाग देता है); harness run_tests, read_file और edit_file tools देता है; और एक scripted stand-in मॉडल की भूमिका निभाता है, उन पाँच कार्रवाइयों का प्रस्ताव करते हुए जो असली मॉडल संभवतः चुनता। यहाँ कोई असली मॉडल नहीं बुलाया गया, इसलिए यह प्रक्रिया (tools, अवलोकन, जाँचें, कठोर चरण सीमा) दिखाता है, बुद्धिमत्ता नहीं। असली agent अपनी कार्रवाइयाँ ख़ुद चुनता है, इसीलिए सत्यापन चरण मायने रखता है।
छोटा coding agent, चलाकर
मैंने यह सादे Python 3 (सिर्फ़ standard library) से चलाया, अस्थायी folder में बनाए throwaway project के साथ। चरण 1 tests चलाकर test_discount_ten_percent को विफल पाता है। चरण 2 और 3 file पढ़ते हैं और / 10 को / 100 में बदलने वाला सटीक-मिलान संपादन लगाते हैं। चरण 4 tests दोबारा चलाता है, जो अब पास होते हैं, और चरण 5 सार के साथ समाप्त होता है। अंतिम स्वतंत्र test run सुधार की पुष्टि करता है। यहाँ मॉडल निश्चित script है, इसलिए यह harness दिखाता है, असली मॉडल नहीं।
import os, tempfile, textwrap
def make_project(root):
files = {
"shop/__init__.py": "",
"shop/pricing.py": textwrap.dedent("""
def apply_discount(price, percent):
\"\"\"Return price after a percentage discount.\"\"\"
return price - price * percent / 10
def add_tax(price, rate=0.18):
return round(price * (1 + rate), 2)
"""),
"shop/cart.py": textwrap.dedent("""
from shop.pricing import apply_discount, add_tax
class Cart:
def __init__(self):
self.items = []
def add(self, name, price):
self.items.append((name, price))
def total(self, discount_percent=0):
subtotal = sum(p for _, p in self.items)
return add_tax(apply_discount(subtotal, discount_percent))
"""),
"tests/__init__.py": "",
"tests/test_pricing.py": textwrap.dedent("""
import unittest
from shop.pricing import apply_discount, add_tax
class PricingTests(unittest.TestCase):
def test_discount_ten_percent(self):
self.assertEqual(apply_discount(200, 10), 180)
def test_discount_zero(self):
self.assertEqual(apply_discount(200, 0), 200)
def test_tax(self):
self.assertEqual(add_tax(100), 118.0)
"""),
}
for rel, text in files.items():
path = os.path.join(root, rel)
os.makedirs(os.path.dirname(path), exist_ok=True)
with open(path, "w") as f:
f.write(text.lstrip("\n"))
import subprocess, sys, re
def run_tests(root):
p = subprocess.run([sys.executable, "-m", "unittest", "discover", "-s", "tests", "-t", "."],
cwd=root, capture_output=True, text=True)
failed = re.findall(r"^(?:FAIL|ERROR): (\S+)", p.stderr, re.M)
return {"passed": p.returncode == 0, "failed": failed}
def read_file(root, rel): return open(os.path.join(root, rel)).read()
def edit_file(root, rel, old, new):
path = os.path.join(root, rel); text = open(path).read()
if text.count(old) != 1: return "error: old text must appear exactly once"
open(path, "w").write(text.replace(old, new)); return "ok"
# A scripted stand-in for the model: it proposes the next action given what it has observed so far.
SCRIPT = [
("run_tests", {}),
("read_file", {"rel": "shop/pricing.py"}),
("edit_file", {"rel": "shop/pricing.py", "old": "price * percent / 10", "new": "price * percent / 100"}),
("run_tests", {}),
("finish", {"message": "Fixed apply_discount: percent was divided by 10 instead of 100."}),
]
def agent(root, max_steps=8):
script = iter(SCRIPT)
for step in range(1, max_steps + 1): # a hard cap on steps
name, args = next(script)
if name == "run_tests": obs = run_tests(root)
elif name == "read_file": obs = "(%d chars)" % len(read_file(root, **args))
elif name == "edit_file": obs = edit_file(root, **args)
else:
print(f"step {step}: finish -> {args['message']}"); return
print(f"step {step}: {name}({', '.join(f'{k}=...' for k in args)}) -> {obs}")
with tempfile.TemporaryDirectory() as root:
make_project(root)
agent(root)
print("final test run:", run_tests(root))
Output:
step 1: run_tests() -> {'passed': False, 'failed': ['test_discount_ten_percent']}
step 2: read_file(rel=...) -> (200 chars)
step 3: edit_file(rel=..., old=..., new=...) -> ok
step 4: run_tests() -> {'passed': True, 'failed': []}
step 5: finish -> Fixed apply_discount: percent was divided by 10 instead of 100.
final test run: {'passed': True, 'failed': []}हर run हरे से शुरू करें
Agent शुरू करने से पहले सुनिश्चित करें कि tests पास हों; वरना आप नहीं जान सकते कि कौन-सी विफलताएँ उसने पैदा कीं।
त्वरित जाँच: Loop अंत में स्वतंत्र test run के साथ क्यों समाप्त होता है?
- क्योंकि मॉडल files नहीं पढ़ सकता
- Run को लंबा करने के लिए
- Tests वैकल्पिक सजावट हैं
- Agent का अपना सफलता-दावा प्रमाण नहीं है; tests हैं
Answer
Agent का अपना सफलता-दावा प्रमाण नहीं है; tests हैं — सत्यापन agent की स्व-रिपोर्ट पर निर्भर नहीं होना चाहिए।