"""Execution backends for benchmark runners — the SHARED model gateway all systems run through. Fairness (standard §9.1): swarm and every baseline must use the same backend on the same task set; only the topology (call pattern) differs. Two backends: - OfflineBackend: deterministic, no network, no key. Echoes the fixture's offline reference solution and charges a notional per-call cost. Used for hermetic CI + pipeline validation. It is IDENTICAL across systems, so offline runs can prove the pipeline (records → G_E → report, no NaN) but CANNOT show a swarm quality advantage — by design (anti-fabrication). - OpenAIBackend: real OpenAI-compatible generation (needs OPENAI_API_KEY / OPENAI_BASE_URL). Produces real files + real token/cost. This is what yields REAL G_E numbers. `generate()` returns the produced implementation files plus measured token/cost/time. """ from __future__ import annotations import json import os import re from abc import ABC, abstractmethod from dataclasses import dataclass, field from typing import List, Optional, Tuple # (path, content) pairs. Files = List[Tuple[str, str]] @dataclass class GenerationResult: files: Files = field(default_factory=list) tokens: int = 0 cost_usd: float = 0.0 elapsed_s: float = 0.0 class ExecutionBackend(ABC): name: str = "abstract" @abstractmethod def generate(self, *, objective: str, reference_files: Files, max_tokens: int) -> GenerationResult: """Produce implementation files for one objective.""" class OfflineBackend(ExecutionBackend): """Deterministic, key-free. Echoes the fixture reference solution; charges a notional cost. cost_per_call / tokens_per_call / time_per_call let the harness model a 'stronger' (costlier) configuration without a real model — see runners/strong.py. """ name = "offline" def __init__(self, *, cost_per_call: float = 0.002, tokens_per_call: int = 800, time_per_call: float = 0.5): self.cost_per_call = cost_per_call self.tokens_per_call = tokens_per_call self.time_per_call = time_per_call def generate(self, *, objective: str, reference_files: Files, max_tokens: int) -> GenerationResult: # Deterministic: same reference output regardless of system/topology (no swarm bias). return GenerationResult( files=list(reference_files), tokens=self.tokens_per_call, cost_usd=self.cost_per_call, elapsed_s=self.time_per_call, ) class OpenAIBackend(ExecutionBackend): """Real OpenAI-compatible generation. Lazy client; raises clearly if no key is configured.""" name = "openai" def __init__(self, *, model: Optional[str] = None, price_per_1k_tokens: float = 0.0): self.model = model or os.getenv("OPENAI_MODEL", "gpt-4o-mini") self.price_per_1k_tokens = price_per_1k_tokens self._client = None def _get_client(self): if self._client is None: from openai import OpenAI # lazy: keep offline/CI import-light if not os.getenv("OPENAI_API_KEY"): raise RuntimeError("OpenAIBackend requires OPENAI_API_KEY (use OfflineBackend for hermetic runs)") self._client = OpenAI(base_url=os.getenv("OPENAI_BASE_URL") or None) return self._client @staticmethod def _parse_files(content: str) -> Files: # Accept a JSON {"files":[{"path","content"}]} (same shape the agent uses), else empty. try: text = re.sub(r"^```(json)?|```$", "", content.strip(), flags=re.MULTILINE).strip() data = json.loads(text) return [(f["path"], f["content"]) for f in data.get("files", []) if f.get("path")] except Exception: return [] def generate(self, *, objective: str, reference_files: Files, max_tokens: int) -> GenerationResult: import time as _time client = self._get_client() prompt = ( f"{objective}\n\nReturn ONLY JSON: " '{"files":[{"path":"relative/path.py","content":"complete file content"}]}' ) start = _time.time() resp = client.chat.completions.create( model=self.model, messages=[{"role": "user", "content": prompt}], max_tokens=max_tokens, ) elapsed = _time.time() - start content = resp.choices[0].message.content or "" usage = getattr(resp, "usage", None) tokens = int(getattr(usage, "total_tokens", 0) or 0) cost = tokens / 1000.0 * self.price_per_1k_tokens return GenerationResult(files=self._parse_files(content), tokens=tokens, cost_usd=cost, elapsed_s=elapsed)