"""Name the two evaluation tracks (user, 2026-10-01): the Research Track (20 scenes) and the Full Benchmark (160 scenes). The page's Text-to-Scene results are the Research Track's; the case-count strip states the two tracks. Each step asserts it found what it edits. python3 scripts/tracks.py """ from __future__ import annotations import re from pathlib import Path from bs4 import BeautifulSoup ROOT = Path(__file__).resolve().parent.parent V = "20261005tracks" def one(found, what): assert found, f"not found: {what}" return found def main(): p = ROOT / "index.html" soup = BeautifulSoup(p.read_text(), "html.parser") frag = lambda html: BeautifulSoup(html, "html.parser") # the case-count strip: the two tracks av = one(soup.find(class_="availability"), "availability strip") av.clear() av.append(frag('20 scenes in the Research Track: affordable, reproducible evaluation ' 'for academic and model-development use' '160 scenes in the Full Benchmark: comprehensive evaluation ' 'for leaderboard and final reporting')) # leaderboard: the Research Track lb = one(soup.find(id="leaderboard"), "leaderboard") one(lb.find(class_="section-desc"), "leaderboard desc").string = \ "Research Track · 20 representative scenes · 14 coding-agent configurations." lb.find(class_="section-desc").insert_after(frag( '
Text-to-scene generation with frontier models is a long-horizon agentic task with substantial ' 'inference cost and runtime, which makes repeated full-scale evaluation difficult for many research groups. The ' 'Research Track supports affordable, reproducible experimentation; the Full Benchmark provides the more comprehensive ' 'evaluation for final model comparison and leaderboard reporting.
')) h3 = one(lb.select_one(".bar-heading h3"), "bar heading") one(h3.find("span", class_="q"), "bar qualifier").string = "Research Track" side = [s for s in lb.select(".bar-heading > span")] assert len(side) == 1, side side[0].string = "20 scenes · score out of 100" svg = one(lb.find(id="bars-t2s"), "bar svg") svg["aria-label"] = "Research Track scores for the 14 Code4Scene configurations" one(lb.find(class_="figure-subtitle"), "pareto subtitle").string = \ "Quality against the cost of one case · Research Track, 20 scenes" note = one(lb.find(class_="board-note"), "board note") t = note.get_text() assert t.startswith("20 public cases."), t note.string = "Research Track, 20 scenes." + t[len("20 public cases."):] # cache key for the stylesheet (one new rule) for link in soup.find_all("link", href=re.compile(r"^assets/buildingbench\.css")): link["href"] = f"assets/buildingbench.css?v={V}" p.write_text(str(soup)) left = re.findall(r"(?i)20 public|original 20|paper evaluation|129 public|full benchmark target", p.read_text()) assert not left, left css = ROOT / "assets/buildingbench.css"; s = css.read_text() rule = ".tracks-why{max-width:860px;margin:8px 0 0;font-size:13px;line-height:1.65;color:var(--text-muted);text-align:left}" if ".tracks-why{" not in s: css.write_text(s.rstrip("\n") + "\n/* evaluation tracks (scripts/tracks.py) */\n" + rule + "\n@media(max-width:650px){.tracks-why{font-size:11px}}\n") print("index.html: tracks strip, Research Track leaderboard") if __name__ == "__main__": main()