Code4Scene / scripts /tracks.py
KoeYe's picture
Claude Opus 5.5
Name the evaluation tracks: Research Track (20 scenes) and Full Benchmark (160 scenes)
7399765 verified
Raw History Blame Contribute Delete
3.62 kB
"""Name the two evaluation tracks (user, 2026-10-01): the Research Track (20 scenes) and the Full Benchmark (160 scenes).
The page's Text-to-Scene results are the Research Track's; the case-count strip states the two tracks.
Each step asserts it found what it edits.
python3 scripts/tracks.py
"""
from __future__ import annotations
import re
from pathlib import Path
from bs4 import BeautifulSoup
ROOT = Path(__file__).resolve().parent.parent
V = "20261005tracks"
def one(found, what):
assert found, f"not found: {what}"
return found
def main():
p = ROOT / "index.html"
soup = BeautifulSoup(p.read_text(), "html.parser")
frag = lambda html: BeautifulSoup(html, "html.parser")
# the case-count strip: the two tracks
av = one(soup.find(class_="availability"), "availability strip")
av.clear()
av.append(frag('<span><b>20</b> scenes in the <strong>Research Track</strong>: affordable, reproducible evaluation '
'for academic and model-development use</span>'
'<span><b>160</b> scenes in the <strong>Full Benchmark</strong>: comprehensive evaluation '
'for leaderboard and final reporting</span>'))
# leaderboard: the Research Track
lb = one(soup.find(id="leaderboard"), "leaderboard")
one(lb.find(class_="section-desc"), "leaderboard desc").string = \
"Research Track · 20 representative scenes · 14 coding-agent configurations."
lb.find(class_="section-desc").insert_after(frag(
'<p class="tracks-why">Text-to-scene generation with frontier models is a long-horizon agentic task with substantial '
'inference cost and runtime, which makes repeated full-scale evaluation difficult for many research groups. The '
'Research Track supports affordable, reproducible experimentation; the Full Benchmark provides the more comprehensive '
'evaluation for final model comparison and leaderboard reporting.</p>'))
h3 = one(lb.select_one(".bar-heading h3"), "bar heading")
one(h3.find("span", class_="q"), "bar qualifier").string = "Research Track"
side = [s for s in lb.select(".bar-heading > span")]
assert len(side) == 1, side
side[0].string = "20 scenes · score out of 100"
svg = one(lb.find(id="bars-t2s"), "bar svg")
svg["aria-label"] = "Research Track scores for the 14 Code4Scene configurations"
one(lb.find(class_="figure-subtitle"), "pareto subtitle").string = \
"Quality against the cost of one case · Research Track, 20 scenes"
note = one(lb.find(class_="board-note"), "board note")
t = note.get_text()
assert t.startswith("20 public cases."), t
note.string = "Research Track, 20 scenes." + t[len("20 public cases."):]
# cache key for the stylesheet (one new rule)
for link in soup.find_all("link", href=re.compile(r"^assets/buildingbench\.css")):
link["href"] = f"assets/buildingbench.css?v={V}"
p.write_text(str(soup))
left = re.findall(r"(?i)20 public|original 20|paper evaluation|129 public|full benchmark target", p.read_text())
assert not left, left
css = ROOT / "assets/buildingbench.css"; s = css.read_text()
rule = ".tracks-why{max-width:860px;margin:8px 0 0;font-size:13px;line-height:1.65;color:var(--text-muted);text-align:left}"
if ".tracks-why{" not in s:
css.write_text(s.rstrip("\n") + "\n/* evaluation tracks (scripts/tracks.py) */\n" + rule +
"\n@media(max-width:650px){.tracks-why{font-size:11px}}\n")
print("index.html: tracks strip, Research Track leaderboard")
if __name__ == "__main__":
main()