Spaces:
Running
Running
Deploy geoguesser environment with train/eval splits
Browse files
geoguesser_env/pyproject.toml
CHANGED
|
@@ -8,7 +8,7 @@ name = "openenv-geoguesser-env"
|
|
| 8 |
# satisfied when an installed distribution matches the version, ignoring the git
|
| 9 |
# revision the spec asked for, so a same-version rebuild is invisible to a
|
| 10 |
# cached environment: a fixed client sat unused through three job runs.
|
| 11 |
-
version = "0.1.
|
| 12 |
description = "GeoGuessr-style visual geolocation environment for OpenEnv"
|
| 13 |
readme = "README.md"
|
| 14 |
requires-python = ">=3.10"
|
|
@@ -48,3 +48,18 @@ packages = [
|
|
| 48 |
"geoguesser_env.server.render",
|
| 49 |
]
|
| 50 |
package-dir = { "geoguesser_env" = ".", "geoguesser_env.server" = "server", "geoguesser_env.server.backends" = "server/backends", "geoguesser_env.server.render" = "server/render" }
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 8 |
# satisfied when an installed distribution matches the version, ignoring the git
|
| 9 |
# revision the spec asked for, so a same-version rebuild is invisible to a
|
| 10 |
# cached environment: a fixed client sat unused through three job runs.
|
| 11 |
+
version = "0.1.4"
|
| 12 |
description = "GeoGuessr-style visual geolocation environment for OpenEnv"
|
| 13 |
readme = "README.md"
|
| 14 |
requires-python = ">=3.10"
|
|
|
|
| 48 |
"geoguesser_env.server.render",
|
| 49 |
]
|
| 50 |
package-dir = { "geoguesser_env" = ".", "geoguesser_env.server" = "server", "geoguesser_env.server.backends" = "server/backends", "geoguesser_env.server.render" = "server/render" }
|
| 51 |
+
|
| 52 |
+
# The map renderer and the reverse geocoder read these at runtime, so an install
|
| 53 |
+
# without them is code that cannot pin, guess or name a country -- which is how
|
| 54 |
+
# it failed: `look` worked and `pin` raised FileNotFoundError on the countries
|
| 55 |
+
# outline. 40 MB of coastlines, places, roads, rivers and urban areas.
|
| 56 |
+
#
|
| 57 |
+
# `data/geo/osm_cache` is deliberately not listed. It is 331 MB of fetched
|
| 58 |
+
# Overpass responses that the environment regenerates on demand, and shipping a
|
| 59 |
+
# cache in a wheel would be eight times the size of the data it caches.
|
| 60 |
+
[tool.setuptools.package-data]
|
| 61 |
+
geoguesser_env = [
|
| 62 |
+
"data/geo/*.geojson",
|
| 63 |
+
"data/geo/detail/*.json",
|
| 64 |
+
"openenv.yaml",
|
| 65 |
+
]
|
geoguesser_env/scripts/render_rollout.py
CHANGED
|
@@ -64,6 +64,8 @@ DURATIONS = {
|
|
| 64 |
"measure": 2.2,
|
| 65 |
"guess": 6.0,
|
| 66 |
"forced_guess": 5.0,
|
|
|
|
|
|
|
| 67 |
}
|
| 68 |
DEFAULT_DURATION = 2.4
|
| 69 |
|
|
@@ -114,6 +116,42 @@ def interpolate(before: dict, after: dict, steps: int) -> list[dict]:
|
|
| 114 |
return out
|
| 115 |
|
| 116 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 117 |
def camera_changed(turn: dict) -> bool:
|
| 118 |
"""Whether the action actually moved the camera, in heading or field of view."""
|
| 119 |
before, after = turn.get("camera_before") or {}, turn.get("camera_after") or {}
|
|
@@ -161,9 +199,24 @@ def render(args: argparse.Namespace) -> None:
|
|
| 161 |
if line.strip()
|
| 162 |
]
|
| 163 |
rows = [r for r in rows if not r.get("error")]
|
| 164 |
-
if
|
| 165 |
-
|
| 166 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 167 |
|
| 168 |
split = record["task"]["split"]
|
| 169 |
index_name = INDEX_FOR_SPLIT.get(split)
|
|
@@ -200,7 +253,20 @@ def render(args: argparse.Namespace) -> None:
|
|
| 200 |
)
|
| 201 |
|
| 202 |
size = (args.width_pano, args.width_pano)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 203 |
counter = 0
|
|
|
|
|
|
|
|
|
|
|
|
|
| 204 |
segments = []
|
| 205 |
# The last panorama the agent saw, kept so a map segment can sit over the
|
| 206 |
# street view it was placed from -- the way the human UI's map inset does --
|
|
@@ -216,15 +282,75 @@ def render(args: argparse.Namespace) -> None:
|
|
| 216 |
image.convert("RGB").save(frames_dir / name, quality=args.quality)
|
| 217 |
return f"frames/{name}"
|
| 218 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 219 |
for turn in record["turns"]:
|
| 220 |
action = turn.get("action") or {}
|
| 221 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 222 |
seconds = DURATIONS.get(kind, DEFAULT_DURATION)
|
| 223 |
if kind in {"look", "pan", "zoom"} and not camera_changed(turn):
|
| 224 |
seconds = min(seconds, STATIC_DURATION)
|
| 225 |
total = max(1, int(round(seconds * args.fps)))
|
| 226 |
paths: list[str] = []
|
|
|
|
| 227 |
cameras: list[list[float]] = []
|
|
|
|
| 228 |
|
| 229 |
if turn.get("image_kind") == "map":
|
| 230 |
# Maps are single stills: re-render at video resolution from the pins
|
|
@@ -258,19 +384,62 @@ def render(args: argparse.Namespace) -> None:
|
|
| 258 |
focus = pins[-1] if pins else None
|
| 259 |
else:
|
| 260 |
focus = pins[-1] if pins else None
|
| 261 |
-
|
| 262 |
-
|
| 263 |
-
|
| 264 |
-
|
| 265 |
-
|
| 266 |
-
|
| 267 |
-
|
| 268 |
-
|
| 269 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 270 |
else:
|
| 271 |
before, after = turn["camera_before"], turn["camera_after"]
|
| 272 |
moving = kind in {"look", "pan", "zoom"} and camera_changed(turn)
|
| 273 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 274 |
for state in states:
|
| 275 |
if state.get("frame_index") is None:
|
| 276 |
continue
|
|
@@ -284,6 +453,20 @@ def render(args: argparse.Namespace) -> None:
|
|
| 284 |
if image.size != size:
|
| 285 |
image = image.resize(size)
|
| 286 |
paths.append(write(image))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 287 |
# The camera the frame was rendered with, so the HUD can read
|
| 288 |
# live through a pan instead of jumping to the end state the
|
| 289 |
# moment the shot begins.
|
|
@@ -319,6 +502,13 @@ def render(args: argparse.Namespace) -> None:
|
|
| 319 |
# Only set on a map segment: the street view it was placed from.
|
| 320 |
"backdrop": backdrop if turn.get("image_kind") == "map" else None,
|
| 321 |
"frames": paths,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 322 |
"cameras": cameras,
|
| 323 |
# A single still is held for the whole segment; a pan has one
|
| 324 |
# frame per video frame.
|
|
@@ -353,6 +543,8 @@ def render(args: argparse.Namespace) -> None:
|
|
| 353 |
# letterboxing it: React sizes the card to the image instead of fitting
|
| 354 |
# the image into a card the wrong shape.
|
| 355 |
"mapAspect": map_aspect,
|
|
|
|
|
|
|
| 356 |
"episode": {
|
| 357 |
"episode_id": record["episode_id"],
|
| 358 |
"model_name": record["model_name"],
|
|
@@ -387,8 +579,20 @@ def render(args: argparse.Namespace) -> None:
|
|
| 387 |
|
| 388 |
#: Uppercase verb for the on-screen action badge, in the human UI's own words
|
| 389 |
#: (the control pad says LOOK / WALK / ZOOM, not `look` / `move` / `zoom`).
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 390 |
VERBS = {
|
| 391 |
"reset": "START",
|
|
|
|
| 392 |
"look": "LOOK",
|
| 393 |
"pan": "LOOK",
|
| 394 |
"zoom": "ZOOM",
|
|
@@ -450,6 +654,8 @@ def action_caption(kind: str, action: dict, turn: dict) -> str:
|
|
| 450 |
separately, and a viewer three seconds into a muted video needs to know
|
| 451 |
"walking forward" before they need to know the argument names.
|
| 452 |
"""
|
|
|
|
|
|
|
| 453 |
before = turn.get("camera_before") or {}
|
| 454 |
after = turn.get("camera_after") or {}
|
| 455 |
if kind == "reset":
|
|
@@ -576,6 +782,42 @@ def main() -> None:
|
|
| 576 |
parser = argparse.ArgumentParser(description=__doc__)
|
| 577 |
parser.add_argument("episodes", type=pathlib.Path)
|
| 578 |
parser.add_argument("--episode", type=int, default=0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 579 |
parser.add_argument("--fps", type=int, default=30)
|
| 580 |
# 4:5 rather than 16:9. The clip is made for a phone timeline, where a
|
| 581 |
# portrait frame occupies about twice the screen a landscape one does, and
|
|
@@ -590,6 +832,21 @@ def main() -> None:
|
|
| 590 |
)
|
| 591 |
parser.add_argument("--quality", type=int, default=88)
|
| 592 |
parser.add_argument("--map-dpi", type=int, default=160)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 593 |
parser.add_argument(
|
| 594 |
"--reply-chars",
|
| 595 |
type=int,
|
|
@@ -605,6 +862,15 @@ def main() -> None:
|
|
| 605 |
help="View size the server used, for the fidelity check.",
|
| 606 |
)
|
| 607 |
parser.add_argument("--skip-verify", action="store_true")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 608 |
parser.add_argument("--cache", type=pathlib.Path, default=ROOT / "data" / "panos")
|
| 609 |
parser.add_argument("--tasks-dir", type=pathlib.Path, default=ROOT / "tasks")
|
| 610 |
parser.add_argument(
|
|
|
|
| 64 |
"measure": 2.2,
|
| 65 |
"guess": 6.0,
|
| 66 |
"forced_guess": 5.0,
|
| 67 |
+
# Nothing moves and nothing was learnt, so it should register and pass.
|
| 68 |
+
"unparseable": 1.6,
|
| 69 |
}
|
| 70 |
DEFAULT_DURATION = 2.4
|
| 71 |
|
|
|
|
| 116 |
return out
|
| 117 |
|
| 118 |
|
| 119 |
+
def walked(turn: dict) -> bool:
|
| 120 |
+
"""Whether a move actually crossed capture points."""
|
| 121 |
+
before = (turn.get("camera_before") or {}).get("frame_index")
|
| 122 |
+
after = (turn.get("camera_after") or {}).get("frame_index")
|
| 123 |
+
return before is not None and after is not None and before != after
|
| 124 |
+
|
| 125 |
+
|
| 126 |
+
def walk_states(before: dict, after: dict, steps: int) -> list[dict]:
|
| 127 |
+
"""One state per video frame, stepping through the frames actually walked.
|
| 128 |
+
|
| 129 |
+
The camera is held at the pose the move ended on -- a move does not turn the
|
| 130 |
+
head -- and only the position advances, eased so the walk starts and stops
|
| 131 |
+
rather than running at a constant rate.
|
| 132 |
+
|
| 133 |
+
Args:
|
| 134 |
+
before (`dict`):
|
| 135 |
+
Camera state before the move, carrying the starting `frame_index`.
|
| 136 |
+
after (`dict`):
|
| 137 |
+
Camera state after it.
|
| 138 |
+
steps (`int`):
|
| 139 |
+
How many video frames the segment lasts.
|
| 140 |
+
|
| 141 |
+
Returns:
|
| 142 |
+
`list[dict]`: `steps` states, each a copy of `after` with the
|
| 143 |
+
`frame_index` of the panorama that should be on screen.
|
| 144 |
+
"""
|
| 145 |
+
start, end = int(before["frame_index"]), int(after["frame_index"])
|
| 146 |
+
direction = 1 if end >= start else -1
|
| 147 |
+
indices = list(range(start, end + direction, direction))
|
| 148 |
+
out = []
|
| 149 |
+
for i in range(steps):
|
| 150 |
+
t = ease(i / max(1, steps - 1))
|
| 151 |
+
out.append({**after, "frame_index": indices[min(len(indices) - 1, int(t * len(indices)))]})
|
| 152 |
+
return out
|
| 153 |
+
|
| 154 |
+
|
| 155 |
def camera_changed(turn: dict) -> bool:
|
| 156 |
"""Whether the action actually moved the camera, in heading or field of view."""
|
| 157 |
before, after = turn.get("camera_before") or {}, turn.get("camera_after") or {}
|
|
|
|
| 199 |
if line.strip()
|
| 200 |
]
|
| 201 |
rows = [r for r in rows if not r.get("error")]
|
| 202 |
+
if args.episode_id:
|
| 203 |
+
# Point at a trace by name rather than by position: an index shifts when
|
| 204 |
+
# a run is resumed or a file is concatenated, so it is not a stable way
|
| 205 |
+
# to refer to the episode you actually watched.
|
| 206 |
+
matches = [r for r in rows if str(r.get("episode_id", "")).startswith(args.episode_id)]
|
| 207 |
+
if not matches:
|
| 208 |
+
raise SystemExit(
|
| 209 |
+
f"no episode id starting {args.episode_id!r} in {args.episodes}. "
|
| 210 |
+
f"Available: {', '.join(str(r.get('episode_id'))[:12] for r in rows[:8])}..."
|
| 211 |
+
)
|
| 212 |
+
if len(matches) > 1:
|
| 213 |
+
raise SystemExit(f"{args.episode_id!r} matches {len(matches)} episodes; be more specific")
|
| 214 |
+
record = matches[0]
|
| 215 |
+
else:
|
| 216 |
+
if not 0 <= args.episode < len(rows):
|
| 217 |
+
raise SystemExit(f"--episode must be 0..{len(rows) - 1} ({len(rows)} usable)")
|
| 218 |
+
record = rows[args.episode]
|
| 219 |
+
logger.info("episode %s · %s", record.get("episode_id"), record.get("model_name"))
|
| 220 |
|
| 221 |
split = record["task"]["split"]
|
| 222 |
index_name = INDEX_FOR_SPLIT.get(split)
|
|
|
|
| 253 |
)
|
| 254 |
|
| 255 |
size = (args.width_pano, args.width_pano)
|
| 256 |
+
# Matched to the composition's left panel, not to 16:9. The panel is about
|
| 257 |
+
# 1.42:1 once the rail and the bars are taken out, so a 16:9 frame had its
|
| 258 |
+
# sides cropped away by object-fit -- throwing away exactly the extra width
|
| 259 |
+
# the wide render exists to provide.
|
| 260 |
+
wide_size = (args.wide_width, int(round(args.wide_width / args.wide_aspect)))
|
| 261 |
+
wide_dir = out_dir / "wide"
|
| 262 |
+
mini_dir = out_dir / "minimap"
|
| 263 |
+
wide_dir.mkdir(parents=True, exist_ok=True)
|
| 264 |
+
mini_dir.mkdir(parents=True, exist_ok=True)
|
| 265 |
counter = 0
|
| 266 |
+
wide_counter = 0
|
| 267 |
+
mini_counter = 0
|
| 268 |
+
# Minimaps repeat whenever the pin set is unchanged, which is most turns.
|
| 269 |
+
minimap_cache: dict[tuple, str] = {}
|
| 270 |
segments = []
|
| 271 |
# The last panorama the agent saw, kept so a map segment can sit over the
|
| 272 |
# street view it was placed from -- the way the human UI's map inset does --
|
|
|
|
| 282 |
image.convert("RGB").save(frames_dir / name, quality=args.quality)
|
| 283 |
return f"frames/{name}"
|
| 284 |
|
| 285 |
+
def write_wide(image) -> str:
|
| 286 |
+
"""Save one wide-context frame. Kept in its own sequence so the two
|
| 287 |
+
panes can be composed side by side without the numbering interleaving."""
|
| 288 |
+
nonlocal wide_counter
|
| 289 |
+
wide_counter += 1
|
| 290 |
+
name = f"{wide_counter:05d}.jpg"
|
| 291 |
+
image.convert("RGB").save(wide_dir / name, quality=args.quality)
|
| 292 |
+
return f"wide/{name}"
|
| 293 |
+
|
| 294 |
+
def write_minimap(image) -> str:
|
| 295 |
+
"""Save one minimap still."""
|
| 296 |
+
nonlocal mini_counter
|
| 297 |
+
mini_counter += 1
|
| 298 |
+
name = f"{mini_counter:05d}.png"
|
| 299 |
+
image.convert("RGB").save(mini_dir / name)
|
| 300 |
+
return f"minimap/{name}"
|
| 301 |
+
|
| 302 |
+
def minimap_for(pins: list[tuple[float, float]]) -> str | None:
|
| 303 |
+
"""A wide map of what the agent has pinned so far, and nothing more.
|
| 304 |
+
|
| 305 |
+
Deliberately never passed `truth`: the minimap is on screen from the
|
| 306 |
+
first turn, so drawing the answer there would spoil every episode before
|
| 307 |
+
the agent has said anything. A run with no pins yet gets no minimap.
|
| 308 |
+
"""
|
| 309 |
+
if not pins:
|
| 310 |
+
return None
|
| 311 |
+
key = tuple(pins)
|
| 312 |
+
if key not in minimap_cache:
|
| 313 |
+
focus = pins[-1]
|
| 314 |
+
minimap_cache[key] = write_minimap(
|
| 315 |
+
render_map(
|
| 316 |
+
list(pins),
|
| 317 |
+
focus=focus,
|
| 318 |
+
span_deg=args.minimap_span,
|
| 319 |
+
dpi=max(70, args.map_dpi // 2),
|
| 320 |
+
)
|
| 321 |
+
)
|
| 322 |
+
return minimap_cache[key]
|
| 323 |
+
|
| 324 |
for turn in record["turns"]:
|
| 325 |
action = turn.get("action") or {}
|
| 326 |
+
# An unparseable reply has no action at all, and defaulting that to
|
| 327 |
+
# "reset" captioned the segment "dropped somewhere on Earth" -- telling
|
| 328 |
+
# a viewer the episode restarted when the model had simply written
|
| 329 |
+
# something the parser could not read. The distinction matters: one is
|
| 330 |
+
# the environment acting, the other is the model failing.
|
| 331 |
+
raw_kind = action.get("action") or action.get("op")
|
| 332 |
+
if raw_kind is None:
|
| 333 |
+
kind = "reset" if turn["turn"] == 0 else "unparseable"
|
| 334 |
+
else:
|
| 335 |
+
# Mirror the environment's own aliases so a video labels the action
|
| 336 |
+
# the environment performed, not the spelling the model used.
|
| 337 |
+
kind = ALIASES.get(str(raw_kind).lower(), str(raw_kind).lower())
|
| 338 |
+
if kind == "unparseable" and args.skip_unparsed:
|
| 339 |
+
# A reply the harness could not read may say more about the parser
|
| 340 |
+
# than about the model, and the environment did nothing with the
|
| 341 |
+
# turn either way. Left in, the segment spends two seconds telling
|
| 342 |
+
# the viewer about our own tooling. The turn is still in the trace;
|
| 343 |
+
# pass --keep-unparsed to show it.
|
| 344 |
+
logger.info("skipping turn %s: reply was not parsed", turn["turn"])
|
| 345 |
+
continue
|
| 346 |
seconds = DURATIONS.get(kind, DEFAULT_DURATION)
|
| 347 |
if kind in {"look", "pan", "zoom"} and not camera_changed(turn):
|
| 348 |
seconds = min(seconds, STATIC_DURATION)
|
| 349 |
total = max(1, int(round(seconds * args.fps)))
|
| 350 |
paths: list[str] = []
|
| 351 |
+
wide_paths: list[str] = []
|
| 352 |
cameras: list[list[float]] = []
|
| 353 |
+
pins_now = [tuple(pt) for pt in (turn.get("pins") or [])]
|
| 354 |
|
| 355 |
if turn.get("image_kind") == "map":
|
| 356 |
# Maps are single stills: re-render at video resolution from the pins
|
|
|
|
| 384 |
focus = pins[-1] if pins else None
|
| 385 |
else:
|
| 386 |
focus = pins[-1] if pins else None
|
| 387 |
+
|
| 388 |
+
if truth is not None and pins and args.reveal_frames > 1:
|
| 389 |
+
# The reveal is animated rather than cut to. A single still
|
| 390 |
+
# showed the guess, the answer and the line between them all at
|
| 391 |
+
# once, which lands as a fact rather than as a result: the
|
| 392 |
+
# viewer has no moment of "where is it, then". So the map opens
|
| 393 |
+
# tight on the guess and widens until the true location comes
|
| 394 |
+
# into frame, and the truth marker is withheld until the
|
| 395 |
+
# framing is wide enough to contain it.
|
| 396 |
+
start_span = max(0.35, span * 0.18)
|
| 397 |
+
reveal_at = max(1, int(args.reveal_frames * 0.42))
|
| 398 |
+
for i in range(args.reveal_frames):
|
| 399 |
+
t = ease(i / (args.reveal_frames - 1))
|
| 400 |
+
step_focus = (
|
| 401 |
+
pins[0][0] + (focus[0] - pins[0][0]) * t,
|
| 402 |
+
pins[0][1] + (focus[1] - pins[0][1]) * t,
|
| 403 |
+
)
|
| 404 |
+
image = render_map(
|
| 405 |
+
pins,
|
| 406 |
+
focus=step_focus,
|
| 407 |
+
span_deg=start_span + (span - start_span) * t,
|
| 408 |
+
dpi=args.reveal_dpi,
|
| 409 |
+
truth=truth if i >= reveal_at else None,
|
| 410 |
+
)
|
| 411 |
+
map_aspect = image.width / image.height
|
| 412 |
+
paths.append(write(image))
|
| 413 |
+
# The last frame at full resolution, because it is held for
|
| 414 |
+
# several seconds and is the shot people screenshot.
|
| 415 |
+
final = render_map(
|
| 416 |
+
pins, focus=focus, span_deg=span, dpi=args.map_dpi, truth=truth
|
| 417 |
+
)
|
| 418 |
+
map_aspect = final.width / final.height
|
| 419 |
+
paths.append(write(final))
|
| 420 |
+
else:
|
| 421 |
+
image = render_map(
|
| 422 |
+
pins,
|
| 423 |
+
focus=focus,
|
| 424 |
+
span_deg=span,
|
| 425 |
+
dpi=args.map_dpi,
|
| 426 |
+
truth=truth,
|
| 427 |
+
)
|
| 428 |
+
map_aspect = image.width / image.height
|
| 429 |
+
paths = [write(image)]
|
| 430 |
else:
|
| 431 |
before, after = turn["camera_before"], turn["camera_after"]
|
| 432 |
moving = kind in {"look", "pan", "zoom"} and camera_changed(turn)
|
| 433 |
+
if kind == "move" and walked(turn):
|
| 434 |
+
# A real walk, not a cut. The capture points are about 3.3 m
|
| 435 |
+
# apart, so a 34 m move crosses roughly ten panoramas -- every
|
| 436 |
+
# one of which exists and can be rendered at the same heading.
|
| 437 |
+
# Holding each for a few video frames gives the forward motion
|
| 438 |
+
# Street View shows when you step down a road, and every frame
|
| 439 |
+
# is something the environment could actually have returned.
|
| 440 |
+
states = walk_states(before, after, total)
|
| 441 |
+
else:
|
| 442 |
+
states = interpolate(before, after, total) if moving else [after]
|
| 443 |
for state in states:
|
| 444 |
if state.get("frame_index") is None:
|
| 445 |
continue
|
|
|
|
| 453 |
if image.size != size:
|
| 454 |
image = image.resize(size)
|
| 455 |
paths.append(write(image))
|
| 456 |
+
# The same moment at a wider field of view. Rendered from the
|
| 457 |
+
# same pose, so the two panes cannot disagree about where the
|
| 458 |
+
# agent is looking -- the left one simply shows more of the
|
| 459 |
+
# street than the model was given.
|
| 460 |
+
wide = backend.render_view(
|
| 461 |
+
task,
|
| 462 |
+
int(state["frame_index"]),
|
| 463 |
+
float(state["heading_deg"]),
|
| 464 |
+
float(state["pitch_deg"]),
|
| 465 |
+
min(120.0, float(state["fov_deg"]) * args.wide_factor),
|
| 466 |
+
)
|
| 467 |
+
if wide.size != wide_size:
|
| 468 |
+
wide = wide.resize(wide_size)
|
| 469 |
+
wide_paths.append(write_wide(wide))
|
| 470 |
# The camera the frame was rendered with, so the HUD can read
|
| 471 |
# live through a pan instead of jumping to the end state the
|
| 472 |
# moment the shot begins.
|
|
|
|
| 502 |
# Only set on a map segment: the street view it was placed from.
|
| 503 |
"backdrop": backdrop if turn.get("image_kind") == "map" else None,
|
| 504 |
"frames": paths,
|
| 505 |
+
# The same pose at a wider field of view, for the left canvas.
|
| 506 |
+
# Empty on a map segment, where there is nothing to widen.
|
| 507 |
+
"wide": wide_paths,
|
| 508 |
+
# What the agent has pinned by the end of this turn, so the
|
| 509 |
+
# minimap and the bubbles agree about the state of the guess.
|
| 510 |
+
"pins": [[round(a, 4), round(b, 4)] for a, b in pins_now],
|
| 511 |
+
"minimap": minimap_for(pins_now),
|
| 512 |
"cameras": cameras,
|
| 513 |
# A single still is held for the whole segment; a pan has one
|
| 514 |
# frame per video frame.
|
|
|
|
| 543 |
# letterboxing it: React sizes the card to the image instead of fitting
|
| 544 |
# the image into a card the wrong shape.
|
| 545 |
"mapAspect": map_aspect,
|
| 546 |
+
"theme": args.theme,
|
| 547 |
+
"wideAspect": args.wide_aspect,
|
| 548 |
"episode": {
|
| 549 |
"episode_id": record["episode_id"],
|
| 550 |
"model_name": record["model_name"],
|
|
|
|
| 579 |
|
| 580 |
#: Uppercase verb for the on-screen action badge, in the human UI's own words
|
| 581 |
#: (the control pad says LOOK / WALK / ZOOM, not `look` / `move` / `zoom`).
|
| 582 |
+
# The spellings a model may emit for an action the environment accepts. Kept in
|
| 583 |
+
# step with `ACTION_ALIASES` in examples/geoguesser_llm_rollout.py.
|
| 584 |
+
ALIASES = {
|
| 585 |
+
"place_pin": "pin",
|
| 586 |
+
"submit_guess": "guess",
|
| 587 |
+
"final_guess": "guess",
|
| 588 |
+
"reverse_geocode": "view_map",
|
| 589 |
+
"turn": "pan",
|
| 590 |
+
"rotate": "pan",
|
| 591 |
+
}
|
| 592 |
+
|
| 593 |
VERBS = {
|
| 594 |
"reset": "START",
|
| 595 |
+
"unparseable": "UNREAD",
|
| 596 |
"look": "LOOK",
|
| 597 |
"pan": "LOOK",
|
| 598 |
"zoom": "ZOOM",
|
|
|
|
| 654 |
separately, and a viewer three seconds into a muted video needs to know
|
| 655 |
"walking forward" before they need to know the argument names.
|
| 656 |
"""
|
| 657 |
+
if kind == "unparseable":
|
| 658 |
+
return "reply could not be parsed as an action"
|
| 659 |
before = turn.get("camera_before") or {}
|
| 660 |
after = turn.get("camera_after") or {}
|
| 661 |
if kind == "reset":
|
|
|
|
| 782 |
parser = argparse.ArgumentParser(description=__doc__)
|
| 783 |
parser.add_argument("episodes", type=pathlib.Path)
|
| 784 |
parser.add_argument("--episode", type=int, default=0)
|
| 785 |
+
parser.add_argument(
|
| 786 |
+
"--episode-id",
|
| 787 |
+
default=None,
|
| 788 |
+
help="Select by episode_id (a unique prefix is enough) instead of by "
|
| 789 |
+
"position, which shifts when a run is resumed.",
|
| 790 |
+
)
|
| 791 |
+
parser.add_argument(
|
| 792 |
+
"--theme",
|
| 793 |
+
choices=("dark", "light"),
|
| 794 |
+
default="dark",
|
| 795 |
+
help="Composition palette. Light reads better on a timeline; dark reads "
|
| 796 |
+
"better as an instrument.",
|
| 797 |
+
)
|
| 798 |
+
parser.add_argument(
|
| 799 |
+
"--wide-factor",
|
| 800 |
+
type=float,
|
| 801 |
+
default=2.0,
|
| 802 |
+
help="How much wider than the agent's own field of view the left canvas "
|
| 803 |
+
"renders. A pan then scrolls through a scene instead of cutting between "
|
| 804 |
+
"two crops, which is what the environment actually models.",
|
| 805 |
+
)
|
| 806 |
+
parser.add_argument("--wide-width", type=int, default=1408)
|
| 807 |
+
parser.add_argument(
|
| 808 |
+
"--wide-aspect",
|
| 809 |
+
type=float,
|
| 810 |
+
default=1.42,
|
| 811 |
+
help="Aspect of the left canvas in the composition. Anything wider is "
|
| 812 |
+
"cropped away by object-fit rather than shown.",
|
| 813 |
+
)
|
| 814 |
+
parser.add_argument(
|
| 815 |
+
"--minimap-span",
|
| 816 |
+
type=float,
|
| 817 |
+
default=60.0,
|
| 818 |
+
help="Half-width in degrees of the minimap. Wide enough to place a "
|
| 819 |
+
"continent without revealing anything the agent has not pinned.",
|
| 820 |
+
)
|
| 821 |
parser.add_argument("--fps", type=int, default=30)
|
| 822 |
# 4:5 rather than 16:9. The clip is made for a phone timeline, where a
|
| 823 |
# portrait frame occupies about twice the screen a landscape one does, and
|
|
|
|
| 832 |
)
|
| 833 |
parser.add_argument("--quality", type=int, default=88)
|
| 834 |
parser.add_argument("--map-dpi", type=int, default=160)
|
| 835 |
+
parser.add_argument(
|
| 836 |
+
"--reveal-frames",
|
| 837 |
+
type=int,
|
| 838 |
+
default=30,
|
| 839 |
+
help="Frames in the guess reveal, widening from the guess until the "
|
| 840 |
+
"true location comes into frame. 1 disables the animation.",
|
| 841 |
+
)
|
| 842 |
+
parser.add_argument(
|
| 843 |
+
"--reveal-dpi",
|
| 844 |
+
type=int,
|
| 845 |
+
default=100,
|
| 846 |
+
help="Resolution of the reveal's intermediate frames. Lower than "
|
| 847 |
+
"--map-dpi because they are on screen for a thirtieth of a second "
|
| 848 |
+
"each, while the final frame is rendered at full resolution.",
|
| 849 |
+
)
|
| 850 |
parser.add_argument(
|
| 851 |
"--reply-chars",
|
| 852 |
type=int,
|
|
|
|
| 862 |
help="View size the server used, for the fidelity check.",
|
| 863 |
)
|
| 864 |
parser.add_argument("--skip-verify", action="store_true")
|
| 865 |
+
parser.add_argument(
|
| 866 |
+
"--keep-unparsed",
|
| 867 |
+
dest="skip_unparsed",
|
| 868 |
+
action="store_false",
|
| 869 |
+
default=True,
|
| 870 |
+
help="Show turns whose reply the harness could not parse. Skipped by "
|
| 871 |
+
"default: such a turn may be a limitation of the parser rather than of "
|
| 872 |
+
"the model, and the environment did nothing with it either way.",
|
| 873 |
+
)
|
| 874 |
parser.add_argument("--cache", type=pathlib.Path, default=ROOT / "data" / "panos")
|
| 875 |
parser.add_argument("--tasks-dir", type=pathlib.Path, default=ROOT / "tasks")
|
| 876 |
parser.add_argument(
|