Bound sample wall clock; a request timeout does not stop a hang

One baseline sample sat on a single model request for 2h15m with an established
connection, 0.1% CPU and no file activity since the init solver wrote func.py. The
per-request timeout and max_retries did not bound it, and it blocked the rest of the run
behind it.

time_limit is the control that actually applies, per sample. Samples that finish take 8
to 15 minutes, so 30 is generous. An unbounded straggler costs more than the sample does.

Claude-Session: https://claude.ai/code/session_01Cq98H7sNoSJdL3W98f18bu
This commit is contained in:
pj committed 2026-09-01 00:01:26 +05:30
1 parent f933ce6d15
commit abacd5c5e1
2 files changed
+14

No files matched your search

+7
View File
@@ -35,6 +35,12 @@ CONCURRENCY = int(os.environ.get("MBB_CONCURRENCY", "12"))
# holding a slot for the length of the run, not to cut short slow thinking. # holding a slot for the length of the run, not to cut short slow thinking.
REQUEST_TIMEOUT = int(os.environ.get("MBB_TIMEOUT", "900")) REQUEST_TIMEOUT = int(os.environ.get("MBB_TIMEOUT", "900"))
# Per-sample wall clock. One baseline sample hung on a single model request for
# 2h15m with no file activity and never came back; the request timeout did not
# bound it. Samples that finish take 8 to 15 minutes, so 30 is generous, and an
# unbounded straggler blocking a whole run is worth more than the sample.
SAMPLE_TIME_LIMIT = int(os.environ.get("MBB_SAMPLE_LIMIT", "1800"))
def credits_used() -> float | None: def credits_used() -> float | None:
"""Total credits spent on the key so far, or None if the endpoint is unavailable.""" """Total credits spent on the key so far, or None if the endpoint is unavailable."""
@@ -71,6 +77,7 @@ def run_split(split: str) -> dict:
fail_on_error=False, fail_on_error=False,
timeout=REQUEST_TIMEOUT, timeout=REQUEST_TIMEOUT,
max_retries=3, max_retries=3,
time_limit=SAMPLE_TIME_LIMIT,
) )
after = credits_used() after = credits_used()
log = logs[0] log = logs[0]
+7
View File
@@ -27,6 +27,12 @@ MODEL = os.environ.get("MBB_MODEL", "openrouter/z-ai/glm-5.3-flash")
LIMIT = int(os.environ.get("MBB_LIMIT", "30")) LIMIT = int(os.environ.get("MBB_LIMIT", "30"))
CONCURRENCY = int(os.environ.get("MBB_CONCURRENCY", "12")) CONCURRENCY = int(os.environ.get("MBB_CONCURRENCY", "12"))
REQUEST_TIMEOUT = int(os.environ.get("MBB_TIMEOUT", "900")) REQUEST_TIMEOUT = int(os.environ.get("MBB_TIMEOUT", "900"))
# Per-sample wall clock. One baseline sample hung on a single model request for
# 2h15m with no file activity and never came back; the request timeout did not
# bound it. Samples that finish take 8 to 15 minutes, so 30 is generous, and an
# unbounded straggler blocking a whole run is worth more than the sample.
SAMPLE_TIME_LIMIT = int(os.environ.get("MBB_SAMPLE_LIMIT", "1800"))
OUT = Path("results") OUT = Path("results")
@@ -88,6 +94,7 @@ if __name__ == "__main__":
fail_on_error=False, fail_on_error=False,
timeout=REQUEST_TIMEOUT, timeout=REQUEST_TIMEOUT,
max_retries=3, max_retries=3,
time_limit=SAMPLE_TIME_LIMIT,
)[0] )[0]
after = credits_used() after = credits_used()