Skip to content

Commit 5603265

Browse files
committed
Make complexity estimates hold up under gVisor
In production two of the four reference solutions got the wrong class: runs there carry some 150 ms of fixed cost and CPU time comes in 10 ms steps, so the small runs the fit leaned on were mostly noise. A size now counts once the solution's own time is at least the run's fixed cost, short runs are timed twice and the faster kept, and only the largest sizes are fitted. Two Sum gets one more input size so it clears the floor at three sizes. A result faster than expected no longer reads as a match.
1 parent 0b43d7d commit 5603265

4 files changed

Lines changed: 44 additions & 9 deletions

File tree

‎apps/backend/app/runner/complexity.py‎

Lines changed: 29 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -25,10 +25,21 @@
2525
BUDGET_SECONDS = 6.0
2626
PER_RUN_SECONDS = 2.5
2727
ENOUGH_SECONDS = 1.0
28-
29-
# Below this a run is mostly noise, not the solution.
28+
# Only runs shorter than this are repeated. Noise is a few tens of
29+
# milliseconds, which does not matter to a longer run, and the time saved
30+
# buys a slow solution one more size.
31+
REPEAT_BELOW_SECONDS = 0.5
32+
33+
# A run counts once the solution's own time is at least the fixed cost taken
34+
# off it, so an error in that estimate stays a fraction of what is measured.
35+
# The fixed cost is some 15 ms on a laptop and 150 ms under gVisor, which also
36+
# reports CPU time in 10 ms steps. This is the floor where it is smaller.
3037
MIN_MEASURABLE_MS = 20.0
3138

39+
# Only the largest sizes are fitted: the smaller a run, the larger the share
40+
# of its time that is noise and error in the fixed cost.
41+
FITTED_POINTS = 3
42+
3243
# Fitted exponent of n -> growth class, split halfway between the powers.
3344
# n log n fits at about 1.1 over the sizes used, too close to n to tell
3445
# apart, so they share a class.
@@ -111,19 +122,32 @@ def analyze(code: str, generator: str, mem_limit_mb: int) -> dict:
111122
# and nothing more; left in, it flattens the growth of every run.
112123
tiny = generate(max(4, sizes[0] // 32))
113124
startup = min(_run(program, workdir, tiny, PER_RUN_SECONDS, mem_limit_mb)[0] or 0.0 for _ in range(3))
125+
floor_ms = max(MIN_MEASURABLE_MS, startup * 1000)
114126

115127
for n in sizes:
116128
remaining = BUDGET_SECONDS - spent
117129
if remaining <= 0.1:
118130
note = f"Stopped before n = {n}: out of time for this analysis."
119131
break
120-
seconds, failure = _run(program, workdir, generate(n), min(PER_RUN_SECONDS, remaining), mem_limit_mb)
132+
stdin = generate(n)
133+
seconds, failure = _run(program, workdir, stdin, min(PER_RUN_SECONDS, remaining), mem_limit_mb)
121134
if seconds is None:
122135
note = f"Stopped at n = {n}: it {failure}."
123136
break
124137
spent += seconds
138+
# Scheduling only ever makes a run slower, so a short measurable
139+
# size is run twice and the faster run kept.
140+
if (
141+
(seconds - startup) * 1000 >= floor_ms
142+
and seconds < REPEAT_BELOW_SECONDS
143+
and BUDGET_SECONDS - spent > seconds
144+
):
145+
again, _ = _run(program, workdir, stdin, min(PER_RUN_SECONDS, BUDGET_SECONDS - spent), mem_limit_mb)
146+
if again is not None:
147+
spent += again
148+
seconds = min(seconds, again)
125149
ms = (seconds - startup) * 1000
126-
if ms >= MIN_MEASURABLE_MS:
150+
if ms >= floor_ms:
127151
points.append((n, ms))
128152
if seconds >= ENOUGH_SECONDS:
129153
break
@@ -135,7 +159,7 @@ def analyze(code: str, generator: str, mem_limit_mb: int) -> dict:
135159
"note": note,
136160
}
137161
if len(points) >= 2:
138-
result["complexity"], slope = classify(points)
162+
result["complexity"], slope = classify(points[-FITTED_POINTS:])
139163
result["slope"] = round(slope, 2)
140164
if len(points) == 2:
141165
result["note"] = (note + " " if note else "") + "Only two sizes were measurable, so this is rough."
Lines changed: 10 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,10 @@
1+
-- V14: One more input size for Two Sum's complexity analysis
2+
--
3+
-- A run only counts once the solution's own time is at least the run's fixed
4+
-- cost (app/runner/complexity.py), and at half a million numbers the
5+
-- reference solution clears that at only one or two sizes. A million numbers
6+
-- fit in the problem's memory limit.
7+
8+
UPDATE problems
9+
SET complexity_generator = replace(complexity_generator, 'range(10)', 'range(11)')
10+
WHERE slug = 'two-sum';

‎apps/web/src/components/ComplexityPanel.tsx‎

Lines changed: 3 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -75,15 +75,16 @@ const ComplexityPanel = ({ submission, expected }: { submission: Submission; exp
7575
const measured = CLASSES.indexOf(analysis.complexity);
7676
const target = expected ? CLASSES.indexOf(expected) : -1;
7777
const slower = target >= 0 && measured > target;
78+
const verdict = target < 0 ? null : measured > target ? "Slower than" : measured < target ? "Faster than" : "Matches";
7879
return (
7980
<div className="space-y-3 border-t-2 border-zinc-800 px-4 py-3 text-sm">
8081
<div className="flex flex-wrap items-baseline gap-x-4 gap-y-1">
8182
<span className={`font-pixel text-3xl leading-none ${slower ? "text-amber-300" : "text-emerald-400"}`}>
8283
<Label name={analysis.complexity} />
8384
</span>
84-
{expected && (
85+
{expected && verdict && (
8586
<span className={muted}>
86-
{slower ? "Slower than" : "Matches"} the expected {LABEL[expected]}
87+
{verdict} the expected {LABEL[expected]}
8788
</span>
8889
)}
8990
</div>

‎docs/judge-and-sandbox.md‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -33,8 +33,8 @@ Without `SANDBOX_IMAGE`, the runner refuses to start unless `ALLOW_UNSANDBOXED=1
3333
On request, an accepted run can be rerun on inputs of growing size to estimate how its running time grows. The problem's generator (`complexity_generator`, see [CONTRIBUTING.md](../CONTRIBUTING.md)) makes the inputs, and `app/runner/complexity.py` runs the program on each size in one sandbox container.
3434

3535
- It measures CPU time from outside the program. Counting executed lines would miss work done inside built-ins, such as `x in some_list`, and would call a quadratic brute force linear. The program also can't report a time of its own.
36-
- The fixed cost of every run (interpreter start, imports, reading input) is measured on a tiny input and subtracted.
37-
- It fits the times to a power of n and names the class. n log n measures too close to n to tell apart, so the two share a class.
36+
- The fixed cost of every run (interpreter start, imports, reading input) is measured on a tiny input and subtracted. Under gVisor it is some 150 ms, ten times a laptop's, and gVisor reports CPU time in 10 ms steps, so a size only counts once the solution's own time is at least that fixed cost. Short runs are timed twice and the faster kept, since scheduling only ever adds time.
37+
- It fits the largest measurable sizes (`FITTED_POINTS`) to a power of n and names the class. n log n measures too close to n to tell apart, so the two share a class.
3838
- Each analysis has a time budget, and each size a limit (`BUDGET_SECONDS` and `PER_RUN_SECONDS` in `complexity.py`). A solution that outgrows them is stopped, and the note says where.
3939
- The runner analyzes only when no submission is waiting, so an analysis never holds up a verdict by more than the one in progress.
4040

0 commit comments

Comments
 (0)