-
-
Notifications
You must be signed in to change notification settings - Fork 29
Expand file tree
/
Copy pathrender_gate.py
More file actions
238 lines (196 loc) · 9.97 KB
/
Copy pathrender_gate.py
File metadata and controls
238 lines (196 loc) · 9.97 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
"""Let a background thread run Python only while the render thread waits on vsync.
With plugin rendering moved to Vegas's prefetch thread (DisplayManager.offscreen,
#630) the render thread no longer stops for it, but it still shares the GIL
with it. The render thread spends most of each refresh inside SwapOnVSync,
which releases the GIL, and needs it back the moment the swap returns. If the
prefetch thread is running Python right then, the render thread waits: up to
the switch interval (5ms) behind bytecode, and for as long as a C call that
keeps the GIL takes. On hdpi that showed up as frames 2-5 refreshes late while
a group was being prepared.
The gate turns that around. The display manager opens it just before each swap,
with a deadline shortly ahead of the refresh the swap will return on, and
closes it when the swap returns. A thread inside ``gate.yielding()`` checks it on
every Python and C call through a profile hook, and once the window has closed
it parks -- blocked on a condition, GIL released -- until the next swap opens
it. The render thread then finds the GIL free when its refresh arrives, and the
background work runs in time the render thread was only spending waiting.
Parking a thread is only safe if nothing the render thread needs is stuck
behind it, so it is never parked:
* while it holds a lock registered with ``guard()`` (the Vegas buffers and
caches the render thread also takes);
* inside logging, threading, importlib or the cache, all of which take locks the
render thread can take too;
* when there is no render loop to protect -- no swap for ``STALE_SECONDS``, as
on a static screen or a stalled frame.
And a parked thread is never held more than ``MAX_WAIT_SECONDS`` at a time, so
whatever the gate gets wrong costs a frame, not a freeze. The render thread
itself is never gated, whatever it calls.
It gates the prefetch thread only. Gating the ESPN fetch threads as well was
tried for the hourly sports refresh, twenty-odd of them at once, and measured
worse on hdpi (0.85% late frames without it, 1.14% with it, across a burst every
five minutes): each parked thread has to take the GIL again just to park at the
end of every window, and the fetches ran two to three times as long.
"""
from __future__ import annotations
import math
import sys
import threading
import time
from collections import deque
from typing import Any, Callable, Deque, List, Optional, cast
from src.common.frame_timing import binding_releases_gil
#: Park background threads this long before the refresh a swap will return on,
#: so a short C call already under way has finished by then.
MARGIN_SECONDS = 0.002
#: The longest a background thread is parked in one go.
MAX_WAIT_SECONDS = 0.05
#: No swap for this long means there is no render loop running to protect.
STALE_SECONDS = 0.05
#: Swaps needed before the refresh period is trusted enough to open a window.
MIN_SAMPLES = 8
#: Parking inside any of these modules could hold a lock the render thread
#: takes: logging handler locks, Condition and Event internals, the module
#: import locks, and the disk and memory cache locks. Matched by module name,
#: not file path: a path can say "cache" or "logging" for reasons of its own --
#: a virtualenv under ~/.cache, or GitHub's /opt/hostedtoolcache, where every
#: stdlib frame would otherwise count and the gate would never park anything.
_UNSAFE_MODULES = frozenset({
"logging", "threading", "importlib", "src.cache_manager", "src.cache",
})
_UNSAFE_PREFIXES = ("logging.", "importlib.", "_frozen_importlib", "src.cache.")
def _unsafe(frame: Any, base: Any) -> bool:
"""True if a frame above ``base`` comes from somewhere parking could deadlock.
``base`` is the frame that entered ``yielding()``; what lies below it (the
thread's own bootstrap in threading.py) holds nothing.
"""
while frame is not None and frame is not base:
name = frame.f_globals.get("__name__") or ""
if name in _UNSAFE_MODULES or name.startswith(_UNSAFE_PREFIXES):
return True
frame = frame.f_back
return False
def swap_releases_gil() -> Optional[bool]:
"""Whether the loaded rgbmatrix binding releases the GIL, or None if none is loaded.
A thin delegate to src.common.frame_timing.binding_releases_gil (#629),
which this used to duplicate line for line. The name stays because the
coordinator calls it here and tests replace it here.
"""
return binding_releases_gil()
def _held(lock: Any) -> bool:
"""Is ``lock`` held? RLocks report this thread's ownership; plain locks, anyone's."""
is_owned = getattr(lock, "_is_owned", None)
if is_owned is not None:
return cast(bool, is_owned())
return cast(bool, lock.locked())
class RenderGate:
"""Opened by the render thread around each swap; honoured by background threads."""
def __init__(self, clock: Callable[[], float] = time.monotonic):
self.clock = clock
self._cond = threading.Condition()
self._generation = 0
self._open_until = 0.0
self._last_return: Optional[float] = None
self._periods: Deque[float] = deque(maxlen=64)
self._period: Optional[float] = None
self._guarded: List[Any] = []
self._local = threading.local()
self._render_ident: Optional[int] = None
#: How often, and for how long in all, background threads were parked.
self.parks = 0
self.parked_seconds = 0.0
def guard(self, *locks: Any) -> None:
"""Never park a thread while it holds (or, for a plain Lock, anyone holds) these."""
self._guarded.extend(lock for lock in locks if lock is not None)
# -- render thread -----------------------------------------------------
def refresh_period(self) -> Optional[float]:
"""The panel's refresh period from recent swaps, or None until known.
The 10th percentile of the gaps between swap returns, each divided by
the hold: a late frame only ever lengthens a gap, so the low end is
the panel's own period.
"""
return self._period
def before_swap(self, hold: int) -> None:
"""The render thread is about to block in SwapOnVSync: open the window."""
hold = max(1, int(hold))
period = self._period
now = self.clock()
last = self._last_return
if period and last is not None and now - last < STALE_SECONDS:
# The swap returns on the first refresh boundary after both the
# current frame's hold is up and this frame has been handed over;
# boundaries fall a whole period apart from the last return.
refreshes = max(hold, math.ceil((now - last) / period))
open_until = last + refreshes * period - MARGIN_SECONDS
else:
open_until = 0.0 # no rhythm to predict from: leave threads be
with self._cond:
self._open_until = open_until
self._generation += 1
self._cond.notify_all()
def after_swap(self, hold: int) -> None:
"""The swap returned and the render thread needs the GIL: close the window."""
now = self.clock()
self._open_until = 0.0
if self._render_ident is None:
# The first thread to swap is the render loop. A plugin pushing a
# live refresh from its update thread swaps too, but must not take
# over its exemption.
self._render_ident = threading.get_ident()
last = self._last_return
if last is not None and now - last < STALE_SECONDS:
self._periods.append((now - last) / max(1, int(hold)))
if len(self._periods) >= MIN_SAMPLES:
ordered = sorted(self._periods)
self._period = ordered[len(ordered) // 10]
self._last_return = now
# -- background threads ------------------------------------------------
def _should_park(self, frame: Any, now: float) -> bool:
if now < self._open_until:
return False # inside the window
last = self._last_return
if last is None or now - last > STALE_SECONDS or self._period is None:
return False # no render loop to protect
for lock in self._guarded:
if _held(lock):
return False
return not _unsafe(frame, getattr(self._local, "base", None))
def _hook(self, frame: Any, _event: str, _arg: Any) -> None:
now = self.clock()
if not self._should_park(frame, now):
return
generation = self._generation
with self._cond:
self._cond.wait_for(lambda: self._generation != generation,
timeout=MAX_WAIT_SECONDS)
self.parks += 1
self.parked_seconds += self.clock() - now
def yielding(self) -> "_Yielding":
"""``with gate.yielding():`` runs the block giving way to the render thread."""
return _Yielding(self)
class _Yielding:
"""Installs a gate's profile hook on the thread for the length of a block."""
def __init__(self, gate: RenderGate):
self.gate = gate
self._previous: Any = None
self._previous_base: Any = None
self._skipped = False
def __enter__(self) -> RenderGate:
gate = self.gate
# pylint: disable=protected-access
if threading.get_ident() == gate._render_ident:
self._skipped = True # parking the render thread parks the display
return gate
local = gate._local
self._previous_base = getattr(local, "base", None)
if self._previous_base is None:
# Nested blocks keep the outermost frame, so everything the thread
# entered since it first gave way is still checked for locks.
local.base = sys._getframe(1)
self._previous = sys.getprofile()
sys.setprofile(gate._hook)
return gate
def __exit__(self, *_exc: Any) -> None:
if self._skipped:
return
sys.setprofile(self._previous)
self.gate._local.base = self._previous_base # pylint: disable=protected-access