added env variables for server limits
This commit is contained in:
@@ -38,8 +38,13 @@ IMAGE_SIZE_CNN = 64
|
|||||||
IMAGE_SIZE_YOLO = 64
|
IMAGE_SIZE_YOLO = 64
|
||||||
|
|
||||||
# ---- Resource limits (the server is weak, keep everything small and bounded) ----
|
# ---- Resource limits (the server is weak, keep everything small and bounded) ----
|
||||||
TORCH_THREADS = int(os.getenv("TORCH_THREADS", "1"))
|
# Number of predictions (bird or face) that may run at the same time.
|
||||||
MAX_PENDING_INFERENCES = int(os.getenv("MAX_PENDING_INFERENCES", "2")) # running + waiting
|
PARALLEL_INFERENCES = max(1, int(os.getenv("PARALLEL_INFERENCES", "1")))
|
||||||
|
# Extra requests allowed to wait for a free slot; anything beyond is rejected.
|
||||||
|
QUEUED_INFERENCES = max(0, int(os.getenv("QUEUED_INFERENCES", "1")))
|
||||||
|
# PyTorch threads used by EACH running prediction. Total CPU use is roughly
|
||||||
|
# PARALLEL_INFERENCES * TORCH_THREADS, so keep the product <= your CPU cores.
|
||||||
|
TORCH_THREADS = max(1, int(os.getenv("TORCH_THREADS", "1")))
|
||||||
MAX_UPLOAD_BYTES = 5 * 1024 * 1024
|
MAX_UPLOAD_BYTES = 5 * 1024 * 1024
|
||||||
MAX_FRAME_BYTES = 1 * 1024 * 1024
|
MAX_FRAME_BYTES = 1 * 1024 * 1024
|
||||||
MAX_IMAGE_PIXELS = 20_000_000
|
MAX_IMAGE_PIXELS = 20_000_000
|
||||||
@@ -49,7 +54,6 @@ FACE_MAX_FPS = float(os.getenv("FACE_MAX_FPS", "5.5")) # per websocket connecti
|
|||||||
FACE_MIN_INTERVAL = 1 / FACE_MAX_FPS
|
FACE_MIN_INTERVAL = 1 / FACE_MAX_FPS
|
||||||
DECODE_DRAFT_SIZE = (256, 256) # JPEG decodes at reduced scale, still larger than the 64px model input
|
DECODE_DRAFT_SIZE = (256, 256) # JPEG decodes at reduced scale, still larger than the 64px model input
|
||||||
|
|
||||||
torch.set_num_threads(TORCH_THREADS)
|
|
||||||
Image.MAX_IMAGE_PIXELS = MAX_IMAGE_PIXELS
|
Image.MAX_IMAGE_PIXELS = MAX_IMAGE_PIXELS
|
||||||
|
|
||||||
origins = [
|
origins = [
|
||||||
@@ -166,8 +170,14 @@ class RateLimiter:
|
|||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
predict_limiter = RateLimiter(limit=10, window=60)
|
predict_limiter = RateLimiter(
|
||||||
review_limiter = RateLimiter(limit=5, window=60)
|
limit=max(1, int(os.getenv("PREDICT_RATE_LIMIT", "10"))),
|
||||||
|
window=max(1.0, float(os.getenv("PREDICT_RATE_WINDOW", "60"))),
|
||||||
|
)
|
||||||
|
review_limiter = RateLimiter(
|
||||||
|
limit=max(1, int(os.getenv("REVIEW_RATE_LIMIT", "5"))),
|
||||||
|
window=max(1.0, float(os.getenv("REVIEW_RATE_WINDOW", "60"))),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def rate_limit(limiter: RateLimiter):
|
def rate_limit(limiter: RateLimiter):
|
||||||
@@ -181,15 +191,28 @@ class Busy(Exception):
|
|||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
||||||
class InferenceGate:
|
def init_inference_thread():
|
||||||
"""One worker thread runs all inference. At most `max_pending` jobs
|
# Must run inside each worker thread: with OpenMP the thread count is a
|
||||||
(running + waiting) are admitted; everything else is rejected at once
|
# per-thread setting, so setting it once in the main thread is not enough.
|
||||||
instead of queueing up and eating memory."""
|
torch.set_num_threads(TORCH_THREADS)
|
||||||
|
|
||||||
def __init__(self, max_pending: int):
|
|
||||||
self.max_pending = max_pending
|
class InferenceGate:
|
||||||
|
"""Runs predictions on `parallel` worker threads. At most
|
||||||
|
`parallel + queued` jobs (running + waiting) are admitted; everything
|
||||||
|
else is rejected at once instead of queueing up and eating memory.
|
||||||
|
|
||||||
|
The models are in eval mode under torch.inference_mode(), so several
|
||||||
|
threads can safely run forward passes on the same model object."""
|
||||||
|
|
||||||
|
def __init__(self, parallel: int, queued: int):
|
||||||
|
self.max_pending = parallel + queued
|
||||||
self.pending = 0
|
self.pending = 0
|
||||||
self.executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="inference")
|
self.executor = ThreadPoolExecutor(
|
||||||
|
max_workers=parallel,
|
||||||
|
thread_name_prefix="inference",
|
||||||
|
initializer=init_inference_thread,
|
||||||
|
)
|
||||||
|
|
||||||
async def run(self, fn, *args):
|
async def run(self, fn, *args):
|
||||||
if self.pending >= self.max_pending:
|
if self.pending >= self.max_pending:
|
||||||
@@ -201,7 +224,7 @@ class InferenceGate:
|
|||||||
self.pending -= 1
|
self.pending -= 1
|
||||||
|
|
||||||
|
|
||||||
gate = InferenceGate(MAX_PENDING_INFERENCES)
|
gate = InferenceGate(PARALLEL_INFERENCES, QUEUED_INFERENCES)
|
||||||
|
|
||||||
|
|
||||||
class InvalidImage(Exception):
|
class InvalidImage(Exception):
|
||||||
|
|||||||
@@ -32,6 +32,15 @@ services:
|
|||||||
- cnn_network
|
- cnn_network
|
||||||
pull_policy: never
|
pull_policy: never
|
||||||
container_name: cnn_api
|
container_name: cnn_api
|
||||||
|
environment:
|
||||||
|
- PARALLEL_INFERENCES=4
|
||||||
|
- QUEUED_INFERENCES=2
|
||||||
|
- TORCH_THREADS=2
|
||||||
|
- FACE_MAX_FPS=5.5
|
||||||
|
- PREDICT_RATE_LIMIT=10
|
||||||
|
- PREDICT_RATE_WINDOW=60
|
||||||
|
- REVIEW_RATE_LIMIT=5
|
||||||
|
- REVIEW_RATE_WINDOW=60
|
||||||
|
|
||||||
|
|
||||||
cnn_website:
|
cnn_website:
|
||||||
|
|||||||
Reference in New Issue
Block a user