Collapse worker_count and max_concurrent into worker_count
This commit is contained in:
parent
6f74fe67b3
commit
1c710433c7
8 changed files with 27 additions and 40 deletions
|
|
@ -148,7 +148,6 @@ Bearer-token auth can replace HTTP Basic via the standard `headers` map:
|
||||||
ingester:
|
ingester:
|
||||||
workers:
|
workers:
|
||||||
worker_count: 4
|
worker_count: 4
|
||||||
max_concurrent: 4
|
|
||||||
poll_idle_interval_s: 1.0
|
poll_idle_interval_s: 1.0
|
||||||
claim_timeout_s: 1800
|
claim_timeout_s: 1800
|
||||||
reaper_interval_s: 60
|
reaper_interval_s: 60
|
||||||
|
|
@ -160,8 +159,9 @@ ingester:
|
||||||
jitter: 0.25 # ±25%
|
jitter: 0.25 # ±25%
|
||||||
```
|
```
|
||||||
|
|
||||||
The worker pool runs `worker_count` async workers behind a shared
|
The worker pool runs `worker_count` async workers, each processing one
|
||||||
`max_concurrent` semaphore. Jobs that hit a `TransientError` are
|
job at a time. `worker_count` is therefore also the maximum number of
|
||||||
|
concurrent in-flight jobs. Jobs that hit a `TransientError` are
|
||||||
rescheduled with exponential backoff plus jitter, up to `max_attempts`,
|
rescheduled with exponential backoff plus jitter, up to `max_attempts`,
|
||||||
then land in the dead-letter queue. `PermanentError` (unsupported
|
then land in the dead-letter queue. `PermanentError` (unsupported
|
||||||
extension, 4xx HTTP except 408/429, etc.) skips retry entirely.
|
extension, 4xx HTTP except 408/429, etc.) skips retry entirely.
|
||||||
|
|
@ -185,9 +185,8 @@ once `claim_timeout_s` elapses.
|
||||||
|
|
||||||
- `claim_timeout_s` must exceed the longest legitimate job duration; a
|
- `claim_timeout_s` must exceed the longest legitimate job duration; a
|
||||||
shorter value lets the reaper resurrect in-flight jobs.
|
shorter value lets the reaper resurrect in-flight jobs.
|
||||||
- `worker_count <= max_concurrent`; extras stall in the semaphore.
|
- `worker_count` should match downstream capacity. docling-serve
|
||||||
- `max_concurrent` should match downstream capacity. docling-serve
|
processes one task per instance, so `worker_count` above the number
|
||||||
processes one task per instance, so `max_concurrent` above the number
|
|
||||||
of `providers.docling_serve.base_url` entries over-subscribes the
|
of `providers.docling_serve.base_url` entries over-subscribes the
|
||||||
fleet — extra submissions queue inside docling-serve and inflate
|
fleet — extra submissions queue inside docling-serve and inflate
|
||||||
`claimed_at` duration toward `claim_timeout_s`.
|
`claimed_at` duration toward `claim_timeout_s`.
|
||||||
|
|
|
||||||
|
|
@ -10,6 +10,11 @@ ingester:
|
||||||
# Queue lives next to the LanceDB so both persist in the data volume.
|
# Queue lives next to the LanceDB so both persist in the data volume.
|
||||||
queue:
|
queue:
|
||||||
path: /data/ingester.db
|
path: /data/ingester.db
|
||||||
|
workers:
|
||||||
|
# Match the docling-serve fleet size below — each instance processes
|
||||||
|
# one task at a time, so more workers would just queue submissions
|
||||||
|
# inside docling-serve.
|
||||||
|
worker_count: 2
|
||||||
api:
|
api:
|
||||||
# Bind to all interfaces inside the container so docker port-mapping works.
|
# Bind to all interfaces inside the container so docker port-mapping works.
|
||||||
host: 0.0.0.0
|
host: 0.0.0.0
|
||||||
|
|
|
||||||
|
|
@ -291,14 +291,10 @@ class WorkerConfig(BaseModel):
|
||||||
worker_count: int = Field(
|
worker_count: int = Field(
|
||||||
default=4,
|
default=4,
|
||||||
description="Number of async worker tasks pulling from the queue. "
|
description="Number of async worker tasks pulling from the queue. "
|
||||||
"Each worker holds at most one job at a time. Should be <= "
|
"Each worker holds at most one job at a time, so worker_count is "
|
||||||
"max_concurrent; extra workers stall in the semaphore queue.",
|
"also the maximum number of concurrent in-flight jobs. Size to the "
|
||||||
)
|
"slowest shared downstream — typically the docling-serve fleet or "
|
||||||
max_concurrent: int = Field(
|
"the embedding endpoint's request budget.",
|
||||||
default=4,
|
|
||||||
description="Upper bound on jobs running concurrently across all "
|
|
||||||
"workers. Sized to the slowest shared downstream — typically the "
|
|
||||||
"docling-serve fleet or the embedding endpoint's request budget.",
|
|
||||||
)
|
)
|
||||||
poll_idle_interval_s: float = Field(
|
poll_idle_interval_s: float = Field(
|
||||||
default=1.0,
|
default=1.0,
|
||||||
|
|
|
||||||
|
|
@ -77,7 +77,6 @@ class IngesterApp:
|
||||||
job_repo=self._jobs,
|
job_repo=self._jobs,
|
||||||
sync_repo=self._sync,
|
sync_repo=self._sync,
|
||||||
worker_count=ingester_cfg.workers.worker_count,
|
worker_count=ingester_cfg.workers.worker_count,
|
||||||
max_concurrent=ingester_cfg.workers.max_concurrent,
|
|
||||||
retry_policy=retry,
|
retry_policy=retry,
|
||||||
poll_idle_interval_s=ingester_cfg.workers.poll_idle_interval_s,
|
poll_idle_interval_s=ingester_cfg.workers.poll_idle_interval_s,
|
||||||
claim_timeout_s=ingester_cfg.workers.claim_timeout_s,
|
claim_timeout_s=ingester_cfg.workers.claim_timeout_s,
|
||||||
|
|
|
||||||
|
|
@ -37,7 +37,6 @@ class WorkerPool:
|
||||||
job_repo: JobRepo,
|
job_repo: JobRepo,
|
||||||
sync_repo: SyncStateRepo,
|
sync_repo: SyncStateRepo,
|
||||||
worker_count: int = 4,
|
worker_count: int = 4,
|
||||||
max_concurrent: int = 4,
|
|
||||||
retry_policy: RetryPolicy | None = None,
|
retry_policy: RetryPolicy | None = None,
|
||||||
poll_idle_interval_s: float = 1.0,
|
poll_idle_interval_s: float = 1.0,
|
||||||
claim_timeout_s: int = 1800,
|
claim_timeout_s: int = 1800,
|
||||||
|
|
@ -48,7 +47,6 @@ class WorkerPool:
|
||||||
self._jobs = job_repo
|
self._jobs = job_repo
|
||||||
self._sync = sync_repo
|
self._sync = sync_repo
|
||||||
self._worker_count = worker_count
|
self._worker_count = worker_count
|
||||||
self._semaphore = asyncio.Semaphore(max_concurrent)
|
|
||||||
self._retry = retry_policy or RetryPolicy()
|
self._retry = retry_policy or RetryPolicy()
|
||||||
self._poll_idle_s = poll_idle_interval_s
|
self._poll_idle_s = poll_idle_interval_s
|
||||||
self._claim_timeout_s = claim_timeout_s
|
self._claim_timeout_s = claim_timeout_s
|
||||||
|
|
@ -132,14 +130,11 @@ class WorkerPool:
|
||||||
if self._breaker.is_open:
|
if self._breaker.is_open:
|
||||||
await self._sleep_or_stop(self._poll_idle_s)
|
await self._sleep_or_stop(self._poll_idle_s)
|
||||||
continue
|
continue
|
||||||
# Semaphore wraps claim + process: at most max_concurrent workers
|
job = await self._jobs.claim_next(worker_id)
|
||||||
# hold a claimed job at any one time.
|
if job is None:
|
||||||
async with self._semaphore:
|
await self._sleep_or_stop(self._poll_idle_s)
|
||||||
job = await self._jobs.claim_next(worker_id)
|
continue
|
||||||
if job is None:
|
await self._process(job)
|
||||||
await self._sleep_or_stop(self._poll_idle_s)
|
|
||||||
continue
|
|
||||||
await self._process(job)
|
|
||||||
|
|
||||||
async def _reaper_loop(self) -> None:
|
async def _reaper_loop(self) -> None:
|
||||||
while not self._stop.is_set():
|
while not self._stop.is_set():
|
||||||
|
|
|
||||||
|
|
@ -114,7 +114,6 @@ ingester:
|
||||||
poll_interval_s: 86400
|
poll_interval_s: 86400
|
||||||
workers:
|
workers:
|
||||||
worker_count: 8
|
worker_count: 8
|
||||||
max_concurrent: 4
|
|
||||||
api:
|
api:
|
||||||
enabled: false
|
enabled: false
|
||||||
"""
|
"""
|
||||||
|
|
|
||||||
|
|
@ -102,7 +102,6 @@ async def test_e2e_initial_sweep_lands_succeeded_jobs(tmp_path, jobs, sync):
|
||||||
job_repo=jobs,
|
job_repo=jobs,
|
||||||
sync_repo=sync,
|
sync_repo=sync,
|
||||||
worker_count=2,
|
worker_count=2,
|
||||||
max_concurrent=2,
|
|
||||||
poll_idle_interval_s=0.05,
|
poll_idle_interval_s=0.05,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
@ -171,7 +170,6 @@ async def test_e2e_handles_url_encoded_special_chars_in_path(tmp_path, jobs, syn
|
||||||
job_repo=jobs,
|
job_repo=jobs,
|
||||||
sync_repo=sync,
|
sync_repo=sync,
|
||||||
worker_count=1,
|
worker_count=1,
|
||||||
max_concurrent=1,
|
|
||||||
poll_idle_interval_s=0.05,
|
poll_idle_interval_s=0.05,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
@ -268,7 +266,6 @@ async def test_pre_existing_job_resolves_through_configured_source(
|
||||||
job_repo=jobs,
|
job_repo=jobs,
|
||||||
sync_repo=sync,
|
sync_repo=sync,
|
||||||
worker_count=1,
|
worker_count=1,
|
||||||
max_concurrent=1,
|
|
||||||
poll_idle_interval_s=0.05,
|
poll_idle_interval_s=0.05,
|
||||||
sources=manager.sources,
|
sources=manager.sources,
|
||||||
)
|
)
|
||||||
|
|
|
||||||
|
|
@ -50,7 +50,6 @@ def _pool(client, jobs, sync, **kwargs) -> WorkerPool:
|
||||||
job_repo=jobs,
|
job_repo=jobs,
|
||||||
sync_repo=sync,
|
sync_repo=sync,
|
||||||
worker_count=kwargs.pop("worker_count", 2),
|
worker_count=kwargs.pop("worker_count", 2),
|
||||||
max_concurrent=kwargs.pop("max_concurrent", 2),
|
|
||||||
poll_idle_interval_s=kwargs.pop("poll_idle_interval_s", 0.05),
|
poll_idle_interval_s=kwargs.pop("poll_idle_interval_s", 0.05),
|
||||||
reaper_interval_s=kwargs.pop("reaper_interval_s", 60),
|
reaper_interval_s=kwargs.pop("reaper_interval_s", 60),
|
||||||
claim_timeout_s=kwargs.pop("claim_timeout_s", 60),
|
claim_timeout_s=kwargs.pop("claim_timeout_s", 60),
|
||||||
|
|
@ -233,7 +232,7 @@ async def test_workers_drain_queue_after_start(client, jobs, sync):
|
||||||
for i in range(5):
|
for i in range(5):
|
||||||
await jobs.enqueue("src", f"u{i}", JobOp.UPSERT)
|
await jobs.enqueue("src", f"u{i}", JobOp.UPSERT)
|
||||||
|
|
||||||
pool = _pool(client, jobs, sync, worker_count=3, max_concurrent=3)
|
pool = _pool(client, jobs, sync, worker_count=3)
|
||||||
await pool.start()
|
await pool.start()
|
||||||
try:
|
try:
|
||||||
# Wait until everything is succeeded or until a deadline.
|
# Wait until everything is succeeded or until a deadline.
|
||||||
|
|
@ -268,7 +267,7 @@ async def test_shutdown_grace_lets_inflight_job_complete(client, jobs, sync):
|
||||||
client.create_document_from_source.side_effect = _slow_then_finish
|
client.create_document_from_source.side_effect = _slow_then_finish
|
||||||
await jobs.enqueue("src", "u", JobOp.UPSERT)
|
await jobs.enqueue("src", "u", JobOp.UPSERT)
|
||||||
|
|
||||||
pool = _pool(client, jobs, sync, worker_count=1, max_concurrent=1)
|
pool = _pool(client, jobs, sync, worker_count=1)
|
||||||
await pool.start()
|
await pool.start()
|
||||||
# Yield long enough for the worker to claim and enter _process.
|
# Yield long enough for the worker to claim and enter _process.
|
||||||
await asyncio.sleep(0.02)
|
await asyncio.sleep(0.02)
|
||||||
|
|
@ -298,7 +297,7 @@ async def test_shutdown_grace_timeout_releases_claim(client, jobs, sync):
|
||||||
job = await jobs.enqueue("src", "u", JobOp.UPSERT)
|
job = await jobs.enqueue("src", "u", JobOp.UPSERT)
|
||||||
assert job is not None
|
assert job is not None
|
||||||
|
|
||||||
pool = _pool(client, jobs, sync, worker_count=1, max_concurrent=1)
|
pool = _pool(client, jobs, sync, worker_count=1)
|
||||||
await pool.start()
|
await pool.start()
|
||||||
await asyncio.sleep(0.05)
|
await asyncio.sleep(0.05)
|
||||||
|
|
||||||
|
|
@ -334,7 +333,7 @@ async def test_worker_loses_claim_to_reaper_does_not_write_sync_state(
|
||||||
await jobs.reap_stale(claim_timeout_seconds=0)
|
await jobs.reap_stale(claim_timeout_seconds=0)
|
||||||
await jobs.claim_next("worker-B")
|
await jobs.claim_next("worker-B")
|
||||||
|
|
||||||
pool = _pool(client, jobs, sync, worker_count=1, max_concurrent=1)
|
pool = _pool(client, jobs, sync, worker_count=1)
|
||||||
# Drive A's _process directly with A's (now stale) Job snapshot.
|
# Drive A's _process directly with A's (now stale) Job snapshot.
|
||||||
await pool._process(claimed_by_a)
|
await pool._process(claimed_by_a)
|
||||||
|
|
||||||
|
|
@ -376,7 +375,7 @@ async def test_cancel_cleanup_survives_second_cancel(client, jobs, sync, monkeyp
|
||||||
job = await jobs.enqueue("src", "u", JobOp.UPSERT)
|
job = await jobs.enqueue("src", "u", JobOp.UPSERT)
|
||||||
assert job is not None
|
assert job is not None
|
||||||
|
|
||||||
pool = _pool(client, jobs, sync, worker_count=1, max_concurrent=1)
|
pool = _pool(client, jobs, sync, worker_count=1)
|
||||||
await pool.start()
|
await pool.start()
|
||||||
try:
|
try:
|
||||||
await asyncio.sleep(0.05)
|
await asyncio.sleep(0.05)
|
||||||
|
|
@ -425,7 +424,7 @@ async def test_drain_pending_releases_waits_for_orphan_releases(
|
||||||
job = await jobs.enqueue("src", "u", JobOp.UPSERT)
|
job = await jobs.enqueue("src", "u", JobOp.UPSERT)
|
||||||
assert job is not None
|
assert job is not None
|
||||||
|
|
||||||
pool = _pool(client, jobs, sync, worker_count=1, max_concurrent=1)
|
pool = _pool(client, jobs, sync, worker_count=1)
|
||||||
await pool.start()
|
await pool.start()
|
||||||
try:
|
try:
|
||||||
await asyncio.sleep(0.05)
|
await asyncio.sleep(0.05)
|
||||||
|
|
@ -454,7 +453,7 @@ async def test_drain_pending_releases_waits_for_orphan_releases(
|
||||||
async def test_drain_pending_releases_with_no_orphans_is_noop(client, jobs, sync):
|
async def test_drain_pending_releases_with_no_orphans_is_noop(client, jobs, sync):
|
||||||
"""Common case: nothing to drain — drain returns 0 immediately, no
|
"""Common case: nothing to drain — drain returns 0 immediately, no
|
||||||
asyncio.wait against an empty set."""
|
asyncio.wait against an empty set."""
|
||||||
pool = _pool(client, jobs, sync, worker_count=1, max_concurrent=1)
|
pool = _pool(client, jobs, sync, worker_count=1)
|
||||||
assert await pool.drain_pending_releases() == 0
|
assert await pool.drain_pending_releases() == 0
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -505,9 +504,7 @@ async def test_breaker_opens_after_n_consecutive_transient_failures(client, jobs
|
||||||
async def test_breaker_pauses_worker_loop_claims(client, jobs, sync):
|
async def test_breaker_pauses_worker_loop_claims(client, jobs, sync):
|
||||||
"""Worker loop honours the breaker: claim_next is not called while
|
"""Worker loop honours the breaker: claim_next is not called while
|
||||||
is_open, so queued jobs stay queued until the breaker closes."""
|
is_open, so queued jobs stay queued until the breaker closes."""
|
||||||
pool = _pool(
|
pool = _pool(client, jobs, sync, worker_count=1, poll_idle_interval_s=0.02)
|
||||||
client, jobs, sync, worker_count=1, max_concurrent=1, poll_idle_interval_s=0.02
|
|
||||||
)
|
|
||||||
# Force the breaker open without touching the queue.
|
# Force the breaker open without touching the queue.
|
||||||
for _ in range(10):
|
for _ in range(10):
|
||||||
pool._breaker.record_failure()
|
pool._breaker.record_failure()
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue