Release claim on jobs when cancelled

This commit is contained in:
Yiorgis Gozadinos 2026-05-25 15:56:23 +03:00
parent 14beba6786
commit da3cfe1a58
No known key found for this signature in database
5 changed files with 96 additions and 9 deletions

View file

@ -15,7 +15,7 @@
- `haiku-rag serve` renamed to `haiku-rag mcp` (only MCP is left). `--mcp-port` renamed to `--port`. Update any `claude_desktop_config.json` from `["serve", "--mcp", "--stdio"]` to `["mcp", "--stdio"]`. - `haiku-rag serve` renamed to `haiku-rag mcp` (only MCP is left). `--mcp-port` renamed to `--port`. Update any `claude_desktop_config.json` from `["serve", "--mcp", "--stdio"]` to `["mcp", "--stdio"]`.
- `document.metadata` now uses source-agnostic keys: `source_revision` (was `etag` — S3-only and never populated for FS, so periodic sweeps re-ingested every file) and `content_type` (was `contentType`, snake_case for consistency). The v0.50.0 startup migration rewrites existing documents. All four source adapters (FS, HTTP, S3, WebDAV) now write their native revision (mtime_ns, ETag, etc.) under the same key, fixing the regression where FS sources never short-circuited on unchanged files. - `document.metadata` now uses source-agnostic keys: `source_revision` (was `etag` — S3-only and never populated for FS, so periodic sweeps re-ingested every file) and `content_type` (was `contentType`, snake_case for consistency). The v0.50.0 startup migration rewrites existing documents. All four source adapters (FS, HTTP, S3, WebDAV) now write their native revision (mtime_ns, ETag, etc.) under the same key, fixing the regression where FS sources never short-circuited on unchanged files.
- Ingester pollers skip their periodic sweep when the source already has queued or claimed jobs in the queue — saves the listing round-trip (`PROPFIND` / `S3 LIST` / FS walk) when work is backed up. FS push events from `watchfiles` keep flowing during skipped sweeps. Visible in Logfire as `ingester.poller.sweep` spans with `skipped=true reason=pending_work`. - Ingester pollers skip their periodic sweep when the source already has queued or claimed jobs in the queue — saves the listing round-trip (`PROPFIND` / `S3 LIST` / FS walk) when work is backed up. FS push events from `watchfiles` keep flowing during skipped sweeps. Visible in Logfire as `ingester.poller.sweep` spans with `skipped=true reason=pending_work`.
- Ingester now drains in-flight jobs on `SIGINT` / `SIGTERM` up to `workers.shutdown_grace_s` (default 60s) before cancelling. Cancelled jobs stay `claimed` and the reaper resets them on next start. Bonus: the pipeline no longer wraps `KeyboardInterrupt` / `SystemExit` / `CancelledError` as `TransientError` — those now propagate as intended. - Ingester now drains in-flight jobs on `SIGINT` / `SIGTERM` up to `workers.shutdown_grace_s` (default 60s) before cancelling. Cancelled jobs release their claim back to `queued` (and decrement `attempts` since a cancel isn't a failure) so the next process picks them up immediately instead of waiting on the reaper's `claim_timeout_s`. Bonus: the pipeline no longer wraps `KeyboardInterrupt` / `SystemExit` / `CancelledError` as `TransientError` — those now propagate as intended.
- `providers.docling_serve.base_url` now accepts a list. Jobs round-robin across the entries with each job's submit/poll/result pinned to one instance (task IDs are instance-local). The counter is per-process; for cross-process load balancing or failover, put an LB in front and pass a single URL here. - `providers.docling_serve.base_url` now accepts a list. Jobs round-robin across the entries with each job's submit/poll/result pinned to one instance (task IDs are instance-local). The counter is per-process; for cross-process load balancing or failover, put an LB in front and pass a single URL here.
- Drop `list_documents` and `get_document` from the default RAG skill's tool set; the skill now exposes only `search` and `cite`. Both tools dumped unbounded content into the agent's context (full document lists, full document bodies) and `get_document` returned no chunk_ids so its output was structurally uncitable. The analysis skill already covers these uses programmatically — `await list_documents()` and `Path('/documents/{id}/content.txt').read_text()` inside `execute_code`. The tool branches remain in `create_skill_tools` and the `skill_generator` `AVAILABLE_TOOLS` set so users can still opt in when building custom skills. - Drop `list_documents` and `get_document` from the default RAG skill's tool set; the skill now exposes only `search` and `cite`. Both tools dumped unbounded content into the agent's context (full document lists, full document bodies) and `get_document` returned no chunk_ids so its output was structurally uncitable. The analysis skill already covers these uses programmatically — `await list_documents()` and `Path('/documents/{id}/content.txt').read_text()` inside `execute_code`. The tool branches remain in `create_skill_tools` and the `skill_generator` `AVAILABLE_TOOLS` set so users can still opt in when building custom skills.

View file

@ -266,6 +266,31 @@ class JobRepo:
rows = await cursor.fetchall() rows = await cursor.fetchall()
return {row["status"]: row["n"] for row in rows} return {row["status"]: row["n"] for row in rows}
async def release_if_claimed(self, job_id: str) -> bool:
"""Reset a still-claimed job back to queued, immediately reclaimable.
Idempotent a no-op if the job already transitioned to
succeeded/dead/rescheduled. Decrements attempts to undo the increment
from `claim_next`, since a cancellation isn't a failed attempt.
Returns True if the row was released."""
now = _utcnow_iso()
async with self._lock:
async with self._conn.execute(
"""
UPDATE jobs
SET status='queued',
claimed_at=NULL,
claimed_by=NULL,
scheduled_at=?,
attempts=MAX(0, attempts - 1)
WHERE id=? AND status='claimed'
RETURNING id
""",
(now, job_id),
) as cursor:
row = await cursor.fetchone()
await self._conn.commit()
return row is not None
async def reap_stale(self, claim_timeout_seconds: int) -> int: async def reap_stale(self, claim_timeout_seconds: int) -> int:
"""Return claimed jobs whose claimed_at is older than the timeout to """Return claimed jobs whose claimed_at is older than the timeout to
the queue. Used by the reaper to recover from crashed workers.""" the queue. Used by the reaper to recover from crashed workers."""

View file

@ -118,6 +118,13 @@ class WorkerPool:
logger.info("Processing %s %s (job %s)", job.op.value, job.uri, job.id) logger.info("Processing %s %s (job %s)", job.op.value, job.uri, job.id)
try: try:
result = await run_job(self._client, job) result = await run_job(self._client, job)
except asyncio.CancelledError:
# Graceful shutdown cancelled us mid-flight. Release the claim so
# the next process can pick the job up immediately instead of
# waiting on the reaper's claim_timeout_s.
await self._jobs.release_if_claimed(job.id)
logger.info("Job %s released back to queue on cancel", job.id)
raise
except PermanentError as e: except PermanentError as e:
await self._jobs.mark_dead(job.id, str(e)) await self._jobs.mark_dead(job.id, str(e))
logger.info("Job %s dead (permanent): %s", job.id, e) logger.info("Job %s dead (permanent): %s", job.id, e)

View file

@ -372,6 +372,56 @@ async def test_reap_stale_leaves_fresh_claims_alone(jobs):
assert refreshed.status is JobStatus.CLAIMED assert refreshed.status is JobStatus.CLAIMED
# --- release_if_claimed ---
@pytest.mark.asyncio
async def test_release_if_claimed_resets_claimed_job_and_decrements_attempts(jobs):
job = await jobs.enqueue("s", "u", JobOp.UPSERT)
assert job is not None
claimed = await jobs.claim_next("w")
assert claimed is not None
assert claimed.attempts == 1
released = await jobs.release_if_claimed(job.id)
assert released is True
refreshed = await jobs.get_job(job.id)
assert refreshed is not None
assert refreshed.status is JobStatus.QUEUED
assert refreshed.claimed_at is None
assert refreshed.claimed_by is None
assert refreshed.attempts == 0
@pytest.mark.asyncio
async def test_release_if_claimed_noop_on_already_queued(jobs):
job = await jobs.enqueue("s", "u", JobOp.UPSERT)
assert job is not None
released = await jobs.release_if_claimed(job.id)
assert released is False
refreshed = await jobs.get_job(job.id)
assert refreshed is not None
assert refreshed.status is JobStatus.QUEUED
assert refreshed.attempts == 0
@pytest.mark.asyncio
async def test_release_if_claimed_noop_on_succeeded(jobs):
job = await jobs.enqueue("s", "u", JobOp.UPSERT)
assert job is not None
claimed = await jobs.claim_next("w")
assert claimed is not None
await jobs.mark_succeeded(job.id)
released = await jobs.release_if_claimed(job.id)
assert released is False
refreshed = await jobs.get_job(job.id)
assert refreshed is not None
assert refreshed.status is JobStatus.SUCCEEDED
# --- list / counts --- # --- list / counts ---

View file

@ -253,9 +253,10 @@ async def test_shutdown_grace_lets_inflight_job_complete(client, jobs, sync):
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_shutdown_grace_timeout_cancels_long_job(client, jobs, sync): async def test_shutdown_grace_timeout_releases_claim(client, jobs, sync):
"""When grace elapses, wait_for raises TimeoutError and the job stays """When grace elapses and the worker is cancelled mid-job, the claim is
'claimed' for the reaper to reset later.""" released back to 'queued' so the next process can pick it up immediately
no waiting on the reaper's claim_timeout_s."""
cancelled = asyncio.Event() cancelled = asyncio.Event()
async def _hangs_forever(*args, **kwargs): async def _hangs_forever(*args, **kwargs):
@ -267,7 +268,8 @@ async def test_shutdown_grace_timeout_cancels_long_job(client, jobs, sync):
return Document(id="doc", content="x", uri="u") return Document(id="doc", content="x", uri="u")
client.create_document_from_source.side_effect = _hangs_forever client.create_document_from_source.side_effect = _hangs_forever
await jobs.enqueue("src", "u", JobOp.UPSERT) job = await jobs.enqueue("src", "u", JobOp.UPSERT)
assert job is not None
pool = _pool(client, jobs, sync, worker_count=1, max_concurrent=1) pool = _pool(client, jobs, sync, worker_count=1, max_concurrent=1)
await pool.start() await pool.start()
@ -277,10 +279,13 @@ async def test_shutdown_grace_timeout_cancels_long_job(client, jobs, sync):
await asyncio.wait_for(pool.stop(), timeout=0.2) await asyncio.wait_for(pool.stop(), timeout=0.2)
assert cancelled.is_set() assert cancelled.is_set()
counts = await jobs.counts_by_status() refreshed = await jobs.get_job(job.id)
# Job was cancelled mid-_process before mark_succeeded/dead could fire, assert refreshed is not None
# so it stays in 'claimed' for the reaper to pick up later. assert refreshed.status is JobStatus.QUEUED
assert counts.get("claimed", 0) == 1 assert refreshed.claimed_by is None
# claim_next incremented attempts to 1; release_if_claimed decremented it
# back to 0 because a cancellation isn't a failed attempt.
assert refreshed.attempts == 0
@pytest.mark.asyncio @pytest.mark.asyncio