diff --git a/.env.example b/.env.example index f228694..99b4f42 100644 --- a/.env.example +++ b/.env.example @@ -77,3 +77,21 @@ MINIO_CONSOLE_PORT=9001 STORAGE_BACKEND=s3 LOCAL_STORAGE_ROOT=./data LOCAL_STORAGE_HOST_PATH=./local_storage_data + +# Operational services (see doc 30) +# Staging→canonical sync: `mc mirror` runs on the worker (Celery task + Beat cron). +# STAGING_LOCATION accepts a local path or an s3:// URI. +STAGING_LOCATION=./data/staging +STAGING_SYNC_INTERVAL_SECONDS=300 +STAGING_SYNC_SOFT_TIME_LIMIT_SECONDS=1800 +# Per-task time-limit contract (doc 30): every Celery task must declare its own +# ceiling explicitly (enforced at class-definition time, see base_task.py). +INGEST_BATCH_SOFT_TIME_LIMIT_SECONDS=7200 +MOSAIC_DRIZZLE_SOFT_TIME_LIMIT_SECONDS=3600 +CLI_TOOL_SOFT_TIME_LIMIT_SECONDS=3600 +DB_BACKUP_SOFT_TIME_LIMIT_SECONDS=3600 +DLQ_DUMP_SOFT_TIME_LIMIT_SECONDS=30 +RECONCILE_STUCK_JOBS_SOFT_TIME_LIMIT_SECONDS=120 +ENABLE_STAGING_SYNC_CRON=true +# Stuck-job watchdog: seconds an IN_PROCESS job may age before it is failed. +JOB_STALENESS_TIMEOUT_SECONDS=3600 diff --git a/docker-compose.prod.yml b/docker-compose.prod.yml index a1450c8..ecf0817 100644 --- a/docker-compose.prod.yml +++ b/docker-compose.prod.yml @@ -104,6 +104,29 @@ services: networks: - diffpype_net + beat: + image: ghcr.io/davecoulter/diffpype_claude-worker:${IMAGE_TAG:-main} + # Single database-backed Celery Beat process (see docker-compose.yml for the + # rationale). Consumes no task queues; publishes scheduled tasks to Redis. + command: celery -A src.worker.celery_app beat -S sqlalchemy_celery_beat.schedulers:DatabaseScheduler --loglevel=info + environment: + DATABASE_URL: postgresql+psycopg://${POSTGRES_USER}:${POSTGRES_PASSWORD}@db:5432/${POSTGRES_DB} + REDIS_URL: redis://redis:6379/0 + LOG_LEVEL: ${LOG_LEVEL:-INFO} + OTEL_EXPORTER_OTLP_ENDPOINT: ${OTEL_EXPORTER_OTLP_ENDPOINT} + OTEL_SERVICE_NAME: diffpype-beat + STAGING_SYNC_INTERVAL_SECONDS: ${STAGING_SYNC_INTERVAL_SECONDS:-300} + ENABLE_STAGING_SYNC_CRON: ${ENABLE_STAGING_SYNC_CRON:-true} + JOB_STALENESS_TIMEOUT_SECONDS: ${JOB_STALENESS_TIMEOUT_SECONDS:-3600} + depends_on: + db: + condition: service_healthy + redis: + condition: service_healthy + restart: unless-stopped + networks: + - diffpype_net + networks: diffpype_net: name: diffpype_net diff --git a/docker-compose.yml b/docker-compose.yml index d7cd0df..7f1ce17 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -170,6 +170,35 @@ services: networks: - diffpype_net + beat: + build: + context: . + dockerfile: docker/worker.Dockerfile + # Dedicated Celery Beat process (exactly one, ever) using the database-backed + # scheduler so schedules edited at runtime via SQLAdmin take effect without a + # restart. Not a --beat flag on a worker: that would double-fire schedules if + # the worker ever scaled past one replica. Requires the scheduler tables to + # exist (run `alembic upgrade head` first); it consumes no task queues. + command: celery -A src.worker.celery_app beat -S sqlalchemy_celery_beat.schedulers:DatabaseScheduler --loglevel=info + environment: + DATABASE_URL: postgresql+psycopg://${POSTGRES_USER}:${POSTGRES_PASSWORD}@db:5432/${POSTGRES_DB} + REDIS_URL: redis://redis:6379/0 + LOG_LEVEL: ${LOG_LEVEL:-INFO} + OTEL_EXPORTER_OTLP_ENDPOINT: ${OTEL_EXPORTER_OTLP_ENDPOINT:-http://jaeger:4317} + OTEL_SERVICE_NAME: diffpype-beat + STAGING_SYNC_INTERVAL_SECONDS: ${STAGING_SYNC_INTERVAL_SECONDS:-300} + ENABLE_STAGING_SYNC_CRON: ${ENABLE_STAGING_SYNC_CRON:-true} + JOB_STALENESS_TIMEOUT_SECONDS: ${JOB_STALENESS_TIMEOUT_SECONDS:-3600} + volumes: + - ./src:/app/src + depends_on: + db: + condition: service_healthy + redis: + condition: service_healthy + networks: + - diffpype_net + flower: image: mher/flower:2.0.1 environment: diff --git a/docker/worker.Dockerfile b/docker/worker.Dockerfile index a3a455e..603a43c 100644 --- a/docker/worker.Dockerfile +++ b/docker/worker.Dockerfile @@ -2,6 +2,17 @@ FROM python:3.12-slim COPY --from=ghcr.io/astral-sh/uv:latest /uv /bin/uv +# MinIO client (`mc`), used by run_staging_sync's `mc mirror` staging->canonical +# sync (doc 30 §1). Arch-aware so the image builds on both CI's linux/amd64 and +# an Apple-Silicon linux/arm64 build. Only the worker image gets `mc` — the api +# image never runs the sync (it is dispatched to the worker). +RUN apt-get update \ + && apt-get install -y --no-install-recommends curl ca-certificates \ + && curl -fsSL "https://dl.min.io/client/mc/release/linux-$(dpkg --print-architecture)/mc" \ + -o /usr/local/bin/mc \ + && chmod +x /usr/local/bin/mc \ + && rm -rf /var/lib/apt/lists/* + WORKDIR /app # Layer 1: install deps only (cached unless pyproject.toml/uv.lock change). diff --git a/docs/architecture/30_operational_services_and_watchdog.md b/docs/architecture/30_operational_services_and_watchdog.md new file mode 100644 index 0000000..4fa4617 --- /dev/null +++ b/docs/architecture/30_operational_services_and_watchdog.md @@ -0,0 +1,158 @@ +##### 30: Operational Services, Tessellation Modes & Stuck-Job Watchdog +**Version:** 0.5 + +###### Preamble +This document establishes four operational service capabilities building directly on top of Document 29's database driver and spatial foundation: decoupled storage staging synchronization via an `mc`-based diff-sync Celery task, configurable survey tile-tessellation modes, an automated stuck-job reconciliation watchdog, and dynamic database-backed Celery Beat management via SQLAdmin. + +###### 1. Storage Model B (Staging → Canonical Sync) +* **Directive:** Decouple ingest pipelines from raw file transfer by providing an explicit, idempotent staging-to-canonical storage synchronization primitive. +* **Behavior:** + * **Settings & Configuration:** Add `STAGING_LOCATION` (`str`, default `"./data/staging"`, accepting local filesystem paths or `s3://` URIs), `STAGING_SYNC_INTERVAL_SECONDS` (`int`, default `300`), and `ENABLE_STAGING_SYNC_CRON` (`bool`, default `True`) to `src/core/config.py`. + * **Service Primitive:** In `src/services/storage_service.py`, implement `sync_staging_to_canonical(staging_location: str, canonical_prefix: str) -> None` that invokes `mc mirror` as a subprocess, reading its `--json` per-file event stream incrementally and logging a structured line per file copied — never `subprocess.run(capture_output=True)`, which would buffer the entire transfer in memory before any progress is visible (the same class of bug already fixed once in `run_ingest_batch`, doc 29). + * **Dispatch & Dependency:** Runs as a Celery task (worker-only; never invoked synchronously inside the API process). Add the `mc` (MinIO client) binary to `docker/worker.Dockerfile` only, not `docker/api.Dockerfile`. + * **No New Status Entity:** `mc mirror` is itself diff-based and idempotent, and Celery's existing `task_acks_late`/`task_reject_on_worker_lost` config already make a redelivered retry after a worker crash safe and cheap — a re-run just skips whatever `mc` already copied. This task deliberately does not get its own DB-tracked status row (unlike `IngestBatch`/`Level3Mosaic`) and is not part of the watchdog's entity registry. + * **LocalStorageService Bug Fixes:** In `LocalStorageService.list_prefix`, fix string-prefix matching so partial-filename prefixes match relative file paths properly instead of requiring an existing directory. Exclude dot-prefixed hidden files (e.g., `.tmp`) from `list_prefix` results to prevent ingesting partial transfers. + * **API/CLI/Beat Parity:** Expose via `POST /api/v1/storage/sync`, `diffpype-manage sync-staging --staging-prefix --canonical-prefix `, and a periodic Celery Beat task `sync_staging_cron`. +* **Testing:** Add unit tests verifying `LocalStorageService.list_prefix` string-prefix filtering and dot-file exclusion. Add an integration test verifying `sync_staging_to_canonical` executes a clean, non-duplicating real `mc mirror` sync against MinIO (already provisioned in CI). + +###### 2. Configurable Tile-Tessellation Modes & Shared Spatial Utilities +* **Directive:** Provide a shared MOC union utility and support flexible region sources and tile keep policies for ongoing survey tiling. +* **Behavior:** + * **Shared Spatial Utility:** In `src/db/spatial_types.py`, extract a public helper `union_mocs(mocs: Sequence[MOC]) -> MOC`. Update `mosaic_service` and `tile_service` to consume this shared function. + * **Request Contract:** In `src/services/tile_service.py`, update `TileTessellationRequest` to accept: + * `region_source`: Enum (`cone` | `project_footprint` | `bounding_box`). + * `overlap_only`: `bool = True` (when `False`, materializes the full bounding grid to pre-provision tiles for incoming survey data). + * Conditional required parameters enforced via Pydantic root validators: + * `cone`: `ra`, `decl`, `radius_deg` + * `project_footprint`: `project_id` + * `bounding_box`: `min_ra`, `max_ra`, `min_decl`, `max_decl` + * **Breaking Change:** This replaces `TileTessellationRequest.moc_to_tile` (previously a pre-computed MOC range list) entirely — no backward-compatible fallback mode. + * **Region Resolution:** When `region_source == "project_footprint"`, `generate_tile_tessellation` queries for all `Level2Calibration` footprints under `project_id`, unions them via `union_mocs`, and derives the bounding region automatically. + * **API/CLI Parity:** Expose parameters across `POST /api/v1/tiles/tessellate` and `diffpype-manage tessellate-tiles` (`--region-source`, `--min-ra`, `--max-ra`, etc.). +* **Testing:** Add Pydantic validation unit tests asserting that missing conditional parameters raise HTTP 422 errors for each mode. Test tessellation against a project with known calibrations, asserting that `overlap_only=False` materializes the full grid including boundary tiles with no calibration overlap, versus `True` (default), which trims to only tiles that actually intersect the resolved region. + +###### 3. Stuck-Job Watchdog & JobConfiguration Linkage +* **Directive:** Automatically detect and fail jobs left in `IN_PROCESS` status due to uncatchable worker crashes or OOM kills, utilizing `JobConfiguration` provenance records. +* **Behavior:** + * **Schema Change:** Add `JobConfiguration.task_name: str | None` — records which Celery task (or CLI tool, for the existing `execute_cli_tool` path) a row's `job_kwargs`/`execution_command` correspond to. `job_configuration_id` on `IngestBatch`/`Level3Mosaic` stays `nullable=True` (no backfill needed) — the "every new dispatch populates it" invariant is enforced at the service layer, not the database. + * **Dispatch Linkage:** Add a shared `job_service.create_job_configuration(db: Session, user_id: int, task_name: str, job_kwargs: dict | None) -> JobConfiguration` helper. Update `ingest_service.create_ingest_batch` and `mosaic_service.create_mosaic` to call it and link the resulting `job_configuration_id` on every newly created `IngestBatch`/`Level3Mosaic` at dispatch time, rather than each service instantiating `JobConfiguration` independently. + * **Scope Note (ties to GitHub issue #37):** Dispatch/queue-level status stays exactly where it already lives — `JobStatus` on each tracked entity — this doc does not consolidate status onto `JobConfiguration`. Object/domain-level pipeline-stage status (e.g. "has this calibration been 1/f-corrected") is a distinct, out-of-scope concept with no current implementation need; it belongs to a future JWST-pipeline-specific doc. + * **Static Entity Registry:** In `src/services/job_service.py`, declare a static registry naming tracked job entities (`IngestBatch`, `Level3Mosaic`) and their corresponding status and `job_configuration_id` columns. + * **Service Method:** Implement `reconcile_stuck_jobs(db: Session, staleness_timeout_seconds: int = settings.job_staleness_timeout_seconds) -> list[dict]`. The function queries all registered entities in `IN_PROCESS` status whose `updated_at` timestamp exceeds the threshold. If an entity's linked `JobConfiguration.job_kwargs` contains a custom `staleness_timeout_seconds` override, honor that specific value over the global default. + * **Per-Job Staleness Override — Write Path (added post-implementation, not yet built — see Logs):** The override above was wired only on the read side; nothing ever wrote it. Add an optional `staleness_timeout_seconds` parameter to every dispatch path that creates a watchdog-tracked entity — today that's `ingest_service.create_ingest_batch` and `mosaic_service.create_mosaic` — threaded through to `create_job_configuration`'s `job_kwargs`, and exposed at both the API (`IngestRequest`, `MosaicCreate`) and CLI (`ingest`, `create-mosaic`) boundaries, per the API/CLI Parity guardrail. **Two open decisions before this is built** (see chat for options/recommendation): (1) whether the override is only written to `job_kwargs` when the caller explicitly supplies a non-default value (deferring to whatever the *current* global `JOB_STALENESS_TIMEOUT_SECONDS` is at reconcile time for every other job), or always written at dispatch time using the global default as of that moment (freezing it for that job even if the global setting changes later); (2) whether this also extends to `dispatch_staging_sync`/`run_staging_sync`, which isn't a watchdog-tracked entity today. + * **Transaction Safety:** `reconcile_stuck_jobs` MUST call `db.rollback()` before executing its status update writes to clear any prior failed transaction state. + * **API/CLI/Beat Parity:** Expose via `POST /api/v1/jobs/reconcile`, `diffpype-manage reconcile-stuck-jobs --threshold-seconds 3600`, and a periodic Celery Beat task `reconcile_stuck_jobs_cron`. +* **Testing:** Create stale `IngestBatch` and `Level3Mosaic` records linked to `JobConfiguration` rows with old `updated_at` timestamps, run `reconcile_stuck_jobs`, and assert both transition to `FAILED`. Test custom `staleness_timeout_seconds` overrides in `JobConfiguration.job_kwargs`. Once the write path above is built: test that `ingest`/`create-mosaic` (API + CLI) with an explicit override actually persists it into `job_kwargs`, and that omitting it produces the doc's decided fallback behavior. + +###### 3a. Per-Task Time-Limit Contract & Stale-Redelivery Guard (added mid-`genTests` — see Logs) + +* **Directive:** Close two gaps surfaced live during this doc's `genTests`: (1) no task in this app has an enforced maximum runtime, so a hung or crashed worker gives Celery's Redis transport no principled value to use for `visibility_timeout` (the broker-level "how long before an unacked message becomes redeliverable" setting — currently unset, defaulting to 3600s); (2) once `visibility_timeout` is tuned using real per-task ceilings, the stuck-job watchdog (§3) will *reliably* mark a crashed job `FAILED` *before* the broker's own redelivery even becomes possible — meaning a later, stale redelivery must not be allowed to silently resume/overwrite a job the watchdog already gave up on. +* **Behavior:** + * **`TimeLimitedTask(celery.Task)`** (`src/worker/base_task.py`): `abstract = True` (Celery's own convention for a non-registered intermediate base). Every *concrete* subclass must declare its own `soft_time_limit_seconds` — enforced via `__init_subclass__` checking `"soft_time_limit_seconds" not in cls.__dict__` (not `getattr`), which specifically rejects a subclass that would otherwise silently inherit a value from the wrong parent (e.g. accidentally subclassing a sibling task instead of the real base) — a plain `getattr`-based check, or `abc.abstractmethod`, would *not* catch that case (verified empirically: `abc.ABC` only requires a concrete value to exist *somewhere* in the MRO, not that *this* class declared it). Exposes `soft_time_limit`/`time_limit` (`= soft_time_limit_seconds + 30`) as properties bridging to Celery's real, recognized attribute names. + * **`DiffpypeTask(TimeLimitedTask)`**: also `abstract = True`; adds the same `cls.__dict__`-enforced contract for `tracked_entity_model` (a real tracked model, or the `NOT_TRACKED` sentinel for tasks that don't own a watchdog-tracked entity — `run_staging_sync`, `dlq_dump`, `db_backup_cron`, `sync_staging_cron`, `reconcile_stuck_jobs_cron`, `execute_cli_tool`). `on_failure` is unchanged. + * **`begin_tracked_job(self, db: Session, entity_id: int) -> Base | None`** (method on `DiffpypeTask`): fetches `self.tracked_entity_model` by id; if its `status` is already `COMPLETE` or `FAILED`, logs `tracked_job_stale_redelivery_skipped` and returns `None` (the caller must bail without doing any work) — this is the guard against a redelivered task silently reviving/overwriting a row the watchdog already resolved. Otherwise transitions it to `IN_PROCESS` and returns it. Raises `TypeError` if called on a task whose `tracked_entity_model` is `NOT_TRACKED`. + * **Migration:** `run_ingest_batch` and `run_mosaic_drizzle` move onto `DiffpypeTask` (`bind=True`, `self` as first param), each via a small dedicated subclass declaring `tracked_entity_model` (`IngestBatch`/`Level3Mosaic`) and `soft_time_limit_seconds`. Their existing `db.get(...)` + manual `status = IN_PROCESS` is replaced by `self.begin_tracked_job(db, entity_id)`, returning early if it yields `None`. This also means these two tasks now get `DiffpypeTask.on_failure`'s structured logging + DLQ dispatch "for free" on any uncaught exception (including a `SoftTimeLimitExceeded` timeout) — a deliberate behavior change, not incidental, since there is no conflict with their own existing inline crash-handling (`on_failure` fires in addition to it, not instead). + * **Every other task** gets its own dedicated subclass declaring `tracked_entity_model = NOT_TRACKED` and its own `soft_time_limit_seconds` — no task is exempt from the contract, even trivial ones (`dlq_dump`). + * **`visibility_timeout`:** set via `celery_app.conf.broker_transport_options = {"visibility_timeout": + 60}`, computed from the now-enforced real ceilings rather than guessed. +* **Environment Variables (all soft-time-limit values, seconds; hard `time_limit` = soft + 30, not separately configured):** + | Name | Default | + | :--- | :--- | + | `STAGING_SYNC_SOFT_TIME_LIMIT_SECONDS` | 1800 (already existed) | + | `INGEST_BATCH_SOFT_TIME_LIMIT_SECONDS` | 7200 | + | `MOSAIC_DRIZZLE_SOFT_TIME_LIMIT_SECONDS` | 3600 | + | `CLI_TOOL_SOFT_TIME_LIMIT_SECONDS` | 3600 | + | `DB_BACKUP_SOFT_TIME_LIMIT_SECONDS` | 3600 | + | `DLQ_DUMP_SOFT_TIME_LIMIT_SECONDS` | 30 | + | `RECONCILE_STUCK_JOBS_SOFT_TIME_LIMIT_SECONDS` | 120 | +* **Testing:** Unit tests for `TimeLimitedTask`/`DiffpypeTask`'s `__init_subclass__` contract (a concrete subclass missing either required attribute raises `TypeError`; an abstract intermediate base does not; a subclass accidentally inheriting from another concrete task without redeclaring its own values raises). Unit tests for `begin_tracked_job` (already-`COMPLETE`/`FAILED` entity returns `None` and logs the skip; a genuinely stuck entity transitions to `IN_PROCESS` and is returned; `NOT_TRACKED` raises). Updated tests for `run_ingest_batch`/`run_mosaic_drizzle` reflecting the `bind=True`/`begin_tracked_job` call shape. + +###### 4. Dynamic Database-Backed Celery Beat & Dedicated Service +* **Directive:** Provision a dedicated Celery Beat container and enable administrators to create, edit, or pause periodic schedules at runtime via SQLAdmin without container restarts. +* **Behavior:** + * **Dependency:** Add `sqlalchemy-celery-beat` as a project dependency in `pyproject.toml`. (The originally-specified `celery-sqlalchemy-scheduler` is unmaintained and incompatible with SQLAlchemy 2.0 — see the implementation log below.) + * **Dedicated Container:** Add a dedicated `beat` service block in `docker-compose.yml` and `docker-compose.prod.yml` running `celery -A src.worker.celery_app beat -S sqlalchemy_celery_beat.schedulers:DatabaseScheduler --loglevel=info`. The scheduler reads its DB URL from `celery_app.conf.beat_dburi` and keeps its tables in a dedicated `celery_schema` Postgres schema. + * **SQLAdmin Views:** Register `PeriodicTask`, `IntervalSchedule`, and `CrontabSchedule` model views in `src/api/admin.py`. + * **Alembic Migration:** A hand-authored migration (0014) creates the `celery_schema` schema and delegates table/enum DDL to `sqlalchemy_celery_beat`'s own `ModelBase.metadata.create_all`, so the migration stays faithful to the package's real schema instead of a hand-transcribed copy. Because the package's tables live in their own schema, autogenerate (default `include_schemas=False`) never sees them, so no `include_object` filter is needed. Downgrade drops the schema with `CASCADE`. + * **Schedule Seeding:** `diffpype-manage seed-db` seeds initial `PeriodicTask` rows into PostgreSQL for `sync_staging_cron` and `reconcile_stuck_jobs_cron` if the schedule table is empty (idempotent no-op once any row exists, so runtime SQLAdmin edits are never clobbered by a later `seed-db`). This is a deliberate operator action only — no container's startup path invokes it today, so a genuinely fresh environment has an empty schedule until `seed-db` is run once (tracked as tech debt, see GitHub issue referenced in Logs). +* **Documentation & Topology Sync:** Update `docs/diagrams/infrastructure_topology.md` to reflect the new `beat` container and staging sync data flow. + +###### 5. Environment Variables +* **Directive:** Ensure all configurations are synchronized across `.env` and `.env.example`. +| Name | Type | Default Value | Description | +| :--- | :--- | :--- | :--- | +| STAGING_LOCATION | str | "./data/staging" | Local filesystem path or s3:// URI for incoming staging files. | +| STAGING_SYNC_INTERVAL_SECONDS | int | 300 | Default execution interval in seconds for the background staging sync task. | +| STAGING_SYNC_SOFT_TIME_LIMIT_SECONDS | int | 1800 | Soft time limit for `run_staging_sync`; on expiry the `mc` subprocess is killed and the task fails (dead-lettered, not retried) rather than hanging indefinitely. | +| ENABLE_STAGING_SYNC_CRON | bool | True | Toggle to enable/disable the periodic staging sync Beat schedule. | +| JOB_STALENESS_TIMEOUT_SECONDS | int | 3600 | Threshold in seconds before an IN_PROCESS job is marked FAILED by the watchdog. | + +###### 6. Dependencies & Packages +* **Packages:** Add `sqlalchemy-celery-beat` to `pyproject.toml` (maintained fork; replaces the originally-specified, unmaintained `celery-sqlalchemy-scheduler`). +* **Mocking:** Add `sqlalchemy_celery_beat` to `autodoc_mock_imports` in `docs/conf.py`. + +###### 7. Testing Mandates +* **Storage Sync:** Write unit tests for string-prefix and dot-file filtering in `LocalStorageService.list_prefix`. Write a unit test for `sync_staging_to_canonical` that mocks the `mc mirror` subprocess and asserts its `--json` event stream is parsed and logged per file. +* **Tessellation Validation:** Every tessellation mode (`cone`, `project_footprint`, `bounding_box`) must have explicit unit tests confirming valid requests succeed and invalid/missing parameter combinations trigger HTTP 422 errors. +* **Watchdog Sweep & Linkage:** Ingest and Mosaic dispatch tests must assert `JobConfiguration` creation and `job_configuration_id` linkage. The watchdog test suite must explicitly cover all registered entity types (`IngestBatch` and `Level3Mosaic`), verifying per-job timeout overrides in `JobConfiguration`. +* **Per-Task Time-Limit Contract & Stale-Redelivery Guard (§3a):** Unit-test `TimeLimitedTask`/`DiffpypeTask`'s `__init_subclass__` contract directly — a concrete subclass missing `soft_time_limit_seconds`/`tracked_entity_model` must raise `TypeError` at class-definition time, an `abstract = True` intermediate base must be exempt, and a subclass accidentally inheriting from another concrete task (instead of the real base) must still be rejected. Unit-test `begin_tracked_job` for all three entity states (`PENDING`→`IN_PROCESS` transition, `COMPLETE`/`FAILED` skip-and-log, `NOT_TRACKED` raises). Unit-test `VISIBILITY_TIMEOUT_SECONDS` is strictly greater than every registered task's hard `time_limit`. Live QA must demonstrate: (1) a real worker actually enforces a task's `soft_time_limit`/`time_limit` (`SoftTimeLimitExceeded` fires and is handled); (2) `begin_tracked_job`'s stale-redelivery skip via direct DB manipulation (flip a tracked entity to `FAILED`, then re-dispatch its task and confirm the log event fires and no state is overwritten) — a true worker-crash-then-redelivery scenario is impractical to demonstrate live given the multi-thousand-second `visibility_timeout` window, so the DB-manipulation shortcut stands in for it. +* **API/CLI Path Coverage:** Add tests covering a success response and at least one error response for every new/changed API route (`POST /api/v1/storage/sync`, `POST /api/v1/tiles/tessellate`, `POST /api/v1/jobs/reconcile`) and every new/changed CLI command (`sync-staging`, the updated `tessellate-tiles`/`create-tiles`, `reconcile-stuck-jobs`). + +###### 8. CLAUDE.md Compliance & Implementation Sequencing +* **Implementation Sequencing:** Implement §1 (Storage Sync & `list_prefix` fixes) and §2 (Shared `union_mocs` & Tessellation Modes) first. Implement §3 (`JobConfiguration` linkage & Stuck-Job Watchdog) second. Finally, implement §4 (Database-Backed Beat, Alembic migration, and SQLAdmin views). +* **Documentation Registration:** + * Ensure docstrings are added to new service methods in `job_service.py` (`job_service.py` is already registered in `docs/index.rst`). + * Add `30_operational_services_and_watchdog` to the toctree in `docs/architecture/index.md`. + * Update `docs/diagrams/infrastructure_topology.md` with the new `beat` container and data flows. + +###### Logs + +###### 2026-07-29 — assessPrompt findings and design decisions (pre-Gemini-revision, v0.1) +* **Finding:** No Celery Beat process runs anywhere in `docker-compose.yml`/`docker-compose.prod.yml` — `ENABLE_DB_BACKUP_CRON`'s `beat_schedule` has been dead code since doc 21. §4's premise (admin-edited schedules take effect at runtime) requires an actual running beat process, which the original draft never specified. **Decision:** add a dedicated `beat` compose service (not a `--beat` flag on `worker_light`, which would risk duplicate schedules if that service ever scales). +* **Finding:** `mc mirror` (§1) had no install path in either Dockerfile. **Decision:** install `mc` on the worker image only (not `api`); run the sync as a Celery task (`run_staging_sync`, same dispatch pattern as `run_ingest_batch`) so the binary and the blocking work both stay off the API process. Stream `mc mirror --json`'s per-file event output rather than buffering the whole run. Not adding a new DB-tracked status entity for this job: `mc mirror` is itself diff-based/idempotent, and Celery's existing `task_acks_late`/`task_reject_on_worker_lost` already make a redelivered retry safe and cheap. +* **Finding:** `TileTessellationRequest` (§2) still took a pre-computed `moc_to_tile` range list, incompatible with the new `region_source` design, and `bounding_box` mode's required fields were never specified. **Decision:** replace `moc_to_tile` entirely with a `region_source`-discriminated schema (`cone`: ra/decl/radius_deg; `project_footprint`: project_id; `bounding_box`: min/max ra/decl, for pre-provisioning tiles ahead of incoming survey data). `generate_tile_tessellation`'s pure-function signature is unchanged — a thin resolver converts whichever mode was chosen into a `MOC` before calling it. The existing preview → create split (`POST /tiles/tessellate` → `POST /tiles`) already matches the desired workflow and needs no redesign. +* **Finding:** `IngestBatch.job_configuration_id`/`Level3Mosaic.job_configuration_id` were never populated by any service, making §3's per-job `staleness_timeout_seconds` override unreachable. Surfaced backlog issue #37 (JobConfiguration's scope was undecided) as directly relevant. **Decision (closes the JobConfiguration-shape part of #37; the object/domain pipeline-stage-status question is explicitly out of scope — no such pipeline stage exists yet, revisit under a future JWST-pipeline doc):** keep dispatch/queue status exactly where it already lives (`JobStatus` per tracked entity); keep `JobConfiguration` as a narrow provenance record, but add `task_name: str | None` and a shared `job_service.create_job_configuration()` helper, called by `ingest_service.create_ingest_batch`, `mosaic_service.create_mosaic`, and the sync dispatch before creating their tracked entity row. `job_configuration_id` stays `nullable=True`. +* **Decision:** `celery-sqlalchemy-scheduler`'s tables (§4) get a hand-authored Alembic migration generated by temporarily merging the package's own `Base.metadata` into `alembic/env.py`'s `target_metadata` for one `--autogenerate` pass, then reviewing/trimming the result into a normal versioned migration — tracked like every other table in this project, but derived from the package's real schema rather than hand-guessed. +* **Process note:** per explicit user instruction, this was intended as the final Gemini revision round for this doc — see the 2026-07-29 (v0.2 review) entry below for how that played out. + +###### 2026-07-29 — assessPrompt on Gemini's v0.2 revision; direct fixes applied (v0.3) +Gemini's revision (v0.1 → v0.2) correctly picked up the tessellation redesign, CLI parity, and toctree/docstring fixes, but regenerated the whole file and in the process (a) reverted §1 to a pure-Python diff-sync, directly contradicting the `mc`-as-a-Celery-task decision logged above — the single most significant finding of this pass, (b) dropped the shared `job_service.create_job_configuration()` helper and the new `JobConfiguration.task_name` column from §3 entirely, (c) narrowed §4's Alembic bullet back to a bare "hand-author," losing the autogenerate-from-package-metadata technique, (d) narrowed §7's testing bullet from API+CLI coverage back to API-only, and (e) wiped this Logs section's prior entry outright. All fixed directly by Claude in this pass (no further Gemini round-trip, per explicit user instruction to keep this session's back-and-forth minimal and to log fixes directly instead) — see the diff for exact wording. Also tightened §2's `overlap_only=False` test-wording (was ambiguously "materializes empty boundary tiles") and added an explicit breaking-change note for `moc_to_tile`'s removal. + +###### 2026-07-29 — Implementation (`runPrompt`) +Implemented all four sections end-to-end on branch `feature/operational-services-watchdog`. +* **§1 Storage sync:** `sync_staging_to_canonical` in `storage_service.py` streams `mc mirror --json` per-file events (via `subprocess.Popen`, never buffered); dispatched as the worker-only `run_staging_sync` Celery task (+ `sync_staging_cron` Beat task), with `dispatch_staging_sync` as the shared API/CLI entry point. `mc` installed on `docker/worker.Dockerfile` only (arch-aware). `LocalStorageService.list_prefix` rewritten for true string-prefix matching + dot-file exclusion. New route `POST /api/v1/storage/sync`, CLI `sync-staging`. +* **§2 Tessellation:** `union_mocs` extracted to `spatial_types.py` (consumed by `mosaic_service` + `tile_service`). `TileTessellationRequest` replaced `moc_to_tile` with a `region_source`-discriminated schema (Pydantic root validator). New service `generate_tessellation_for_region` + `_resolve_region_moc` (cone/bounding_box/project_footprint) keeps the pure `generate_tile_tessellation` DB-free; it gained an `overlap_only` param. CLI `--region-source` + mode args added; the dead `_cone_moc` helper removed. +* **§3 Watchdog:** `JobConfiguration.task_name` column added (migration 0013). `job_service.create_job_configuration` + `reconcile_stuck_jobs` (static `_STUCK_JOB_ENTITIES` registry, per-job override, `db.rollback()` first). `ingest_service`/`mosaic_service` now create + link a `JobConfiguration` at dispatch. New route `POST /api/v1/jobs/reconcile`, CLI `reconcile-stuck-jobs`, Beat task `reconcile_stuck_jobs_cron`. +* **§4 Beat — package pivot (QA discovery):** The originally-specified `celery-sqlalchemy-scheduler` (last released 2021) is **broken against SQLAlchemy 2.0**: its `after_insert`/`after_update`/`after_delete` event listeners call `update_changed`, which uses 1.x `select([Model])` list-syntax that 2.0 rejects — so *every* schedule insert/update/delete crashes (our `seed-db`, every runtime SQLAdmin edit, and every `last_run_at` write from the beat process). Also imports `pytz` without declaring it. Proposed the options to the user per the QA/Debugging Fix Confirmation guardrail; the user chose to **switch to the maintained fork `sqlalchemy-celery-beat==0.8.4`** rather than carry a compatibility shim (explicit preference: no tech debt during the design phase). The fork works cleanly against our stack (validated end-to-end against the live DB before rewiring), isolates its tables in a dedicated `celery_schema` Postgres schema, and uses a polymorphic `schedule_model` association + a `Period` enum. `pyproject.toml`, `docs/conf.py` `autodoc_mock_imports`, `admin.py` views, `seed.py` seeding, `celery_app.py` `beat_dburi`, and both compose files' `beat` service commands all point at the fork. Dedicated `beat` compose service added (dev + prod). §4/§6 doc text + this log updated to match. +* **Migrations:** `0013` (hand-authored, `task_name`); `0014` creates `celery_schema` and delegates the scheduler tables/enums to the fork's own `ModelBase.metadata.create_all` (faithful to the package, downgrade = `DROP SCHEMA … CASCADE`). Because the fork's tables live in their own schema, autogenerate (default `include_schemas=False`) never sees them, so `migrations/env.py` needs no exclusion filter. Both applied to the dev DB; `0014` round-trips cleanly; `alembic check` reports no drift. +* **Env vars:** `STAGING_LOCATION`, `STAGING_SYNC_INTERVAL_SECONDS`, `ENABLE_STAGING_SYNC_CRON`, `JOB_STALENESS_TIMEOUT_SECONDS` synced across `.env` + `.env.example` (key sets verified identical). +* **Tests/docs:** full suite 287 passed at 99.3% coverage (≥90 gate); Sphinx `-W` clean; infrastructure topology diagram updated with the `beat` node + schedule/publish edges; `docs/index.rst` (storage/jobs route automodules), `docs/architecture/index.md` toctree, and `docs/cli_guide.rst` updated. +* **Bug fixed during QA:** the `create_mosaic` integration test's teardown didn't delete the `JobConfiguration` the new dispatch-linkage creates, so its FK blocked the owning-user delete and leaked a duplicate-username row across runs; the test's cleanup now deletes it (and `_cleanup_seeded_rows` now also clears the seeded `PeriodicTask`/`IntervalSchedule` rows). Regression covered by the new watchdog + seeding integration tests. + +###### 2026-07-29 — Scope amendment during `stratSesh`: per-job staleness override write path (v0.4, not yet implemented) +During the post-implementation `stratSesh`, tracing the actual code confirmed a gap: `reconcile_stuck_jobs` reads a per-job `staleness_timeout_seconds` override from `JobConfiguration.job_kwargs`, but no dispatch path (`create_ingest_batch`, `create_mosaic`) ever writes that key — every job today uses the global default uniformly, and the override is unreachable. Per explicit user instruction, this is being closed as in-scope for doc 30 rather than left as a known limitation: §3's "Per-Job Staleness Override — Write Path" bullet above specifies the addition (optional param threaded from API/CLI through to `job_kwargs`). Two design forks are still open (bake the default in at dispatch time vs. defer to whatever the global setting is at reconcile time; whether to extend this to the untracked `run_staging_sync`) — flagged as Decision Required in chat, not yet answered, so no code has been written for this yet. + +###### 2026-07-29 — Bug found via empirical `mc` validation: trailing summary/error events miscounted as file copies +`sync_staging_to_canonical`'s original event-parsing loop treated every parsed JSON line from `mc --json mirror` as a file-copy event, based on mocked test JSON (`{"target": "..."}`) rather than the real binary's output — flagged as an open honesty gap during `stratSesh`. Per user suggestion, validated empirically: installed `mc` locally, copied real FITS files from `diffpype_claude_data/raw/` into a scratch staging dir, and ran a real `mc --json mirror` against the live MinIO instance (both a success case and a deliberate failure against a nonexistent bucket), then cleaned up the test objects/alias afterward. Confirmed `mc`'s real `--json` stream has three distinct shapes: per-file success (`target`/`source`/`size`/cumulative `totalCount`/`totalSize`), an error event (`status: "error"`, no `target` for a whole-job failure), and a single trailing job-summary event with no `target`/`source` at all (`total`/`transferred`/`duration`/`speed`). The original code counted the trailing summary line (and any error line) as an extra "file copied" with `key=None`, over-counting `copied` by one and misreporting failures as successes in the log. Fixed: `sync_staging_to_canonical` now discriminates on the presence of `target` and `status == "error"` before counting/logging a line as a real file copy; error events log `staging_sync_file_error` and the summary event logs `staging_sync_job_summary`, neither incrementing the copied count. Regression tests added using the real, empirically-captured event shapes (`test_storage_service.py`), including one asserting the summary event is not miscounted. + +###### 2026-07-29 — `run_staging_sync` hang mitigation: Celery soft time limit + subprocess kill +Closed the "no timeout wrapper on the mc subprocess" gap flagged in `stratSesh`. Added `STAGING_SYNC_SOFT_TIME_LIMIT_SECONDS` (default 1800s) and set `soft_time_limit`/`time_limit` (soft + 30s) on `run_staging_sync`'s task decorator — the idiomatic Celery-native mechanism (not a custom watchdog) per explicit user preference for standard, non-idiosyncratic patterns. `sync_staging_to_canonical` now wraps its read loop in `try/except SoftTimeLimitExceeded`, killing and reaping the `mc` subprocess before re-raising, so a genuine hang can't leave an orphaned process. Not added to `DiffpypeTask.autoretry_for` — a hang that already exceeded a generous soft limit likely hangs again immediately, so it dead-letters for operator attention rather than retrying blindly. Env var synced across `.env`/`.env.example`; unit test added simulating a mid-loop `SoftTimeLimitExceeded` and asserting `process.kill()`/`process.wait()` are called. + +###### 2026-07-29 — Per-task time-limit contract & watchdog/broker race guard (§3a, added mid-`genTests`) +Deep-dived, during `genTests`, into what actually happens when a task genuinely hangs past its Celery time limit vs. when a worker process is killed outright (SIGKILL/OOM): the two failure modes need two independent, uncoordinated systems (the stuck-job watchdog reads `JobStatus` on a polling interval; Redis's own `visibility_timeout` redelivers an unacked message on a completely separate clock) — and nothing previously stopped them from racing each other. If the watchdog flips an entity to `FAILED` and then Redis redelivers the same message before its `visibility_timeout` catches up, the redelivered task must recognize the entity is already resolved and bail, or it silently overwrites the watchdog's verdict. Fixing this required knowing every task's real time ceiling first, since `visibility_timeout` must be set above the slowest task's hard limit or Redis would redeliver still-healthy, legitimately-running tasks. +* **Per-task time-limit contract:** `TimeLimitedTask(celery.Task)` (`src/worker/base_task.py`) requires every concrete subclass to declare its own `soft_time_limit_seconds` via `__init_subclass__` inspecting `cls.__dict__` directly (never `getattr`), so a subclass that accidentally inherits from another concrete task instead of the true base is rejected at class-definition time rather than silently inheriting the wrong ceiling. `time_limit` is derived as `soft_time_limit_seconds + HARD_LIMIT_BUFFER_SECONDS` (30s grace for cleanup code to run before Celery force-kills). Empirically confirmed a plain `abc.ABC`/`abstractmethod` does **not** catch the wrong-sibling mistake — ABC only requires a concrete value to exist somewhere in the MRO, not that the subclass itself declared it — which is the actual justification for the `__init_subclass__`+`cls.__dict__` approach (an earlier, incorrect justification — that `abc.ABCMeta` conflicts with Celery's task metaclass — was raised, then disproved empirically: `type(celery.Task) is type`, no conflict exists). +* **`DiffpypeTask` extends the same contract** with a required `tracked_entity_model` (a real model, or the new `NOT_TRACKED` sentinel — a distinct value, never bare `None`, so "deliberately untracked" can't be confused with "forgot to declare") and a new `begin_tracked_job(db, entity_id)` method: fetches the entity, and if its status is already `COMPLETE`/`FAILED`, logs `tracked_job_stale_redelivery_skipped` and returns `None` without writing anything — this is the actual race guard. Callers (`run_ingest_batch`, `run_mosaic_drizzle`) must check for `None` and bail. Both tasks migrated to `bind=True` to access `self.begin_tracked_job`. +* **Two empirical bugs found and fixed while building this:** (1) `DiffpypeTask` itself first failed its own parent's contract check, since intermediate abstract bases were never exempted — fixed by checking `cls.__dict__.get("abstract", False)`, reusing Celery's own `abstract = True` convention for non-registered base classes. (2) After fixing that, running the real `tasks.py` against `celery_app.tasks` still raised `TypeError` for every single real task, even ones with correctly-declared attributes — traced to Celery's own `_task_from_fun`, which always wraps `base=` in an auto-generated subclass (`type(fun.__name__, (base,), {"_decorated": True, ...})`) that never has the attribute in *its own* `__dict__`. Fixed by also exempting `cls.__dict__.get("_decorated", False)`. Re-verified this still correctly rejects the wrong-sibling mistake (in fact earlier — at class-definition time rather than waiting for Celery's decorator to run). +* **`visibility_timeout`:** `celery_app.py` computes `VISIBILITY_TIMEOUT_SECONDS = max(all seven tasks' soft_time_limit_seconds) + HARD_LIMIT_BUFFER_SECONDS + 60s safety margin` and sets it via `broker_transport_options`, deriving the number from the same enforced ceilings rather than a hand-picked constant. Currently 7290s (driven by `ingest_batch`'s 7200s default). +* **New settings:** `INGEST_BATCH_SOFT_TIME_LIMIT_SECONDS` (7200), `MOSAIC_DRIZZLE_SOFT_TIME_LIMIT_SECONDS` (3600), `CLI_TOOL_SOFT_TIME_LIMIT_SECONDS` (3600), `DB_BACKUP_SOFT_TIME_LIMIT_SECONDS` (3600), `DLQ_DUMP_SOFT_TIME_LIMIT_SECONDS` (30), `RECONCILE_STUCK_JOBS_SOFT_TIME_LIMIT_SECONDS` (120) — synced across `.env`/`.env.example`, all overridable without a code change (worker/beat restart required, since `pydantic-settings` reads env once at process startup). +* **Tests:** 309 passed at 99.59% coverage; ruff and mypy clean. §7 above amended with the new testing mandate; QA plan extended accordingly (see next QA steps). + +###### 2026-07-30 — Doc-accuracy gap found during genTests: schedule seeding is CLI-only, not "on startup" +QA Step 16/19 found `celery_schema.celery_periodictask` empty despite the `beat` container running cleanly — §4's "On startup / `diffpype-manage seed-db`" phrasing was checked against the actual code and found inaccurate: `_seed_periodic_tasks()` (`src/db/seed.py`) is only ever invoked by the `seed-db` CLI command; no container (`api`, `beat`, or either worker) calls it at startup. Running `diffpype-manage seed-db` seeded both rows correctly, confirming the seeding logic itself is fine — only the doc's "on startup" claim was wrong. Per user decision, §4 corrected to describe the actual (CLI-only, idempotent) behavior rather than adding real startup-time seeding now; the gap (a fresh environment silently has no periodic schedule until an operator remembers to run `seed-db` once) is tracked as tech debt in GitHub issue #42 rather than fixed in this session. + +###### 2026-07-30 — Bug found during genTests QA Step 17/19: SQLAdmin blocked every PeriodicTask edit +Attempting the "dynamic edit via SQLAdmin, no restart" QA step (disabling the seeded `reconcile_stuck_jobs_cron` row) failed with "Not a valid choice" validation errors on three unrelated form fields (`Model Crontabschedule`, `Model Solarschedule`, `Model Clockedschedule`), blocking the save entirely — even a trivial `enabled` toggle. Root cause: `PeriodicTaskAdmin` (`src/api/admin.py`) had no `form_columns`/`form_excluded_columns`, so SQLAdmin auto-generated a required dropdown for all four of `sqlalchemy_celery_beat`'s schedule-type relationships (`model_intervalschedule`/`model_crontabschedule`/`model_solarschedule`/`model_clockedschedule`), even though a given `PeriodicTask` row only ever uses one (matching its `discriminator` column) — both currently-seeded tasks use `IntervalSchedule` only. First fix attempt (`form_excluded_columns = [PeriodicTask.model_crontabschedule, ...]`, referencing the relationship attributes directly) crashed the `api` container outright at import time: `AttributeError: type object 'PeriodicTask' has no attribute 'model_crontabschedule'`. Root cause: `PeriodicTask` comes from `sqlalchemy_celery_beat`'s own separate declarative registry (its own `ModelBase`, not this app's `Base`), and accessing its relationship attributes directly at `admin.py`'s class-body-evaluation time raced that registry's own mapper configuration — a plain isolated `python3 -c` import of just `PeriodicTask` didn't reproduce it, only importing it as part of the full app's module graph did. Fixed by passing plain **strings** instead (`form_excluded_columns = ["model_crontabschedule", "model_solarschedule", "model_clockedschedule"]`) — confirmed `sqladmin`'s own source (`sqladmin/models.py`) explicitly supports strings alongside attribute objects, and strings defer resolution to request time, well after the whole app (and both declarative registries) have finished configuring. `api` container rebuilt; confirmed it stays up and `/admin/` responds normally (302 redirect to login) rather than crash-looping. + +###### 2026-07-30 — Post-genTests admin polish: IntervalSchedule display bug + verified schedule-reassignment safety +After `genTests` closed, the user noticed the seeded `IntervalSchedule` rows (id=1, id=2) are visually indistinguishable in the admin list view — both read "every 300 seconds" with no indication of which `PeriodicTask` owns which (they're never shared; each task gets its own dedicated schedule row, per `_seed_periodic_tasks`, and just happen to share the same default interval value). Added a `periodic_tasks` column to `IntervalScheduleAdmin` via its reverse relationship, with a `column_formatters` entry to render task names instead of raw ORM reprs. +* **Bug found and fixed:** first formatter attempt returned one joined string (`", ".join(pt.name for pt in model.periodic_tasks)`), which rendered as `(s)`/`(r)` — just the first letter of each name. Root cause, traced into `sqladmin`'s own `templates/sqladmin/list.html`: for a to-many relation column, the template does `zip(value, formatted_value)`, pairing each related object with the *corresponding element* of whatever the formatter returns — a plain string is itself iterable, so `zip()` silently paired the single related `PeriodicTask` with just the string's first character. Fixed by returning a list of per-element labels (`[pt.name for pt in model.periodic_tasks]`) instead of one joined string, matching what the template actually expects to zip against. +* **Investigated, not a bug:** the user separately asked how `CrontabScheduleAdmin`'s own "Periodic Tasks" field (its reverse relationship, not excluded from that form) interacts with a task currently driven by an `IntervalSchedule` — worried that selecting an existing task there could silently corrupt its schedule (write `schedule_id` without updating `discriminator` to match, since `PeriodicTask`'s four forward `model_*` relationships are `viewonly=True` while the reverse `periodic_tasks` relationships are not, an asymmetry that looked exactly like it could cause partial writes). Verified empirically (disposable in-session test row, rolled back, not committed) that `crontab.periodic_tasks.append(task)` correctly updates **both** `schedule_id` and `discriminator` together — `sqlalchemy_celery_beat` handles this correctly, it isn't a footgun. No code change needed for this part. \ No newline at end of file diff --git a/docs/architecture/index.md b/docs/architecture/index.md index 9976469..2a650db 100644 --- a/docs/architecture/index.md +++ b/docs/architecture/index.md @@ -36,4 +36,5 @@ work for that stage. 27_schema_storage_spatial_types 28_domain_graph_population 29_psycopg3_healpix_dummy_cleanup +30_operational_services_and_watchdog ``` diff --git a/docs/cli_guide.rst b/docs/cli_guide.rst index bf1dd15..d04a8d4 100644 --- a/docs/cli_guide.rst +++ b/docs/cli_guide.rst @@ -29,10 +29,18 @@ Overview .. note:: This guide covers the foundational database-management commands. The domain - commands added in later stages (``create-project``, ``ingest``, - ``tessellate-tiles``, ``create-mosaic``, ``populate-demo-project``, and their - ``*-status`` pollers) share the same Service Layer and follow the same - API/CLI-parity contract; run ``diffpype-manage --help`` for the full list. + and operational commands added in later stages (``create-project``, + ``ingest``, ``tessellate-tiles``, ``create-mosaic``, ``sync-staging``, + ``reconcile-stuck-jobs``, ``populate-demo-project``, and their ``*-status`` + pollers) share the same Service Layer and follow the same API/CLI-parity + contract; run ``diffpype-manage --help`` for the full list. + + ``tessellate-tiles``/``create-tiles`` take a ``--region-source`` + (``cone`` | ``project_footprint`` | ``bounding_box``) with the fields that + mode needs (e.g. ``--ra/--decl/--radius-deg`` for ``cone``, + ``--min-ra/--max-ra/--min-decl/--max-decl`` for ``bounding_box``), plus + ``--overlap-only/--no-overlap-only`` to trim the grid to the region or + materialize it fully. ``seed-db`` ----------- @@ -62,3 +70,30 @@ usable. Intended for local development only. Schema reset complete. Auto-seeding foundational records... Seeding database: inserting foundational sysadmin + reference records... Done. + +``sync-staging`` +---------------- + +Dispatches a staging→canonical storage sync to the worker (which runs +``mc mirror`` in a streamed, restart-safe Celery task). ``--staging-prefix`` +accepts a local path or an ``s3://`` URI; ``--canonical-prefix`` defaults to the +bucket root. + +.. code-block:: console + + $ docker compose run --rm api diffpype-manage sync-staging --staging-prefix ./data/staging --canonical-prefix raw + Dispatched staging sync. job_id= + +``reconcile-stuck-jobs`` +------------------------ + +Fails any job left in ``IN_PROCESS`` past the staleness threshold (an +uncatchable worker crash or OOM kill can't run a task's own failure handler). +``--threshold-seconds`` overrides ``JOB_STALENESS_TIMEOUT_SECONDS`` for this +sweep; it also runs automatically on a Celery Beat schedule. + +.. code-block:: console + + $ docker compose run --rm api diffpype-manage reconcile-stuck-jobs --threshold-seconds 3600 + Reconciled 1 stuck job(s). + IngestBatch id=9 (age=7200s) -> FAILED diff --git a/docs/conf.py b/docs/conf.py index 34b0cbf..a01032f 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -67,6 +67,7 @@ "slugify", "sqladmin", "sqlalchemy", + "sqlalchemy_celery_beat", "src.api.admin", "starlette", "starlette_exporter", diff --git a/docs/diagrams/infrastructure_topology.md b/docs/diagrams/infrastructure_topology.md index 76f8782..943e063 100644 --- a/docs/diagrams/infrastructure_topology.md +++ b/docs/diagrams/infrastructure_topology.md @@ -63,6 +63,10 @@ flowchart TB w_app --> w_core end + subgraph beat_c["beat (Celery Beat)"] + beat_app["Celery Beat
DatabaseScheduler"]:::workerLayer + end + subgraph ui_c["ui (React / Vite)"] direction TB ui_view["Dashboard
DashboardPage.tsx"]:::uiLayer @@ -112,6 +116,11 @@ flowchart TB w_tasks -->|store/fetch FITS| minio_node portainer_node -.-> minio_c + %% Beat scheduler (declared last so existing link indices stay valid) + beat_app -->|read/write schedules| db + beat_app -->|publish scheduled tasks| redis + portainer_node -.-> beat_c + classDef apiLayer fill:#3B6EA5,stroke:#1F4066,color:#fff classDef workerLayer fill:#C97A3D,stroke:#8A4F24,color:#fff classDef repo fill:#4C8C6B,stroke:#2E5842,color:#fff @@ -123,7 +132,7 @@ flowchart TB classDef spacer fill:transparent,stroke:transparent,color:transparent classDef networkBg fill:transparent,stroke:#5A6C77,color:#fff - class db_c,redis_c,minio_c,api_c,worker_c,ui_c,jaeger_c,flower_c,portainer_c containerBg + class db_c,redis_c,minio_c,api_c,worker_c,beat_c,ui_c,jaeger_c,flower_c,portainer_c containerBg class network networkBg %% Tier 1 (default): solid amber — configured application data flow between components @@ -131,7 +140,7 @@ flowchart TB %% Invisible spacer links used only to center nodes within db_c, worker_c, portainer_c — must override the amber default or they render as dangling solid lines linkStyle 0,1,11,17,18 stroke:none,stroke-width:0 %% Tier 3: dashed cool steel gray — Portainer manages containers at the Docker daemon level, independent of what's running inside them - linkStyle 25,26,27,28,29,30,31,38 stroke:#8F97A0,stroke-width:4px + linkStyle 25,26,27,28,29,30,31,38,41 stroke:#8F97A0,stroke-width:4px %% Tier 2: dashed rose — Jaeger/Flower/DBeaver observe a specific process's exposed port/protocol (OTLP, Celery/Redis state, Postgres wire protocol) linkStyle 32,33,34,35 stroke:#C0546A,stroke-width:4px ``` @@ -167,6 +176,7 @@ flowchart TB | `minio` | Local S3-compatible object store (mock) for FITS payloads; the api and workers read/write files here via `S3StorageService` | | `api` | Serves the HTTP API, admin panel, and CLI; validates input and dispatches work | | `worker` (×2: `worker_light`, `worker_heavy`) | Same image/codebase, deployed as two instances with different queue subscriptions and resource limits — `light` for fast I/O-bound tasks, `heavy_memory` for memory/compute-intensive ones | +| `beat` | Single Celery Beat process (same worker image) using the database-backed scheduler; reads/writes runtime-editable schedules in Postgres (`celery_schema`) and publishes due tasks to Redis. Consumes no task queues | | `ui` | Frontend for dispatching jobs and viewing status | | `jaeger` | Collects and visualizes distributed traces via OTLP | | `flower` | Real-time Celery task/worker monitoring dashboard | diff --git a/docs/index.rst b/docs/index.rst index 0db4274..61be779 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -87,6 +87,14 @@ API (FastAPI) :members: :undoc-members: +.. automodule:: src.api.routes.storage + :members: + :undoc-members: + +.. automodule:: src.api.routes.jobs + :members: + :undoc-members: + .. automodule:: src.api.routes.tiles :members: :undoc-members: diff --git a/migrations/env.py b/migrations/env.py index a4a092c..6930119 100644 --- a/migrations/env.py +++ b/migrations/env.py @@ -12,6 +12,12 @@ target_metadata = Base.metadata +# Note: sqlalchemy-celery-beat's tables (doc 30 §4) live in a dedicated +# `celery_schema` Postgres schema, created by migration 0014. Autogenerate runs +# with the default include_schemas=False, so it only ever reflects the public +# schema and never sees — or tries to drop — those package-owned tables. No +# include_object filter is needed. + def get_url() -> str: # Allow callers (e.g. the integration-test fixture) to override the URL @@ -38,7 +44,10 @@ def run_migrations_online() -> None: cfg["sqlalchemy.url"] = get_url() connectable = engine_from_config(cfg, prefix="sqlalchemy.", poolclass=pool.NullPool) with connectable.connect() as connection: - context.configure(connection=connection, target_metadata=target_metadata) + context.configure( + connection=connection, + target_metadata=target_metadata, + ) with context.begin_transaction(): context.run_migrations() diff --git a/migrations/versions/20260729_0013_add_job_configuration_task_name.py b/migrations/versions/20260729_0013_add_job_configuration_task_name.py new file mode 100644 index 0000000..858b948 --- /dev/null +++ b/migrations/versions/20260729_0013_add_job_configuration_task_name.py @@ -0,0 +1,29 @@ +"""add JobConfiguration.task_name provenance column + +Revision ID: 0013 +Revises: 0012 +Create Date: 2026-07-29 + +Doc 30 §3. Records which Celery task (or CLI tool) a JobConfiguration row's +kwargs/command correspond to, e.g. "src.worker.tasks.run_ingest_batch". Nullable, +so existing provenance rows need no backfill. +""" + +import sqlalchemy as sa +from alembic import op + +revision = "0013" +down_revision = "0012" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + op.add_column( + "job_configurations", + sa.Column("task_name", sa.String(), nullable=True), + ) + + +def downgrade() -> None: + op.drop_column("job_configurations", "task_name") diff --git a/migrations/versions/20260729_0014_add_beat_scheduler_tables.py b/migrations/versions/20260729_0014_add_beat_scheduler_tables.py new file mode 100644 index 0000000..135fc38 --- /dev/null +++ b/migrations/versions/20260729_0014_add_beat_scheduler_tables.py @@ -0,0 +1,38 @@ +"""add sqlalchemy-celery-beat scheduler schema and tables + +Revision ID: 0014 +Revises: 0013 +Create Date: 2026-07-29 + +Doc 30 §4. Creates the dedicated ``celery_schema`` Postgres schema and the +relational tables `sqlalchemy-celery-beat` uses for its database-backed Celery +Beat scheduler (DatabaseScheduler), so schedules can be created/edited/paused at +runtime via SQLAdmin. + +Table (and Enum type) creation is delegated to the package's own +``ModelBase.metadata`` rather than transcribing six tables + two schema-qualified +enums by hand: that keeps this migration faithful to the package's real schema and +lets a future package upgrade re-run the same delegation instead of drifting from +a hand-copied definition. The tables live in their own schema, so autogenerate +(which runs with the default ``include_schemas=False``) never sees them and this +migration is the sole authority over their lifecycle. Downgrade drops the whole +schema (tables + enum types) in one cascade. +""" + +from alembic import op + +revision = "0014" +down_revision = "0013" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + from sqlalchemy_celery_beat.models import ModelBase as BeatBase + + op.execute("CREATE SCHEMA IF NOT EXISTS celery_schema") + BeatBase.metadata.create_all(op.get_bind()) + + +def downgrade() -> None: + op.execute("DROP SCHEMA IF EXISTS celery_schema CASCADE") diff --git a/pyproject.toml b/pyproject.toml index b0694ad..6b84cfd 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -33,6 +33,7 @@ dependencies = [ "scipy==1.18.0", "scikit-learn==1.9.0", "pandas==3.0.5", + "sqlalchemy-celery-beat==0.8.4", ] [project.scripts] diff --git a/src/api/admin.py b/src/api/admin.py index e605ba3..346081f 100644 --- a/src/api/admin.py +++ b/src/api/admin.py @@ -2,6 +2,11 @@ import bcrypt from sqladmin import ModelView +from sqlalchemy_celery_beat.models import ( + CrontabSchedule, + IntervalSchedule, + PeriodicTask, +) from sqladmin.authentication import AuthenticationBackend from starlette.requests import Request @@ -105,6 +110,7 @@ class JobConfigurationAdmin(ModelView, model=JobConfiguration): column_list = [ JobConfiguration.id, JobConfiguration.user_id, + JobConfiguration.task_name, JobConfiguration.execution_command, JobConfiguration.created_at, ] @@ -112,3 +118,77 @@ class JobConfigurationAdmin(ModelView, model=JobConfiguration): JobConfiguration.created_at, JobConfiguration.updated_at, ] + + +class PeriodicTaskAdmin(ModelView, model=PeriodicTask): + """Admin view for creating, editing, and pausing database-backed Celery Beat schedules.""" + + name = "Periodic Task" + name_plural = "Periodic Tasks" + column_list = [ + PeriodicTask.id, + PeriodicTask.name, + PeriodicTask.task, + PeriodicTask.enabled, + PeriodicTask.last_run_at, + ] + # Only IntervalSchedule is used today (see doc 30 §4) — excluding the other + # three schedule-type relations avoids SQLAdmin auto-generating a required + # "Not a valid choice" dropdown for schedule types no PeriodicTask actually uses. + # Plain strings (not PeriodicTask.model_crontabschedule etc.) are deliberate: + # PeriodicTask comes from sqlalchemy_celery_beat's own separate declarative + # registry, and referencing its relationship attributes directly at this + # class body's evaluation time raced that registry's mapper configuration, + # raising a spurious AttributeError. Strings defer resolution to request time. + form_excluded_columns = [ + "model_crontabschedule", + "model_solarschedule", + "model_clockedschedule", + ] + + +class IntervalScheduleAdmin(ModelView, model=IntervalSchedule): + """Admin view for the interval (every-N-period) schedules a PeriodicTask can reference.""" + + name = "Interval Schedule" + name_plural = "Interval Schedules" + # Each PeriodicTask gets its own dedicated IntervalSchedule row (never + # shared), so two rows with identical every/period are easy to mistake for + # duplicates. Surfacing the owning task name(s) disambiguates them. + # String keys (not IntervalSchedule.periodic_tasks) for the same reason + # PeriodicTaskAdmin.form_excluded_columns uses strings above: this model's + # relationships live in sqlalchemy_celery_beat's own declarative registry. + column_list = [ + IntervalSchedule.id, + IntervalSchedule.every, + IntervalSchedule.period, + "periodic_tasks", + ] + column_formatters = { + # sqladmin's own ClassVar annotation types the formatter's first arg as + # bare `type`, but it's actually the model instance at render time. + # Must return a list aligned 1:1 with the relation's items, not a + # single joined string: for to-many columns, list.html zips the raw + # related-object list against whatever this returns, one link per + # pair — a joined string got zipped character-by-character against a + # single-item list, silently rendering only its first letter. + "periodic_tasks": lambda model, _attr: [ + pt.name + for pt in model.periodic_tasks # type: ignore[attr-defined] + ], + } + + +class CrontabScheduleAdmin(ModelView, model=CrontabSchedule): + """Admin view for the crontab schedules a PeriodicTask can reference.""" + + name = "Crontab Schedule" + name_plural = "Crontab Schedules" + column_list = [ + CrontabSchedule.id, + CrontabSchedule.minute, + CrontabSchedule.hour, + CrontabSchedule.day_of_week, + CrontabSchedule.day_of_month, + CrontabSchedule.month_of_year, + ] diff --git a/src/api/main.py b/src/api/main.py index a6a0fd3..891e9cb 100644 --- a/src/api/main.py +++ b/src/api/main.py @@ -9,17 +9,22 @@ from sqladmin import Admin from src.api.admin import ( + CrontabScheduleAdmin, DiffpypeAuthBackend, + IntervalScheduleAdmin, JobConfigurationAdmin, + PeriodicTaskAdmin, ProjectAdmin, StepDefinitionAdmin, UserAdmin, ) from src.api.routes.epochs import router as epochs_router from src.api.routes.ingest import router as ingest_router +from src.api.routes.jobs import router as jobs_router from src.api.routes.meta import router as meta_router from src.api.routes.mosaics import router as mosaics_router from src.api.routes.projects import router as projects_router +from src.api.routes.storage import router as storage_router from src.api.routes.tiles import router as tiles_router from src.core.config import settings from src.core.logger import configure_logging, get_logger @@ -36,6 +41,9 @@ admin.add_view(ProjectAdmin) admin.add_view(StepDefinitionAdmin) admin.add_view(JobConfigurationAdmin) +admin.add_view(PeriodicTaskAdmin) +admin.add_view(IntervalScheduleAdmin) +admin.add_view(CrontabScheduleAdmin) def sqladmin_exception_handler(request: Request, exc: Exception) -> NoReturn: @@ -85,9 +93,11 @@ async def unhandled_exception_handler(request: Request, exc: Exception) -> JSONR app.include_router(meta_router, prefix="/api/v1") app.include_router(projects_router, prefix="/api/v1") app.include_router(ingest_router, prefix="/api/v1") +app.include_router(storage_router, prefix="/api/v1") app.include_router(tiles_router, prefix="/api/v1") app.include_router(epochs_router, prefix="/api/v1") app.include_router(mosaics_router, prefix="/api/v1") +app.include_router(jobs_router, prefix="/api/v1") # Instrument the app, the SQLAlchemy engine, and Celery. Called last so the router # and middleware stack are fully assembled before OTel wraps the ASGI app. diff --git a/src/api/routes/jobs.py b/src/api/routes/jobs.py new file mode 100644 index 0000000..f3179ea --- /dev/null +++ b/src/api/routes/jobs.py @@ -0,0 +1,20 @@ +from fastapi import APIRouter, Depends +from sqlalchemy.orm import Session + +from src.api.schemas import JobReconcileRequest, JobReconcileResponse +from src.db.session import get_db +from src.services import job_service + +router = APIRouter(prefix="/jobs", tags=["jobs"]) + + +@router.post("/reconcile", response_model=JobReconcileResponse) +def reconcile_stuck_jobs( + body: JobReconcileRequest, db: Session = Depends(get_db) +) -> JobReconcileResponse: + """Fail any job stuck IN_PROCESS past the staleness threshold; return what changed.""" + if body.threshold_seconds is not None: + reconciled = job_service.reconcile_stuck_jobs(db, body.threshold_seconds) + else: + reconciled = job_service.reconcile_stuck_jobs(db) + return JobReconcileResponse(reconciled=reconciled) diff --git a/src/api/routes/storage.py b/src/api/routes/storage.py new file mode 100644 index 0000000..b7b49b1 --- /dev/null +++ b/src/api/routes/storage.py @@ -0,0 +1,15 @@ +from fastapi import APIRouter + +from src.api.schemas import StorageSyncRequest, StorageSyncResponse +from src.services import storage_service + +router = APIRouter(prefix="/storage", tags=["storage"]) + + +@router.post("/sync", response_model=StorageSyncResponse) +def sync_storage(body: StorageSyncRequest) -> StorageSyncResponse: + """Dispatch a staging→canonical storage sync Celery task and return its job id.""" + job_id = storage_service.dispatch_staging_sync( + body.staging_location, body.canonical_prefix + ) + return StorageSyncResponse(job_id=job_id) diff --git a/src/api/routes/tiles.py b/src/api/routes/tiles.py index 45be634..5837c99 100644 --- a/src/api/routes/tiles.py +++ b/src/api/routes/tiles.py @@ -15,10 +15,23 @@ @router.post("/tessellate", response_model=list[TileCreate]) -def tessellate_tiles(body: TileTessellationRequest) -> list[TileCreate]: - moc_to_tile = ranges_to_moc(body.moc_to_tile) - tiles = tile_service.generate_tile_tessellation( - body.tile_side_length_arc_min, moc_to_tile, body.overlap_in_arc_min +def tessellate_tiles( + body: TileTessellationRequest, db: Session = Depends(get_db) +) -> list[TileCreate]: + tiles = tile_service.generate_tessellation_for_region( + db, + body.region_source, + body.tile_side_length_arc_min, + body.overlap_in_arc_min, + body.overlap_only, + ra=body.ra, + decl=body.decl, + radius_deg=body.radius_deg, + project_id=body.project_id, + min_ra=body.min_ra, + max_ra=body.max_ra, + min_decl=body.min_decl, + max_decl=body.max_decl, ) return [ TileCreate( diff --git a/src/api/schemas.py b/src/api/schemas.py index dd9869b..f3dd719 100644 --- a/src/api/schemas.py +++ b/src/api/schemas.py @@ -1,8 +1,8 @@ from datetime import datetime -from pydantic import BaseModel, Field, field_validator +from pydantic import BaseModel, Field, field_validator, model_validator -from src.db.enums import JobStatus +from src.db.enums import JobStatus, RegionSource from src.db.spatial_types import moc_to_ranges @@ -44,6 +44,35 @@ class IngestRequest(BaseModel): s3_prefix: str +class StorageSyncRequest(BaseModel): + """Request body dispatching a staging→canonical storage sync.""" + + staging_location: str + canonical_prefix: str = "" + + +class StorageSyncResponse(BaseModel): + """Celery task id for a dispatched staging→canonical storage sync.""" + + job_id: str + + +class JobReconcileRequest(BaseModel): + """Request body for a stuck-job reconciliation sweep. + + ``threshold_seconds`` overrides the global staleness default for this sweep; + omit it to use ``JOB_STALENESS_TIMEOUT_SECONDS``. + """ + + threshold_seconds: int | None = None + + +class JobReconcileResponse(BaseModel): + """Result of a stuck-job reconciliation sweep: the entities transitioned to FAILED.""" + + reconciled: list[dict] + + class IngestDispatchResponse(BaseModel): job_id: str batch_id: int @@ -152,12 +181,49 @@ class EpochRead(BaseModel): class TileTessellationRequest(BaseModel): - """Request body for a no-DB tile-tessellation preview over a target region.""" + """Request body for a tile-tessellation preview over a region_source-specified region. - project_id: int + Replaces the previous pre-computed ``moc_to_tile`` range list: the region is + now specified declaratively via ``region_source`` plus its mode-specific + fields, which the service layer resolves into a MOC. A root validator enforces + that exactly the fields required by the chosen ``region_source`` are present. + """ + + region_source: RegionSource tile_side_length_arc_min: float - moc_to_tile: list[tuple[int, int]] overlap_in_arc_min: float = 0.0 + overlap_only: bool = True + + # cone + ra: float | None = None + decl: float | None = None + radius_deg: float | None = None + + # project_footprint + project_id: int | None = None + + # bounding_box + min_ra: float | None = None + max_ra: float | None = None + min_decl: float | None = None + max_decl: float | None = None + + @model_validator(mode="after") + def _require_fields_for_region_source(self) -> "TileTessellationRequest": + """Ensure the fields required by the chosen region_source are all present.""" + required_by_source = { + RegionSource.CONE: ("ra", "decl", "radius_deg"), + RegionSource.PROJECT_FOOTPRINT: ("project_id",), + RegionSource.BOUNDING_BOX: ("min_ra", "max_ra", "min_decl", "max_decl"), + } + required = required_by_source[self.region_source] + missing = [name for name in required if getattr(self, name) is None] + if missing: + raise ValueError( + f"region_source={self.region_source.value} requires: " + f"{', '.join(missing)}" + ) + return self class TileBulkCreateRequest(BaseModel): diff --git a/src/api/tests/test_cli.py b/src/api/tests/test_cli.py index b390c4d..1aa53d7 100644 --- a/src/api/tests/test_cli.py +++ b/src/api/tests/test_cli.py @@ -20,8 +20,10 @@ cmd_ingest_status, cmd_mosaic_status, cmd_populate_demo_project, + cmd_reconcile_stuck_jobs, cmd_reset_db, cmd_seed_db, + cmd_sync_staging, cmd_tessellate_tiles, main, ) @@ -446,25 +448,38 @@ def test_main_routes_tessellate_tiles_to_cmd_tessellate_tiles(mocker): mock_cmd.assert_called_once() +def _tessellation_namespace(command, **overrides): + """A full tessellation argparse.Namespace with region fields defaulted to cone.""" + base = dict( + command=command, + region_source="cone", + tile_side_arcmin=6.0, + overlap_arcmin=0.0, + overlap_only=True, + ra=10.0, + decl=20.0, + radius_deg=0.1, + region_project_id=None, + min_ra=None, + max_ra=None, + min_decl=None, + max_decl=None, + ) + base.update(overrides) + return argparse.Namespace(**base) + + def test_cmd_tessellate_tiles_prints_generated_tiles(mocker, capsys): mocker.patch( - "src.services.tile_service.generate_tile_tessellation", + "src.services.tile_service.generate_tessellation_for_region", return_value=[ {"name": "Tile_1", "ra": 10.0, "decl": 20.0}, {"name": "Tile_2", "ra": 10.1, "decl": 20.1}, ], ) + mocker.patch("src.db.session.SessionLocal", return_value=MagicMock()) - cmd_tessellate_tiles( - argparse.Namespace( - command="tessellate-tiles", - tile_side_arcmin=6.0, - ra=10.0, - decl=20.0, - radius_deg=0.1, - overlap_arcmin=0.0, - ) - ) + cmd_tessellate_tiles(_tessellation_namespace("tessellate-tiles")) out = capsys.readouterr().out assert "Generated 2 tile(s)" in out @@ -488,7 +503,7 @@ def test_main_routes_create_tiles_to_cmd_create_tiles(mocker): def test_cmd_create_tiles_persists_and_prints_count(mocker, capsys): mocker.patch( - "src.services.tile_service.generate_tile_tessellation", + "src.services.tile_service.generate_tessellation_for_region", return_value=[{"name": "Tile_1"}, {"name": "Tile_2"}], ) mock_create = mocker.patch( @@ -498,17 +513,7 @@ def test_cmd_create_tiles_persists_and_prints_count(mocker, capsys): mock_session = MagicMock() mocker.patch("src.db.session.SessionLocal", return_value=mock_session) - cmd_create_tiles( - argparse.Namespace( - command="create-tiles", - project_id=3, - tile_side_arcmin=6.0, - ra=10.0, - decl=20.0, - radius_deg=0.1, - overlap_arcmin=0.0, - ) - ) + cmd_create_tiles(_tessellation_namespace("create-tiles", project_id=3)) mock_create.assert_called_once() assert mock_create.call_args[0][1] == 3 @@ -517,6 +522,89 @@ def test_cmd_create_tiles_persists_and_prints_count(mocker, capsys): assert "Created 2 tile(s) for project_id=3" in out +def test_cmd_create_tiles_bounding_box_passes_region_params(mocker): + resolve = mocker.patch( + "src.services.tile_service.generate_tessellation_for_region", + return_value=[{"name": "Tile_1"}], + ) + mocker.patch("src.services.tile_service.create_tiles", return_value=[MagicMock()]) + mocker.patch("src.db.session.SessionLocal", return_value=MagicMock()) + + cmd_create_tiles( + _tessellation_namespace( + "create-tiles", + project_id=3, + region_source="bounding_box", + ra=None, + decl=None, + radius_deg=None, + min_ra=10.0, + max_ra=11.0, + min_decl=20.0, + max_decl=21.0, + ) + ) + + assert resolve.call_args.kwargs["min_ra"] == 10.0 + assert resolve.call_args.kwargs["max_decl"] == 21.0 + + +def test_main_routes_sync_staging_to_cmd_sync_staging(mocker): + mock_cmd = mocker.patch("src.cli.cmd_sync_staging") + main(["sync-staging", "--staging-prefix", "/staging"]) + mock_cmd.assert_called_once() + + +def test_main_routes_reconcile_stuck_jobs_to_cmd(mocker): + mock_cmd = mocker.patch("src.cli.cmd_reconcile_stuck_jobs") + main(["reconcile-stuck-jobs"]) + mock_cmd.assert_called_once() + + +def test_cmd_sync_staging_dispatches_and_prints_job_id(mocker, capsys): + mocker.patch( + "src.services.storage_service.dispatch_staging_sync", return_value="sync-7" + ) + + cmd_sync_staging( + argparse.Namespace( + command="sync-staging", staging_prefix="/staging", canonical_prefix="raw" + ) + ) + + assert "Dispatched staging sync. job_id=sync-7" in capsys.readouterr().out + + +def test_cmd_reconcile_stuck_jobs_prints_failed_entities(mocker, capsys): + mocker.patch( + "src.services.job_service.reconcile_stuck_jobs", + return_value=[{"entity": "IngestBatch", "id": 9, "age_seconds": 9000.0}], + ) + mocker.patch("src.db.session.SessionLocal", return_value=MagicMock()) + + cmd_reconcile_stuck_jobs( + argparse.Namespace(command="reconcile-stuck-jobs", threshold_seconds=3600) + ) + + out = capsys.readouterr().out + assert "Reconciled 1 stuck job(s)." in out + assert "IngestBatch id=9" in out + + +def test_cmd_reconcile_stuck_jobs_default_threshold(mocker): + reconcile = mocker.patch( + "src.services.job_service.reconcile_stuck_jobs", return_value=[] + ) + mocker.patch("src.db.session.SessionLocal", return_value=MagicMock()) + + cmd_reconcile_stuck_jobs( + argparse.Namespace(command="reconcile-stuck-jobs", threshold_seconds=None) + ) + + # No explicit threshold => called with the session only. + assert reconcile.call_args[0][1:] == () + + EPOCH_ARGS = [ "--project-id", "1", diff --git a/src/api/tests/test_jobs.py b/src/api/tests/test_jobs.py new file mode 100644 index 0000000..22d15d6 --- /dev/null +++ b/src/api/tests/test_jobs.py @@ -0,0 +1,52 @@ +from unittest.mock import MagicMock + +import pytest +from fastapi.testclient import TestClient + +from src.api.main import app +from src.db.session import get_db + + +@pytest.fixture +def mock_db(): + db = MagicMock() + app.dependency_overrides[get_db] = lambda: db + yield db + app.dependency_overrides.pop(get_db, None) + + +@pytest.fixture +def client(): + return TestClient(app) + + +def test_reconcile_returns_failed_entities(client, mock_db, mocker): + mocker.patch( + "src.services.job_service.reconcile_stuck_jobs", + return_value=[{"entity": "IngestBatch", "id": 1, "age_seconds": 9000.0}], + ) + + response = client.post("/api/v1/jobs/reconcile", json={"threshold_seconds": 3600}) + + assert response.status_code == 200 + assert response.json()["reconciled"][0]["id"] == 1 + + +def test_reconcile_uses_default_threshold_when_omitted(client, mock_db, mocker): + reconcile = mocker.patch( + "src.services.job_service.reconcile_stuck_jobs", return_value=[] + ) + + response = client.post("/api/v1/jobs/reconcile", json={}) + + assert response.status_code == 200 + assert response.json() == {"reconciled": []} + # Called with the session only (no explicit threshold => service default). + assert reconcile.call_args[0][1:] == () + + +def test_reconcile_invalid_threshold_type_is_422(client, mock_db): + response = client.post( + "/api/v1/jobs/reconcile", json={"threshold_seconds": "not-an-int"} + ) + assert response.status_code == 422 diff --git a/src/api/tests/test_schemas.py b/src/api/tests/test_schemas.py index 4c61538..e67d95e 100644 --- a/src/api/tests/test_schemas.py +++ b/src/api/tests/test_schemas.py @@ -59,3 +59,71 @@ class _FakeTile: result = TileRead.model_validate(_FakeTile(), from_attributes=True) assert result.footprint == moc_to_ranges(moc) + + +# --- TileTessellationRequest region_source validation --- + + +def test_tessellation_request_cone_valid(): + from src.api.schemas import TileTessellationRequest + + req = TileTessellationRequest( + region_source="cone", + tile_side_length_arc_min=6.0, + ra=10.0, + decl=20.0, + radius_deg=0.3, + ) + assert req.overlap_only is True + assert req.region_source.value == "cone" + + +def test_tessellation_request_bounding_box_valid(): + from src.api.schemas import TileTessellationRequest + + req = TileTessellationRequest( + region_source="bounding_box", + tile_side_length_arc_min=6.0, + min_ra=1.0, + max_ra=2.0, + min_decl=3.0, + max_decl=4.0, + overlap_only=False, + ) + assert req.overlap_only is False + + +def test_tessellation_request_project_footprint_valid(): + from src.api.schemas import TileTessellationRequest + + req = TileTessellationRequest( + region_source="project_footprint", tile_side_length_arc_min=6.0, project_id=7 + ) + assert req.project_id == 7 + + +def test_tessellation_request_cone_missing_fields_rejected(): + from src.api.schemas import TileTessellationRequest + + with pytest.raises(ValidationError, match="region_source=cone requires"): + TileTessellationRequest(region_source="cone", tile_side_length_arc_min=6.0) + + +def test_tessellation_request_project_footprint_missing_project_id_rejected(): + from src.api.schemas import TileTessellationRequest + + with pytest.raises( + ValidationError, match="region_source=project_footprint requires" + ): + TileTessellationRequest( + region_source="project_footprint", tile_side_length_arc_min=6.0 + ) + + +def test_tessellation_request_bounding_box_missing_fields_rejected(): + from src.api.schemas import TileTessellationRequest + + with pytest.raises(ValidationError, match="region_source=bounding_box requires"): + TileTessellationRequest( + region_source="bounding_box", tile_side_length_arc_min=6.0, min_ra=1.0 + ) diff --git a/src/api/tests/test_storage.py b/src/api/tests/test_storage.py new file mode 100644 index 0000000..fc41258 --- /dev/null +++ b/src/api/tests/test_storage.py @@ -0,0 +1,41 @@ +import pytest +from fastapi.testclient import TestClient + +from src.api.main import app + + +@pytest.fixture +def client(): + return TestClient(app) + + +def test_storage_sync_dispatches_and_returns_job_id(client, mocker): + mocker.patch( + "src.services.storage_service.dispatch_staging_sync", return_value="job-99" + ) + + response = client.post( + "/api/v1/storage/sync", + json={"staging_location": "/staging", "canonical_prefix": "raw"}, + ) + + assert response.status_code == 200 + assert response.json() == {"job_id": "job-99"} + + +def test_storage_sync_defaults_canonical_prefix_to_empty(client, mocker): + dispatch = mocker.patch( + "src.services.storage_service.dispatch_staging_sync", return_value="job-1" + ) + + response = client.post( + "/api/v1/storage/sync", json={"staging_location": "/staging"} + ) + + assert response.status_code == 200 + dispatch.assert_called_once_with("/staging", "") + + +def test_storage_sync_missing_staging_location_is_422(client): + response = client.post("/api/v1/storage/sync", json={"canonical_prefix": "raw"}) + assert response.status_code == 422 diff --git a/src/api/tests/test_tiles.py b/src/api/tests/test_tiles.py index 3008607..78304d9 100644 --- a/src/api/tests/test_tiles.py +++ b/src/api/tests/test_tiles.py @@ -22,10 +22,10 @@ def client(): return TestClient(app) -def test_tessellate_tiles_returns_preview_without_writing(client, mocker): +def test_tessellate_tiles_returns_preview_without_writing(client, mock_db, mocker): fake_moc = MOC.new_empty(10) - mocker.patch( - "src.services.tile_service.generate_tile_tessellation", + resolve = mocker.patch( + "src.services.tile_service.generate_tessellation_for_region", return_value=[ { "name": "Tile_1", @@ -41,10 +41,11 @@ def test_tessellate_tiles_returns_preview_without_writing(client, mocker): response = client.post( "/api/v1/tiles/tessellate", json={ - "project_id": 1, + "region_source": "cone", "tile_side_length_arc_min": 6.0, - "moc_to_tile": [[0, 10]], - "overlap_in_arc_min": 0.0, + "ra": 10.0, + "decl": 20.0, + "radius_deg": 0.3, }, ) @@ -53,6 +54,21 @@ def test_tessellate_tiles_returns_preview_without_writing(client, mocker): assert len(body) == 1 assert body[0]["name"] == "Tile_1" assert body[0]["footprint"] == [] # new_empty MOC has no ranges + # region_source + params were passed through to the service resolver. + assert resolve.call_args.kwargs["radius_deg"] == 0.3 + + +def test_tessellate_tiles_cone_missing_radius_is_422(client, mock_db): + response = client.post( + "/api/v1/tiles/tessellate", + json={ + "region_source": "cone", + "tile_side_length_arc_min": 6.0, + "ra": 10.0, + "decl": 20.0, + }, + ) + assert response.status_code == 422 def test_create_tiles_returns_persisted_tiles(client, mock_db, mocker): diff --git a/src/cli.py b/src/cli.py index f538b9d..1c74f06 100644 --- a/src/cli.py +++ b/src/cli.py @@ -159,41 +159,88 @@ def cmd_ingest_status(args: argparse.Namespace) -> None: _print_entity_table([batch]) -def _cone_moc(ra: float, decl: float, radius_deg: float): - """Build a MOC covering a cone region, the CLI's way to specify a tile-tessellation target.""" - import astropy.units as u - from mocpy import MOC +def cmd_sync_staging(args: argparse.Namespace) -> None: + """Dispatch a staging→canonical storage sync through the service layer and print its job id.""" + from src.services import storage_service - return MOC.from_cone( - lon=ra * u.deg, lat=decl * u.deg, radius=radius_deg * u.deg, max_depth=10 + job_id = storage_service.dispatch_staging_sync( + args.staging_prefix, args.canonical_prefix ) + print(f"Dispatched staging sync. job_id={job_id}") + + +def cmd_reconcile_stuck_jobs(args: argparse.Namespace) -> None: + """Fail any job stuck IN_PROCESS past the staleness threshold and print what changed.""" + from src.db.session import SessionLocal + from src.services import job_service + + db = SessionLocal() + try: + if args.threshold_seconds is not None: + reconciled = job_service.reconcile_stuck_jobs(db, args.threshold_seconds) + else: + reconciled = job_service.reconcile_stuck_jobs(db) + finally: + db.close() + + print(f"Reconciled {len(reconciled)} stuck job(s).") + for r in reconciled: + print(f" {r['entity']} id={r['id']} (age={r['age_seconds']:.0f}s) -> FAILED") + + +def _tessellation_region_kwargs(args: argparse.Namespace) -> dict: + """Collect the region-resolution kwargs from parsed tessellation CLI args.""" + return { + "ra": args.ra, + "decl": args.decl, + "radius_deg": args.radius_deg, + "project_id": args.region_project_id, + "min_ra": args.min_ra, + "max_ra": args.max_ra, + "min_decl": args.min_decl, + "max_decl": args.max_decl, + } def cmd_tessellate_tiles(args: argparse.Namespace) -> None: - """Preview a tile tessellation over a cone region and print it, without writing to the DB.""" + """Preview a tile tessellation over the requested region and print it, without writing to the DB.""" + from src.db.enums import RegionSource + from src.db.session import SessionLocal from src.services import tile_service - moc_to_tile = _cone_moc(args.ra, args.decl, args.radius_deg) - tiles = tile_service.generate_tile_tessellation( - args.tile_side_arcmin, moc_to_tile, args.overlap_arcmin - ) + db = SessionLocal() + try: + tiles = tile_service.generate_tessellation_for_region( + db, + RegionSource(args.region_source), + args.tile_side_arcmin, + args.overlap_arcmin, + args.overlap_only, + **_tessellation_region_kwargs(args), + ) + finally: + db.close() print(f"Generated {len(tiles)} tile(s):") for t in tiles: print(f" {t['name']}: ra={t['ra']:.4f}, decl={t['decl']:.4f}") def cmd_create_tiles(args: argparse.Namespace) -> None: - """Generate a tile tessellation over a cone region and persist it through the service layer.""" + """Generate a tile tessellation over the requested region and persist it through the service layer.""" + from src.db.enums import RegionSource from src.db.session import SessionLocal from src.services import tile_service - moc_to_tile = _cone_moc(args.ra, args.decl, args.radius_deg) - tiles = tile_service.generate_tile_tessellation( - args.tile_side_arcmin, moc_to_tile, args.overlap_arcmin - ) - db = SessionLocal() try: + tiles = tile_service.generate_tessellation_for_region( + db, + RegionSource(args.region_source), + args.tile_side_arcmin, + args.overlap_arcmin, + args.overlap_only, + **_tessellation_region_kwargs(args), + ) created = tile_service.create_tiles(db, args.project_id, tiles) finally: db.close() @@ -341,9 +388,16 @@ def _ingest_status(): return print("Ingest complete.") - moc_to_tile = _cone_moc(args.ra, args.decl, args.radius_deg) - tessellation = tile_service.generate_tile_tessellation( - args.tile_side_arcmin, moc_to_tile, args.overlap_arcmin + from src.db.enums import RegionSource + + tessellation = tile_service.generate_tessellation_for_region( + db, + RegionSource.CONE, + args.tile_side_arcmin, + args.overlap_arcmin, + ra=args.ra, + decl=args.decl, + radius_deg=args.radius_deg, ) tiles = tile_service.create_tiles(db, project.id, tessellation) print(f"Created {len(tiles)} tile(s).") @@ -399,6 +453,8 @@ def _mosaic_status(): def build_parser() -> argparse.ArgumentParser: """Construct the argparse parser with all diffpype-manage subcommands.""" + from src.db.enums import RegionSource + parser = argparse.ArgumentParser( prog="diffpype-manage", description="DevOps CLI for Diffpype administrative tasks.", @@ -443,7 +499,39 @@ def build_parser() -> argparse.ArgumentParser: "--id", type=int, required=True, metavar="ID", help="IngestBatch integer ID." ) + sync_staging = subparsers.add_parser( + "sync-staging", + help="Dispatch a staging→canonical storage sync (mc mirror) via the worker.", + ) + sync_staging.add_argument( + "--staging-prefix", + required=True, + help="Staging location (local path or s3:// URI) to mirror from.", + ) + sync_staging.add_argument( + "--canonical-prefix", + default="", + help="Canonical bucket prefix to mirror into (default: bucket root).", + ) + + reconcile = subparsers.add_parser( + "reconcile-stuck-jobs", + help="Fail any job stuck IN_PROCESS past the staleness threshold.", + ) + reconcile.add_argument( + "--threshold-seconds", + type=int, + default=None, + help="Staleness threshold in seconds (default: JOB_STALENESS_TIMEOUT_SECONDS).", + ) + def _add_tessellation_args(subparser: argparse.ArgumentParser) -> None: + subparser.add_argument( + "--region-source", + choices=[r.value for r in RegionSource], + default=RegionSource.CONE.value, + help="Region specification mode (default: cone).", + ) subparser.add_argument( "--tile-side-arcmin", type=float, @@ -451,16 +539,44 @@ def _add_tessellation_args(subparser: argparse.ArgumentParser) -> None: help="Tile side length (arcmin).", ) subparser.add_argument( - "--ra", type=float, required=True, help="Target cone center RA (deg)." + "--overlap-arcmin", type=float, default=0.0, help="Tile overlap (arcmin)." ) subparser.add_argument( - "--decl", type=float, required=True, help="Target cone center Dec (deg)." + "--overlap-only", + action=argparse.BooleanOptionalAction, + default=True, + help="Keep only tiles intersecting the region (default); " + "--no-overlap-only materializes the full grid.", ) + # cone subparser.add_argument( - "--radius-deg", type=float, required=True, help="Target cone radius (deg)." + "--ra", type=float, help="Cone center RA (deg) [region-source=cone]." ) subparser.add_argument( - "--overlap-arcmin", type=float, default=0.0, help="Tile overlap (arcmin)." + "--decl", type=float, help="Cone center Dec (deg) [region-source=cone]." + ) + subparser.add_argument( + "--radius-deg", type=float, help="Cone radius (deg) [region-source=cone]." + ) + # project_footprint + subparser.add_argument( + "--region-project-id", + type=int, + help="Project whose calibration footprints define the region " + "[region-source=project_footprint].", + ) + # bounding_box + subparser.add_argument( + "--min-ra", type=float, help="Min RA (deg) [region-source=bounding_box]." + ) + subparser.add_argument( + "--max-ra", type=float, help="Max RA (deg) [region-source=bounding_box]." + ) + subparser.add_argument( + "--min-decl", type=float, help="Min Dec (deg) [region-source=bounding_box]." + ) + subparser.add_argument( + "--max-decl", type=float, help="Max Dec (deg) [region-source=bounding_box]." ) tessellate_tiles = subparsers.add_parser( @@ -572,6 +688,10 @@ def main(argv: list[str] | None = None) -> None: cmd_ingest(args) elif args.command == "ingest-status": cmd_ingest_status(args) + elif args.command == "sync-staging": + cmd_sync_staging(args) + elif args.command == "reconcile-stuck-jobs": + cmd_reconcile_stuck_jobs(args) elif args.command == "tessellate-tiles": cmd_tessellate_tiles(args) elif args.command == "create-tiles": diff --git a/src/core/config.py b/src/core/config.py index 2a16b28..fb51f9d 100644 --- a/src/core/config.py +++ b/src/core/config.py @@ -29,6 +29,17 @@ class Settings(BaseSettings): s3_region: str = "us-east-1" storage_backend: str = "s3" local_storage_root: str = "./data" + staging_location: str = "./data/staging" + staging_sync_interval_seconds: int = 300 + enable_staging_sync_cron: bool = True + staging_sync_soft_time_limit_seconds: int = 1800 + ingest_batch_soft_time_limit_seconds: int = 7200 + mosaic_drizzle_soft_time_limit_seconds: int = 3600 + cli_tool_soft_time_limit_seconds: int = 3600 + db_backup_soft_time_limit_seconds: int = 3600 + dlq_dump_soft_time_limit_seconds: int = 30 + reconcile_stuck_jobs_soft_time_limit_seconds: int = 120 + job_staleness_timeout_seconds: int = 3600 settings = Settings() # type: ignore[call-arg] # database_url/redis_url are populated from the environment at runtime; mypy can't see that. diff --git a/src/core/tests/test_docker_compose_prod.py b/src/core/tests/test_docker_compose_prod.py index a32d601..0c15065 100644 --- a/src/core/tests/test_docker_compose_prod.py +++ b/src/core/tests/test_docker_compose_prod.py @@ -20,7 +20,8 @@ def test_prod_compose_images_match_ghcr_package_names(): """docker-compose.prod.yml must reference the same image names ci.yml pushes to ghcr.io.""" content = COMPOSE_PATH.read_text() assert content.count(f"image: {EXPECTED_IMAGE_PREFIX}api:") == 1 - assert content.count(f"image: {EXPECTED_IMAGE_PREFIX}worker:") == 2 + # worker_light, worker_heavy, and beat all run the same worker image (doc 30 §4). + assert content.count(f"image: {EXPECTED_IMAGE_PREFIX}worker:") == 3 assert content.count(f"image: {EXPECTED_IMAGE_PREFIX}db:") == 1 diff --git a/src/db/enums.py b/src/db/enums.py index ca8ad79..2ec1fc9 100644 --- a/src/db/enums.py +++ b/src/db/enums.py @@ -12,3 +12,11 @@ class CeleryQueue(str, enum.Enum): LIGHT = "light" HEAVY_MEMORY = "heavy_memory" GPU = "gpu" + + +class RegionSource(str, enum.Enum): + """How a tile-tessellation request specifies the sky region to cover.""" + + CONE = "cone" + PROJECT_FOOTPRINT = "project_footprint" + BOUNDING_BOX = "bounding_box" diff --git a/src/db/models.py b/src/db/models.py index a751562..4da0ddc 100644 --- a/src/db/models.py +++ b/src/db/models.py @@ -134,6 +134,10 @@ class JobConfiguration(TimestampMixin, Base): id: Mapped[int] = mapped_column(sa.Integer, primary_key=True) job_kwargs: Mapped[dict | None] = mapped_column(sa.JSON, nullable=True) execution_command: Mapped[str | None] = mapped_column(sa.String, nullable=True) + # Which Celery task (or CLI tool) this row's kwargs/command correspond to, + # e.g. "src.worker.tasks.run_ingest_batch". Nullable for provenance rows that + # predate the column or aren't tied to a named task. + task_name: Mapped[str | None] = mapped_column(sa.String, nullable=True) user_id: Mapped[int] = mapped_column(sa.ForeignKey("users.id"), nullable=False) user: Mapped["User"] = relationship(back_populates="job_configurations") diff --git a/src/db/seed.py b/src/db/seed.py index 7359a39..cc01164 100644 --- a/src/db/seed.py +++ b/src/db/seed.py @@ -1,10 +1,16 @@ import bcrypt from sqlalchemy.orm import Session +from sqlalchemy_celery_beat.models import IntervalSchedule, PeriodicTask, Period from src.core.config import settings from src.db.models import Band, Instrument, User from src.db.session import SessionLocal +# Check cadence for the stuck-job watchdog Beat task. Distinct from the staleness +# *threshold* (JOB_STALENESS_TIMEOUT_SECONDS): this is how often the sweep runs, +# not how old a job must be to be failed. +_RECONCILE_CHECK_INTERVAL_SECONDS = 300 + # Baseline JWST reference data so a fresh sandbox is immediately usable. Central # wavelengths are the filter pivot wavelengths in microns. The complete standard # NIRCam (wide/medium/narrow) + MIRI imaging filter sets — not a curated subset — @@ -66,8 +72,46 @@ def _seed_reference_data(db: Session) -> None: db.add(Band(name=name, central_lambda=central_lambda)) +def _seed_periodic_tasks(db: Session) -> None: + """Seed the initial staging-sync and stuck-job-reconcile Beat schedules if none exist. + + A no-op once any PeriodicTask exists, so runtime edits made via SQLAdmin are + never clobbered by a later ``seed-db``. + """ + if db.query(PeriodicTask).count() > 0: + return + + sync_interval = IntervalSchedule( + every=settings.staging_sync_interval_seconds, period=Period.SECONDS + ) + reconcile_interval = IntervalSchedule( + every=_RECONCILE_CHECK_INTERVAL_SECONDS, period=Period.SECONDS + ) + # Flush so each schedule gets its id before PeriodicTask.schedule_model reads + # it to populate the polymorphic (discriminator, schedule_id) association. + db.add_all([sync_interval, reconcile_interval]) + db.flush() + + db.add_all( + [ + PeriodicTask( + schedule_model=sync_interval, + name="sync-staging-cron", + task="src.worker.tasks.sync_staging_cron", + enabled=settings.enable_staging_sync_cron, + ), + PeriodicTask( + schedule_model=reconcile_interval, + name="reconcile-stuck-jobs-cron", + task="src.worker.tasks.reconcile_stuck_jobs_cron", + enabled=True, + ), + ] + ) + + def seed_step_definitions() -> None: - """Upsert the sysadmin User and baseline Instrument/Band reference data.""" + """Upsert the sysadmin User, baseline Instrument/Band reference data, and Beat schedules.""" db = SessionLocal() try: hashed = bcrypt.hashpw( @@ -88,6 +132,7 @@ def seed_step_definitions() -> None: db.flush() _seed_reference_data(db) + _seed_periodic_tasks(db) db.commit() finally: db.close() diff --git a/src/db/spatial_types.py b/src/db/spatial_types.py index f06a76b..f976c71 100644 --- a/src/db/spatial_types.py +++ b/src/db/spatial_types.py @@ -40,6 +40,8 @@ from __future__ import annotations +from collections.abc import Sequence + import astropy.units as u import numpy as np import sqlalchemy as sa @@ -139,6 +141,20 @@ def _point_to_depth29_cell(ra_deg: float, decl_deg: float) -> int: return int(moc.to_depth29_ranges[0][0]) +def union_mocs(mocs: Sequence[MOC]) -> MOC: + """Return the union of a non-empty sequence of MOCs. + + The single shared implementation of "combine these footprints into one" — + consumed by both ``mosaic_service`` (constituent-calibration footprint union) + and ``tile_service`` (project-footprint region derivation), replacing the + prototype's ``DataUtils.Get_Unioned_MOC``. + """ + union = mocs[0] + for moc in mocs[1:]: + union = union.union(moc) + return union + + def moc_to_ranges(moc: MOC | None) -> list[tuple[int, int]] | None: """Convert a MOC to a plain list of depth-29 ``[lo, hi)`` integer range pairs. diff --git a/src/db/tests/test_integration.py b/src/db/tests/test_integration.py index 2648d38..987bc87 100644 --- a/src/db/tests/test_integration.py +++ b/src/db/tests/test_integration.py @@ -255,6 +255,8 @@ def test_sysadmin_seeding_creates_sysadmin_user(mocker, test_engine): def _cleanup_seeded_rows(TestSession): """Delete every row seed_step_definitions() commits outside the transactional fixture.""" + from sqlalchemy_celery_beat.models import IntervalSchedule, PeriodicTask + cleanup = TestSession() try: sysadmin_id = cleanup.query(User.id).filter_by(username="sysadmin").scalar() @@ -267,6 +269,11 @@ def _cleanup_seeded_rows(TestSession): cleanup.query(Band).filter(Band.name.in_(["F150W", "F277W"])).delete( synchronize_session=False ) + # seed_step_definitions also seeds the Beat schedules (doc 30 §4), committed + # outside the fixture. Delete PeriodicTask before IntervalSchedule (the + # former references the latter via the polymorphic schedule association). + cleanup.query(PeriodicTask).delete() + cleanup.query(IntervalSchedule).delete() cleanup.commit() finally: cleanup.close() @@ -1212,6 +1219,13 @@ def _make_associated_cal(base_filename, ra, band_row): Band.name.in_(["F150W-mosaicsvc", "F277W-mosaicsvc"]) ).delete(synchronize_session=False) db.query(Instrument).filter_by(name="NIRCam-mosaicsvc").delete() + # create_mosaic now commits a JobConfiguration referencing this user + # (doc 30 §3); it must be deleted before the user or the FK blocks it. + db.query(JobConfiguration).filter( + JobConfiguration.user_id.in_( + db.query(User.id).filter_by(username="mosaicserviceowner") + ) + ).delete(synchronize_session=False) db.query(User).filter_by(username="mosaicserviceowner").delete() db.commit() db.close() @@ -1321,3 +1335,148 @@ def test_bulk_upsert_images_and_calibrations_is_idempotent(test_engine): db.query(User).filter_by(username="bulkupsertowner").delete() db.commit() db.close() + + +# --- Operational services (doc 30) --- + + +def test_seeding_creates_beat_schedules(mocker, test_engine): + """seed_step_definitions seeds the sync/reconcile Beat schedules idempotently.""" + from sqlalchemy_celery_beat.models import PeriodicTask + + from src.db.seed import seed_step_definitions + + TestSession = sessionmaker(bind=test_engine) + mocker.patch("src.db.seed.SessionLocal", side_effect=TestSession) + + seed_step_definitions() + + db = TestSession() + try: + names = {t.name for t in db.query(PeriodicTask).all()} + assert "sync-staging-cron" in names + assert "reconcile-stuck-jobs-cron" in names + # A second seed must not duplicate the schedules. + seed_step_definitions() + assert db.query(PeriodicTask).count() == 2 + finally: + db.close() + + _cleanup_seeded_rows(TestSession) + + +def test_reconcile_stuck_jobs_fails_stale_in_process_records(test_engine): + """Real-DB watchdog sweep across both tracked entity types + a per-job override. + + Own session (reconcile_stuck_jobs commits internally), with explicit cleanup + per the integration-test isolation rule. + """ + from src.services.job_service import reconcile_stuck_jobs + + TestSession = sessionmaker(bind=test_engine) + db = TestSession() + user = project = None + try: + user = User( + username="watchdogowner", + email="watchdogowner@diffpype.local", + is_active=True, + hashed_password="dummy_hash_for_testing", + ) + db.add(user) + db.flush() + project = _make_project(db, user, name="WatchdogProject") + instrument, band = _make_ref(db, "NIRCam-wd", "F150W-wd") + tile = _make_tile(db, project, name="WD-Tile") + epoch = _make_epoch(db, project, tile, band) + + stale_batch = IngestBatch( + project_id=project.id, s3_prefix="raw/", status=JobStatus.IN_PROCESS + ) + fresh_batch = IngestBatch( + project_id=project.id, s3_prefix="raw/", status=JobStatus.IN_PROCESS + ) + override_config = JobConfiguration( + user_id=user.id, + task_name="src.worker.tasks.run_ingest_batch", + job_kwargs={"staleness_timeout_seconds": 60}, + ) + db.add(override_config) + db.flush() + override_batch = IngestBatch( + project_id=project.id, + s3_prefix="raw/", + status=JobStatus.IN_PROCESS, + job_configuration_id=override_config.id, + ) + stale_mosaic = Level3Mosaic( + filename="wd.fits", + target_plate_scale=0.03, + instrument_id=instrument.id, + band_id=band.id, + epoch_id=epoch.id, + tile_id=tile.id, + project_id=project.id, + status=JobStatus.IN_PROCESS, + ) + db.add_all([stale_batch, fresh_batch, override_batch, stale_mosaic]) + db.commit() + + # updated_at is server-managed; force the stale rows' clocks into the past. + db.execute( + text( + "UPDATE ingest_batches SET updated_at = now() - interval '2 hours' " + "WHERE id = :i" + ).bindparams(i=stale_batch.id) + ) + # override_batch: aged 120s -> survives the 3600s global default, but the + # per-job 60s override makes it stale. + db.execute( + text( + "UPDATE ingest_batches SET updated_at = now() - interval '120 seconds' " + "WHERE id = :i" + ).bindparams(i=override_batch.id) + ) + db.execute( + text( + "UPDATE level3_mosaics SET updated_at = now() - interval '2 hours' " + "WHERE id = :i" + ).bindparams(i=stale_mosaic.id) + ) + db.commit() + + reconciled = reconcile_stuck_jobs(db, staleness_timeout_seconds=3600) + + for row in (stale_batch, fresh_batch, override_batch, stale_mosaic): + db.refresh(row) + assert stale_batch.status == JobStatus.FAILED + assert fresh_batch.status == JobStatus.IN_PROCESS + assert override_batch.status == JobStatus.FAILED + assert stale_mosaic.status == JobStatus.FAILED + reconciled_ids = {(r["entity"], r["id"]) for r in reconciled} + assert ("IngestBatch", stale_batch.id) in reconciled_ids + assert ("Level3Mosaic", stale_mosaic.id) in reconciled_ids + finally: + if project is not None: + db.query(Level3Mosaic).filter_by(project_id=project.id).delete( + synchronize_session=False + ) + db.query(IngestBatch).filter_by(project_id=project.id).delete( + synchronize_session=False + ) + db.query(Epoch).filter_by(project_id=project.id).delete( + synchronize_session=False + ) + db.query(Tile).filter_by(project_id=project.id).delete( + synchronize_session=False + ) + db.query(Project).filter_by(id=project.id).delete() + if user is not None: + db.query(JobConfiguration).filter_by(user_id=user.id).delete( + synchronize_session=False + ) + db.query(User).filter_by(id=user.id).delete() + db.query(Band).filter_by(name="F150W-wd").delete() + db.query(Instrument).filter_by(name="NIRCam-wd").delete() + db.commit() + db.close() diff --git a/src/db/tests/test_spatial_types.py b/src/db/tests/test_spatial_types.py index 93dbad3..ab79167 100644 --- a/src/db/tests/test_spatial_types.py +++ b/src/db/tests/test_spatial_types.py @@ -8,9 +8,30 @@ _point_to_depth29_cell, moc_to_ranges, ranges_to_moc, + union_mocs, ) +def test_union_mocs_single_moc_is_a_no_op(): + moc = MOC.from_cone( + lon=10 * u.deg, lat=20 * u.deg, radius=0.1 * u.deg, max_depth=10 + ) + result = union_mocs([moc]) + assert result == moc + + +def test_union_mocs_combines_disjoint_regions(): + a = MOC.from_cone(lon=10 * u.deg, lat=0 * u.deg, radius=0.1 * u.deg, max_depth=10) + b = MOC.from_cone(lon=200 * u.deg, lat=0 * u.deg, radius=0.1 * u.deg, max_depth=10) + + result = union_mocs([a, b]) + + # The union covers both cones' sky and is at least as large as either alone. + assert result.sky_fraction >= a.sky_fraction + assert result.sky_fraction >= b.sky_fraction + assert result == a.union(b) + + def test_point_to_depth29_cell_is_a_stable_positive_int(): cell = _point_to_depth29_cell(180.0, 0.0) assert isinstance(cell, int) diff --git a/src/services/ingest_service.py b/src/services/ingest_service.py index 13a9766..91e12fe 100644 --- a/src/services/ingest_service.py +++ b/src/services/ingest_service.py @@ -22,7 +22,15 @@ from src.core.logger import get_logger from src.db.enums import JobStatus -from src.db.models import Band, IngestBatch, Instrument, Level2Calibration, Level2Image +from src.db.models import ( + Band, + IngestBatch, + Instrument, + Level2Calibration, + Level2Image, + Project, +) +from src.services import job_service # HEALPix order used for per-image footprints computed from a WCS polygon — matches # the prototype's ported convention for detector-scale (not all-sky-tile-scale) @@ -171,8 +179,20 @@ def create_ingest_batch( """Persist a PENDING IngestBatch, dispatch the ingest Celery task, return (job_id, batch_id).""" from src.worker.tasks import run_ingest_batch # lazy: avoids a circular import + project = db.get(Project, project_id) + assert project is not None, f"Project {project_id} not found" + job_config = job_service.create_job_configuration( + db, + user_id=project.user_id, + task_name="src.worker.tasks.run_ingest_batch", + job_kwargs={"project_id": project_id, "s3_prefix": s3_prefix}, + ) + batch = IngestBatch( - project_id=project_id, s3_prefix=s3_prefix, status=JobStatus.PENDING + project_id=project_id, + s3_prefix=s3_prefix, + status=JobStatus.PENDING, + job_configuration_id=job_config.id, ) db.add(batch) db.commit() diff --git a/src/services/job_service.py b/src/services/job_service.py index aae8f8e..8694fe9 100644 --- a/src/services/job_service.py +++ b/src/services/job_service.py @@ -1,7 +1,94 @@ -"""Shared job-status service layer for API and CLI boundaries. +"""Shared job-provenance and stuck-job-reconciliation service for API and CLI. -The Stage-0 dummy-job dispatch/lookup helpers that originally lived here were -removed when the ``DummyImage`` scaffolding was decommissioned (doc 29). The -stuck-job reconciliation/watchdog service (``reconcile_stuck_jobs``) is scoped -to doc 30 and will land here. +Two responsibilities live here: + +* ``create_job_configuration`` — the single place a dispatch path records its + provenance (``task_name`` + ``job_kwargs`` + owning user) as a + ``JobConfiguration`` row, linked from the tracked entity it dispatches. +* ``reconcile_stuck_jobs`` — the watchdog that fails any tracked job left in + ``IN_PROCESS`` past its staleness threshold (an uncatchable worker crash / OOM + kill can't run the task's own ``except`` block, so the row would otherwise sit + ``IN_PROCESS`` forever). """ + +from datetime import datetime, timezone +from typing import cast + +import sqlalchemy as sa +from sqlalchemy.orm import Session + +from src.core.config import settings +from src.db.enums import JobStatus +from src.db.models import IngestBatch, JobConfiguration, Level3Mosaic + +# Static registry of the job entities the watchdog reconciles. Each shares the +# same tracked columns: ``status`` (JobStatus), ``updated_at`` (staleness clock), +# and ``job_configuration`` (per-job override source). Add an entity here to +# bring it under the watchdog. +_STUCK_JOB_ENTITIES = (IngestBatch, Level3Mosaic) + + +def create_job_configuration( + db: Session, + user_id: int, + task_name: str, + job_kwargs: dict | None = None, +) -> JobConfiguration: + """Create and flush a JobConfiguration provenance row, returning it with its id assigned. + + Flushed (not committed) so it participates in the caller's transaction — the + dispatch path commits it atomically with the tracked entity row it links to. + """ + job_config = JobConfiguration( + user_id=user_id, task_name=task_name, job_kwargs=job_kwargs + ) + db.add(job_config) + db.flush() + return job_config + + +def reconcile_stuck_jobs( + db: Session, + staleness_timeout_seconds: int = settings.job_staleness_timeout_seconds, +) -> list[dict]: + """Fail every tracked job stuck IN_PROCESS past its staleness threshold; return what changed. + + Honors a per-job ``staleness_timeout_seconds`` override in the linked + ``JobConfiguration.job_kwargs`` over the global default. Calls ``db.rollback()`` + first so a prior failed transaction on this session can't poison the status + writes (mirrors the framework error-handler pattern). + """ + db.rollback() + now = datetime.now(timezone.utc) + reconciled: list[dict] = [] + + for model in _STUCK_JOB_ENTITIES: + rows = ( + db.execute(sa.select(model).where(model.status == JobStatus.IN_PROCESS)) + .scalars() + .all() + ) + for raw_row in rows: + # Both registered entities share these columns (status/updated_at/id + # via TimestampMixin, plus job_configuration); the cast tells the type + # checker that, since the registry widens the row to the ORM Base. + row = cast("IngestBatch | Level3Mosaic", raw_row) + timeout = staleness_timeout_seconds + job_config = row.job_configuration + if job_config is not None and job_config.job_kwargs: + override = job_config.job_kwargs.get("staleness_timeout_seconds") + if override is not None: + timeout = override + age_seconds = (now - row.updated_at).total_seconds() + if age_seconds > timeout: + row.status = JobStatus.FAILED + reconciled.append( + { + "entity": model.__name__, + "id": row.id, + "age_seconds": age_seconds, + } + ) + + db.commit() + return reconciled diff --git a/src/services/mosaic_service.py b/src/services/mosaic_service.py index a0f42ad..62732c8 100644 --- a/src/services/mosaic_service.py +++ b/src/services/mosaic_service.py @@ -17,21 +17,22 @@ Level2Calibration, Level2Image, Level3Mosaic, + Project, epoch_level2_calibration_association, tile_level2_calibration_association, ) +from src.db.spatial_types import union_mocs +from src.services import job_service def _unioned_footprint_and_barycenter(mocs: list[MOC]) -> tuple[MOC, float, float]: """Union a list of MOCs and return (union, ra_barycenter_deg, decl_barycenter_deg). - Ports the prototype's ``Get_Unioned_MOC``, plus barycenter extraction — - addresses GitHub issue #27 in the same write path that already has to - compute the union. + Delegates the union to the shared ``spatial_types.union_mocs`` helper, plus + barycenter extraction — addresses GitHub issue #27 in the same write path that + already has to compute the union. """ - union = mocs[0] - for moc in mocs[1:]: - union = union.union(moc) + union = union_mocs(mocs) barycenter = union.barycenter() return union, float(barycenter.ra.degree), float(barycenter.dec.degree) @@ -93,6 +94,21 @@ def create_mosaic( if footprints: footprint, ra, decl = _unioned_footprint_and_barycenter(footprints) + project = db.get(Project, project_id) + assert project is not None, f"Project {project_id} not found" + job_config = job_service.create_job_configuration( + db, + user_id=project.user_id, + task_name="src.worker.tasks.run_mosaic_drizzle", + job_kwargs={ + "project_id": project_id, + "tile_id": tile_id, + "epoch_id": epoch_id, + "band_id": band_id, + "instrument_id": instrument_id, + }, + ) + mosaic = Level3Mosaic( filename=filename, target_plate_scale=target_plate_scale, @@ -107,6 +123,7 @@ def create_mosaic( epoch_id=epoch_id, tile_id=tile_id, project_id=project_id, + job_configuration_id=job_config.id, status=JobStatus.PENDING, ) db.add(mosaic) diff --git a/src/services/storage_service.py b/src/services/storage_service.py index 7ead455..946bc70 100644 --- a/src/services/storage_service.py +++ b/src/services/storage_service.py @@ -6,13 +6,21 @@ selects between them via the `STORAGE_BACKEND` setting. """ +import json import shutil +import subprocess from pathlib import Path from typing import Protocol, runtime_checkable import boto3 +from celery.exceptions import SoftTimeLimitExceeded from src.core.config import Settings, settings +from src.core.logger import get_logger + +# Reused, idempotently-registered `mc` alias name for the canonical S3/MinIO +# endpoint. `mc alias set` is a no-op overwrite when re-run with the same values. +_MC_CANONICAL_ALIAS = "diffpype-canonical" @runtime_checkable @@ -98,13 +106,31 @@ def download_file(self, key: str, local_path: str) -> None: shutil.copyfile(source, local_path) def list_prefix(self, prefix: str) -> list[str]: - """Return all relative keys under the given prefix in the storage root.""" - base = self._root / prefix - if not base.is_dir(): + """Return all relative keys whose path string starts with the given prefix. + + Matches S3's pure string-prefix semantics: a partial-filename prefix like + ``raw/jw0123`` matches ``raw/jw0123_nrca.fits`` rather than requiring + ``prefix`` to name an existing directory (the previous behavior, which + silently returned nothing for a narrow prefix). Dot-prefixed hidden files + (e.g. a partially-downloaded ``.tmp``) are excluded so an in-flight + transfer is never surfaced as an ingestable key. + """ + # Walk from the deepest real directory at/above the prefix (so a narrow + # prefix doesn't force a full-root scan), then keep only files whose + # root-relative path string actually starts with the prefix. + search_base = self._root / prefix + if not search_base.is_dir(): + search_base = search_base.parent + if not search_base.is_dir(): return [] - return sorted( - str(p.relative_to(self._root)) for p in base.rglob("*") if p.is_file() - ) + keys: list[str] = [] + for path in search_base.rglob("*"): + if not path.is_file() or path.name.startswith("."): + continue + rel = str(path.relative_to(self._root)) + if rel.startswith(prefix): + keys.append(rel) + return sorted(keys) def get_storage_service(config: Settings = settings) -> StorageBackend: @@ -112,3 +138,147 @@ def get_storage_service(config: Settings = settings) -> StorageBackend: if config.storage_backend == "local": return LocalStorageService(config) return S3StorageService(config) + + +def _resolve_mc_canonical_target(config: Settings, canonical_prefix: str) -> str: + """Return the ``mc`` target path for the canonical store, registering an S3 alias if needed. + + For the ``local`` backend the canonical store is a filesystem path under + ``LOCAL_STORAGE_ROOT``; for S3/MinIO it is ``//`` after + an idempotent ``mc alias set`` against the configured endpoint/credentials. + """ + if config.storage_backend == "local": + return str(Path(config.local_storage_root) / canonical_prefix) + subprocess.run( + [ + "mc", + "alias", + "set", + _MC_CANONICAL_ALIAS, + config.s3_endpoint_url, + config.aws_access_key_id, + config.aws_secret_access_key, + ], + check=True, + capture_output=True, + text=True, + ) + target = f"{_MC_CANONICAL_ALIAS}/{config.s3_bucket_name}" + return f"{target}/{canonical_prefix}" if canonical_prefix else target + + +def sync_staging_to_canonical( + staging_location: str, canonical_prefix: str, config: Settings = settings +) -> None: + """Mirror new/changed files from a staging location into the canonical bucket via ``mc mirror``. + + Consumes ``mc mirror``'s ``--json`` per-file event stream line by line, logging + one structured record per copied object — never ``subprocess.run(capture_output=True)``, + which would buffer the entire transfer in memory before any progress is visible. + ``mc mirror`` is itself diff-based and idempotent, so a redelivered Celery retry + after a worker crash simply skips whatever was already copied. + + ``mc``'s real ``--json`` stream (confirmed against a live MinIO instance with + real FITS files, not assumed) emits three distinct shapes: a per-file success + event (``target``/``source``/``size``/``totalCount``/``totalSize`` — the latter + two are cumulative running totals across the whole mirror, giving real + progress even though there's no intra-file byte-level percentage); a per-file + or whole-job error event (``status: "error"``, an ``error`` object, no + ``target``); and a single trailing job-summary event with no ``target``/ + ``source`` at all (``total``/``transferred``/``duration``/``speed``). These are + discriminated explicitly below rather than treating every parsed line as a + file copy. + """ + log = get_logger() + canonical_target = _resolve_mc_canonical_target(config, canonical_prefix) + log.info( + "staging_sync_started", + staging_location=staging_location, + canonical_target=canonical_target, + ) + + # `--json` before the subcommand: mc emits one JSON object per transferred + # object on stdout. stderr is folded into stdout so a failure diagnostic is + # captured in the same stream we already read incrementally. + process = subprocess.Popen( + ["mc", "--json", "mirror", staging_location, canonical_target], + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + ) + copied = 0 + assert process.stdout is not None + try: + for line in process.stdout: + line = line.strip() + if not line: + continue + try: + event = json.loads(line) + except json.JSONDecodeError: + log.warning("staging_sync_unparsed_line", line=line) + continue + + target = event.get("target") + if target is not None and event.get("status") != "error": + # Real per-file success event. + copied += 1 + log.info( + "staging_sync_file_copied", + key=target, + size=event.get("size"), + index=copied, + total_count=event.get("totalCount"), + total_size=event.get("totalSize"), + ) + elif event.get("status") == "error": + # Either a per-file error (target present) or a whole-job error + # (target absent, e.g. the destination bucket doesn't exist). + log.warning( + "staging_sync_file_error", + key=target, + source=event.get("source"), + error=event.get("error"), + ) + else: + # The trailing job-summary event: no target/source at all. + log.info( + "staging_sync_job_summary", + total_bytes=event.get("total"), + transferred_bytes=event.get("transferred"), + duration_ns=event.get("duration"), + speed_bytes_per_sec=event.get("speed"), + ) + return_code = process.wait() + except SoftTimeLimitExceeded: + # A genuine hang (network partition mid-transfer, not a clean crash) — + # kill the subprocess so it doesn't outlive the task as an orphan, then + # let Celery's own timeout failure handling (on_failure + DLQ) proceed. + log.error("staging_sync_timed_out", files_copied_before_timeout=copied) + process.kill() + process.wait() + raise + if return_code != 0: + raise RuntimeError(f"mc mirror exited with code {return_code}") + log.info( + "staging_sync_completed", files_copied=copied, canonical_target=canonical_target + ) + + +def dispatch_staging_sync(staging_location: str, canonical_prefix: str) -> str: + """Dispatch the staging→canonical sync Celery task and return its job id. + + The shared entry point for both the API route and the CLI command — neither + boundary runs ``mc`` in-process; the worker (where the ``mc`` binary lives) + does. Returns the Celery task id. + """ + from src.worker.tasks import run_staging_sync # lazy: avoids a circular import + + async_result = run_staging_sync.delay(staging_location, canonical_prefix) + get_logger().info( + "staging_sync_dispatched", + job_id=async_result.id, + staging_location=staging_location, + canonical_prefix=canonical_prefix, + ) + return async_result.id diff --git a/src/services/tests/test_ingest_service.py b/src/services/tests/test_ingest_service.py index 8be9472..f294379 100644 --- a/src/services/tests/test_ingest_service.py +++ b/src/services/tests/test_ingest_service.py @@ -5,7 +5,7 @@ import pytest from src.db.enums import JobStatus -from src.db.models import IngestBatch +from src.db.models import IngestBatch, JobConfiguration from src.services.ingest_service import ( _resolve_reference_ids, bulk_upsert_images_and_calibrations, @@ -137,6 +137,7 @@ def test_bulk_upsert_returns_zero_for_empty_dataframe(): def test_create_ingest_batch_dispatches_and_returns_ids(mocker): mock_db = MagicMock() + mock_db.get.return_value = MagicMock(user_id=42) # the owning Project mock_db.refresh.side_effect = lambda obj: setattr(obj, "id", 11) fake_result = MagicMock(id="ingest-task-id") mock_delay = mocker.patch( @@ -147,11 +148,14 @@ def test_create_ingest_batch_dispatches_and_returns_ids(mocker): assert job_id == "ingest-task-id" assert batch_id == 11 - mock_db.add.assert_called_once() - added = mock_db.add.call_args[0][0] - assert isinstance(added, IngestBatch) - assert added.project_id == 1 - assert added.s3_prefix == "raw/" + # Two rows added: the JobConfiguration provenance row, then the IngestBatch. + added = [c.args[0] for c in mock_db.add.call_args_list] + job_config = next(o for o in added if isinstance(o, JobConfiguration)) + assert job_config.task_name == "src.worker.tasks.run_ingest_batch" + assert job_config.user_id == 42 + batch = next(o for o in added if isinstance(o, IngestBatch)) + assert batch.project_id == 1 + assert batch.s3_prefix == "raw/" mock_delay.assert_called_once_with(11) diff --git a/src/services/tests/test_job_service.py b/src/services/tests/test_job_service.py new file mode 100644 index 0000000..219a6e9 --- /dev/null +++ b/src/services/tests/test_job_service.py @@ -0,0 +1,90 @@ +import datetime as dt +from types import SimpleNamespace +from unittest.mock import MagicMock + +from src.db.enums import JobStatus +from src.services.job_service import create_job_configuration, reconcile_stuck_jobs + + +def _execute_result(rows): + result = MagicMock() + result.scalars.return_value.all.return_value = rows + return result + + +def test_create_job_configuration_flushes_and_returns_row(): + db = MagicMock() + + jc = create_job_configuration( + db, + user_id=1, + task_name="src.worker.tasks.run_ingest_batch", + job_kwargs={"a": 1}, + ) + + db.add.assert_called_once_with(jc) + db.flush.assert_called_once() + assert jc.user_id == 1 + assert jc.task_name == "src.worker.tasks.run_ingest_batch" + assert jc.job_kwargs == {"a": 1} + + +def test_reconcile_marks_only_stale_in_process_jobs_failed(): + now = dt.datetime.now(dt.timezone.utc) + stale = SimpleNamespace( + id=1, + status=JobStatus.IN_PROCESS, + updated_at=now - dt.timedelta(seconds=7200), + job_configuration=None, + ) + fresh = SimpleNamespace( + id=2, + status=JobStatus.IN_PROCESS, + updated_at=now - dt.timedelta(seconds=10), + job_configuration=None, + ) + db = MagicMock() + # First query -> IngestBatch rows; second -> Level3Mosaic rows (none). + db.execute.side_effect = [_execute_result([stale, fresh]), _execute_result([])] + + result = reconcile_stuck_jobs(db, staleness_timeout_seconds=3600) + + db.rollback.assert_called_once() + db.commit.assert_called_once() + assert stale.status == JobStatus.FAILED + assert fresh.status == JobStatus.IN_PROCESS + assert result == [ + {"entity": "IngestBatch", "id": 1, "age_seconds": result[0]["age_seconds"]} + ] + assert result[0]["age_seconds"] > 3600 + + +def test_reconcile_honors_per_job_staleness_override(): + now = dt.datetime.now(dt.timezone.utc) + # Aged 200s: under the global 3600 default (would survive), but a per-job + # override of 100s makes it stale. + job_config = SimpleNamespace(job_kwargs={"staleness_timeout_seconds": 100}) + row = SimpleNamespace( + id=5, + status=JobStatus.IN_PROCESS, + updated_at=now - dt.timedelta(seconds=200), + job_configuration=job_config, + ) + db = MagicMock() + db.execute.side_effect = [_execute_result([row]), _execute_result([])] + + result = reconcile_stuck_jobs(db, staleness_timeout_seconds=3600) + + assert row.status == JobStatus.FAILED + assert len(result) == 1 + + +def test_reconcile_no_stale_jobs_returns_empty_but_still_commits(): + db = MagicMock() + db.execute.side_effect = [_execute_result([]), _execute_result([])] + + result = reconcile_stuck_jobs(db) + + assert result == [] + db.rollback.assert_called_once() + db.commit.assert_called_once() diff --git a/src/services/tests/test_storage_service.py b/src/services/tests/test_storage_service.py index 349b87b..bab7dc1 100644 --- a/src/services/tests/test_storage_service.py +++ b/src/services/tests/test_storage_service.py @@ -189,6 +189,227 @@ def test_local_storage_service_list_prefix_empty_when_prefix_missing(tmp_path): assert svc.list_prefix("nope") == [] +def test_local_list_prefix_matches_partial_filename_prefix(tmp_path): + """A narrow prefix like 'raw/jw0123' matches by string, not by requiring a directory.""" + config = SimpleNamespace(local_storage_root=str(tmp_path / "root")) + svc = LocalStorageService(config=config) + svc.upload_file(str(_write(tmp_path, "a")), "raw/jw0123_nrca.fits") + svc.upload_file(str(_write(tmp_path, "b")), "raw/jw0999_nrcb.fits") + + assert svc.list_prefix("raw/jw0123") == ["raw/jw0123_nrca.fits"] + + +def test_local_list_prefix_empty_when_prefix_and_parent_both_missing(tmp_path): + """A deep prefix whose parent directory also doesn't exist returns [].""" + config = SimpleNamespace(local_storage_root=str(tmp_path / "root")) + svc = LocalStorageService(config=config) + + assert svc.list_prefix("no/such/deep") == [] + + +def test_local_list_prefix_excludes_dot_files(tmp_path): + """A partially-downloaded hidden file (dot-prefixed) is never surfaced as a key.""" + config = SimpleNamespace(local_storage_root=str(tmp_path / "root")) + svc = LocalStorageService(config=config) + svc.upload_file(str(_write(tmp_path, "a")), "raw/good.fits") + svc.upload_file(str(_write(tmp_path, "b")), "raw/.inflight.tmp") + + assert svc.list_prefix("raw") == ["raw/good.fits"] + + +# --- Staging → canonical sync (mc mirror) --- + +_LOCAL_SYNC_CONFIG = SimpleNamespace( + storage_backend="local", local_storage_root="/data" +) +_S3_SYNC_CONFIG = SimpleNamespace( + storage_backend="s3", + s3_endpoint_url="http://minio:9000", + aws_access_key_id="key", + aws_secret_access_key="secret", + s3_bucket_name="bucket", +) + + +def _fake_mc_process(lines, return_code=0): + proc = MagicMock() + proc.stdout = iter(lines) + proc.wait.return_value = return_code + return proc + + +def test_resolve_mc_canonical_target_local_is_a_filesystem_path(): + from src.services.storage_service import _resolve_mc_canonical_target + + assert _resolve_mc_canonical_target(_LOCAL_SYNC_CONFIG, "raw") == "/data/raw" + + +def test_resolve_mc_canonical_target_s3_registers_alias_and_qualifies_bucket(mocker): + run = mocker.patch("src.services.storage_service.subprocess.run") + from src.services.storage_service import _resolve_mc_canonical_target + + target = _resolve_mc_canonical_target(_S3_SYNC_CONFIG, "raw") + + assert target == "diffpype-canonical/bucket/raw" + assert run.call_args[0][0][:3] == ["mc", "alias", "set"] + + +def test_resolve_mc_canonical_target_s3_empty_prefix_is_bucket_root(mocker): + mocker.patch("src.services.storage_service.subprocess.run") + from src.services.storage_service import _resolve_mc_canonical_target + + assert ( + _resolve_mc_canonical_target(_S3_SYNC_CONFIG, "") == "diffpype-canonical/bucket" + ) + + +# Real event shapes, empirically confirmed by running `mc --json mirror` against +# a live MinIO instance with real FITS files (not assumed/fabricated) — see doc +# 30's Logs. Per-file success events carry cumulative totalCount/totalSize; the +# trailing summary event has no target/source at all; errors carry status=="error". +_REAL_FILE_EVENT_A = ( + '{"status":"success","source":"/staging/a.fits","target":"canonical/a.fits",' + '"size":100,"totalCount":1,"totalSize":100,"eventTime":"","eventType":""}' +) +_REAL_FILE_EVENT_B = ( + '{"status":"success","source":"/staging/b.fits","target":"canonical/b.fits",' + '"size":50,"totalCount":2,"totalSize":150,"eventTime":"","eventType":""}' +) +_REAL_SUMMARY_EVENT = ( + '{"status":"success","total":150,"transferred":150,' + '"duration":1971842875,"speed":178878025.46}' +) +_REAL_ERROR_EVENT = ( + '{"status":"error","error":{"message":"Failed to perform mirroring",' + '"cause":{"message":"The specified bucket does not exist"},"type":"error"}}' +) + + +def test_sync_staging_to_canonical_streams_mc_json_events(mocker): + """The sync invokes `mc --json mirror` and consumes its per-file event stream.""" + popen = mocker.patch( + "src.services.storage_service.subprocess.Popen", + return_value=_fake_mc_process([_REAL_FILE_EVENT_A, "", _REAL_FILE_EVENT_B]), + ) + from src.services.storage_service import sync_staging_to_canonical + + sync_staging_to_canonical("/staging", "canonical", config=_LOCAL_SYNC_CONFIG) + + cmd = popen.call_args[0][0] + assert cmd[:3] == ["mc", "--json", "mirror"] + assert cmd[3] == "/staging" + assert cmd[4] == "/data/canonical" + + +def test_sync_staging_to_canonical_does_not_miscount_the_trailing_summary_event(mocker): + """Regression: the trailing job-summary line has no target/source and must + not be counted as a copied file (it previously was, over-counting by one).""" + mocker.patch( + "src.services.storage_service.subprocess.Popen", + return_value=_fake_mc_process( + [_REAL_FILE_EVENT_A, _REAL_FILE_EVENT_B, _REAL_SUMMARY_EVENT] + ), + ) + mock_log = MagicMock() + mocker.patch("src.services.storage_service.get_logger", return_value=mock_log) + from src.services.storage_service import sync_staging_to_canonical + + sync_staging_to_canonical("/staging", "canonical", config=_LOCAL_SYNC_CONFIG) + + copied_calls = [ + c + for c in mock_log.info.call_args_list + if c.args[0] == "staging_sync_file_copied" + ] + summary_calls = [ + c + for c in mock_log.info.call_args_list + if c.args[0] == "staging_sync_job_summary" + ] + assert len(copied_calls) == 2 # exactly the two real files, not three + assert copied_calls[-1].kwargs["index"] == 2 + assert len(summary_calls) == 1 + assert summary_calls[0].kwargs["transferred_bytes"] == 150 + + +def test_sync_staging_to_canonical_logs_error_events_without_counting_them(mocker): + mocker.patch( + "src.services.storage_service.subprocess.Popen", + return_value=_fake_mc_process( + [_REAL_ERROR_EVENT, _REAL_SUMMARY_EVENT], return_code=1 + ), + ) + mock_log = MagicMock() + mocker.patch("src.services.storage_service.get_logger", return_value=mock_log) + from src.services.storage_service import sync_staging_to_canonical + + with pytest.raises(RuntimeError, match="mc mirror exited with code 1"): + sync_staging_to_canonical("/staging", "canonical", config=_LOCAL_SYNC_CONFIG) + + error_calls = [ + c + for c in mock_log.warning.call_args_list + if c.args[0] == "staging_sync_file_error" + ] + assert len(error_calls) == 1 + assert "does not exist" in error_calls[0].kwargs["error"]["cause"]["message"] + + +def test_sync_staging_to_canonical_raises_on_nonzero_exit(mocker): + mocker.patch( + "src.services.storage_service.subprocess.Popen", + return_value=_fake_mc_process([_REAL_FILE_EVENT_A], return_code=1), + ) + from src.services.storage_service import sync_staging_to_canonical + + with pytest.raises(RuntimeError, match="mc mirror exited with code 1"): + sync_staging_to_canonical("/staging", "canonical", config=_LOCAL_SYNC_CONFIG) + + +def test_sync_staging_to_canonical_kills_subprocess_on_soft_time_limit(mocker): + """A hang (SoftTimeLimitExceeded, raised by Celery inside the loop) must kill + the mc subprocess rather than leave it orphaned, then re-raise.""" + from celery.exceptions import SoftTimeLimitExceeded + + def _hanging_stdout(): + yield _REAL_FILE_EVENT_A + raise SoftTimeLimitExceeded() + + proc = MagicMock() + proc.stdout = _hanging_stdout() + mocker.patch("src.services.storage_service.subprocess.Popen", return_value=proc) + from src.services.storage_service import sync_staging_to_canonical + + with pytest.raises(SoftTimeLimitExceeded): + sync_staging_to_canonical("/staging", "canonical", config=_LOCAL_SYNC_CONFIG) + + proc.kill.assert_called_once() + proc.wait.assert_called_once() + + +def test_sync_staging_to_canonical_skips_unparseable_lines(mocker): + """A non-JSON progress line is logged and skipped, never fatal.""" + mocker.patch( + "src.services.storage_service.subprocess.Popen", + return_value=_fake_mc_process(["not-json-noise", _REAL_FILE_EVENT_A]), + ) + from src.services.storage_service import sync_staging_to_canonical + + # Must not raise despite the malformed first line. + sync_staging_to_canonical("/staging", "", config=_LOCAL_SYNC_CONFIG) + + +def test_dispatch_staging_sync_delays_task_and_returns_job_id(mocker): + fake_result = MagicMock(id="sync-job-1") + delay = mocker.patch( + "src.worker.tasks.run_staging_sync.delay", return_value=fake_result + ) + from src.services.storage_service import dispatch_staging_sync + + assert dispatch_staging_sync("/staging", "raw") == "sync-job-1" + delay.assert_called_once_with("/staging", "raw") + + def _write(tmp_path, name: str): path = tmp_path / name path.write_text("x") diff --git a/src/services/tests/test_tile_service.py b/src/services/tests/test_tile_service.py index 59eeaab..1c18c47 100644 --- a/src/services/tests/test_tile_service.py +++ b/src/services/tests/test_tile_service.py @@ -1,10 +1,14 @@ from unittest.mock import MagicMock import astropy.units as u +import pytest from mocpy import MOC +from src.db.enums import RegionSource from src.services.tile_service import ( + _resolve_region_moc, create_tiles, + generate_tessellation_for_region, generate_tile_tessellation, tile_with_most_calibrations, ) @@ -90,3 +94,94 @@ def test_tile_with_most_calibrations_returns_none_when_no_tile_has_any(): result = tile_with_most_calibrations(mock_db, [28, 29, 30]) assert result is None + + +# --- Region resolution (region_source dispatch) --- + + +def test_resolve_region_moc_cone_builds_a_cone(): + moc = _resolve_region_moc( + MagicMock(), RegionSource.CONE, ra=180.0, decl=0.0, radius_deg=0.1 + ) + assert isinstance(moc, MOC) + assert moc.sky_fraction > 0 + + +def test_resolve_region_moc_bounding_box_builds_a_polygon(): + moc = _resolve_region_moc( + MagicMock(), + RegionSource.BOUNDING_BOX, + min_ra=10.0, + max_ra=11.0, + min_decl=20.0, + max_decl=21.0, + ) + assert isinstance(moc, MOC) + assert moc.sky_fraction > 0 + + +def test_resolve_region_moc_project_footprint_unions_calibration_footprints(mocker): + footprint = MOC.from_cone( + lon=45 * u.deg, lat=10 * u.deg, radius=0.1 * u.deg, max_depth=10 + ) + mock_db = MagicMock() + mock_db.execute.return_value.all.return_value = [(footprint,)] + + moc = _resolve_region_moc(mock_db, RegionSource.PROJECT_FOOTPRINT, project_id=3) + + assert isinstance(moc, MOC) + assert moc.sky_fraction > 0 + + +def test_resolve_region_moc_project_footprint_raises_when_no_footprints(): + mock_db = MagicMock() + mock_db.execute.return_value.all.return_value = [] + + with pytest.raises(ValueError, match="at least one calibration"): + _resolve_region_moc(mock_db, RegionSource.PROJECT_FOOTPRINT, project_id=3) + + +def test_resolve_region_moc_cone_missing_fields_raises(): + with pytest.raises(ValueError, match="region_source=cone requires"): + _resolve_region_moc(MagicMock(), RegionSource.CONE, ra=180.0) + + +def test_resolve_region_moc_bounding_box_missing_fields_raises(): + with pytest.raises(ValueError, match="region_source=bounding_box requires"): + _resolve_region_moc( + MagicMock(), RegionSource.BOUNDING_BOX, min_ra=1.0, max_ra=2.0 + ) + + +def test_generate_tessellation_for_region_resolves_then_tessellates(mocker): + mock_resolve = mocker.patch( + "src.services.tile_service._resolve_region_moc", + return_value=MOC.from_cone( + lon=180 * u.deg, lat=0 * u.deg, radius=0.1 * u.deg, max_depth=10 + ), + ) + + tiles = generate_tessellation_for_region( + MagicMock(), + RegionSource.CONE, + tile_side_length_arc_min=6.0, + ra=180.0, + decl=0.0, + radius_deg=0.1, + ) + + assert len(tiles) >= 1 + mock_resolve.assert_called_once() + + +def test_generate_tile_tessellation_overlap_only_false_keeps_more_tiles(): + """overlap_only=False materializes the full grid; True trims to intersecting tiles.""" + moc_to_tile = MOC.from_cone( + lon=180 * u.deg, lat=0 * u.deg, radius=0.05 * u.deg, max_depth=10 + ) + + trimmed = generate_tile_tessellation(6.0, moc_to_tile, 0.0, overlap_only=True) + full = generate_tile_tessellation(6.0, moc_to_tile, 0.0, overlap_only=False) + + assert len(full) >= len(trimmed) + assert len(full) > 0 diff --git a/src/services/tile_service.py b/src/services/tile_service.py index 1d5ce7a..0019901 100644 --- a/src/services/tile_service.py +++ b/src/services/tile_service.py @@ -18,19 +18,25 @@ from sqlalchemy.dialects.postgresql import insert as pg_insert from sqlalchemy.orm import Session +from src.db.enums import RegionSource from src.db.models import Level2Calibration, Tile, tile_level2_calibration_association -from src.db.spatial_types import MOCType +from src.db.spatial_types import MOCType, union_mocs # HEALPix order for generated tile footprints, matching the prototype's ported # convention for tile-scale (not detector-scale) precision. MOCType normalizes # any input depth to depth 29 on persistence, so this only affects computation. _TILE_FOOTPRINT_MAX_DEPTH = 21 +# HEALPix order for the resolved region MOC (cone/bounding-box/project-footprint), +# matching the CLI's existing cone precision (`_cone_moc` uses max_depth=10). +_REGION_MAX_DEPTH = 10 + def generate_tile_tessellation( tile_side_length_arc_min: float, moc_to_tile: MOC, overlap_in_arc_min: float = 0.0, + overlap_only: bool = True, ) -> list[dict]: """Generate a regular tile grid covering ``moc_to_tile``. Pure computation, no DB access. @@ -41,6 +47,11 @@ def generate_tile_tessellation( (matching the CLI/API parameter names); ``delta_ra``/``delta_decl`` in the returned dicts (and the persisted ``Tile`` columns) are degrees, not arcmin — converted once here and never converted back. + + ``overlap_only`` (default ``True``) keeps only tiles that actually intersect + ``moc_to_tile``; ``False`` keeps the full rectangular grid regardless of + intersection, to pre-provision tiles over a region ahead of incoming survey + data. """ orig_deg_height = tile_side_length_arc_min / 60.0 orig_deg_width = tile_side_length_arc_min / 60.0 @@ -132,7 +143,7 @@ def generate_tile_tessellation( corner_coords, complement=False, max_depth=_TILE_FOOTPRINT_MAX_DEPTH ) - if moc_to_tile.intersection(tile_moc).sky_fraction > 0: + if not overlap_only or moc_to_tile.intersection(tile_moc).sky_fraction > 0: tiles.append( { "name": f"Tile_{tile_num}", @@ -148,6 +159,113 @@ def generate_tile_tessellation( return tiles +def _resolve_region_moc( + db: Session, + region_source: RegionSource, + *, + ra: float | None = None, + decl: float | None = None, + radius_deg: float | None = None, + project_id: int | None = None, + min_ra: float | None = None, + max_ra: float | None = None, + min_decl: float | None = None, + max_decl: float | None = None, +) -> MOC: + """Convert a ``region_source`` and its mode-specific parameters into a single MOC. + + ``cone`` and ``bounding_box`` are pure geometry; ``project_footprint`` queries + every ``Level2Calibration`` footprint under ``project_id`` and unions them via + ``union_mocs``. Raises ``ValueError`` if the fields required by ``region_source`` + are missing, or if a ``project_footprint`` request names a project with no + footprints to derive a region from. (The API boundary validates required fields + up front via Pydantic; this guard gives the CLI boundary the same protection.) + """ + + def _require(**fields: object) -> None: + missing = [name for name, value in fields.items() if value is None] + if missing: + raise ValueError( + f"region_source={region_source.value} requires: {', '.join(missing)}" + ) + + if region_source == RegionSource.CONE: + _require(ra=ra, decl=decl, radius_deg=radius_deg) + return MOC.from_cone( + lon=ra * u.deg, + lat=decl * u.deg, + radius=radius_deg * u.deg, + max_depth=_REGION_MAX_DEPTH, + ) + if region_source == RegionSource.BOUNDING_BOX: + _require(min_ra=min_ra, max_ra=max_ra, min_decl=min_decl, max_decl=max_decl) + corners = SkyCoord( + [min_ra, max_ra, max_ra, min_ra], + [min_decl, min_decl, max_decl, max_decl], + unit="deg", + frame="icrs", + ) + return MOC.from_polygon_skycoord( + corners, complement=False, max_depth=_REGION_MAX_DEPTH + ) + + _require(project_id=project_id) + footprints = [ + row[0] + for row in db.execute( + sa.select(Level2Calibration.footprint).where( + Level2Calibration.project_id == project_id, + Level2Calibration.footprint.isnot(None), + ) + ).all() + ] + if not footprints: + raise ValueError( + "project_footprint tessellation requires at least one calibration with " + f"a footprint under project_id={project_id}" + ) + return union_mocs(footprints) + + +def generate_tessellation_for_region( + db: Session, + region_source: RegionSource, + tile_side_length_arc_min: float, + overlap_in_arc_min: float = 0.0, + overlap_only: bool = True, + *, + ra: float | None = None, + decl: float | None = None, + radius_deg: float | None = None, + project_id: int | None = None, + min_ra: float | None = None, + max_ra: float | None = None, + min_decl: float | None = None, + max_decl: float | None = None, +) -> list[dict]: + """Resolve a ``region_source`` into a MOC, then generate its tile tessellation. + + The single service entry point both the API route and CLI delegate to, keeping + the pure ``generate_tile_tessellation`` free of DB access while supporting all + three region modes. + """ + moc_to_tile = _resolve_region_moc( + db, + region_source, + ra=ra, + decl=decl, + radius_deg=radius_deg, + project_id=project_id, + min_ra=min_ra, + max_ra=max_ra, + min_decl=min_decl, + max_decl=max_decl, + ) + return generate_tile_tessellation( + tile_side_length_arc_min, moc_to_tile, overlap_in_arc_min, overlap_only + ) + + def create_tiles(db: Session, project_id: int, tiles: list[dict]) -> list[Tile]: """Bulk-insert Tile rows and associate each with its overlapping Level2Calibrations. diff --git a/src/worker/base_task.py b/src/worker/base_task.py index 87a3b7d..c07a47b 100644 --- a/src/worker/base_task.py +++ b/src/worker/base_task.py @@ -4,21 +4,96 @@ out of a task body. It is deliberately entity-agnostic: it logs the crash with structlog and dispatches the failed payload to the dead-letter queue, but does not write any domain entity's status. Tasks that track a domain row (e.g. -``run_ingest_batch``, ``run_mosaic_drizzle``) own their own crash-safe FAILED -transition inside the task body; entities orphaned in ``IN_PROCESS`` by an -uncatchable crash (SIGKILL/OOM, where ``on_failure`` never runs at all) are -reconciled by the stuck-job watchdog (doc 30). +``run_ingest_batch``, ``run_mosaic_drizzle``) use ``begin_tracked_job`` for their +crash-safe ``IN_PROCESS``/stale-redelivery handling; entities orphaned in +``IN_PROCESS`` by an uncatchable crash (SIGKILL/OOM, where ``on_failure`` never +runs at all) are reconciled by the stuck-job watchdog (doc 30). + +Every concrete task in this app must declare an explicit time ceiling and +(if applicable) which watchdog-tracked entity it owns — see ``TimeLimitedTask`` +and ``DiffpypeTask`` below. Both contracts are enforced via ``__init_subclass__`` +checking ``cls.__dict__`` (never ``getattr``), so a task that accidentally +subclasses another concrete task instead of the real base — silently inheriting +its ceiling/tracked entity — is rejected at class-definition time rather than +quietly doing the wrong thing. (Verified empirically during design: a plain +``abc.ABC``/``abstractmethod`` does *not* catch this specific mistake, since ABC +only requires a concrete value to exist somewhere in the MRO, not that the +subclass itself declared it.) """ import celery from sqlalchemy.exc import OperationalError as SAOperationalError +from sqlalchemy.orm import Session from src.core.config import settings from src.core.logger import get_logger +from src.db.enums import JobStatus +from src.db.models import Base + + +class NotTracked: + """Sentinel: this task deliberately does not own a watchdog-tracked entity.""" + + def __repr__(self) -> str: + return "NOT_TRACKED" + + +NOT_TRACKED = NotTracked() +"""Explicit ``tracked_entity_model`` value for tasks with no tracked entity — a +real, distinct value (never bare ``None``) so "deliberately untracked" can't be +confused with "the developer forgot to declare this".""" +HARD_LIMIT_BUFFER_SECONDS = 30 +"""Grace window between a task's soft and hard time limit, giving +``SoftTimeLimitExceeded`` cleanup code (e.g. killing a subprocess) a chance to +run before Celery forcibly kills the task outright. Reused by ``celery_app.py`` +to compute ``broker_transport_options["visibility_timeout"]`` from the same +real, enforced ceilings, rather than duplicating the number.""" -class DiffpypeTask(celery.Task): - """Base task that guarantees failure logging and dead-letter dispatch.""" + +class TimeLimitedTask(celery.Task): + """Requires every concrete subclass to declare its own ``soft_time_limit_seconds``. + + ``abstract = True`` follows Celery's own convention for a non-registered + intermediate base, exempting it (and any other abstract base) from its own + contract. A missing declaration on a concrete subclass raises ``TypeError`` + at class-definition time (Celery's ``@app.task(base=...)`` decorator + dynamically creates a real subclass per task, so this fires no later than + worker/beat startup). + """ + + abstract = True + + def __init_subclass__(cls, **kwargs) -> None: + super().__init_subclass__(**kwargs) + if cls.__dict__.get("abstract", False) or cls.__dict__.get("_decorated", False): + # `_decorated` is set by Celery's own `_task_from_fun` on the dynamic + # wrapper subclass it creates for every `@app.task(...)`-decorated + # function (`type(fun.__name__, (base,), {"_decorated": True, ...})`) + # — this is Celery's own internal wrapping, not a human-authored + # subclass, so it's legitimately exempt: it inherits from whatever + # `base=` was passed, which has already satisfied this contract. + return + if "soft_time_limit_seconds" not in cls.__dict__: + raise TypeError( + f"{cls.__name__} must declare its own soft_time_limit_seconds " + "(never inherited from a parent class)." + ) + + @property + def soft_time_limit(self) -> int: + return self.soft_time_limit_seconds + + @property + def time_limit(self) -> int: + return self.soft_time_limit_seconds + HARD_LIMIT_BUFFER_SECONDS + + +class DiffpypeTask(TimeLimitedTask): + """Base task guaranteeing failure logging, dead-letter dispatch, and (for + tracked tasks) crash-safe, stale-redelivery-proof entity transitions.""" + + abstract = True # Include SAOperationalError so transient DB connection drops are retried, # not just raw socket-level IOError/ConnectionError. @@ -32,6 +107,16 @@ class DiffpypeTask(celery.Task): max_retries = settings.celery_task_max_retries default_retry_delay = settings.celery_task_retry_delay + def __init_subclass__(cls, **kwargs) -> None: + super().__init_subclass__(**kwargs) + if cls.__dict__.get("abstract", False) or cls.__dict__.get("_decorated", False): + return # see TimeLimitedTask.__init_subclass__ for why `_decorated` is exempt + if "tracked_entity_model" not in cls.__dict__: + raise TypeError( + f"{cls.__name__} must declare tracked_entity_model explicitly " + "(a real model, or base_task.NOT_TRACKED) — never inherited." + ) + def on_failure(self, exc, task_id, args, kwargs, einfo) -> None: # The active OTel task span supplies the correlation_id to every log line # via the structlog processor, so no manual context binding is needed here. @@ -62,3 +147,32 @@ def on_failure(self, exc, task_id, args, kwargs, einfo) -> None: ) except Exception: log.error("on_failure_dlq_dispatch_failed", task_id=task_id, exc_info=True) + + def begin_tracked_job(self, db: Session, entity_id: int) -> Base | None: + """Fetch this task's declared tracked entity and transition it to + IN_PROCESS — unless it's already resolved (a stale redelivery arriving + after the stuck-job watchdog already gave up on it), in which case log + the skip and return None. The caller must bail without doing any work + when this returns None; this is what prevents a late, uncoordinated + Celery/Redis redelivery from silently reviving or overwriting a job the + watchdog already marked FAILED. + """ + if self.tracked_entity_model is NOT_TRACKED: + raise TypeError( + f"{type(self).__name__} declared NOT_TRACKED; cannot call begin_tracked_job" + ) + entity = db.get(self.tracked_entity_model, entity_id) + assert ( + entity is not None + ), f"{self.tracked_entity_model.__name__} {entity_id} not found" + if entity.status in (JobStatus.COMPLETE, JobStatus.FAILED): + get_logger().warning( + "tracked_job_stale_redelivery_skipped", + entity=self.tracked_entity_model.__name__, + id=entity_id, + status=entity.status.value, + ) + return None + entity.status = JobStatus.IN_PROCESS + db.commit() + return entity diff --git a/src/worker/celery_app.py b/src/worker/celery_app.py index 0e4e7f1..42fa620 100644 --- a/src/worker/celery_app.py +++ b/src/worker/celery_app.py @@ -6,6 +6,28 @@ from src.core.tracing import setup_tracing from src.db.enums import CeleryQueue from src.db.session import engine +from src.worker.base_task import HARD_LIMIT_BUFFER_SECONDS + +# Every task's enforced hard time_limit (soft + HARD_LIMIT_BUFFER_SECONDS) that +# broker-level redelivery must never precede — otherwise a still-healthy, +# legitimately-running task could be falsely treated as abandoned and handed to +# a second worker concurrently. 60s extra margin for SIGTERM/SIGKILL cleanup and +# broker round-trip time on top of the largest real ceiling (currently ingest). +_ALL_TASK_SOFT_TIME_LIMITS = ( + settings.staging_sync_soft_time_limit_seconds, + settings.ingest_batch_soft_time_limit_seconds, + settings.mosaic_drizzle_soft_time_limit_seconds, + settings.cli_tool_soft_time_limit_seconds, + settings.db_backup_soft_time_limit_seconds, + settings.dlq_dump_soft_time_limit_seconds, + settings.reconcile_stuck_jobs_soft_time_limit_seconds, +) +_VISIBILITY_TIMEOUT_SAFETY_MARGIN_SECONDS = 60 +VISIBILITY_TIMEOUT_SECONDS = ( + max(_ALL_TASK_SOFT_TIME_LIMITS) + + HARD_LIMIT_BUFFER_SECONDS + + _VISIBILITY_TIMEOUT_SAFETY_MARGIN_SECONDS +) # Configure JSON logging for the worker process (and any CLI path that imports # the service/task layer) so all components stream structured logs to stdout. @@ -25,10 +47,24 @@ celery_app.conf.update( task_acks_late=True, task_reject_on_worker_lost=True, + # How long an unacked message stays invisible before Redis considers the + # worker holding it dead and redelivers it. Tied to the largest real, + # enforced task ceiling (see VISIBILITY_TIMEOUT_SECONDS above) rather than + # left at kombu's 3600s default, which had no relationship to any task's + # actual runtime before every task declared an explicit time limit (doc 30). + broker_transport_options={"visibility_timeout": VISIBILITY_TIMEOUT_SECONDS}, + # sqlalchemy-celery-beat's DatabaseScheduler reads/writes schedules here; it + # defaults to the `celery_schema` Postgres schema (created by migration 0014). + beat_dburi=settings.database_url, task_routes={ "src.worker.tasks.dlq_dump": {"queue": "dead_letter"}, "src.worker.tasks.run_ingest_batch": {"queue": CeleryQueue.HEAVY_MEMORY}, "src.worker.tasks.run_mosaic_drizzle": {"queue": CeleryQueue.HEAVY_MEMORY}, + # Beat-triggered and dispatched I/O tasks land on the light queue; without + # an explicit route they would go to the default queue no worker consumes. + "src.worker.tasks.run_staging_sync": {"queue": CeleryQueue.LIGHT}, + "src.worker.tasks.sync_staging_cron": {"queue": CeleryQueue.LIGHT}, + "src.worker.tasks.reconcile_stuck_jobs_cron": {"queue": CeleryQueue.LIGHT}, }, ) diff --git a/src/worker/tasks.py b/src/worker/tasks.py index 9c79d50..7c02a72 100644 --- a/src/worker/tasks.py +++ b/src/worker/tasks.py @@ -5,35 +5,42 @@ import pandas as pd +from src.core.config import settings from src.core.logger import get_logger from src.db.enums import JobStatus from src.db.models import IngestBatch, JobConfiguration, Level3Mosaic from src.db.session import SessionLocal -from src.services import ingest_service +from src.services import ingest_service, job_service, storage_service from src.services.storage_service import get_storage_service -from src.worker.base_task import DiffpypeTask +from src.worker.base_task import NOT_TRACKED, DiffpypeTask, TimeLimitedTask from src.worker.celery_app import celery_app from src.worker.utils import build_cli_command -@celery_app.task(name="src.worker.tasks.run_ingest_batch") -def run_ingest_batch(batch_id: int) -> None: +class _IngestBatchTask(DiffpypeTask): + tracked_entity_model = IngestBatch + soft_time_limit_seconds = settings.ingest_batch_soft_time_limit_seconds + + +@celery_app.task( + base=_IngestBatchTask, bind=True, name="src.worker.tasks.run_ingest_batch" +) +def run_ingest_batch(self, batch_id: int) -> None: """Scan an IngestBatch's storage prefix and bulk-upsert Level2Image/Level2Calibration rows. - Not built on DiffpypeTask: that base's on_failure is entity-agnostic - (logging + DLQ dispatch only), so it can't transition this batch's row out - of IN_PROCESS on crash. This task handles its own crash-safety instead, - guaranteeing no orphaned IN_PROCESS row for the entity it actually tracks. + Uses ``begin_tracked_job`` (DiffpypeTask) for its IN_PROCESS transition: + guarantees no orphaned IN_PROCESS row for the entity it actually tracks, + and refuses to resume/overwrite a row the stuck-job watchdog already + resolved — a stale redelivery arriving after the watchdog gave up (doc 30). """ log = get_logger() log.info("ingest_batch_started", batch_id=batch_id) db = SessionLocal() try: - batch = db.get(IngestBatch, batch_id) - assert batch is not None, f"IngestBatch {batch_id} not found" - batch.status = JobStatus.IN_PROCESS - db.commit() + batch = self.begin_tracked_job(db, batch_id) + if batch is None: + return storage = get_storage_service() keys = storage.list_prefix(batch.s3_prefix) @@ -94,26 +101,31 @@ def run_ingest_batch(batch_id: int) -> None: db.close() -@celery_app.task(name="src.worker.tasks.run_mosaic_drizzle") -def run_mosaic_drizzle(mosaic_id: int) -> None: +class _MosaicDrizzleTask(DiffpypeTask): + tracked_entity_model = Level3Mosaic + soft_time_limit_seconds = settings.mosaic_drizzle_soft_time_limit_seconds + + +@celery_app.task( + base=_MosaicDrizzleTask, bind=True, name="src.worker.tasks.run_mosaic_drizzle" +) +def run_mosaic_drizzle(self, mosaic_id: int) -> None: """Placeholder drizzle execution: sleeps and transitions to COMPLETE. The Level3Mosaic row itself (footprint, barycenter, and its M2M-derived constituent calibrations) is already the complete job specification — only the actual pixel-level drizzle is deferred, to a future - JWST-pipeline-specific doc. Not built on DiffpypeTask for the same reason - as run_ingest_batch: that base's on_failure is entity-agnostic and cannot - transition this mosaic's row out of IN_PROCESS on crash. + JWST-pipeline-specific doc. Uses ``begin_tracked_job`` for the same + stale-redelivery protection as ``run_ingest_batch``. """ log = get_logger() log.info("mosaic_drizzle_started", mosaic_id=mosaic_id) db = SessionLocal() try: - mosaic = db.get(Level3Mosaic, mosaic_id) - assert mosaic is not None, f"Level3Mosaic {mosaic_id} not found" - mosaic.status = JobStatus.IN_PROCESS - db.commit() + mosaic = self.begin_tracked_job(db, mosaic_id) + if mosaic is None: + return time.sleep(5) @@ -134,7 +146,69 @@ def run_mosaic_drizzle(mosaic_id: int) -> None: db.close() -@celery_app.task(name="src.worker.tasks.dlq_dump") +class _StagingSyncTask(DiffpypeTask): + tracked_entity_model = NOT_TRACKED + soft_time_limit_seconds = settings.staging_sync_soft_time_limit_seconds + + +@celery_app.task(base=_StagingSyncTask, name="src.worker.tasks.run_staging_sync") +def run_staging_sync(staging_location: str, canonical_prefix: str) -> None: + """Mirror a staging location into the canonical bucket (streamed ``mc mirror``). + + Built on DiffpypeTask (unlike ``run_ingest_batch``): it owns no DB status + entity (``tracked_entity_model = NOT_TRACKED``), so the base's + entity-agnostic retry/DLQ behavior is exactly right — a transient failure + retries, a permanent one dead-letters, and the underlying ``mc mirror`` is + idempotent so a redelivered run is always safe. Bounds a hung ``mc`` + subprocess (network partition mid-transfer, not a clean crash) via + ``_StagingSyncTask.soft_time_limit_seconds`` — ``sync_staging_to_canonical`` + catches ``SoftTimeLimitExceeded`` to kill the subprocess before re-raising. + Not auto-retried on a timeout (``SoftTimeLimitExceeded`` isn't in + ``DiffpypeTask.autoretry_for``): a hang that already exceeded a generous + soft limit likely hangs again immediately, so it dead-letters for operator + attention instead of retrying blindly. + """ + storage_service.sync_staging_to_canonical(staging_location, canonical_prefix) + + +class _SyncStagingCronTask(DiffpypeTask): + tracked_entity_model = NOT_TRACKED + soft_time_limit_seconds = settings.staging_sync_soft_time_limit_seconds + + +@celery_app.task(base=_SyncStagingCronTask, name="src.worker.tasks.sync_staging_cron") +def sync_staging_cron() -> None: + """Celery Beat entry point: mirror the configured staging location into the canonical bucket root.""" + storage_service.sync_staging_to_canonical(settings.staging_location, "") + + +class _ReconcileStuckJobsCronTask(DiffpypeTask): + tracked_entity_model = NOT_TRACKED + soft_time_limit_seconds = settings.reconcile_stuck_jobs_soft_time_limit_seconds + + +@celery_app.task( + base=_ReconcileStuckJobsCronTask, name="src.worker.tasks.reconcile_stuck_jobs_cron" +) +def reconcile_stuck_jobs_cron() -> None: + """Celery Beat entry point: fail any job left stuck in IN_PROCESS past the staleness threshold.""" + db = SessionLocal() + try: + reconciled = job_service.reconcile_stuck_jobs(db) + get_logger().info("reconcile_stuck_jobs_cron_completed", reconciled=reconciled) + finally: + db.close() + + +class _DlqDumpTask(TimeLimitedTask): + """Deliberately NOT DiffpypeTask: on_failure would try to dispatch another + dlq_dump on failure, risking a self-referential dispatch loop if dlq_dump + itself is ever persistently broken.""" + + soft_time_limit_seconds = settings.dlq_dump_soft_time_limit_seconds + + +@celery_app.task(base=_DlqDumpTask, name="src.worker.tasks.dlq_dump") def dlq_dump(failed_task_name: str, task_kwargs: dict, error_msg: str) -> None: """Log a permanently failed task payload to the dead letter queue.""" get_logger().warning( @@ -145,13 +219,23 @@ def dlq_dump(failed_task_name: str, task_kwargs: dict, error_msg: str) -> None: ) -@celery_app.task(name="src.worker.tasks.db_backup_cron") +class _DbBackupCronTask(DiffpypeTask): + tracked_entity_model = NOT_TRACKED + soft_time_limit_seconds = settings.db_backup_soft_time_limit_seconds + + +@celery_app.task(base=_DbBackupCronTask, name="src.worker.tasks.db_backup_cron") def db_backup_cron() -> None: """Placeholder for nightly database backup, triggered by Celery Beat.""" get_logger().info("db_backup_cron_triggered", detail="Nightly backup triggered") -@celery_app.task(base=DiffpypeTask, name="src.worker.tasks.execute_cli_tool") +class _CliToolTask(DiffpypeTask): + tracked_entity_model = NOT_TRACKED + soft_time_limit_seconds = settings.cli_tool_soft_time_limit_seconds + + +@celery_app.task(base=_CliToolTask, name="src.worker.tasks.execute_cli_tool") def execute_cli_tool(job_config_id: int, executable: str) -> None: """Execute an external CLI tool using the kwargs stored in JobConfiguration.""" log = get_logger() diff --git a/src/worker/tests/test_base_task.py b/src/worker/tests/test_base_task.py index 797064f..76d6d32 100644 --- a/src/worker/tests/test_base_task.py +++ b/src/worker/tests/test_base_task.py @@ -1,9 +1,11 @@ from unittest.mock import MagicMock +import pytest from sqlalchemy.exc import OperationalError as SAOperationalError from src.core.config import settings -from src.worker.base_task import DiffpypeTask +from src.db.enums import JobStatus +from src.worker.base_task import NOT_TRACKED, DiffpypeTask, TimeLimitedTask def test_retryable_exceptions_are_configured(): @@ -104,3 +106,133 @@ def test_on_failure_handles_empty_args(mocker): task.on_failure(ValueError("x"), "task-0", (), {}, None) mock_dlq.apply_async.assert_called_once() + + +# --------------------------------------------------------------------------- +# TimeLimitedTask / DiffpypeTask contract enforcement +# --------------------------------------------------------------------------- + + +def test_time_limited_task_requires_soft_time_limit_seconds(): + with pytest.raises(TypeError, match="must declare its own soft_time_limit_seconds"): + + class _Missing(TimeLimitedTask): + pass + + +def test_time_limited_task_abstract_base_is_exempt(): + """An abstract=True intermediate base doesn't need to satisfy its own contract.""" + + class _AbstractBase(TimeLimitedTask): + abstract = True + + assert _AbstractBase.abstract is True + + +def test_diffpype_task_requires_tracked_entity_model(): + with pytest.raises(TypeError, match="must declare tracked_entity_model"): + + class _Missing(DiffpypeTask): + soft_time_limit_seconds = 60 + + +def test_diffpype_task_rejects_wrong_sibling_inheritance(): + """Regression: a task accidentally subclassing another concrete task instead + of DiffpypeTask directly must not silently inherit its declared values.""" + + class _Real(DiffpypeTask): + tracked_entity_model = NOT_TRACKED + soft_time_limit_seconds = 100 + + with pytest.raises(TypeError, match="must declare its own soft_time_limit_seconds"): + + class _Accidental(_Real): + pass + + +def test_soft_time_limit_and_time_limit_properties_bridge_correctly(): + class _Real(DiffpypeTask): + tracked_entity_model = NOT_TRACKED + soft_time_limit_seconds = 100 + + task = _Real() + assert task.soft_time_limit == 100 + assert task.time_limit == 130 # soft + HARD_LIMIT_BUFFER_SECONDS (30) + + +# --------------------------------------------------------------------------- +# begin_tracked_job +# --------------------------------------------------------------------------- + + +class _FakeTrackedModel: + __name__ = "_FakeTrackedModel" + + +class _TrackedTask(DiffpypeTask): + tracked_entity_model = _FakeTrackedModel + soft_time_limit_seconds = 100 + + +class _UntrackedTask(DiffpypeTask): + tracked_entity_model = NOT_TRACKED + soft_time_limit_seconds = 100 + + +def test_begin_tracked_job_transitions_pending_entity_to_in_process(mocker): + entity = MagicMock(status=JobStatus.PENDING) + db = MagicMock() + db.get.return_value = entity + + result = _TrackedTask().begin_tracked_job(db, 1) + + assert result is entity + assert entity.status == JobStatus.IN_PROCESS + db.commit.assert_called_once() + + +def test_begin_tracked_job_skips_already_complete_entity(mocker): + entity = MagicMock(status=JobStatus.COMPLETE) + db = MagicMock() + db.get.return_value = entity + mock_log = MagicMock() + mocker.patch("src.worker.base_task.get_logger", return_value=mock_log) + + result = _TrackedTask().begin_tracked_job(db, 1) + + assert result is None + db.commit.assert_not_called() + mock_log.warning.assert_called_once_with( + "tracked_job_stale_redelivery_skipped", + entity="_FakeTrackedModel", + id=1, + status="complete", + ) + + +def test_begin_tracked_job_skips_already_failed_entity(): + entity = MagicMock(status=JobStatus.FAILED) + db = MagicMock() + db.get.return_value = entity + + result = _TrackedTask().begin_tracked_job(db, 1) + + assert result is None + db.commit.assert_not_called() + + +def test_begin_tracked_job_raises_for_not_tracked_task(): + with pytest.raises(TypeError, match="declared NOT_TRACKED"): + _UntrackedTask().begin_tracked_job(MagicMock(), 1) + + +def test_begin_tracked_job_raises_assertion_for_missing_entity(): + db = MagicMock() + db.get.return_value = None + + with pytest.raises(AssertionError, match="_FakeTrackedModel 99 not found"): + _TrackedTask().begin_tracked_job(db, 99) + + +def test_not_tracked_repr(): + assert repr(NOT_TRACKED) == "NOT_TRACKED" diff --git a/src/worker/tests/test_celery_app.py b/src/worker/tests/test_celery_app.py index 7747109..e50bdb1 100644 --- a/src/worker/tests/test_celery_app.py +++ b/src/worker/tests/test_celery_app.py @@ -4,7 +4,13 @@ from celery import Celery -from src.worker.celery_app import _configure_beat_schedule +from src.core.config import settings +from src.worker.base_task import HARD_LIMIT_BUFFER_SECONDS +from src.worker.celery_app import ( + VISIBILITY_TIMEOUT_SECONDS, + _configure_beat_schedule, + celery_app, +) def test_beat_schedule_populated_when_cron_enabled(): @@ -25,3 +31,27 @@ def test_beat_schedule_absent_when_cron_disabled(): cfg = MagicMock(enable_db_backup_cron=False) _configure_beat_schedule(app, cfg) assert not app.conf.beat_schedule + + +def test_visibility_timeout_is_at_least_the_largest_task_hard_time_limit(): + """visibility_timeout must never be shorter than any task's own enforced + hard time_limit — otherwise Redis could redeliver a still-healthy, + legitimately-running task to a second worker concurrently (doc 30 §3a).""" + largest_soft_limit = max( + settings.staging_sync_soft_time_limit_seconds, + settings.ingest_batch_soft_time_limit_seconds, + settings.mosaic_drizzle_soft_time_limit_seconds, + settings.cli_tool_soft_time_limit_seconds, + settings.db_backup_soft_time_limit_seconds, + settings.dlq_dump_soft_time_limit_seconds, + settings.reconcile_stuck_jobs_soft_time_limit_seconds, + ) + largest_hard_limit = largest_soft_limit + HARD_LIMIT_BUFFER_SECONDS + assert VISIBILITY_TIMEOUT_SECONDS > largest_hard_limit + + +def test_broker_transport_options_configured_on_the_real_app(): + assert ( + celery_app.conf.broker_transport_options["visibility_timeout"] + == VISIBILITY_TIMEOUT_SECONDS + ) diff --git a/src/worker/tests/test_tasks.py b/src/worker/tests/test_tasks.py index 6a17465..d5a0652 100644 --- a/src/worker/tests/test_tasks.py +++ b/src/worker/tests/test_tasks.py @@ -7,10 +7,14 @@ from src.db.enums import JobStatus from src.db.models import IngestBatch, JobConfiguration, Level3Mosaic from src.worker.tasks import ( + db_backup_cron, dlq_dump, execute_cli_tool, + reconcile_stuck_jobs_cron, run_ingest_batch, run_mosaic_drizzle, + run_staging_sync, + sync_staging_cron, ) @@ -279,6 +283,20 @@ def test_run_ingest_batch_missing_batch_raises_assertion(mocker): mock_session.close.assert_called_once() +def test_run_ingest_batch_skips_already_resolved_stale_redelivery(mocker): + """Regression: a redelivered task must not resume/overwrite a batch the + stuck-job watchdog already marked FAILED (doc 30 §3a).""" + batch = MagicMock(spec=IngestBatch, id=1, status=JobStatus.FAILED) + mock_session, _ = _make_ingest_session(mocker, fake_batch=batch) + mock_storage_factory = mocker.patch("src.worker.tasks.get_storage_service") + + run_ingest_batch(1) + + mock_storage_factory.assert_not_called() + assert batch.status == JobStatus.FAILED # unchanged, never touched + mock_session.close.assert_called_once() + + # --------------------------------------------------------------------------- # run_mosaic_drizzle # --------------------------------------------------------------------------- @@ -327,6 +345,20 @@ def test_run_mosaic_drizzle_missing_mosaic_raises_assertion(mocker): mock_session.close.assert_called_once() +def test_run_mosaic_drizzle_skips_already_resolved_stale_redelivery(mocker): + """Regression: a redelivered task must not resume/overwrite a mosaic the + stuck-job watchdog already marked FAILED (doc 30 §3a).""" + mosaic = MagicMock(spec=Level3Mosaic, id=1, status=JobStatus.FAILED) + mock_session, _ = _make_mosaic_session(mocker, fake_mosaic=mosaic) + mock_sleep = mocker.patch("src.worker.tasks.time.sleep") + + run_mosaic_drizzle(1) + + mock_sleep.assert_not_called() + assert mosaic.status == JobStatus.FAILED + mock_session.close.assert_called_once() + + def test_execute_cli_tool_handles_none_job_kwargs(mocker): mock_session, _ = _make_cli_session(mocker, None) mock_session.get.return_value = MagicMock(spec=JobConfiguration, job_kwargs=None) @@ -339,3 +371,48 @@ def test_execute_cli_tool_handles_none_job_kwargs(mocker): mock_run.assert_called_once_with( ["mytool"], capture_output=True, text=True, check=True ) + + +# --- Operational service tasks (doc 30) --- + + +def test_run_staging_sync_delegates_to_service(mocker): + sync = mocker.patch("src.services.storage_service.sync_staging_to_canonical") + + run_staging_sync("/staging", "raw") + + sync.assert_called_once_with("/staging", "raw") + + +def test_sync_staging_cron_syncs_configured_location_to_bucket_root(mocker): + sync = mocker.patch("src.services.storage_service.sync_staging_to_canonical") + mocker.patch("src.worker.tasks.settings.staging_location", "/data/staging") + + sync_staging_cron() + + sync.assert_called_once_with("/data/staging", "") + + +def test_reconcile_stuck_jobs_cron_runs_reconcile_and_closes_session(mocker): + fake_session = MagicMock() + mocker.patch("src.worker.tasks.SessionLocal", return_value=fake_session) + reconcile = mocker.patch( + "src.services.job_service.reconcile_stuck_jobs", + return_value=[{"entity": "IngestBatch", "id": 1, "age_seconds": 9000.0}], + ) + + reconcile_stuck_jobs_cron() + + reconcile.assert_called_once_with(fake_session) + fake_session.close.assert_called_once() + + +def test_db_backup_cron_logs_trigger(mocker): + mock_logger = MagicMock() + mocker.patch("src.worker.tasks.get_logger", return_value=mock_logger) + + db_backup_cron() + + mock_logger.info.assert_called_once_with( + "db_backup_cron_triggered", detail="Nightly backup triggered" + ) diff --git a/uv.lock b/uv.lock index 1f75b1d..7f077b2 100644 --- a/uv.lock +++ b/uv.lock @@ -498,6 +498,7 @@ dependencies = [ { name = "setuptools" }, { name = "sqladmin" }, { name = "sqlalchemy" }, + { name = "sqlalchemy-celery-beat" }, { name = "starlette-exporter" }, { name = "structlog" }, { name = "tabulate" }, @@ -543,6 +544,7 @@ requires-dist = [ { name = "setuptools", specifier = "<81" }, { name = "sqladmin", specifier = "==0.18.0" }, { name = "sqlalchemy", specifier = "==2.0.35" }, + { name = "sqlalchemy-celery-beat", specifier = "==0.8.4" }, { name = "starlette-exporter", specifier = "==0.23.0" }, { name = "structlog", specifier = "==24.4.0" }, { name = "tabulate", specifier = "==0.9.0" }, @@ -580,6 +582,62 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/8f/d7/9322c609343d929e75e7e5e6255e614fcc67572cfd083959cdef3b7aad79/docutils-0.21.2-py3-none-any.whl", hash = "sha256:dafca5b9e384f0e419294eb4d2ff9fa826435bf15f15b7bd45723e8ad76811b2", size = 587408, upload-time = "2024-04-23T18:57:14.835Z" }, ] +[[package]] +name = "ephem" +version = "4.2.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/37/f0/a38e882d3c73bb8bcc37a0a27c9a7b8fceb7906312006fb03c01d0a098b3/ephem-4.2.1.tar.gz", hash = "sha256:920cb30369c79fde1088c2060d555ea5f8a50fdc80a9265832fd5bf195cf147f", size = 1261529, upload-time = "2026-02-28T09:48:25.935Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b8/59/5849e612e6460c82b36396d91fd63d0126f72bbe9fe2f6a990bd1d03068f/ephem-4.2.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:de9b0ed3836f4b2765e0556058e0dcc595c8226f7aa9d0f04f09b5775b792525", size = 1431686, upload-time = "2026-02-28T09:46:32.396Z" }, + { url = "https://files.pythonhosted.org/packages/a1/0f/345f5801f90d1e0888bdaeb0ad23559cd1bcb50841247c90aa5edf243ed4/ephem-4.2.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:0017a0ff0eac2632943d37b9f12cfea65062e9dc0b3476d17a11319409f0b82f", size = 1433118, upload-time = "2026-02-28T09:46:33.779Z" }, + { url = "https://files.pythonhosted.org/packages/5b/37/f5d55fe605493d4509b35805c0a4b89ffb687fbecd02d58adea398e92344/ephem-4.2.1-cp312-cp312-manylinux2010_i686.manylinux_2_12_i686.manylinux_2_28_i686.whl", hash = "sha256:f3ee0fdb7a0c19d0d4d6043f3f52f216fcf38619783c4ffad285f8f71ec26cd9", size = 1768826, upload-time = "2026-02-28T09:46:35.472Z" }, + { url = "https://files.pythonhosted.org/packages/79/b7/ce41706b23a572f1c9658262ec164eea7e74a2519b866dddf656332f7d88/ephem-4.2.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:17c09bc88fca62284bd5f055dc3fcad8a5fc38de35f59f451ade82761512a45f", size = 1788922, upload-time = "2026-02-28T09:46:36.675Z" }, + { url = "https://files.pythonhosted.org/packages/9f/44/a07a6247b5b75e9177e92a1e83076ba9bea62f3d54d0ff25d1597707076c/ephem-4.2.1-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:dd8ec19883545c7ff97fab95035862c37827723eae7135e6d78c32d3ac1b413b", size = 1834037, upload-time = "2026-02-28T09:46:38.055Z" }, + { url = "https://files.pythonhosted.org/packages/1b/b4/116a78a309c975c05f69e6d26208d8e607fb84de8f0fa0090306d3cc63f0/ephem-4.2.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:f16ed34b60056eeae3d8afc78ec528981d5a90977e1bc734b9308279eb016791", size = 1784685, upload-time = "2026-02-28T09:46:39.426Z" }, + { url = "https://files.pythonhosted.org/packages/03/be/65d2c768b66b18f9a7f255a3c73963e80001bcb3218af40fe785bc802c31/ephem-4.2.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:5df38b498eaa779385f2422263af0f9688120dee4daddb47f58a03ffd3eaac56", size = 1787128, upload-time = "2026-02-28T09:46:40.751Z" }, + { url = "https://files.pythonhosted.org/packages/64/2d/99d409df0b750036a150a6a7e962811b38c0b02e7f77f30b1aa73bda73ff/ephem-4.2.1-cp312-cp312-musllinux_1_2_i686.whl", hash = "sha256:bd1927af82de44f0c6d6690c3d6ec3a57a619ecdba20706480ffd2b65ee59bec", size = 1798850, upload-time = "2026-02-28T09:46:41.962Z" }, + { url = "https://files.pythonhosted.org/packages/ce/5e/5cf9d592ae14d2a46a0b8429381e0ea1436e62d044ce1031ffe25f4d79b7/ephem-4.2.1-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:05a216cbb739de6c412f447da32f5f613d15c582424dd07d5341cda798cb80f1", size = 1817530, upload-time = "2026-02-28T09:46:44.005Z" }, + { url = "https://files.pythonhosted.org/packages/96/b3/bb285a1a14abdf630939809538dfe92531881492b23e1d108a2ff2ba464e/ephem-4.2.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:9bbda62115c9d1c2d967b6b3b80d4ab3b2c7662c2dfd587ba50fc55a6c48aa80", size = 1784631, upload-time = "2026-02-28T09:46:45.222Z" }, + { url = "https://files.pythonhosted.org/packages/40/4c/5297a1cd52d4b1ab7d96e941ebec2086ffef75e42d4df663900c31691994/ephem-4.2.1-cp312-cp312-win32.whl", hash = "sha256:979b2ef09eb29674ca2742124c1ffc231dcbbcb0fcc7f312b5e70cb7e43a62fc", size = 1399811, upload-time = "2026-02-28T09:46:46.86Z" }, + { url = "https://files.pythonhosted.org/packages/06/b4/f210bbb64fbc59e4d2e72433e767396a3471fde26f49fd56ac6930f36532/ephem-4.2.1-cp312-cp312-win_amd64.whl", hash = "sha256:c35502ba6c87cf1d3ff7ecd3690fadeea39583ef783d7d365037b385d3d72eb0", size = 1414939, upload-time = "2026-02-28T09:46:48.078Z" }, + { url = "https://files.pythonhosted.org/packages/bb/8a/2af7194252902b8f9bd6f282051864d86311b90e846a9f9a6c33798d60a9/ephem-4.2.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:19ad36d7a5bf63cd424576ff392d7bd9bf579d903b14ecd98173d577a51051e1", size = 1431651, upload-time = "2026-02-28T09:46:49.335Z" }, + { url = "https://files.pythonhosted.org/packages/ef/3f/d28fb26afacb07c81268a5f7b6a1c2b4a14d0304899c20502205dfccdeaf/ephem-4.2.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:458637fd487cdc897a86912089086161f042c3525a268245490565255bcf46a0", size = 1433094, upload-time = "2026-02-28T09:46:50.736Z" }, + { url = "https://files.pythonhosted.org/packages/43/b9/008cbb78827b145b7afaa152af6991ba7fc5baa99cf16cf64ec9f0fef9ce/ephem-4.2.1-cp313-cp313-manylinux2010_i686.manylinux_2_12_i686.manylinux_2_28_i686.whl", hash = "sha256:b99bdfa5372d83f43c052c845c05260e9355824dc0a88bbb0b9b4bdd438305fc", size = 1768840, upload-time = "2026-02-28T09:46:52.974Z" }, + { url = "https://files.pythonhosted.org/packages/b4/b2/2e6674089d4eb0e4cc4bde6e30d841ff9bcd2d513deebb4ae41b83eaef84/ephem-4.2.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:9ae26bc5abc4073edf6898e5969f3e7b77c47d3c88521177cb3df14fe8963259", size = 1788881, upload-time = "2026-02-28T09:46:54.655Z" }, + { url = "https://files.pythonhosted.org/packages/b9/b5/c04ddeea27c093146d064c994bd795ce5244533b839ef7ccfe35850fb289/ephem-4.2.1-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:078894d72e38cfaee8d00959d6720506af61c0fba5ef42f9e2ea1528296a4320", size = 1834059, upload-time = "2026-02-28T09:46:56.083Z" }, + { url = "https://files.pythonhosted.org/packages/53/1c/c7149f3be5e9852316907621e6d6e2f8833b2044c12f041157495eb74358/ephem-4.2.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:3b37ef161ca287f025c7c299b612fbb7e83530840c938d06a61fcd7889a66db5", size = 1784679, upload-time = "2026-02-28T09:46:57.908Z" }, + { url = "https://files.pythonhosted.org/packages/eb/54/3f8575295de6c07b151d73dfa4d7ed0dd3efe6a02b7b633fbd74381e13d2/ephem-4.2.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:5e9934d309fb3c6d79373259b2a9fc71874a512b90d61523e334720805b6fa5c", size = 1787191, upload-time = "2026-02-28T09:46:59.71Z" }, + { url = "https://files.pythonhosted.org/packages/0d/72/b961fa4b1efb39ac0021e1e31fb3c0a6c5fad45d7fde4908748252e2ac17/ephem-4.2.1-cp313-cp313-musllinux_1_2_i686.whl", hash = "sha256:5530ca3c85a22d31838844427b96abcbe96da6048b6e83a675a5bcc23e5af43c", size = 1798895, upload-time = "2026-02-28T09:47:01.067Z" }, + { url = "https://files.pythonhosted.org/packages/74/ce/93ad503431136611187a60c986441510cd69908add782003526e0d7bb24b/ephem-4.2.1-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:61e38e2affe676fc9f9a8fbd4f575f05ca7db4f87a6e5fc589a09c14fe68850e", size = 1817637, upload-time = "2026-02-28T09:47:02.712Z" }, + { url = "https://files.pythonhosted.org/packages/59/ff/fbcd7a7795741e5770c73b39dd1318215b357f6e8d503233ca3b3fb10437/ephem-4.2.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:46427d35a2fbf4a95fcf6de9d36dc0a29e49b96a63654856285f0262c4082fc7", size = 1784648, upload-time = "2026-02-28T09:47:04.151Z" }, + { url = "https://files.pythonhosted.org/packages/8f/c9/340dd18a288ecdf079ae15711a84d8fb82bad63dc7f1e985b88aeaeea81c/ephem-4.2.1-cp313-cp313-win32.whl", hash = "sha256:a075014299467d29598019fb11972a8f3f476d28962987231b24a234ea8764f2", size = 1399821, upload-time = "2026-02-28T09:47:06.162Z" }, + { url = "https://files.pythonhosted.org/packages/57/70/323fd7145e7a9abb4878c1abb5822b9578db3fa4d8d952806941d7c7910d/ephem-4.2.1-cp313-cp313-win_amd64.whl", hash = "sha256:f22a797ab3658f5f1ae56dd32627e9ef4766aa3e9dbddc817cbb2741893a8a28", size = 1414942, upload-time = "2026-02-28T09:47:07.781Z" }, + { url = "https://files.pythonhosted.org/packages/9d/42/0a94728aaf67b9bb7ee2f0f300380ebaea3362f188c1d9956c0d88ada6ec/ephem-4.2.1-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:6aa46747d0866afcaf64c25855546ecbeccb58456a93b368c963a8e18e414794", size = 1431765, upload-time = "2026-02-28T09:47:09.229Z" }, + { url = "https://files.pythonhosted.org/packages/45/45/e3528adb5e496b4f2c6853313b4bb65f3ea665b41cd4553847ece39040f1/ephem-4.2.1-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:8339dcc7187c99b169d3b0f650b9e1f1047f6d7687bda025acbdeb049ae3e9a2", size = 1433099, upload-time = "2026-02-28T09:47:11.941Z" }, + { url = "https://files.pythonhosted.org/packages/1f/aa/e8b86b27357eafbbdc15ab315014c15651c9009d504d73b9e1fab7ac7062/ephem-4.2.1-cp314-cp314-manylinux2010_i686.manylinux_2_12_i686.manylinux_2_28_i686.whl", hash = "sha256:560a417a4252f4c77a8ebc325a081c684f24c2287f82579ee4eb62d382dcfa89", size = 1769064, upload-time = "2026-02-28T09:47:13.632Z" }, + { url = "https://files.pythonhosted.org/packages/38/28/971ada8acdf52afafb1866cc474dba1cc3deb631f7092e7dc10faba88198/ephem-4.2.1-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:9ce6b0b076b18e9b7a7a9ff5f37573455849079ec232eae0419c172304e7d2aa", size = 1788690, upload-time = "2026-02-28T09:47:15.182Z" }, + { url = "https://files.pythonhosted.org/packages/c1/b3/75194b704da47e649a449d7797ce78f18b3147dfee761fa7dea83eb23fd9/ephem-4.2.1-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:ec91aa7735a10bfd8a4f79afb39567500483a1b5812a3ba675fb10abb5aa95d1", size = 1833795, upload-time = "2026-02-28T09:47:16.993Z" }, + { url = "https://files.pythonhosted.org/packages/e0/c8/578499dc1383ddec58a79be525e54d47a23210c8147a65610ee7cf477406/ephem-4.2.1-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:3610fd501549fe8e29417ecd9fecdbe19e1bc416bd76ee24e40ae81ce3049760", size = 1784463, upload-time = "2026-02-28T09:47:18.323Z" }, + { url = "https://files.pythonhosted.org/packages/8c/21/c76b805d70df3e45cf4795ec19bdb91b0fe42d473b56ea8c80f7706af1f6/ephem-4.2.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:27b8d6893cc47765ab55b2eb4f4bad97b8ffa844ff6b67b0dcad44e48d0b6e77", size = 1786832, upload-time = "2026-02-28T09:47:19.687Z" }, + { url = "https://files.pythonhosted.org/packages/66/1e/79e7eb46c1bda35187a4cacc767f3d6707247a4afc759981c3c69ed6af6b/ephem-4.2.1-cp314-cp314-musllinux_1_2_i686.whl", hash = "sha256:e29c7aabcc32af070d99fd7a2268e7c854d7b68df3414133b5581eb9996939ce", size = 1799002, upload-time = "2026-02-28T09:47:21.594Z" }, + { url = "https://files.pythonhosted.org/packages/61/ef/6ab516d889986a7c16731f9b7a89a9a689ca9a3196dc544d187f2f82f9da/ephem-4.2.1-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:b831567c5db7b101e9e647cfac38da458f48b4c66bee949cff11a875f260d153", size = 1817236, upload-time = "2026-02-28T09:47:22.97Z" }, + { url = "https://files.pythonhosted.org/packages/4c/12/9c4d2014ddad6a9c4e3a7cc43ed9f9585d60a04475dc47b8822b78fbb3dd/ephem-4.2.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:7359e573798357318738f628660946e4cee89f39e84940264e428504d541921b", size = 1784500, upload-time = "2026-02-28T09:47:24.319Z" }, + { url = "https://files.pythonhosted.org/packages/47/b5/59b5554ba2f8a850a65e3beb52cf290b4616e2e94827261b3a3b826e1ac4/ephem-4.2.1-cp314-cp314-win32.whl", hash = "sha256:c0bc8c928b7ef10167b7a2f2f20c7ee079c3b988b9a3a943bda123949acef79f", size = 1406790, upload-time = "2026-02-28T09:47:25.96Z" }, + { url = "https://files.pythonhosted.org/packages/ec/a1/a40f57eae2589933943369d0aa4de024fbba18f5689c06c5f012b59627de/ephem-4.2.1-cp314-cp314-win_amd64.whl", hash = "sha256:936d7a8745fbb18869e238d541f027cd7fc086e3fdfb826cb76fcfb4bed3e35a", size = 1423041, upload-time = "2026-02-28T09:47:27.286Z" }, + { url = "https://files.pythonhosted.org/packages/84/bb/55932750261f35d84c944941dff761112eb9582b608b314fd8c4b87e89a9/ephem-4.2.1-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:1e5806b5a633fb25480486d5419604e82a6932f9f9e6855f1e07da852b8de570", size = 1432847, upload-time = "2026-02-28T09:47:28.603Z" }, + { url = "https://files.pythonhosted.org/packages/d1/8e/c6bbb724f1a51ec79dc9c167dfde27246095abecdf45057fc6de50822e94/ephem-4.2.1-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:5ad67a86ffeeda5173d88b04b9477b5ca3e78331acc168a084d18779452c0f7a", size = 1433840, upload-time = "2026-02-28T09:47:30.009Z" }, + { url = "https://files.pythonhosted.org/packages/0c/2e/84b07818630298656dbae0c95c0e99b89ff98d6a87daf9819a5c5aee191b/ephem-4.2.1-cp314-cp314t-manylinux2010_i686.manylinux_2_12_i686.manylinux_2_28_i686.whl", hash = "sha256:2b8792e60114a07253160e81b63f28b3c021f4a5dfbee91a6db526965bdc039f", size = 1782457, upload-time = "2026-02-28T09:47:31.827Z" }, + { url = "https://files.pythonhosted.org/packages/be/fc/357df051cdc848b482c823f18dc94632d40717c64fc0068d79b91cf12f75/ephem-4.2.1-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7812f16600ecec6d7b21194a38948ea3a0c406332481375e735924a3f82f5e29", size = 1804033, upload-time = "2026-02-28T09:47:33.19Z" }, + { url = "https://files.pythonhosted.org/packages/72/cc/4348e26e0ba41a5e5e6e4ecc1fca81ceaccc656c9ad3d4251c301bf34d78/ephem-4.2.1-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:e020cb6976ff3badb303bc55026366fa3886280f15f6d4d46c52fa36356fe16b", size = 1847470, upload-time = "2026-02-28T09:47:34.926Z" }, + { url = "https://files.pythonhosted.org/packages/a4/9c/77f0007679d0de12372db61d863bee604441b4827b27dbcbed07f48a815c/ephem-4.2.1-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:788e61fa779c54e6dd745f57adaeeff14eda2b8dd737ef578af6ac8415d31a48", size = 1796560, upload-time = "2026-02-28T09:47:36.363Z" }, + { url = "https://files.pythonhosted.org/packages/68/6f/30751b504be2c55b9a2a77f678003f0e065554d1f2f75f3ec85162f2ccd0/ephem-4.2.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:e8c6598443375f119c9f7c586f3198c6ac60a38bac3fbc08781c55c8e102cbba", size = 1800619, upload-time = "2026-02-28T09:47:37.755Z" }, + { url = "https://files.pythonhosted.org/packages/fa/79/49981b3ad843f2180a17ea6f89874ff5ec4d26b6d6fad754be4cbaf97d13/ephem-4.2.1-cp314-cp314t-musllinux_1_2_i686.whl", hash = "sha256:d62cea09046153052cc15c85938fadf7648a7c8b1133d729b76f20d4cabcfefc", size = 1811800, upload-time = "2026-02-28T09:47:39.506Z" }, + { url = "https://files.pythonhosted.org/packages/16/c9/235acf25dfa74e041a4d95b3fbe6835ddd152d6479ebb762da31818bee09/ephem-4.2.1-cp314-cp314t-musllinux_1_2_s390x.whl", hash = "sha256:bfdcc68b2a5ac84305d792ca8ffac23bed0b5be21edbf45b33dd00423f7a6fdb", size = 1829542, upload-time = "2026-02-28T09:47:41.284Z" }, + { url = "https://files.pythonhosted.org/packages/0e/70/429ee91d941b4969571fca658f3350eb5e217f04e4aa9bc295b77613ec3b/ephem-4.2.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:15bc51d5c926b65ec15742042de79f6181a979809a6f9da1bd8d6abfb23415b7", size = 1795831, upload-time = "2026-02-28T09:47:42.703Z" }, + { url = "https://files.pythonhosted.org/packages/9b/92/cc35c78c00894e64d312960507ee84c9fa76919ffca28a7f92ffa59bda20/ephem-4.2.1-cp314-cp314t-win32.whl", hash = "sha256:ce86665355d88e4e7ccc3b9e9c74905a2c6ad51c27433e5c98371a7ea38ad502", size = 1407887, upload-time = "2026-02-28T09:47:44.374Z" }, + { url = "https://files.pythonhosted.org/packages/da/c8/b0f1788a6a87a0a18e799b73486329f33f0ae4c152d62bca46476e76fc35/ephem-4.2.1-cp314-cp314t-win_amd64.whl", hash = "sha256:c4e6e15dc6ccc22504e27aea58a06f8308fd7eb007947d16dfcea6a3aec28b4d", size = 1424698, upload-time = "2026-02-28T09:47:46.38Z" }, +] + [[package]] name = "fastapi" version = "0.115.0" @@ -2144,6 +2202,21 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/0e/c6/33c706449cdd92b1b6d756b247761e27d32230fd6b2de5f44c4c3e5632b2/SQLAlchemy-2.0.35-py3-none-any.whl", hash = "sha256:2ab3f0336c0387662ce6221ad30ab3a5e6499aab01b9790879b6578fd9b8faa1", size = 1881276, upload-time = "2024-09-16T23:14:28.324Z" }, ] +[[package]] +name = "sqlalchemy-celery-beat" +version = "0.8.4" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "celery" }, + { name = "ephem" }, + { name = "sqlalchemy" }, + { name = "tzdata" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/10/e7/06b395aad83a994e6110b66b07f582c634ceb141f5ae84c9f38aeb85ba7d/sqlalchemy_celery_beat-0.8.4.tar.gz", hash = "sha256:c5f3f7b456f76072ea7dcd3f9363d45b05ae915b17496d9f49225fd0e918a623", size = 31407, upload-time = "2025-05-20T17:45:29.116Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/0f/13/97fb283a448178d8aaa01c9d0f51927ce4072cd10f53bf7b891941ded42f/sqlalchemy_celery_beat-0.8.4-py3-none-any.whl", hash = "sha256:0bba9e821c9152cc5b1d406df8dd814162b7f44aad62b9f5636c934a2d007514", size = 21407, upload-time = "2025-05-20T17:45:27.163Z" }, +] + [[package]] name = "starlette" version = "0.38.6"