From de39a3748d715a866bbf817023c050c77482b42b Mon Sep 17 00:00:00 2001 From: ibrog Date: Thu, 17 Sep 2026 16:49:30 +0300 Subject: [PATCH] feat(pipeline): add safe review replicas Generated-with: Codex --- README.md | 87 +- promptpilot/api.py | 11 + promptpilot/bot.py | 50 +- promptpilot/db.py | 762 ++++++++++- promptpilot/fallback_handoff.py | 48 +- promptpilot/herdr_exec.py | 189 ++- promptpilot/pipeline_insights.py | 670 +++++++-- promptpilot/process_tree.py | 24 +- promptpilot/project_pipeline.py | 280 +++- promptpilot/static/index.html | 18 +- promptpilot/worker.py | 1116 +++++++++++++-- tests/test_herdr_workflow_completion.py | 11 +- tests/test_pipeline_replicas.py | 1655 +++++++++++++++++++++++ tests/test_process_tree.py | 53 +- tests/test_project_pipeline.py | 42 + tests/test_schedule_series.py | 17 +- tests/test_worker_admission.py | 8 +- 17 files changed, 4683 insertions(+), 358 deletions(-) create mode 100644 tests/test_pipeline_replicas.py diff --git a/README.md b/README.md index d6274de..64af43f 100644 --- a/README.md +++ b/README.md @@ -682,15 +682,17 @@ Web UI: **⚙ Providers** → «Изменить» у нужного прова атомарно сохраняет полный последний успешный ответ и исторический снимок по `profile_id`. Обычное открытие читает только SQLite и не обращается к GitHub — в том числе после перезапуска server или bot. Возраст, источник и признак -устаревания снимка видны в интерфейсе. Узкое место — этап с максимальным ETA: учитываются -`backlog / capacity`, интервал серии и средняя длительность её выполнения. +устаревания снимка видны в интерфейсе. Узкое место — этап с максимальным ETA: +учитываются `backlog / (capacity × активные реплики)`, интервал серии и средняя +длительность её выполнения. Без явного `replicas` число реплик равно единице. PromptPilot не поставляет профили конкретных проектов. Метки, запросы и лимиты задаются пользователем в `~/.promptpilot/pipeline_profiles.json` (или файле из `PP_PIPELINE_PROFILES`), поэтому чужие репозитории не появляются в новой установке. Если в профиле включён `priority_control`, дашборд показывает элементы очередей и позволяет поставить P0…P3 либо вернуть автоматическую оценку. Кнопка -«⚡ Следующим» ставит P0 и переносит будущий запуск соответствующей серии на -ближайшее время; поставленная на паузу серия при этом остаётся на паузе. +«⚡ Следующим» ставит P0 и переносит будущий запуск всех серий-реплик +соответствующей очереди на ближайшее время; поставленная на паузу серия при +этом остаётся на паузе. Одинаковые диагностические сигналы в окне конвейера сворачиваются в одну строку со списком и количеством PR; владелец single-flight барьера всегда показывается первым, чтобы было видно, какой PR сейчас должен пройти REVIEW → MERGE. В основной @@ -736,8 +738,9 @@ GitHub CLI и Go в `C:\Program Files`. Поэтому tray/служба не з останавливает дерево. Поэтому штатный stop не оставляет скрытый worker, который продолжает расходовать токены после «остановки» старого launcher PID. -Рекомендация считается прозрачно: `backlog / capacity` даёт число необходимых -прогонов, а один цикл равен `интервал + средняя длительность прогона` — повтор +Рекомендация считается прозрачно: `backlog / (capacity × активные реплики)` +даёт число необходимых параллельных прогонов, а один цикл равен +`интервал + средняя длительность прогона` — повтор назначается после завершения предыдущего запуска. Поэтому серия `15m` со средним выполнением 30 минут даёт примерно один результат за 45 минут, а не четыре в час. Профиль задаёт желаемое время очистки (`target_clear_hours`, обычно 8 часов), @@ -901,6 +904,75 @@ claim одной задачи происходят в одной SQLite-тран получат одно вхождение. Без `scheduler` остаётся прежний порядок `priority → created_at → id`. +#### Параллельные реплики REVIEW + +Очередь REVIEW может явно владеть несколькими независимыми сериями. Для этого +используется отдельное поле `replicas`; `capacity` по-прежнему означает, сколько +целей обрабатывает **один** запуск, и не подменяет число исполнителей: + +```json +{ + "id": "review", + "capacity": 1, + "replicas": 2, + "series_contains": "MyProject - REVIEW", + "execution": { + "mode": "auto", + "stage": "review", + "command": [ + "{python}", "-m", "promptpilot.project_pipeline", + "--config", "pipelinectl.json", "next", "{stage}" + ], + "required_paths": ["pipelinectl.json"] + } +} +``` + +Нужно создать ровно две активные повторяющиеся серии, например +`MyProject - REVIEW 1` и `MyProject - REVIEW 2`. Каждой задаётся существующий +абсолютный и отличный от остальных `working_dir` — обычно два git worktree. +Отличие проверяется по identity каталога (`device/inode`), поэтому symlink и +варианты регистра одного APFS/NTFS-каталога не считаются разными репликами. +Реплики выполняются на той же машине и используют общую локальную SQLite: +удалённые `machine`, динамический `worktree` задачи, отсутствующие каталоги и +дубли путей блокируются до запуска провайдера. Если настроен `scheduler`, REVIEW +должен присутствовать минимум в двух lanes; без scheduler достаточно общей +конкурентности worker не меньше двух. + +В `pipelinectl.json` для параллельного REVIEW обязательны +`review_completion_gate: "target-v1"` и `fallback_handoff: "target-v1"`. +Необязательный `target_reservation_ttl_seconds` задаёт TTL локальной резервации +(по умолчанию равен `review_lease_seconds`, допустимо 300–28800 секунд): + +```json +{ + "review_completion_gate": "target-v1", + "fallback_handoff": "target-v1", + "review_lease_seconds": 7200, + "target_reservation_ttl_seconds": 7200 +} +``` + +До запуска модели `next review` атомарно резервирует точный +`repository/stage/PR/HEAD` за конкретной попыткой `PP_TASK_ID`; mutex действует +на весь PR даже при смене stage или HEAD. Если первая цель уже занята другой +репликой, election берёт следующую, не начиная повторное ревью. Обычный +`complete review` и `gate-fallback` проверяют и продлевают ту же резервацию +непосредственно перед первой GitHub-мутацией. Обычный terminal exit освобождает +резервацию. TTL задаёт частоту lease heartbeat, но истёкшая строка намеренно не +крадётся, пока её exact task attempt всё ещё `running`: после жёсткого падения +startup recovery сначала по сохранённому ownership descriptor подтверждает +остановку headless/Herdr-провайдера, затем атомарно requeue/cancel задачи и +освобождает fence. Если остановку доказать нельзя, задача и PR остаются в +quarantine; вручную удалять такую резервацию небезопасно. Резервация — только +локальный scheduling fence и не заменяет GraphQL/HEAD/timeline/CAS-гейты. + +`run_now`, `wake_when`, `wake_after_success` и `adaptive_cadence` применяются ко +всем совпавшим репликам. Дашборд показывает каждую серию, её каталог и состояние, +а ETA использует суммарную параллельную ёмкость. Сейчас `replicas > 1` +намеренно разрешено только для REVIEW; MERGE и прочие изменяющие этапы остаются +single-flight. + Полоса выбирает только серию TRIAGE. Порядок целей *внутри* неё — recovery, ручной P0, затем обычный backlog — остаётся контрактом репозиторного `next` и его свежих gate-проверок. Так машинное расписание не может обойти recovery или @@ -1116,6 +1188,9 @@ HEAD в `review_candidates`. Для двухполосной схемы он т дашборде и боте для каждого этапа. Это позволяет заметить незапланированный fallback сразу, а не по суточному расходу токенов. Пустой REVIEW/MERGE закрывается самим preflight с явной записью «провайдер не запускался». +Для `replicas > 1` небезопасный нетаргетированный `auto → skill` отключён: +недоступный CLI, malformed lease или fallback без signed target блокирует запуск +до модели, а доказанный `target-v1` handoff остаётся разрешён. Интеграционный владелец на `integration-review` или `legacy-integration-review` всегда оставляет MERGE на полном skill-маршруте. diff --git a/promptpilot/api.py b/promptpilot/api.py index 2e5423c..cb38be9 100644 --- a/promptpilot/api.py +++ b/promptpilot/api.py @@ -338,6 +338,9 @@ def api_pipeline_item_priority(profile_id: str, queue_id: str, kind: str, number @app.delete("/api/tasks/{task_id}", response_model=dict) def api_delete_task(task_id: int): + task = db.get_task(task_id) + if task and task.status.value == "running": + raise HTTPException(409, "Cancel the running task and wait for it to stop first") if not db.delete_task(task_id): raise HTTPException(404, "Task not found") return {"ok": True} @@ -346,6 +349,14 @@ def api_delete_task(task_id: int): @app.post("/api/tasks/{task_id}/reset", response_model=dict) def api_reset_task(task_id: int): if not db.reset_task(task_id): + task = db.get_task(task_id) + if (task and task.status.value == "running" + and db.task_has_live_pipeline_target_reservation(task_id)): + raise HTTPException( + 409, + "This pipeline task still owns a live provider/target; cancel it " + "and wait for cleanup before resetting", + ) raise HTTPException(400, "Task not found or not in running state") return {"ok": True} diff --git a/promptpilot/bot.py b/promptpilot/bot.py index 96e8bd7..d2416f7 100644 --- a/promptpilot/bot.py +++ b/promptpilot/bot.py @@ -199,8 +199,11 @@ def _task_detail_keyboard(task) -> InlineKeyboardMarkup: rows.append(action_row) if task.result and len(task.result) > 800: rows.append([InlineKeyboardButton("📄 Полный вывод", callback_data=f"full_result:{task.id}")]) - rows.append([InlineKeyboardButton("← К списку", callback_data="tasklist"), - InlineKeyboardButton("🗑 Удалить", callback_data=f"delete_task:{task.id}")]) + final_row = [InlineKeyboardButton("← К списку", callback_data="tasklist")] + if status != "running": + final_row.append(InlineKeyboardButton( + "🗑 Удалить", callback_data=f"delete_task:{task.id}")) + rows.append(final_row) return InlineKeyboardMarkup(rows) @@ -657,7 +660,13 @@ async def cb_reset_task(update: Update, context: ContextTypes.DEFAULT_TYPE): await query.edit_message_text(f"Задача #{task_id} возвращена в очередь.", reply_markup=_after_action_keyboard(task_id)) else: - await query.answer("Задача не в статусе running.", show_alert=True) + if db.task_has_live_pipeline_target_reservation(task_id): + await query.answer( + "Сначала останови задачу и дождись завершения очистки провайдера.", + show_alert=True, + ) + else: + await query.answer("Задача не в статусе running.", show_alert=True) @require_auth @@ -666,8 +675,13 @@ async def cb_delete_task(update: Update, context: ContextTypes.DEFAULT_TYPE): the button sits next to the frequently-used ones and a stray tap would destroy the task together with its result.""" query = update.callback_query - await query.answer() task_id = int(query.data.split(":")[1]) + task = db.get_task(task_id) + if task and task.status.value == "running": + await query.answer( + "Сначала отмените задачу и дождитесь её остановки.", show_alert=True) + return + await query.answer() await query.edit_message_reply_markup(reply_markup=InlineKeyboardMarkup([[ InlineKeyboardButton("🗑 Точно удалить", callback_data=f"del_yes:{task_id}"), InlineKeyboardButton("↩ Отмена", callback_data=f"task:{task_id}"), @@ -685,7 +699,11 @@ async def cb_delete_task_confirm(update: Update, context: ContextTypes.DEFAULT_T reply_markup=InlineKeyboardMarkup([[ InlineKeyboardButton("← К списку", callback_data="tasklist")]])) else: - await query.answer("Задача не найдена.", show_alert=True) + task = db.get_task(task_id) + reason = ("Сначала отмените задачу и дождитесь её остановки." + if task and task.status.value == "running" + else "Задача не найдена.") + await query.answer(reason, show_alert=True) @require_auth @@ -986,12 +1004,30 @@ def _pipeline_text(data: dict) -> str: cadence = queue.get("adaptive_cadence") or {} cadence_text = (f"; adaptive {cadence.get('mode')}" if cadence else "") + replica_count = int(queue.get("replica_count") or 1) + replicas_active = int(queue.get("replicas_active") or 0) + parallel_capacity = queue.get("parallel_capacity") + parallel_capacity = int(queue["capacity"] if parallel_capacity is None + else parallel_capacity) + capacity_text = ( + f"{queue['capacity']} × {replicas_active} = {parallel_capacity}") + replica_status = queue.get("replica_status") or {} + replica_details = "" + if replica_count > 1: + issues = replica_status.get("issues") or [] + replica_details = ( + f"\n Реплики: {replicas_active}/{replica_count} активны, " + f"найдено {replica_status.get('present', 0)}" + + (f"; ошибка: {'; '.join(issues)}" if issues else "") + ) lines.append(f"{marker} {queue['title']}: {backlog if backlog is not None else '—'} / " - f"{queue['capacity']} за прогон = {runs_needed if runs_needed is not None else '—'} прогонов; " + f"{capacity_text} за параллельный прогон = " + f"{runs_needed if runs_needed is not None else '—'} прогонов; " f"сейчас {queue['interval'] or 'не настроено'}{cadence_text}; " f"средний запуск {duration_text}; ETA {eta if eta is not None else '—'} ч\n" f" Маршрут: {route_text}\n" - f" Рекомендация: {queue['recommendation']}") + f" Рекомендация: {queue['recommendation']}" + f"{replica_details}") runs = recent.get("runs", {}) lines.extend(["", f"Прогоны за окно: {runs.get('runs', 0)}; готово {runs.get('ready', 0)}, " f"нужен человек {runs.get('human', 0)}, не смог {runs.get('unable', 0)}, " diff --git a/promptpilot/db.py b/promptpilot/db.py index d01abf7..f643b5f 100644 --- a/promptpilot/db.py +++ b/promptpilot/db.py @@ -137,11 +137,37 @@ payload_json TEXT NOT NULL ); +CREATE TABLE IF NOT EXISTS pipeline_target_reservations ( + repository TEXT NOT NULL, + stage TEXT NOT NULL, + number INTEGER NOT NULL, + head TEXT NOT NULL, + task_id INTEGER NOT NULL, + task_started_at TEXT NOT NULL DEFAULT '', + ownership_kind TEXT NOT NULL DEFAULT '', + herdr_session_state TEXT NOT NULL DEFAULT '', + herdr_pane_id TEXT NOT NULL DEFAULT '', + herdr_tab_id TEXT NOT NULL DEFAULT '', + herdr_workspace_id TEXT NOT NULL DEFAULT '', + token TEXT NOT NULL, + reserved_at TEXT NOT NULL, + expires_at TEXT NOT NULL, + PRIMARY KEY (repository, stage, number, head) +); + CREATE INDEX IF NOT EXISTS idx_tasks_status ON tasks(status); CREATE INDEX IF NOT EXISTS idx_tasks_runnable ON tasks(status, priority, next_run_at); CREATE INDEX IF NOT EXISTS idx_prompt_log_project ON prompt_log(project); CREATE INDEX IF NOT EXISTS idx_pipeline_snapshots_profile_time ON pipeline_snapshots(profile_id, captured_at); +CREATE UNIQUE INDEX IF NOT EXISTS idx_pipeline_target_reservations_task + ON pipeline_target_reservations(repository, task_id); +CREATE UNIQUE INDEX IF NOT EXISTS idx_pipeline_target_reservations_task_global + ON pipeline_target_reservations(task_id); +CREATE UNIQUE INDEX IF NOT EXISTS idx_pipeline_target_reservations_pr + ON pipeline_target_reservations(repository, number); +CREATE INDEX IF NOT EXISTS idx_pipeline_target_reservations_expiry + ON pipeline_target_reservations(expires_at); CREATE TABLE IF NOT EXISTS workflows ( id TEXT PRIMARY KEY, @@ -348,10 +374,30 @@ captured_at TEXT NOT NULL, payload_json TEXT NOT NULL )""", "CREATE INDEX IF NOT EXISTS idx_pipeline_snapshots_profile_time ON pipeline_snapshots(profile_id, captured_at)", + """CREATE TABLE IF NOT EXISTS pipeline_target_reservations ( + repository TEXT NOT NULL, stage TEXT NOT NULL, number INTEGER NOT NULL, + head TEXT NOT NULL, task_id INTEGER NOT NULL, token TEXT NOT NULL, + reserved_at TEXT NOT NULL, expires_at TEXT NOT NULL, + PRIMARY KEY (repository, stage, number, head) + )""", + "CREATE UNIQUE INDEX IF NOT EXISTS idx_pipeline_target_reservations_task ON pipeline_target_reservations(repository, task_id)", + "CREATE UNIQUE INDEX IF NOT EXISTS idx_pipeline_target_reservations_task_global ON pipeline_target_reservations(task_id)", + "CREATE UNIQUE INDEX IF NOT EXISTS idx_pipeline_target_reservations_pr ON pipeline_target_reservations(repository, number)", + "CREATE INDEX IF NOT EXISTS idx_pipeline_target_reservations_expiry ON pipeline_target_reservations(expires_at)", + "ALTER TABLE pipeline_target_reservations ADD COLUMN task_started_at TEXT NOT NULL DEFAULT ''", + "ALTER TABLE pipeline_target_reservations ADD COLUMN ownership_kind TEXT NOT NULL DEFAULT ''", + "ALTER TABLE pipeline_target_reservations ADD COLUMN herdr_session_state TEXT NOT NULL DEFAULT ''", + "ALTER TABLE pipeline_target_reservations ADD COLUMN herdr_pane_id TEXT NOT NULL DEFAULT ''", + "ALTER TABLE pipeline_target_reservations ADD COLUMN herdr_tab_id TEXT NOT NULL DEFAULT ''", + "ALTER TABLE pipeline_target_reservations ADD COLUMN herdr_workspace_id TEXT NOT NULL DEFAULT ''", ] WORKFLOW_SCHEMA_VERSION = "workflow_orchestrator_w0_v1" WORKFLOW_STAGE_SCHEMA_VERSION = "workflow_stage_planner_w3_v1" +PIPELINE_TARGET_RESERVATION_SCHEMA_VERSION = "pipeline_target_reservations_v1" +PIPELINE_TARGET_ATTEMPT_SCHEMA_VERSION = "pipeline_target_reservations_attempt_v2" +PIPELINE_TARGET_OWNERSHIP_SCHEMA_VERSION = "pipeline_target_reservations_ownership_v3" +PIPELINE_TARGET_HERDR_SCHEMA_VERSION = "pipeline_target_reservations_herdr_v4" def _now() -> str: @@ -440,6 +486,26 @@ def _init_db_once(): VALUES (?, ?)""", (WORKFLOW_STAGE_SCHEMA_VERSION, _now()), ) + conn.execute( + """INSERT OR IGNORE INTO schema_migrations (version, applied_at) + VALUES (?, ?)""", + (PIPELINE_TARGET_RESERVATION_SCHEMA_VERSION, _now()), + ) + conn.execute( + """INSERT OR IGNORE INTO schema_migrations (version, applied_at) + VALUES (?, ?)""", + (PIPELINE_TARGET_ATTEMPT_SCHEMA_VERSION, _now()), + ) + conn.execute( + """INSERT OR IGNORE INTO schema_migrations (version, applied_at) + VALUES (?, ?)""", + (PIPELINE_TARGET_OWNERSHIP_SCHEMA_VERSION, _now()), + ) + conn.execute( + """INSERT OR IGNORE INTO schema_migrations (version, applied_at) + VALUES (?, ?)""", + (PIPELINE_TARGET_HERDR_SCHEMA_VERSION, _now()), + ) _backfill_task_series(conn) @@ -462,6 +528,416 @@ def init_db(): time.sleep(INIT_DB_BUSY_DELAYS[attempt]) +def _pipeline_target_identity(repository: str, stage: str, number: int, + head: str, task_id: int) -> tuple[str, str, int, str, int]: + repository = str(repository or "").strip().lower() + stage = str(stage or "").strip().lower() + head = str(head or "").strip().lower() + if (not repository or "/" not in repository or not stage + or isinstance(number, bool) or not isinstance(number, int) or number <= 0 + or not re.fullmatch(r"[0-9a-f]{40}", head) + or isinstance(task_id, bool) or not isinstance(task_id, int) or task_id <= 0): + raise ValueError("invalid pipeline target reservation identity") + return repository, stage, number, head, task_id + + +def _pipeline_reservation_ttl(ttl_seconds: int) -> int: + if (isinstance(ttl_seconds, bool) or not isinstance(ttl_seconds, int) + or not 60 <= ttl_seconds <= 86400): + raise ValueError("pipeline target reservation TTL must be 60..86400 seconds") + return ttl_seconds + + +def _reservation_dict(row: sqlite3.Row) -> dict: + return { + "repository": row["repository"], "stage": row["stage"], + "number": int(row["number"]), "head": row["head"], + "task_id": int(row["task_id"]), + "task_started_at": row["task_started_at"], + "ownership_kind": row["ownership_kind"], + "herdr_session_state": row["herdr_session_state"], + "herdr_pane_id": row["herdr_pane_id"], + "herdr_tab_id": row["herdr_tab_id"], + "herdr_workspace_id": row["herdr_workspace_id"], + "token": row["token"], + "reserved_at": row["reserved_at"], "expires_at": row["expires_at"], + } + + +def reserve_pipeline_target(repository: str, stage: str, number: int, + head: str, task_id: int, ttl_seconds: int, + *, task_started_at: str | None = None, + ownership_kind: str | None = None, + now: Optional[datetime] = None) -> Optional[dict]: + """Atomically reserve one exact project-pipeline target for a task. + + A task owns at most one target per repository. Repeating election for + the same task is idempotent and renews its existing token; another task + never steals a live target. Expired rows are crash-recovery leases. + """ + repository, stage, number, head, task_id = _pipeline_target_identity( + repository, stage, number, head, task_id) + ttl_seconds = _pipeline_reservation_ttl(ttl_seconds) + current = (now or datetime.now(timezone.utc)).astimezone(timezone.utc) + now_value = current.isoformat() + expires_at = (current + timedelta(seconds=ttl_seconds)).isoformat() + ownership_kind = str(ownership_kind or "").strip().lower() + if ownership_kind not in {"", "headless", "herdr"}: + raise ValueError("invalid pipeline provider ownership kind") + with _connect(immediate=True) as conn: + attempt = "" + if task_started_at is not None: + attempt = str(task_started_at).strip() + try: + parsed_attempt = datetime.fromisoformat(attempt) + except (TypeError, ValueError): + raise ValueError("invalid pipeline task attempt timestamp") + if parsed_attempt.tzinfo is None: + raise ValueError("pipeline task attempt timestamp must include timezone") + task_row = conn.execute( + "SELECT status, started_at FROM tasks WHERE id = ?", + (task_id,), + ).fetchone() + if (task_row is None or task_row["status"] != "running" + or task_row["started_at"] != attempt): + return None + conn.execute( + """DELETE FROM pipeline_target_reservations + WHERE expires_at <= ? + AND NOT EXISTS ( + SELECT 1 FROM tasks + WHERE tasks.id = pipeline_target_reservations.task_id + AND tasks.status = 'running' + AND (pipeline_target_reservations.task_started_at = '' + OR tasks.started_at = pipeline_target_reservations.task_started_at) + )""", + (now_value,), + ) + # A PR is the mutation boundary. Stage and head belong to its immutable + # election envelope, but a transition between them must not admit a + # second provider while the first exact attempt is still live. + existing = conn.execute( + """SELECT * FROM pipeline_target_reservations + WHERE repository = ? AND number = ?""", + (repository, number), + ).fetchone() + if existing is not None: + if (existing["stage"] != stage or existing["head"] != head + or int(existing["task_id"]) != task_id + or (attempt and existing["task_started_at"] != attempt) + or (ownership_kind and existing["ownership_kind"] != ownership_kind)): + return None + conn.execute( + """UPDATE pipeline_target_reservations SET expires_at = ? + WHERE repository = ? AND stage = ? AND number = ? AND head = ? + AND task_id = ? AND token = ?""", + (expires_at, repository, stage, number, head, task_id, + existing["token"]), + ) + existing = conn.execute( + """SELECT * FROM pipeline_target_reservations + WHERE repository = ? AND stage = ? AND number = ? AND head = ?""", + (repository, stage, number, head), + ).fetchone() + return _reservation_dict(existing) + + owned = conn.execute( + """SELECT * FROM pipeline_target_reservations + WHERE task_id = ?""", + (task_id,), + ).fetchone() + if owned is not None: + # A live task may call election more than once (or two injected + # calls may race). Never swap its immutable target envelope. A + # terminal transition releases the old row before a real retry. + return None + token = uuid.uuid4().hex + conn.execute( + """INSERT INTO pipeline_target_reservations + (repository, stage, number, head, task_id, task_started_at, + ownership_kind, herdr_session_state, token, reserved_at, expires_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", + (repository, stage, number, head, task_id, attempt, ownership_kind, + "reserved" if ownership_kind == "herdr" else "", token, + now_value, expires_at), + ) + row = conn.execute( + """SELECT * FROM pipeline_target_reservations + WHERE repository = ? AND stage = ? AND number = ? AND head = ?""", + (repository, stage, number, head), + ).fetchone() + return _reservation_dict(row) + + +def _exact_pipeline_reservation_parts(reservation: dict): + if not isinstance(reservation, dict): + raise ValueError("invalid pipeline target reservation") + repository, stage, number, head, task_id = _pipeline_target_identity( + reservation.get("repository"), reservation.get("stage"), + reservation.get("number"), reservation.get("head"), + reservation.get("task_id"), + ) + token = reservation.get("token") + attempt = reservation.get("task_started_at") + if (not isinstance(token, str) or not re.fullmatch(r"[0-9a-f]{32}", token) + or not isinstance(attempt, str) or not attempt): + raise ValueError("invalid exact pipeline target reservation") + return repository, stage, number, head, task_id, token, attempt + + +def begin_pipeline_target_herdr_session(reservation: dict) -> Optional[dict]: + """Persist the crash-safe pre-start state for one exact Herdr attempt.""" + parts = _exact_pipeline_reservation_parts(reservation) + repository, stage, number, head, task_id, token, attempt = parts + with _connect(immediate=True) as conn: + row = conn.execute( + """SELECT * FROM pipeline_target_reservations + WHERE repository = ? AND stage = ? AND number = ? AND head = ? + AND task_id = ? AND token = ? AND task_started_at = ? + AND ownership_kind = 'herdr' + AND EXISTS ( + SELECT 1 FROM tasks + WHERE tasks.id = pipeline_target_reservations.task_id + AND tasks.status = 'running' + AND tasks.started_at = pipeline_target_reservations.task_started_at + )""", + (repository, stage, number, head, task_id, token, attempt), + ).fetchone() + if row is None: + return None + state = row["herdr_session_state"] + if state == "creating": + if row["herdr_pane_id"] or row["herdr_tab_id"] or row["herdr_workspace_id"]: + return None + return _reservation_dict(row) + if state != "reserved": + return None + conn.execute( + """UPDATE pipeline_target_reservations + SET herdr_session_state = 'creating' + WHERE repository = ? AND stage = ? AND number = ? AND head = ? + AND task_id = ? AND token = ? AND task_started_at = ? + AND herdr_session_state = 'reserved'""", + (repository, stage, number, head, task_id, token, attempt), + ) + row = conn.execute( + """SELECT * FROM pipeline_target_reservations + WHERE repository = ? AND stage = ? AND number = ? AND head = ?""", + (repository, stage, number, head), + ).fetchone() + return _reservation_dict(row) if row is not None else None + + +def bind_pipeline_target_herdr_session( + reservation: dict, pane_id: str, tab_id: str, + workspace_id: str = "") -> Optional[dict]: + """Durably bind immutable Herdr IDs before its agent may be started.""" + parts = _exact_pipeline_reservation_parts(reservation) + repository, stage, number, head, task_id, token, attempt = parts + values = (pane_id, tab_id, workspace_id) + if any(not isinstance(value, str) or len(value) > 512 for value in values): + raise ValueError("invalid Herdr ownership descriptor") + if not pane_id or not tab_id: + raise ValueError("Herdr ownership descriptor requires pane and tab ids") + with _connect(immediate=True) as conn: + row = conn.execute( + """SELECT * FROM pipeline_target_reservations + WHERE repository = ? AND stage = ? AND number = ? AND head = ? + AND task_id = ? AND token = ? AND task_started_at = ? + AND ownership_kind = 'herdr' + AND EXISTS ( + SELECT 1 FROM tasks + WHERE tasks.id = pipeline_target_reservations.task_id + AND tasks.status = 'running' + AND tasks.started_at = pipeline_target_reservations.task_started_at + )""", + (repository, stage, number, head, task_id, token, attempt), + ).fetchone() + if row is None: + return None + state = row["herdr_session_state"] + descriptor = ( + row["herdr_pane_id"], row["herdr_tab_id"], + row["herdr_workspace_id"], + ) + if state == "owned": + return _reservation_dict(row) if descriptor == values else None + if state != "creating" or any(descriptor): + return None + conn.execute( + """UPDATE pipeline_target_reservations + SET herdr_session_state = 'owned', herdr_pane_id = ?, + herdr_tab_id = ?, herdr_workspace_id = ? + WHERE repository = ? AND stage = ? AND number = ? AND head = ? + AND task_id = ? AND token = ? AND task_started_at = ? + AND herdr_session_state = 'creating'""", + (pane_id, tab_id, workspace_id, repository, stage, number, head, + task_id, token, attempt), + ) + row = conn.execute( + """SELECT * FROM pipeline_target_reservations + WHERE repository = ? AND stage = ? AND number = ? AND head = ?""", + (repository, stage, number, head), + ).fetchone() + return _reservation_dict(row) if row is not None else None + + +def renew_pipeline_target_reservation(reservation: dict, ttl_seconds: int, + *, now: Optional[datetime] = None) -> Optional[dict]: + """Renew only the still-live exact lease represented by ``reservation``.""" + if not isinstance(reservation, dict) or not isinstance(reservation.get("token"), str): + raise ValueError("invalid pipeline target reservation") + identity = _pipeline_target_identity( + reservation.get("repository"), reservation.get("stage"), + reservation.get("number"), reservation.get("head"), + reservation.get("task_id"), + ) + repository, stage, number, head, task_id = identity + token = reservation["token"] + task_started_at = reservation.get("task_started_at", "") + if not isinstance(task_started_at, str): + raise ValueError("invalid pipeline target reservation attempt") + if not re.fullmatch(r"[0-9a-f]{32}", token): + raise ValueError("invalid pipeline target reservation token") + ttl_seconds = _pipeline_reservation_ttl(ttl_seconds) + current = (now or datetime.now(timezone.utc)).astimezone(timezone.utc) + now_value = current.isoformat() + expires_at = (current + timedelta(seconds=ttl_seconds)).isoformat() + with _connect(immediate=True) as conn: + updated = conn.execute( + """UPDATE pipeline_target_reservations + SET expires_at = CASE WHEN expires_at < ? THEN ? ELSE expires_at END + WHERE repository = ? AND stage = ? AND number = ? AND head = ? + AND task_id = ? AND token = ? + AND (expires_at > ? OR EXISTS ( + SELECT 1 FROM tasks + WHERE tasks.id = pipeline_target_reservations.task_id + AND tasks.status = 'running' + AND (pipeline_target_reservations.task_started_at = '' + OR tasks.started_at = pipeline_target_reservations.task_started_at) + )) + AND pipeline_target_reservations.task_started_at = ? + AND (pipeline_target_reservations.task_started_at = '' OR EXISTS ( + SELECT 1 FROM tasks + WHERE tasks.id = pipeline_target_reservations.task_id + AND tasks.status = 'running' + AND tasks.started_at = pipeline_target_reservations.task_started_at + ))""", + (expires_at, expires_at, repository, stage, number, head, task_id, token, + now_value, task_started_at), + ) + if not updated.rowcount: + conn.execute( + """DELETE FROM pipeline_target_reservations + WHERE expires_at <= ? + AND NOT EXISTS ( + SELECT 1 FROM tasks + WHERE tasks.id = pipeline_target_reservations.task_id + AND tasks.status = 'running' + AND (pipeline_target_reservations.task_started_at = '' + OR tasks.started_at = pipeline_target_reservations.task_started_at) + )""", + (now_value,), + ) + return None + row = conn.execute( + """SELECT * FROM pipeline_target_reservations + WHERE repository = ? AND stage = ? AND number = ? AND head = ?""", + (repository, stage, number, head), + ).fetchone() + return _reservation_dict(row) + + +def renew_task_pipeline_target_reservations( + task_id: int, ttl_seconds: int = 600, + *, now: Optional[datetime] = None) -> int: + """Heartbeat every still-live target owned by one running task. + + Expired rows are never resurrected, and a shorter heartbeat never reduces + an operator-configured longer TTL. This keeps a live/uncertain provider's + target fenced while preserving TTL recovery after a hard process crash. + """ + if isinstance(task_id, bool) or not isinstance(task_id, int) or task_id <= 0: + return 0 + ttl_seconds = _pipeline_reservation_ttl(ttl_seconds) + current = (now or datetime.now(timezone.utc)).astimezone(timezone.utc) + now_value = current.isoformat() + heartbeat_expiry = (current + timedelta(seconds=ttl_seconds)).isoformat() + with _connect(immediate=True) as conn: + updated = conn.execute( + """UPDATE pipeline_target_reservations + SET expires_at = CASE WHEN expires_at < ? THEN ? ELSE expires_at END + WHERE task_id = ? + AND (expires_at > ? OR EXISTS ( + SELECT 1 FROM tasks + WHERE tasks.id = pipeline_target_reservations.task_id + AND tasks.status = 'running' + AND (pipeline_target_reservations.task_started_at = '' + OR tasks.started_at = pipeline_target_reservations.task_started_at) + )) + AND (task_started_at = '' OR EXISTS ( + SELECT 1 FROM tasks + WHERE tasks.id = pipeline_target_reservations.task_id + AND tasks.status = 'running' + AND tasks.started_at = pipeline_target_reservations.task_started_at + ))""", + (heartbeat_expiry, heartbeat_expiry, task_id, now_value), + ) + conn.execute( + """DELETE FROM pipeline_target_reservations + WHERE expires_at <= ? + AND NOT EXISTS ( + SELECT 1 FROM tasks + WHERE tasks.id = pipeline_target_reservations.task_id + AND tasks.status = 'running' + AND (pipeline_target_reservations.task_started_at = '' + OR tasks.started_at = pipeline_target_reservations.task_started_at) + )""", + (now_value,), + ) + return updated.rowcount + + +def release_pipeline_target_reservations(task_id: int) -> int: + """Release every target owned by one completed/aborted task attempt.""" + if isinstance(task_id, bool) or not isinstance(task_id, int) or task_id <= 0: + return 0 + with _connect(immediate=True) as conn: + deleted = conn.execute( + "DELETE FROM pipeline_target_reservations WHERE task_id = ?", + (task_id,), + ) + return deleted.rowcount + + +def list_pipeline_target_reservations(*, repository: str | None = None, + stage: str | None = None, + now: Optional[datetime] = None) -> list[dict]: + """Return live reservations for diagnostics without extending their TTL.""" + current = (now or datetime.now(timezone.utc)).astimezone(timezone.utc).isoformat() + clauses = ["(expires_at > ? OR EXISTS (" + "SELECT 1 FROM tasks " + "WHERE tasks.id = pipeline_target_reservations.task_id " + "AND tasks.status = 'running' " + "AND (pipeline_target_reservations.task_started_at = '' " + "OR tasks.started_at = pipeline_target_reservations.task_started_at)))"] + values: list[object] = [current] + if repository is not None: + clauses.append("repository = ?") + values.append(str(repository).strip().lower()) + if stage is not None: + clauses.append("stage = ?") + values.append(str(stage).strip().lower()) + with _connect() as conn: + rows = conn.execute( + "SELECT * FROM pipeline_target_reservations WHERE " + + " AND ".join(clauses) + + " ORDER BY repository, stage, number, head", + values, + ).fetchall() + return [_reservation_dict(row) for row in rows] + + def _series_title(prompt: str) -> str: return (prompt or "").splitlines()[0].strip()[:160] or "Повторяющаяся задача" @@ -754,29 +1230,70 @@ def _drop_cancel_flag(conn, task_id: int): conn.execute("DELETE FROM settings WHERE key = ?", (f"cancel_task:{task_id}",)) +def _attempt_iso(started_at) -> str | None: + if isinstance(started_at, datetime): + return _to_utc_iso(started_at) + return str(started_at) if started_at else None + + +def _release_task_pipeline_targets(conn, task_id: int, started_at=None): + attempt = _attempt_iso(started_at) + if attempt is None: + conn.execute( + "DELETE FROM pipeline_target_reservations WHERE task_id = ?", + (task_id,), + ) + else: + conn.execute( + """DELETE FROM pipeline_target_reservations + WHERE task_id = ? + AND (task_started_at = '' OR task_started_at = ?)""", + (task_id, attempt), + ) + + def mark_completed(task_id: int, result: str, exit_code: int = 0, model_used: str = None, session_id: str = None, - verdict: str = None): + verdict: str = None, *, expected_started_at=None) -> bool: """Finalize a task, optionally committing its verdict atomically.""" with _connect() as conn: - conn.execute( + attempt = _attempt_iso(expected_started_at) + where = "id = ?" + values = [result, exit_code, _now(), model_used, session_id, verdict, task_id] + if attempt is not None: + where += " AND status = 'running' AND started_at = ?" + values.append(attempt) + cur = conn.execute( "UPDATE tasks SET status = 'completed', result = ?, error = NULL, " "next_run_at = NULL, exit_code = ?, completed_at = ?, model_used = ?, " "session_id = COALESCE(?, session_id), " - "verdict = COALESCE(?, verdict), note = NULL WHERE id = ?", - (result, exit_code, _now(), model_used, session_id, verdict, task_id), + f"verdict = COALESCE(?, verdict), note = NULL WHERE {where}", + values, ) - _drop_cancel_flag(conn, task_id) + if cur.rowcount: + _drop_cancel_flag(conn, task_id) + _release_task_pipeline_targets(conn, task_id, attempt) + return cur.rowcount > 0 -def mark_failed(task_id: int, error: str, exit_code: int = 1): +def mark_failed(task_id: int, error: str, exit_code: int = 1, + *, expected_started_at=None) -> bool: with _connect() as conn: - conn.execute( + attempt = _attempt_iso(expected_started_at) + where = "id = ?" + values = [error, exit_code, _now(), task_id] + if attempt is not None: + where += " AND status = 'running' AND started_at = ?" + values.append(attempt) + cur = conn.execute( "UPDATE tasks SET status = 'failed', error = ?, exit_code = ?, " - "completed_at = ?, note = NULL WHERE id = ?", - (error, exit_code, _now(), task_id), + f"completed_at = ?, note = NULL WHERE {where}", + values, ) - _drop_cancel_flag(conn, task_id) + if cur.rowcount: + _drop_cancel_flag(conn, task_id) + _release_task_pipeline_targets(conn, task_id, attempt) + return cur.rowcount > 0 def fail_running_attempt(task_id: int, started_at, error: str, @@ -801,25 +1318,37 @@ def fail_running_attempt(task_id: int, started_at, error: str, ) if cur.rowcount: _drop_cancel_flag(conn, task_id) + _release_task_pipeline_targets(conn, task_id, started_at) return cur.rowcount > 0 -def mark_rate_limited(task_id: int, next_run_at: datetime, error: str = None): +def mark_rate_limited(task_id: int, next_run_at: datetime, error: str = None, + *, expected_started_at=None) -> bool: with _connect() as conn: - conn.execute( + attempt = _attempt_iso(expected_started_at) + where = "id = ?" + values = [_to_utc_iso(next_run_at), error, task_id] + if attempt is not None: + where += " AND status = 'running' AND started_at = ?" + values.append(attempt) + cur = conn.execute( """UPDATE tasks SET status = 'rate_limited', next_run_at = ?, retry_count = retry_count + 1, error = COALESCE(?, error) - WHERE id = ?""", - (_to_utc_iso(next_run_at), error, task_id), + WHERE """ + where, + values, ) - _drop_cancel_flag(conn, task_id) + if cur.rowcount: + _drop_cancel_flag(conn, task_id) + _release_task_pipeline_targets(conn, task_id, attempt) + return cur.rowcount > 0 def defer_task(task_id: int, next_run_at: datetime, reason: str = None, - *, hard_not_before: bool = False): + *, hard_not_before: bool = False, + expected_started_at=None) -> bool: """Return a claimed task to pending without consuming a retry attempt. Used by deterministic pipeline dependency gates. This is waiting, not a @@ -830,14 +1359,23 @@ def defer_task(task_id: int, next_run_at: datetime, reason: str = None, """ deadline = _to_utc_iso(next_run_at) with _connect() as conn: - conn.execute( + attempt = _attempt_iso(expected_started_at) + where = "id = ? AND status = 'running'" + values = [deadline, deadline if hard_not_before else None, reason, task_id] + if attempt is not None: + where += " AND started_at = ?" + values.append(attempt) + cur = conn.execute( """UPDATE tasks SET status = 'pending', scheduled_at = ?, next_run_at = ?, started_at = NULL, error = COALESCE(?, error) - WHERE id = ? AND status = 'running'""", - (deadline, deadline if hard_not_before else None, reason, task_id), + WHERE """ + where, + values, ) - _drop_cancel_flag(conn, task_id) + if cur.rowcount: + _drop_cancel_flag(conn, task_id) + _release_task_pipeline_targets(conn, task_id, attempt) + return cur.rowcount > 0 def request_cancel(task_id: int) -> bool: @@ -864,14 +1402,24 @@ def clear_cancel_request(task_id: int): conn.execute("DELETE FROM settings WHERE key = ?", (f"cancel_task:{task_id}",)) -def mark_cancelled(task_id: int, note: str = None): - clear_note(task_id) +def mark_cancelled(task_id: int, note: str = None, + *, expected_started_at=None) -> bool: with _connect() as conn: - conn.execute( - "UPDATE tasks SET status = 'cancelled', completed_at = ?, error = COALESCE(?, error) WHERE id = ?", - (_now(), note, task_id), + attempt = _attempt_iso(expected_started_at) + where = "id = ?" + values = [_now(), note, task_id] + if attempt is not None: + where += " AND status = 'running' AND started_at = ?" + values.append(attempt) + cur = conn.execute( + "UPDATE tasks SET status = 'cancelled', completed_at = ?, " + f"error = COALESCE(?, error), note = NULL WHERE {where}", + values, ) - _drop_cancel_flag(conn, task_id) + if cur.rowcount: + _drop_cancel_flag(conn, task_id) + _release_task_pipeline_targets(conn, task_id, attempt) + return cur.rowcount > 0 def cancel_task(task_id: int) -> bool: @@ -880,6 +1428,8 @@ def cancel_task(task_id: int) -> bool: "UPDATE tasks SET status = 'cancelled', completed_at = ? WHERE id = ? AND status IN ('pending', 'rate_limited')", (_now(), task_id), ) + if cur.rowcount: + _release_task_pipeline_targets(conn, task_id) return cur.rowcount > 0 @@ -947,13 +1497,16 @@ def list_series() -> list: GROUP BY series_id ) SELECT 'active' AS kind, t.series_id, t.id, t.machine, - t.scheduled_at, t.error + t.scheduled_at, t.error, t.worktree, t.detached, + t.herdr_target FROM tasks AS t INNER JOIN picked ON picked.active_id = t.id INNER JOIN task_series AS s ON s.id = t.series_id UNION ALL SELECT 'last' AS kind, t.series_id, t.id, t.machine, - NULL AS scheduled_at, NULL AS error + NULL AS scheduled_at, NULL AS error, + NULL AS worktree, NULL AS detached, + NULL AS herdr_target FROM tasks AS t INNER JOIN picked ON picked.last_id = t.id INNER JOIN task_series AS s ON s.id = t.series_id"""): @@ -989,6 +1542,9 @@ def list_series() -> list: "priority": s["priority"], "task_timeout": s["task_timeout"], "machine": (active_detail["machine"] if active_detail else (last_detail["machine"] if last_detail else None)), + "worktree": bool(active_detail["worktree"]) if active_detail else False, + "detached": bool(active_detail["detached"]) if active_detail else False, + "herdr_target": active_detail["herdr_target"] if active_detail else None, "paused": bool(s["paused"]), "ended": bool(s["ended_at"]), "next_task_id": active["id"] if active else None, "next_status": active["status"] if active else None, @@ -1587,6 +2143,91 @@ def wake_series_once(series_id: Optional[int], latch_key: str, return True +def wake_series_group_once(series_ids: list[int], latch_key: str, + fingerprint: Optional[str], *, cache_guard: dict) -> list[int]: + """Atomically apply one diagnostic wake to every configured replica.""" + normalized = [] + for value in series_ids: + if type(value) is int and value > 0 and value not in normalized: + normalized.append(value) + with _connect(immediate=True) as conn: + paused = conn.execute( + "SELECT value FROM settings WHERE key = 'worker_paused'" + ).fetchone() + if paused and paused["value"] == "1": + return [] + if not _pipeline_cache_guard_matches(conn, cache_guard): + return [] + if fingerprint is None: + conn.execute("DELETE FROM settings WHERE key = ?", (latch_key,)) + return [] + previous = conn.execute( + "SELECT value FROM settings WHERE key = ?", (latch_key,), + ).fetchone() + if previous and previous["value"] == fingerprint: + return [] + now = _now() + woken = [] + for series_id in normalized: + series = conn.execute( + "SELECT * FROM task_series WHERE id = ?", + (series_id,), + ).fetchone() + if not series or series["paused"] or series["ended_at"]: + continue + task = conn.execute( + """UPDATE tasks SET scheduled_at = ?, next_run_at = NULL + WHERE series_id = ? AND status IN ('pending', 'rate_limited') + AND (next_run_at IS NULL OR next_run_at <= ?)""", + (now, series_id, now), + ) + if not task.rowcount: + deferred = conn.execute( + """SELECT 1 FROM tasks + WHERE series_id = ? + AND status IN ('pending', 'rate_limited') + AND next_run_at > ?""", + (series_id, now), + ).fetchone() + running = conn.execute( + """SELECT 1 FROM tasks + WHERE series_id = ? AND status = 'running'""", + (series_id,), + ).fetchone() + if deferred is None and running is None: + # A recurring worker commits its terminal result before + # ``_recur_after_run`` creates the successor. Do not let + # the shared diagnostic latch make this replica miss the + # wake permanently when the group wake lands in that + # narrow gap. Cancellation is an intentional human stop, + # so only repair the same terminal states as + # ``repair_active_series_occurrences``. + latest = conn.execute( + """SELECT status FROM tasks WHERE series_id = ? + ORDER BY id DESC LIMIT 1""", + (series_id,), + ).fetchone() + if (latest is None + or latest["status"] not in ("completed", "failed")): + continue + if not _recreate_series_occurrence( + conn, series_id, series, datetime.now(timezone.utc)): + continue + woken.append(series_id) + continue + conn.execute( + "INSERT OR REPLACE INTO settings (key, value) VALUES (?, '1')", + (_pipeline_series_wake_intent_key(series_id),), + ) + woken.append(series_id) + if woken: + conn.execute( + "INSERT OR REPLACE INTO settings (key, value) VALUES (?, ?)", + (latch_key, fingerprint), + ) + return woken + + def _pipeline_stale_reselect_guard_key(series_id: int) -> str: return f"pipeline_stale_reselect_guard:v1:{int(series_id)}" @@ -1669,7 +2310,16 @@ def update_task_fields(task_id: int, fields: dict) -> bool: def delete_task(task_id: int) -> bool: with _connect() as conn: - cur = conn.execute("DELETE FROM tasks WHERE id = ?", (task_id,)) + # A running provider may still own a checkout and an exact pipeline + # target. Deletion must follow cancel + confirmed terminal cleanup; + # otherwise removing the row would also free the reservation while the + # old agent can still mutate that PR. + cur = conn.execute( + "DELETE FROM tasks WHERE id = ? AND status != 'running'", + (task_id,), + ) + if cur.rowcount: + _release_task_pipeline_targets(conn, task_id) return cur.rowcount > 0 @@ -2982,16 +3632,68 @@ def recover_running(keep_ids=()): conn.execute(sql, keep) +def recover_running_attempt(task_id: int, started_at) -> bool: + """Settle one exact orphan after its provider was proven stopped. + + A cancellation requested before the crash remains cancellation; every + other orphan is requeued. The status change, flag consumption and exact + reservation release share one transaction. + """ + attempt = _attempt_iso(started_at) + if attempt is None: + return False + with _connect(immediate=True) as conn: + cancelled = conn.execute( + "SELECT 1 FROM settings WHERE key = ?", + (f"cancel_task:{task_id}",), + ).fetchone() is not None + if cancelled: + cur = conn.execute( + """UPDATE tasks + SET status = 'cancelled', completed_at = ?, + error = COALESCE(error, ?), note = NULL + WHERE id = ? AND status = 'running' AND started_at = ?""", + (_now(), "Отменена пользователем до перезапуска worker", + task_id, attempt), + ) + else: + cur = conn.execute( + """UPDATE tasks SET status = 'pending', started_at = NULL + WHERE id = ? AND status = 'running' AND started_at = ?""", + (task_id, attempt), + ) + if cur.rowcount: + _drop_cancel_flag(conn, task_id) + _release_task_pipeline_targets(conn, task_id, attempt) + return cur.rowcount > 0 + + def reset_task(task_id: int) -> bool: """Reset a single stuck 'running' task back to 'pending'.""" with _connect() as conn: cur = conn.execute( - "UPDATE tasks SET status = 'pending', started_at = NULL WHERE id = ? AND status = 'running'", + """UPDATE tasks SET status = 'pending', started_at = NULL + WHERE id = ? AND status = 'running' + AND NOT EXISTS ( + SELECT 1 FROM pipeline_target_reservations AS r + WHERE r.task_id = tasks.id + )""", (task_id,), ) return cur.rowcount > 0 +def task_has_live_pipeline_target_reservation(task_id: int) -> bool: + """Whether reset must wait for an owned provider to stop safely.""" + with _connect() as conn: + row = conn.execute( + """SELECT 1 FROM pipeline_target_reservations + WHERE task_id = ? LIMIT 1""", + (task_id,), + ).fetchone() + return row is not None + + def purge_old(before_days: int = 7) -> int: with _connect() as conn: cutoff = datetime.now(timezone.utc) diff --git a/promptpilot/fallback_handoff.py b/promptpilot/fallback_handoff.py index 0f81871..b9c9d0b 100644 --- a/promptpilot/fallback_handoff.py +++ b/promptpilot/fallback_handoff.py @@ -198,20 +198,36 @@ def config_digest(config: dict) -> str: return digest(config) -def create(config: dict, health: dict, stage: str, target: dict, reason: str) -> dict: - from .project_pipeline import encode_signed_lease +def create(config: dict, health: dict, stage: str, target: dict, reason: str, + *, target_reservation: dict | None = None) -> dict: + from . import project_pipeline as pp health_gate(health, stage, target, election=True) target = identity(target) issued = int(time.time()) + replicas = pp._configured_replica_count() + if replicas > 1 and target_reservation is None: + raise pp.PipelineError( + "replicated REVIEW fallback has no target reservation") lease = {"version": 1, "purpose": PROTOCOL, "stage": stage, "repository": config["repository"], "target": target, "config_sha256": config_digest(config), "issued_at": issued, "expires_at": issued + TTL} + if target_reservation is not None: + repository, target_stage, number, head, task_id, _token = \ + pp._reservation_identity(target_reservation) + if (repository != str(config["repository"]).lower() + or target_stage != target["stage"] or number != target["number"] + or head != target["head"] + or pp._pipeline_task_id(required=True) != task_id): + raise pp.PipelineError( + "pipeline target reservation contradicts fallback target") + lease["target_reservation"] = target_reservation + lease["pipeline_replicas"] = replicas return {"action": "fallback", "reason": reason, "target": target, "handoff": {"protocol": PROTOCOL, "stage": stage, "repository": config["repository"], "target": target, - "lease": encode_signed_lease(lease)}} + "lease": pp.encode_signed_lease(lease)}} def validate(preflight: dict, stage: str) -> dict: @@ -233,6 +249,7 @@ def validate(preflight: dict, stage: str) -> dict: def validate_lease(lease: dict, stage: str) -> None: + from . import project_pipeline as pp from .project_pipeline import PipelineError if (lease.get("purpose") != PROTOCOL or lease.get("stage") != stage @@ -250,6 +267,21 @@ def validate_lease(lease: dict, stage: str) -> None: if (type(issued) is not int or type(expires) is not int or issued > now + 60 or expires <= now or not 0 < expires - issued <= TTL): raise PipelineError("fallback lease is expired or has an invalid validity window") + reservation = lease.get("target_reservation") + replicas = lease.get("pipeline_replicas", 1) + if type(replicas) is not int or not 1 <= replicas <= 16: + raise PipelineError("fallback lease has an invalid replica count") + if replicas > 1 and reservation is None: + raise PipelineError( + "replicated REVIEW fallback lease has no target reservation") + if reservation is not None: + repository, target_stage, number, head, _task_id, _token = \ + pp._reservation_identity(reservation) + if (repository != str(lease["repository"]).lower() + or target_stage != target["stage"] or number != target["number"] + or head != target["head"]): + raise PipelineError( + "pipeline target reservation contradicts fallback lease") def gate(gh, config: dict, stage: str, lease_value: str, *, config_path=None) -> dict: @@ -274,6 +306,16 @@ def check_config(): raise pp.PipelineError("pending merge cleanup takes precedence; start a new task") health_gate(health, stage, lease["target"], election=False) validate_lease(lease, stage) + if lease.get("target_reservation") is not None: + pp.renew_lease_target_reservation({ + "stage": stage, + "target_stage": lease["target"]["stage"], + "repository": lease["repository"], + "number": lease["target"]["number"], + "head": lease["target"]["head"], + "target_reservation": lease["target_reservation"], + "pipeline_replicas": lease.get("pipeline_replicas", 1), + }, config) return {"action": "validated", "stage": stage, "repository": lease["repository"], "target": lease["target"], "mutation_authorized": False, "reason": "fresh scheduling gate passed; full repository mutation gates still required"} diff --git a/promptpilot/herdr_exec.py b/promptpilot/herdr_exec.py index 9734331..f85fa54 100644 --- a/promptpilot/herdr_exec.py +++ b/promptpilot/herdr_exec.py @@ -172,23 +172,69 @@ def _run(args, host=None, timeout=None): return proc.returncode, data, raw -def _close_owned_session(name: str, close_args: list, host=None) -> str: +def _owned_container_probe(close_args: list) -> tuple[list[str], str, str]: + """Return list command, result collection, and immutable id to verify.""" + if close_args[:2] == ["tab", "close"] and len(close_args) >= 3: + return ["tab", "list"], "tabs", str(close_args[2]) + if close_args[:2] == ["workspace", "close"] and len(close_args) >= 3: + return ["workspace", "list"], "workspaces", str(close_args[2]) + if close_args[:2] == ["worktree", "remove"]: + try: + index = close_args.index("--workspace") + workspace_id = str(close_args[index + 1]) + except (ValueError, IndexError): + return [], "", "" + return ["workspace", "list"], "workspaces", workspace_id + return [], "", "" + + +def _exact_herdr_ids_absent(close_args: list, pane_id: str, host=None) -> tuple[bool, str]: + """Prove immutable container and pane IDs disappeared after cleanup.""" + list_args, collection, container_id = _owned_container_probe(close_args) + if not list_args or not container_id: + return False, "invalid owned-session close descriptor" + rc, data, raw = _run(list_args, host=host, timeout=10) + values = _dig(data, "result", collection, default=None) + if rc != 0 or not isinstance(values, list): + return False, raw or f"herdr {list_args[0]} list failed" + id_field = "tab_id" if collection == "tabs" else "workspace_id" + if any(not isinstance(value, dict) + or not isinstance(value.get(id_field), str) + or not value.get(id_field) for value in values): + return False, f"herdr {collection} list is malformed" + if any(str(value.get(id_field) or "") == container_id for value in values): + return False, f"owned {id_field} {container_id} is still present" + if pane_id: + rc, data, raw = _run(["agent", "list"], host=host, timeout=10) + agents = _dig(data, "result", "agents", default=None) + if rc != 0 or not isinstance(agents, list): + return False, raw or "herdr agent list failed" + if any(not isinstance(agent, dict) + or not isinstance(agent.get("pane_id"), str) + or not agent.get("pane_id") for agent in agents): + return False, "herdr agent list is malformed" + if any(str(agent.get("pane_id") or "") == pane_id for agent in agents): + return False, f"owned pane_id {pane_id} is still present" + return True, "" + + +def _close_owned_session(name: str, close_args: list, host=None, *, + pane_id: str = "") -> str: """Close and verify a PromptPilot-owned herdr session. A failed best-effort close must not be reported as a successful task - cancellation: the agent could still be changing its checkout. Repeating - the exact tab/workspace command is safe, and never targets the shared - server or a user-provided ``herdr_target``. + cancellation: the agent could still be changing its checkout. Repeating + the exact tab/workspace command is safe. Success is proved by immutable + container/pane IDs, never by the mutable agent/tab/workspace label. """ last_raw = "" for _ in range(3): # Let _run choose the operation-aware bound: worktree removal gets the # longer configured allowance, while tab/workspace close stays short. _, _, close_raw = _run(close_args, host=host) - rc, data, probe_raw = _run(["agent", "get", name], host=host, timeout=10) - # Only herdr's explicit not-found response proves absence. A successful - # but incomplete/unknown schema is not evidence that the agent stopped. - if rc != 0 and _error_code(data) == "agent_not_found": + absent, probe_raw = _exact_herdr_ids_absent( + close_args, pane_id, host) + if absent: return "" last_raw = probe_raw or close_raw time.sleep(0.2) @@ -563,7 +609,8 @@ def _wait_settled(name, until_args, deadline, cancel_check, host=None): def run_in_herdr(task, provider_cfg: dict, on_blocked=None, timeout: int = None, cancel_check=None, keep_pane: bool = None, host: str = None, - on_worktree=None, on_pane=None, on_started=None, + on_worktree=None, on_pane=None, on_session=None, + on_started=None, prompt_override: str = None, require_closing_verdict: bool = False, allow_targeted_stale: bool = False) -> dict: @@ -572,6 +619,8 @@ def run_in_herdr(task, provider_cfg: dict, on_blocked=None, timeout: int = None, on_blocked(pane_id) is called once when the agent first enters ``blocked``. on_pane(pane_id) is called as soon as the pane is known — the bot's task card wants a «📺 Экран» button while the run is still going. + on_session(pane_id, tab_id, workspace_id) durably binds immutable owned + session IDs before its agent is allowed to start. on_started(pane_id) is called only after an existing target was verified or a newly owned provider was successfully started. on_worktree(path, branch) is called as soon as a ``worktree`` task has its @@ -593,6 +642,8 @@ def run_in_herdr(task, provider_cfg: dict, on_blocked=None, timeout: int = None, "output": "", "error": "", "pane_id": "", "env_failure": "", "verdict": "", "worktree_path": "", "worktree_branch": ""} + strict_ownership = bool(provider_cfg.get("strict_owned_session")) + strict_cleanup = None attach = attach_hint(host) from .worker import effective_prompt prompt = prompt_override if prompt_override is not None else effective_prompt(task) @@ -602,6 +653,10 @@ def run_in_herdr(task, provider_cfg: dict, on_blocked=None, timeout: int = None, if require_closing_verdict: prompt = ensure_closing_verdict_contract( prompt, allow_targeted_stale=allow_targeted_stale) + if strict_ownership and getattr(task, "herdr_target", None): + outcome["error"] = ( + "strict pipeline ownership is incompatible with herdr_target") + return outcome if requires_verdict and getattr(task, "detached", False): outcome["error"] = ( "herdr detached mode is incompatible with a required closing " @@ -643,14 +698,33 @@ def run_in_herdr(task, provider_cfg: dict, on_blocked=None, timeout: int = None, ) return outcome else: - _close_stale_tabs(task.id, host) + def close_stale_owned_tabs(): + try: + _close_stale_tabs(task.id, host) + except Exception as exc: + return ( + "could not confirm stale PromptPilot herdr sessions " + f"were closed: {type(exc).__name__}: {exc}" + ) + return "" + + stale_close_error = close_stale_owned_tabs() + if stale_close_error: + outcome["error"] = stale_close_error + if strict_ownership: + # A stale pp-t-* session may still be running the same + # task from a prior attempt. Keep the target reservation + # fenced until a later worker recovery confirms cleanup. + outcome["ownership_uncertain"] = True + outcome["_ownership_cleanup"] = close_stale_owned_tabs + return outcome name = f"pp-t{task.id}-{int(time.time()) % 100000}" # Remote: the working dir must exist on THAT machine; without one # herdr falls back to its own default (the user's home there). cwd = task.working_dir or (None if host else ".") # PP_TASK_ID marks the run in the pane's environment and is # inherited by the agent, so a live run can be found by process. - env_args = ["--env", f"PP_TASK_ID={task.id}"] + tab_env = {"PP_TASK_ID": str(task.id)} # A local herdr pane is created by the long-lived server rather # than as our child process, so it does not inherit PromptPilot's # runtime paths. Forward only the non-secret paths required by the @@ -661,10 +735,15 @@ def run_in_herdr(task, provider_cfg: dict, on_blocked=None, timeout: int = None, "PP_DATA_DIR", "PP_PIPELINE_LEASE_KEY_FILE", "PP_GH_EXE", "PP_GO_EXE"): if os.environ.get(key): - env_args += ["--env", f"{key}={os.environ[key]}"] + tab_env[key] = os.environ[key] for k, v in (provider_cfg.get("env") or {}).items(): if v: - env_args += ["--env", f"{k}={v}"] + tab_env[k] = str(v) + # Scheduler identity is reserved per claimed attempt. A stale + # provider-level setting must never retag this owned session. + tab_env["PP_TASK_ID"] = str(task.id) + env_args = [part for key, value in tab_env.items() + for part in ("--env", f"{key}={value}")] if getattr(task, "worktree", False): wt = _open_worktree(task, cwd, name, host) @@ -696,6 +775,21 @@ def run_in_herdr(task, provider_cfg: dict, on_blocked=None, timeout: int = None, outcome["error"] = f"herdr tab create failed: {raw}" return outcome outcome["pane_id"] = pane_id + if on_session: + try: + on_session(pane_id, tab_id, workspace_id) + except Exception as exc: + close_args = (["workspace", "close", workspace_id] + if workspace_id else ["tab", "close", tab_id]) + close_error = _close_owned_session( + name, close_args, host, pane_id=pane_id) + message = ( + "herdr session bookkeeping failed before agent start: " + f"{type(exc).__name__}: {exc}" + ) + outcome["error"] = ( + f"{message}\n{close_error}" if close_error else message) + return outcome if on_pane: try: on_pane(pane_id) @@ -703,7 +797,7 @@ def run_in_herdr(task, provider_cfg: dict, on_blocked=None, timeout: int = None, close_args = (["workspace", "close", workspace_id] if workspace_id else ["tab", "close", tab_id]) close_error = _close_owned_session( - name, close_args, host) + name, close_args, host, pane_id=pane_id) message = ( "herdr pane bookkeeping failed before agent start: " f"{type(exc).__name__}: {exc}" @@ -721,12 +815,18 @@ def close_tab(remove_untouched=True): close_args = ["worktree", "remove", "--workspace", workspace_id] else: close_args = ["workspace", "close", workspace_id] - return _close_owned_session(name, close_args, host) + return _close_owned_session( + name, close_args, host, pane_id=pane_id) if tab_id: - return _close_owned_session(name, ["tab", "close", tab_id], host) + return _close_owned_session( + name, ["tab", "close", tab_id], host, pane_id=pane_id) return "" # herdr_target: foreign pane, intentionally untouched. + strict_cleanup = lambda: close_tab(remove_untouched=False) + def fail(msg, keep_pane=True): + if strict_ownership: + keep_pane = False if keep_pane: outcome["error"] = f"{msg}\n(панель {pane_id} оставлена — подключись командой: {attach})" else: @@ -854,20 +954,26 @@ def fail(msg, keep_pane=True): ) if state == "__cancel__": + cancel_intent = { + "cancelled": True, + "error": "", + "cancel_note": "Отменена пользователем во время выполнения", + } if target: - outcome["cancel_note"] = ( + cancel_intent["cancel_note"] = ( "Отменено ожидание PromptPilot; пользовательская herdr-сессия " f"{name} не закрывалась и агент в ней может продолжать работу" ) else: close_error = close_tab(remove_untouched=False) if close_error: + outcome["_terminal_intent"] = cancel_intent outcome["error"] = ( "herdr: cancellation requested, but the owned session " f"could not be stopped safely\n{close_error}" ) return outcome - outcome["cancelled"] = True + outcome.update(cancel_intent) return outcome if state == "__timeout__": message = f"herdr: таймаут задачи ({timeout}s)" @@ -888,23 +994,27 @@ def fail(msg, keep_pane=True): reason = _retry_reason(cleaned) if reason: + retry_intent = { + "rate_limited": True, "retry_reason": reason, + "error": cleaned, + } close_error = close_tab(remove_untouched=False) if close_error: + outcome["_terminal_intent"] = retry_intent outcome["error"] = f"{cleaned}\n{close_error}" return outcome - outcome["rate_limited"] = True - outcome["retry_reason"] = reason - outcome["error"] = cleaned + outcome.update(retry_intent) return outcome env_marker = _looks_env_failure(cleaned) if env_marker: + env_intent = {"env_failure": env_marker, "error": cleaned} close_error = close_tab(remove_untouched=False) if close_error: + outcome["_terminal_intent"] = env_intent outcome["error"] = f"{cleaned}\n{close_error}" return outcome - outcome["env_failure"] = env_marker - outcome["error"] = cleaned + outcome.update(env_intent) return outcome closing_verdict = _closing_workflow_verdict( @@ -934,7 +1044,8 @@ def fail(msg, keep_pane=True): if changes: wt_meta += f"\nИзменения: {changes}" - keep = bool(keep_pane) or provider_cfg.get("keep_pane") or HERDR_KEEP_PANE + keep = (not strict_ownership and ( + bool(keep_pane) or provider_cfg.get("keep_pane") or HERDR_KEEP_PANE)) if keep: # Keep the live session for follow-up work. Rename the agent out of # the pp-t* namespace so the Telegram bridge watches the continued @@ -951,14 +1062,18 @@ def fail(msg, keep_pane=True): ) return outcome + success_intent = { + "ok": True, "error": "", + "output": (f"{cleaned}\n\n--- Meta ---\n" + f"Executor: herdr (pane {pane_id}{where}){wt_meta}"), + } close_error = close_tab() if close_error: + outcome["_terminal_intent"] = success_intent outcome["error"] = ("herdr task finished, but its owned session could not " f"be stopped safely\n{close_error}") return outcome - outcome["ok"] = True - outcome["output"] = (f"{cleaned}\n\n--- Meta ---\n" - f"Executor: herdr (pane {pane_id}{where}){wt_meta}") + outcome.update(success_intent) return outcome except HerdrError as e: @@ -967,3 +1082,23 @@ def fail(msg, keep_pane=True): except Exception as e: # never crash the worker loop over a herdr hiccup outcome["error"] = f"herdr executor error: {type(e).__name__}: {e}" return outcome + finally: + if strict_ownership and not outcome["ok"] and callable(strict_cleanup): + try: + close_error = strict_cleanup() + except Exception as exc: + close_error = ( + "strict herdr cleanup failed: " + f"{type(exc).__name__}: {exc}" + ) + if close_error: + outcome["ownership_uncertain"] = True + outcome["_ownership_cleanup"] = strict_cleanup + if close_error not in outcome["error"]: + outcome["error"] = ( + f"{outcome['error']}\n{close_error}" + if outcome["error"] else close_error) + else: + terminal_intent = outcome.pop("_terminal_intent", None) + if isinstance(terminal_intent, dict): + outcome.update(terminal_intent) diff --git a/promptpilot/pipeline_insights.py b/promptpilot/pipeline_insights.py index 2919ff6..eb92b4d 100644 --- a/promptpilot/pipeline_insights.py +++ b/promptpilot/pipeline_insights.py @@ -19,7 +19,8 @@ from urllib.parse import quote from . import db -from .config import DB_DIR, POLL_INTERVAL, TASK_TIMEOUT +from .config import (CONCURRENCY, DB_DIR, DEFAULT_CLI, POLL_INTERVAL, + TASK_TIMEOUT, load_providers) DEFAULT_PROFILES: dict = {} @@ -1548,20 +1549,21 @@ def set_item_priority(profile_id: str, queue_id: str, kind: str, number: int, woke = False paused = False + woken_series = [] if run_now: - marker = str(queue.get("series_contains") or "").lower() - target = next((entry for entry in series - if marker and marker in str(entry.get("title", "")).lower() - and not entry.get("ended")), None) - if target: - woke = db.series_action(int(target["id"]), "run_now") - paused = bool(target.get("paused")) + targets = _series_replicas_for_queue(queue, series) + for target in targets: + if db.series_action(int(target["id"]), "run_now"): + woken_series.append(int(target["id"])) + woke = bool(woken_series) + paused = any(bool(target.get("paused")) for target in targets) _discard_cache(profile_id) return { "ok": True, "profile_id": profile_id, "queue_id": queue_id, "kind": kind, "number": number, "level": level, "label": selected, "run_now": run_now, "series_woken": woke, - "series_paused": paused, + "series_woken_count": len(woken_series), + "series_woken_ids": woken_series, "series_paused": paused, } @@ -1625,60 +1627,77 @@ def _adaptive_cadence_policy(queue: dict) -> dict | None: } -def _adaptive_cadence_status(queue: dict, matching: dict | None, +def _matching_series_list(matching) -> list[dict]: + if isinstance(matching, dict): + return [matching] + if isinstance(matching, list): + return [item for item in matching if isinstance(item, dict)] + return [] + + +def _adaptive_cadence_status(queue: dict, matching, backlog: int | None) -> dict | None: policy = _adaptive_cadence_policy(queue) if policy is None: return None + matches = _matching_series_list(matching) busy = bool( policy["busy_recurrence"] is not None and isinstance(backlog, int) and backlog > policy["backlog_above"] ) - temporary = matching.get("temporary_recurrence") if matching else None - empty_count = int(matching.get("temporary_empty_count") or 0) if matching else 0 + temporary = [item.get("temporary_recurrence") for item in matches] + empty_counts = [int(item.get("temporary_empty_count") or 0) + for item in matches] draining = bool( not busy and policy["empty_runs_before_idle"] - and temporary == policy["busy_recurrence"] + and any(value == policy["busy_recurrence"] for value in temporary) ) mode = "busy" if busy else "draining" if draining else ( "event" if policy["event_wake"] else "idle") + recurrences = list(dict.fromkeys( + item.get("effective_recurrence") for item in matches + if item.get("effective_recurrence"))) return { **policy, "mode": mode, "effective_recurrence": ( - matching.get("effective_recurrence") if matching else None), - "empty_runs": empty_count, - "series_present": matching is not None, + recurrences[0] if len(recurrences) == 1 else + ", ".join(recurrences) if recurrences else None), + "empty_runs": min(empty_counts) if empty_counts else 0, + "series_present": bool(matches), + "series_count": len(matches), } def _reconcile_adaptive_cadence( - queue: dict, matching: dict | None, backlog: int | None, + queue: dict, matching, backlog: int | None, publication_guard: dict | None = None) -> dict | None: """Apply cadence only during a successful live queue observation.""" policy = _adaptive_cadence_policy(queue) - if policy is None or matching is None or not isinstance(backlog, int): + matches = _matching_series_list(matching) + if policy is None or not matches or not isinstance(backlog, int): return _adaptive_cadence_status(queue, matching, backlog) boost = bool( policy["busy_recurrence"] is not None and backlog > policy["backlog_above"]) - result = db.apply_pipeline_series_cadence( - int(matching["id"]), - idle_recurrence=policy["idle_recurrence"], - busy_recurrence=policy["busy_recurrence"], - boost=boost, - empty_runs_before_idle=policy["empty_runs_before_idle"], - publication_guard=publication_guard, - ) - if result is not None: - matching.update({ - "recurrence": result["base_recurrence"], - "effective_recurrence": result["effective_recurrence"], - "temporary_recurrence": result["temporary_recurrence"], - "temporary_empty_limit": result["temporary_empty_limit"], - "temporary_empty_count": result["temporary_empty_count"], - }) + for item in matches: + result = db.apply_pipeline_series_cadence( + int(item["id"]), + idle_recurrence=policy["idle_recurrence"], + busy_recurrence=policy["busy_recurrence"], + boost=boost, + empty_runs_before_idle=policy["empty_runs_before_idle"], + publication_guard=publication_guard, + ) + if result is not None: + item.update({ + "recurrence": result["base_recurrence"], + "effective_recurrence": result["effective_recurrence"], + "temporary_recurrence": result["temporary_recurrence"], + "temporary_empty_limit": result["temporary_empty_limit"], + "temporary_empty_count": result["temporary_empty_count"], + }) return _adaptive_cadence_status(queue, matching, backlog) @@ -1980,6 +1999,19 @@ def worker_lane_policy() -> dict | None: "queues": list(queues), "borrow": borrow, "order": len(lanes), }) + for queue in profile.get("queues", []): + replicas = _queue_replica_count(queue) + if replicas <= 1: + continue + queue_id = str(queue.get("id")) + eligible = sum( + lane["profile_id"] == profile_id and queue_id in lane["queues"] + for lane in lanes + ) + if eligible < replicas: + raise ValueError( + f"очередь {profile_id}/{queue_id} настроена на {replicas} " + f"реплики, но доступна только в {eligible} scheduler lanes") return {"profiles": profiles, "lanes": lanes} if lanes else None @@ -2130,16 +2162,27 @@ def _wake_ready_queues(profile_id: str, profile: dict, data: dict, continue latch_key = _wake_latch_key(profile_id, str(queue.get("id"))) fingerprint = _wake_fingerprint(diagnostics, condition) + targets = [item for item in _series_replicas_for_queue(queue, series) + if not item.get("paused")] + replicated = _queue_replica_count(queue) > 1 if fingerprint is None: - db.wake_series_once( - None, latch_key, None, cache_guard=cache_guard) + if replicated: + db.wake_series_group_once( + [], latch_key, None, cache_guard=cache_guard) + else: + db.wake_series_once( + None, latch_key, None, cache_guard=cache_guard) continue - target = next((item for item in series - if marker in str(item.get("title", "")).lower() - and not item.get("ended") and not item.get("paused")), None) - if target and db.wake_series_once( + if replicated: + woke = db.wake_series_group_once( + [int(item["id"]) for item in targets], latch_key, fingerprint, + cache_guard=cache_guard) + else: + target = targets[0] if targets else None + woke = bool(target and db.wake_series_once( int(target["id"]), latch_key, fingerprint, - cache_guard=cache_guard): + cache_guard=cache_guard)) + if woke: woken.append(str(queue.get("id"))) return woken @@ -2172,12 +2215,13 @@ def _wake_configured_successors(profile: dict, queue: dict, raise ValueError( "wake_after_success может будить только очередь с project " f"execution preflight: {queue_id}") - target = _series_for_queue(target_queue, series) - if (not target or target.get("ended") or target.get("ended_at") - or target.get("paused")): - continue - wake = db.request_pipeline_series_wake(int(target["id"])) - if wake.get("accepted"): + targets = [item for item in _series_replicas_for_queue(target_queue, series) + if not item.get("paused")] + accepted = False + for target in targets: + wake = db.request_pipeline_series_wake(int(target["id"])) + accepted = accepted or bool(wake.get("accepted")) + if accepted: woken.append(queue_id) return woken @@ -2283,14 +2327,26 @@ def _tool_available(execution: dict, command: list[str], working_dir: str | None return True, "инструмент доступен" -def _tool_preflight(execution: dict, command: list[str], working_dir: str | None) -> dict: +def _tool_preflight(execution: dict, command: list[str], working_dir: str | None, + *, env_extra: dict[str, str] | None = None) -> dict: root = Path(working_dir or os.getcwd()) + environment = os.environ.copy() + # Replica identity is per claimed task, never process-global operator + # configuration. A stale shell variable must not opt a legacy queue into + # replicated election without the scheduler's reservation contract. + environment.pop("PP_TASK_ID", None) + environment.pop("PP_TASK_STARTED_AT", None) + environment.pop("PP_PIPELINE_REPLICAS", None) + environment.pop("PP_PROVIDER_OWNERSHIP_KIND", None) + environment.pop("PP_PIPELINE_TARGET_TOKEN", None) + if env_extra: + environment.update(env_extra) try: _ensure_github_scan_lease(renew=True) result = subprocess.run( command, cwd=str(root), capture_output=True, text=True, timeout=max(1, min(int(execution.get("timeout_seconds", 180)), 900)), - encoding="utf-8", errors="replace", + encoding="utf-8", errors="replace", env=environment, ) except (OSError, subprocess.TimeoutExpired) as exc: raise RuntimeError(f"pipeline preflight не выполнен: {exc}") from exc @@ -2408,8 +2464,62 @@ def execution_route(task, fallback_prompt: str, working_dir: str | None = None, if matched is None: return {"action": "prompt", "mode": "skill", "prompt": fallback_prompt} profile_id, profile, queue = matched + try: + replica_count = _queue_replica_count(queue) + except ValueError as exc: + return { + "action": "block", "mode": "tool", "reason": str(exc), + "profile_id": profile_id, "queue_id": queue.get("id"), + } + replicated = replica_count > 1 + if replicated: + replica_status = _queue_replica_status(queue, db.list_series()) + stage_name = str((queue.get("execution") or {}).get("stage") + or queue.get("id") or "").lower() + issues = list(replica_status["issues"]) + if stage_name != "review": + issues.append("replicas > 1 сейчас поддерживаются только для REVIEW") + if getattr(task, "machine", None): + issues.append("реплицированная REVIEW-задача должна выполняться локально") + if getattr(task, "worktree", False): + issues.append( + "реплицированная REVIEW-задача требует постоянный отдельный working_dir, " + "а не динамический worktree") + if getattr(task, "detached", False): + issues.append( + "реплицированная REVIEW-задача не может запускаться detached") + if getattr(task, "herdr_target", None): + issues.append( + "реплицированная REVIEW-задача не может использовать пользовательский herdr_target") + current_replica = next(( + item for item in replica_status["matching"] + if int(item.get("id")) == int(task.series_id) + ), None) + if current_replica is None: + issues.append("текущая серия не входит в настроенный набор реплик") + elif os.path.normcase(os.path.realpath(str(working_dir or ""))) != \ + os.path.normcase(os.path.realpath( + str(current_replica.get("working_dir") or ""))): + issues.append("working_dir текущей задачи не совпадает с её replica series") + if issues: + return { + "action": "block", "mode": "tool", + "reason": "небезопасная конфигурация pipeline replicas: " + + "; ".join(dict.fromkeys(issues)), + "profile_id": profile_id, "queue_id": queue.get("id"), + "replica_status": { + key: value for key, value in replica_status.items() + if key != "matching" + }, + } execution = queue.get("execution") if not isinstance(execution, dict): + if replicated: + return { + "action": "block", "mode": "tool", + "reason": "replicas > 1 требуют project execution preflight", + "profile_id": profile_id, "queue_id": queue.get("id"), + } with _github_scan_admission( profile, f"pipeline task {profile_id}/{queue.get('id')}", profile_id=profile_id, budget_route="skill") as admission: @@ -2435,6 +2545,12 @@ def execution_route(task, fallback_prompt: str, working_dir: str | None = None, "profile_id": profile_id, "queue_id": queue.get("id"), } if mode == "skill": + if replicated: + return { + "action": "block", "mode": "skill", + "reason": "replicas > 1 несовместимы с execution.mode=skill", + "profile_id": profile_id, "queue_id": queue.get("id"), + } with _github_scan_admission( profile, f"pipeline task {profile_id}/{queue.get('id')}", profile_id=profile_id, budget_route="skill") as admission: @@ -2474,7 +2590,7 @@ def execution_route(task, fallback_prompt: str, working_dir: str | None = None, profile_id, profile, queue) if not available: - if mode == "auto": + if mode == "auto" and not replicated: admission = _reserve_execution_admission( task, profile_id, profile, queue, admission, "skill", retain_budget=retain_budget) @@ -2491,7 +2607,41 @@ def execution_route(task, fallback_prompt: str, working_dir: str | None = None, "profile_id": profile_id, "queue_id": queue.get("id"), } try: - preflight = _tool_preflight(execution, command, working_dir) + if replicated: + # Both worktrees must reserve in the scheduler's actual DB. + # Do not let a project-local .env silently select one DB per + # checkout and destroy cross-replica atomicity. + scheduler_data_dir = str(Path(db.DB_PATH).resolve().parent) + configured_key = os.environ.get("PP_PIPELINE_LEASE_KEY_FILE") + scheduler_lease_key = str( + (Path(configured_key) if configured_key else + Path(scheduler_data_dir) / "pipeline-lease.key").resolve()) + task_started_at = getattr(task, "started_at", None) + if not isinstance(task_started_at, datetime): + raise RuntimeError( + "replicated pipeline task has no claimed attempt timestamp") + if task_started_at.tzinfo is None: + task_started_at = task_started_at.astimezone() + scheduler_task_started_at = task_started_at.astimezone( + timezone.utc).isoformat() + provider_cfg = load_providers().get( + getattr(task, "provider", None) or DEFAULT_CLI, {}) + provider_ownership_kind = ( + "herdr" if provider_cfg.get("executor") == "herdr" + else "headless") + preflight = _tool_preflight( + execution, command, working_dir, + env_extra={ + "PP_TASK_ID": str(task.id), + "PP_TASK_STARTED_AT": scheduler_task_started_at, + "PP_PIPELINE_REPLICAS": str(replica_count), + "PP_PROVIDER_OWNERSHIP_KIND": provider_ownership_kind, + "PP_DATA_DIR": scheduler_data_dir, + "PP_PIPELINE_LEASE_KEY_FILE": scheduler_lease_key, + }, + ) + else: + preflight = _tool_preflight(execution, command, working_dir) except _GitHubScanLeaseFailure as exc: return _budget_defer_route( _replace_admission_with_lease_failure( @@ -2500,7 +2650,7 @@ def execution_route(task, fallback_prompt: str, working_dir: str | None = None, profile, queue) except RuntimeError as exc: reason = str(exc) - if mode == "auto": + if mode == "auto" and not replicated: admission = _reserve_execution_admission( task, profile_id, profile, queue, admission, "skill", retain_budget=retain_budget) @@ -2518,6 +2668,50 @@ def execution_route(task, fallback_prompt: str, working_dir: str | None = None, } preflight_action = preflight["action"].lower() + target_reservation = None + if replicated and preflight_action in {"audit", "fallback"}: + try: + from . import project_pipeline + + if preflight_action == "audit": + lease = project_pipeline.decode_signed_lease( + str(preflight.get("lease") or "")) + else: + handoff = preflight.get("handoff") or {} + lease = project_pipeline.decode_signed_lease( + str(handoff.get("lease") or "")) + reservation = lease.get("target_reservation") + repository, target_stage, number, head, owner_task, _token = \ + project_pipeline._reservation_identity(reservation) + target = (lease.get("target") if preflight_action == "fallback" + else { + "stage": lease.get("target_stage") or lease.get("stage"), + "number": lease.get("number"), "head": lease.get("head"), + }) + preflight_target = preflight.get("target") or {} + if (owner_task != int(task.id) + or reservation.get("task_started_at") != scheduler_task_started_at + or reservation.get("ownership_kind") != provider_ownership_kind + or lease.get("pipeline_replicas") != replica_count + or repository != str(profile.get("repository") or "").lower() + or repository != str(lease.get("repository") or "").lower() + or (target_stage, number, head) != ( + str(target.get("stage") or "").lower(), + target.get("number"), str(target.get("head") or "").lower()) + or (target_stage, number, head) != ( + str(preflight_target.get("stage") or "").lower(), + preflight_target.get("number"), + str(preflight_target.get("head") or "").lower())): + raise project_pipeline.PipelineError( + "pipeline target reservation contradicts task, repository, or target") + target_reservation = reservation + except (project_pipeline.PipelineError, TypeError, ValueError) as exc: + return { + "action": "block", "mode": "tool", + "reason": f"replicated REVIEW preflight has no valid reservation: {exc}", + "profile_id": profile_id, "queue_id": queue.get("id"), + "preflight": preflight, + } provider_route = None if preflight_action == "fallback" and mode == "auto": provider_route = ("fallback_targeted" if "handoff" in preflight @@ -2632,8 +2826,18 @@ def execution_route(task, fallback_prompt: str, working_dir: str | None = None, "gate_command": gate_command, "target": preflight["target"], "fallback_reason": preflight_reason, "profile_id": profile_id, "queue_id": queue.get("id"), "preflight": preflight, + **({"pipeline_replicas": replica_count} if replicated else {}), + **({"pipeline_data_dir": scheduler_data_dir} if replicated else {}), + **({"pipeline_lease_key_file": scheduler_lease_key} + if replicated else {}), + **({"pipeline_target_reservation": target_reservation} + if replicated else {}), + **({"pipeline_task_started_at": scheduler_task_started_at} + if replicated else {}), + **({"pipeline_provider_ownership_kind": provider_ownership_kind} + if replicated else {}), } - if mode == "auto": + if mode == "auto" and not replicated: return { "action": "prompt", "mode": "skill", "prompt": fallback_prompt, "fallback_reason": preflight_reason, "profile_id": profile_id, @@ -2646,7 +2850,7 @@ def execution_route(task, fallback_prompt: str, working_dir: str | None = None, } if preflight_action not in {"audit", "merge", "cleanup"}: reason = f"pipeline preflight вернул неподдерживаемое action={preflight_action}" - if mode == "auto": + if mode == "auto" and not replicated: return { "action": "prompt", "mode": "skill", "prompt": fallback_prompt, "fallback_reason": reason, "profile_id": profile_id, @@ -2684,6 +2888,16 @@ def execution_route(task, fallback_prompt: str, working_dir: str | None = None, "action": "prompt", "mode": "tool", "prompt": prompt, "command": command, "preflight": preflight, "profile_id": profile_id, "queue_id": queue.get("id"), + **({"pipeline_replicas": replica_count} if replicated else {}), + **({"pipeline_data_dir": scheduler_data_dir} if replicated else {}), + **({"pipeline_lease_key_file": scheduler_lease_key} + if replicated else {}), + **({"pipeline_target_reservation": target_reservation} + if replicated else {}), + **({"pipeline_task_started_at": scheduler_task_started_at} + if replicated else {}), + **({"pipeline_provider_ownership_kind": provider_ownership_kind} + if replicated else {}), } @@ -2781,7 +2995,8 @@ def _pipeline_runtime(matching_series: list[dict], now: datetime) -> dict: def _health(backlog: int, windows: dict, broken_series: int, paused_series: int = 0, - diagnostics: dict | None = None, runtime: dict | None = None) -> dict: + diagnostics: dict | None = None, runtime: dict | None = None, + invalid_replica_queues: list[dict] | None = None) -> dict: if runtime and runtime.get("required") and runtime.get("state") != "online": age = runtime.get("age_seconds") detail = f"; последний heartbeat {age} сек назад" if age is not None else "" @@ -2791,6 +3006,17 @@ def _health(backlog: int, windows: dict, broken_series: int, paused_series: int tasks = ", ".join(f"#{item['task_id']}" for item in runtime["stalled"]) return {"state": "red", "label": "зависший запуск", "reason": f"превышен task timeout: {tasks}"} + if invalid_replica_queues: + details = [] + for queue in invalid_replica_queues: + issues = (queue.get("replica_status") or {}).get("issues") or [] + suffix = f": {'; '.join(str(issue) for issue in issues)}" if issues else "" + details.append(f"{queue.get('id') or '?'}{suffix}") + return { + "state": "red", "label": "невалидная конфигурация реплик", + "reason": "очереди без безопасной исполнимой ёмкости — " + + " | ".join(details), + } if broken_series: return {"state": "red", "label": "требует внимания", "reason": f"оборванных серий: {broken_series}"} @@ -2848,12 +3074,196 @@ def _profile_active(profile: dict, series: list[dict]) -> bool: if queue.get("series_contains")) -def _series_for_queue(queue: dict, series: list[dict]) -> dict | None: +def _queue_replica_count(queue: dict) -> int: + value = queue.get("replicas", 1) + if (isinstance(value, bool) or not isinstance(value, int) + or not 1 <= value <= 16): + raise ValueError("replicas должен быть целым числом от 1 до 16") + return value + + +def _series_replicas_for_queue(queue: dict, series: list[dict]) -> list[dict]: marker = str(queue.get("series_contains") or "").lower() if not marker: - return None - return next((item for item in series - if marker in str(item.get("title") or "").lower()), None) + return [] + matching = [item for item in series + if marker in str(item.get("title") or "").lower() + and not item.get("ended") and not item.get("ended_at")] + # Preserve the historical first-match behavior unless the operator opted + # into replicas explicitly. A broad legacy marker must not wake multiple + # unreserved tasks merely because an accidental duplicate series exists. + return matching if _queue_replica_count(queue) > 1 else matching[:1] + + +def _local_directory_identity(path: Path) -> tuple[int, int]: + """Filesystem identity, including case-insensitive and symlink aliases.""" + stat_result = os.stat(path) + return int(stat_result.st_dev), int(stat_result.st_ino) + + +def _queue_replica_status(queue: dict, series: list[dict]) -> dict: + """Validate an opt-in local replica set and expose every matched series.""" + configured = _queue_replica_count(queue) + matching = _series_replicas_for_queue(queue, series) + issues = [] + if configured > 1 and len(matching) != configured: + issues.append( + f"ожидалось реплик: {configured}, найдено активных серий: {len(matching)}") + if configured > 1 and CONCURRENCY < configured: + issues.append( + f"PP_CONCURRENCY={CONCURRENCY} меньше числа реплик {configured}") + seen_paths = {} + rendered = [] + for item in matching: + working_dir = str(item.get("working_dir") or "").strip() + normalized = None + if configured > 1: + if item.get("machine"): + issues.append( + f"серия #{item.get('id')} удалённая; реплики требуют общий локальный SQLite") + if item.get("worktree"): + issues.append( + f"серия #{item.get('id')} использует динамический worktree; " + "репликам нужен постоянный отдельный working_dir") + if item.get("detached"): + issues.append( + f"серия #{item.get('id')} запускается detached; " + "репликам требуется управляемый lifetime провайдера") + if item.get("herdr_target"): + issues.append( + f"серия #{item.get('id')} использует пользовательский herdr_target") + if not working_dir: + issues.append(f"у серии #{item.get('id')} не задан working_dir") + else: + path = Path(working_dir).expanduser() + if not path.is_absolute(): + issues.append( + f"working_dir серии #{item.get('id')} должен быть абсолютным") + else: + if not path.is_dir(): + issues.append( + f"working_dir серии #{item.get('id')} не существует: {working_dir}") + else: + try: + normalized = _local_directory_identity(path) + except OSError as exc: + issues.append( + f"working_dir серии #{item.get('id')} недоступен: {exc}") + if normalized is not None: + previous = seen_paths.get(normalized) + if previous is not None: + issues.append( + f"серии #{previous} и #{item.get('id')} " + "используют один working_dir") + else: + seen_paths[normalized] = item.get("id") + rendered.append({ + "series_id": item.get("id"), "title": item.get("title"), + "working_dir": working_dir or None, + "task_id": item.get("next_task_id"), + "task_status": item.get("next_status"), + "interval": item.get("effective_recurrence"), + "paused": bool(item.get("paused")), + "broken": bool(item.get("broken")), + "worktree": bool(item.get("worktree")), + "detached": bool(item.get("detached")), + "herdr_target": item.get("herdr_target"), + "failure_rate": item.get("failure_rate"), + "empty_rate": item.get("empty_rate"), + "avg_duration_seconds": item.get("avg_duration_seconds"), + }) + return { + "configured": configured, + "present": len(matching), + "active": sum(not item.get("paused") and not item.get("broken") + for item in matching), + "valid": not issues, + "issues": issues, + "series": rendered, + "matching": matching, + } + + +def _queue_replica_projection(queue: dict, series: list[dict], + capacity: int) -> dict: + status = _queue_replica_status(queue, series) + matching = status["matching"] + primary = matching[0] if matching else None + active = int(status["active"]) + replicated = int(status["configured"]) > 1 + effective_replicas = active if (not replicated or status["valid"]) else 0 + recurrences = list(dict.fromkeys( + item.get("effective_recurrence") for item in matching + if item.get("effective_recurrence"))) + parseable = [(value, _interval_hours(value)) for value in recurrences] + parseable = [(value, hours) for value, hours in parseable if hours is not None] + interval = (max(parseable, key=lambda pair: pair[1])[0] if parseable + else recurrences[0] if recurrences else None) + durations = [int(item["avg_duration_seconds"]) for item in matching + if item.get("avg_duration_seconds") is not None] + failures = [float(item.get("failure_rate") or 0) for item in matching] + empties = [float(item.get("empty_rate") or 0) for item in matching] + public_status = {key: value for key, value in status.items() + if key != "matching"} + return { + "matching": matching, + "primary": primary, + "replica_count": status["configured"], + "replicas_present": status["present"], + "replicas_active": active, + "replica_status": public_status, + "series_replicas": public_status["series"], + "parallel_capacity": capacity * effective_replicas, + "interval": interval, + "avg_duration_seconds": ( + round(sum(durations) / len(durations)) if durations else None), + "failure_rate": ( + round(sum(failures) / len(failures), 3) if failures else None), + "empty_rate": round(sum(empties) / len(empties), 3) if empties else None, + } + + +def _replica_recommendation(queue: dict, backlog: int, projection: dict, + target_hours: float) -> dict: + capacity = int(projection["parallel_capacity"]) + if capacity > 0: + return _recommendation( + queue, backlog, capacity, projection["interval"], target_hours, + projection["avg_duration_seconds"], + ) + invalid = not bool(projection["replica_status"].get("valid", True)) + return { + "recommended_interval": None, "eta_hours": None, + "recommendation": ( + "невалидная конфигурация реплик — исправьте series/working_dir" + if invalid else + "нет активных реплик — восстановите или возобновите серию" + ), + "avg_duration_seconds": projection["avg_duration_seconds"], + "cycle_hours": None, "throughput_per_hour": 0, + } + + +def _replica_runs_needed(backlog: int, projection: dict) -> float | None: + capacity = int(projection["parallel_capacity"]) + return round(backlog / capacity, 1) if capacity > 0 else None + + +def _bottleneck_rank(queue: dict) -> float: + """Rank a positive backlog with no executable capacity as a hard stall.""" + backlog = queue.get("backlog") + if not isinstance(backlog, int) or backlog <= 0: + return 0 + if queue.get("eta_hours") is not None: + return float(queue["eta_hours"]) + if queue.get("runs_needed") is not None: + return float(queue["runs_needed"]) + return math.inf + + +def _series_for_queue(queue: dict, series: list[dict]) -> dict | None: + matching = _series_replicas_for_queue(queue, series) + return matching[0] if matching else None def _unknown_recommendation() -> dict: @@ -2912,28 +3322,27 @@ def _refresh_local_state(result: dict, profile: dict, series: list[dict], *, for queue in data.get("queues", []): config = queue_configs.get(str(queue.get("id")), {}) - matching = _series_for_queue(config, series) - if matching: - matching_series.append(matching) - series_ids.append(int(matching["id"])) - broken_series += int(bool(matching.get("broken"))) - paused_series += int(bool(matching.get("paused"))) + capacity = max(1, int(config.get("capacity", queue.get("capacity", 1)))) + projection = _queue_replica_projection(config, series, capacity) + matches = projection.pop("matching") + matching = projection.pop("primary") + if matches: + matching_series.extend(matches) + series_ids.extend(int(item["id"]) for item in matches) + broken_series += sum(int(bool(item.get("broken"))) for item in matches) + paused_series += sum(int(bool(item.get("paused"))) for item in matches) queue.update({ - "capacity": max(1, int(config.get("capacity", queue.get("capacity", 1)))), + "capacity": capacity, "series_id": matching["id"] if matching else None, "task_id": matching.get("next_task_id") if matching else None, "task_status": matching.get("next_status") if matching else None, - "interval": matching.get("effective_recurrence") if matching else None, - "failure_rate": matching.get("failure_rate") if matching else None, - "empty_rate": matching.get("empty_rate") if matching else None, + **projection, }) backlog = queue.get("backlog") if isinstance(backlog, int): - queue["runs_needed"] = round(backlog / queue["capacity"], 1) - queue.update(_recommendation( - config, backlog, queue["capacity"], queue["interval"], target_hours, - matching.get("avg_duration_seconds") if matching else None, - )) + queue["runs_needed"] = _replica_runs_needed(backlog, queue) + queue.update(_replica_recommendation( + config, backlog, queue, target_hours)) else: queue["runs_needed"] = None queue.update(_unknown_recommendation()) @@ -2946,14 +3355,12 @@ def _refresh_local_state(result: dict, profile: dict, series: list[dict], *, queue["wake"] = (_wake_status(data["profile_id"], config, diagnostics) if isinstance(diagnostics, dict) else None) queue["adaptive_cadence"] = _adaptive_cadence_status( - config, matching, backlog if isinstance(backlog, int) else None) + config, matches, backlog if isinstance(backlog, int) else None) bottleneck = max( (queue for queue in data.get("queues", []) if isinstance(queue.get("backlog"), int)), - key=lambda queue: (queue["eta_hours"] - if queue.get("eta_hours") is not None - else queue["runs_needed"]), + key=_bottleneck_rank, default=None, ) data["bottleneck"] = ( @@ -2961,18 +3368,22 @@ def _refresh_local_state(result: dict, profile: dict, series: list[dict], *, activity = db.pipeline_series_activity(series_ids) empty_runs = db.pipeline_run_metrics([], now - timedelta(hours=5)) - metrics_by_series = {} aggregate_runs = dict(empty_runs) for queue in data.get("queues", []): - queue["last_run"] = activity.get(queue.get("series_id")) - series_id = queue.get("series_id") - if series_id is not None and series_id not in metrics_by_series: - metrics = db.pipeline_run_metrics( - [series_id], now - timedelta(hours=5)) - metrics_by_series[series_id] = metrics - for key, value in metrics.items(): - aggregate_runs[key] = aggregate_runs.get(key, 0) + value - queue["runs_5h"] = metrics_by_series.get(series_id, dict(empty_runs)) + queue_series_ids = [int(item["series_id"]) + for item in queue.get("series_replicas", []) + if item.get("series_id") is not None] + last_runs = [activity[value] for value in queue_series_ids + if value in activity] + queue["last_run"] = max( + last_runs, key=lambda value: value.get("at") or "", default=None) + metrics = db.pipeline_run_metrics( + queue_series_ids, now - timedelta(hours=5)) + queue["runs_5h"] = metrics + for replica in queue.get("series_replicas", []): + replica["last_run"] = activity.get(replica.get("series_id")) + for key, value in metrics.items(): + aggregate_runs[key] = aggregate_runs.get(key, 0) + value history = data.get("history") if not isinstance(history, dict): @@ -3013,9 +3424,16 @@ def _refresh_local_state(result: dict, profile: dict, series: list[dict], *, runtime = _pipeline_runtime(matching_series, now) backlog_total = data.get("backlog_total") + invalid_replica_queues = [ + queue for queue in data.get("queues", []) + if isinstance(queue.get("backlog"), int) and queue["backlog"] > 0 + and isinstance(queue.get("replica_status"), dict) + and not queue["replica_status"].get("valid", True) + ] health = _health( backlog_total if isinstance(backlog_total, int) else 0, history, broken_series, paused_series, diagnostics, runtime, + invalid_replica_queues, ) if source == "none" and health.get("state") in {"green", "warming"}: health = { @@ -3068,14 +3486,13 @@ def _snapshot_fallback(profile_id: str, profile: dict, for member in members: if member.get("key"): all_items[member["key"]] = member - matching = _series_for_queue(config, series) capacity = max(1, int(config.get("capacity", 1))) - recommendation = (_recommendation( - config, backlog, capacity, - matching.get("effective_recurrence") if matching else None, - target_hours, - matching.get("avg_duration_seconds") if matching else None, - ) if isinstance(backlog, int) else _unknown_recommendation()) + projection = _queue_replica_projection(config, series, capacity) + projection.pop("matching") + projection.pop("primary") + recommendation = (_replica_recommendation( + config, backlog, projection, target_hours) + if isinstance(backlog, int) else _unknown_recommendation()) ordered_members = copy.deepcopy(members) if priority_settings: for member in ordered_members: @@ -3087,12 +3504,13 @@ def _snapshot_fallback(profile_id: str, profile: dict, queues.append({ "id": config["id"], "title": config["title"], "backlog": backlog, "capacity": capacity, - "runs_needed": round(backlog / capacity, 1) + "runs_needed": _replica_runs_needed(backlog, projection) if isinstance(backlog, int) else None, "membership_complete": bool(saved.get("membership_complete")), "age": _age_stats(members, now), "items": ordered_members[:priority_settings["max_items"]] if priority_settings else [], + **projection, **recommendation, }) @@ -3120,8 +3538,7 @@ def _snapshot_fallback(profile_id: str, profile: dict, if complete_backlog: bottleneck = max( queues, - key=lambda queue: queue["eta_hours"] if queue.get("eta_hours") is not None - else queue["runs_needed"], + key=_bottleneck_rank, default=None, ) result["bottleneck"] = ( @@ -3475,20 +3892,21 @@ def _analyze_without_budget(profile_id: str, series: list[dict], *, and int(member["number"]) in allowed_numbers} all_items.update(members) capacity = max(1, int(item.get("capacity", 1))) - matching = next((s for s in series - if item.get("series_contains", "").lower() in s["title"].lower()), None) - if matching: - matching_series.append(matching) - series_ids.append(matching["id"]) - broken_series += int(bool(matching.get("broken"))) - paused_series += int(bool(matching.get("paused"))) - cadence = _adaptive_cadence_status(item, matching, backlog) - cadence_reconciliations.append((item, matching, backlog)) - runs_needed = round(backlog / capacity, 1) - recommendation = _recommendation( - item, backlog, capacity, - matching["effective_recurrence"] if matching else None, target_hours, - matching.get("avg_duration_seconds") if matching else None) + projection = _queue_replica_projection(item, series, capacity) + matches = projection.pop("matching") + matching = projection.pop("primary") + if matches: + matching_series.extend(matches) + series_ids.extend(int(value["id"]) for value in matches) + broken_series += sum(int(bool(value.get("broken"))) + for value in matches) + paused_series += sum(int(bool(value.get("paused"))) + for value in matches) + cadence = _adaptive_cadence_status(item, matches, backlog) + cadence_reconciliations.append((item, matches, backlog)) + runs_needed = _replica_runs_needed(backlog, projection) + recommendation = _replica_recommendation( + item, backlog, projection, target_hours) age = _age_stats(list(members.values()), now) membership_complete = all(search["membership_complete"] for search in searches) ordered_members = list(members.values()) @@ -3507,9 +3925,7 @@ def _analyze_without_budget(profile_id: str, series: list[dict], *, "series_id": matching["id"] if matching else None, "task_id": matching.get("next_task_id") if matching else None, "task_status": matching.get("next_status") if matching else None, - "interval": matching["effective_recurrence"] if matching else None, - "failure_rate": matching["failure_rate"] if matching else None, - "empty_rate": matching["empty_rate"] if matching else None, + **projection, "membership_complete": membership_complete, "age": age, "items": ordered_members[:priority_settings["max_items"]] if priority_settings else [], @@ -3541,19 +3957,32 @@ def _analyze_without_budget(profile_id: str, series: list[dict], *, bottleneck = max( queues, - key=lambda q: q["eta_hours"] if q.get("eta_hours") is not None - else q["runs_needed"], + key=_bottleneck_rank, default=None, ) backlog_total = sum(queue["backlog"] for queue in queues) activity = db.pipeline_series_activity(series_ids) for queue in queues: - queue["last_run"] = activity.get(queue.get("series_id")) + queue_series_ids = [int(item["series_id"]) + for item in queue.get("series_replicas", []) + if item.get("series_id") is not None] + last_runs = [activity[value] for value in queue_series_ids + if value in activity] + queue["last_run"] = max( + last_runs, key=lambda value: value.get("at") or "", default=None) queue["runs_5h"] = db.pipeline_run_metrics( - [queue["series_id"]] if queue.get("series_id") else [], + queue_series_ids, now - timedelta(hours=5), ) + for replica in queue.get("series_replicas", []): + replica["last_run"] = activity.get(replica.get("series_id")) runtime = _pipeline_runtime(matching_series, now) + invalid_replica_queues = [ + queue for queue in queues + if queue.get("backlog", 0) > 0 + and isinstance(queue.get("replica_status"), dict) + and not queue["replica_status"].get("valid", True) + ] if db.is_paused(): github_rate_limit = (cached[1].get("github_rate_limit") if cached else None) @@ -3574,7 +4003,8 @@ def _analyze_without_budget(profile_id: str, series: list[dict], *, "age": _age_stats(list(all_items.values()), now), "history": windows, "health": _health( - backlog_total, windows, broken_series, paused_series, diagnostics, runtime), + backlog_total, windows, broken_series, paused_series, + diagnostics, runtime, invalid_replica_queues), "diagnostics": diagnostics, "diagnostics_generated_at": diagnostics_generated_at, "github_rate_limit": github_rate_limit, diff --git a/promptpilot/process_tree.py b/promptpilot/process_tree.py index 4f436cc..8fa81f8 100644 --- a/promptpilot/process_tree.py +++ b/promptpilot/process_tree.py @@ -248,18 +248,19 @@ def terminate(self) -> None: pass raise error return - group_id, self._group_id = self._group_id, None + group_id = self._group_id if group_id is None: return try: os.killpg(group_id, signal.SIGKILL) except ProcessLookupError: pass + self._group_id = None def close(self) -> None: """Release the boundary, killing descendants still inside it.""" if os.name == "nt": - job, self._job = self._job, None + job = self._job if job: # Do not rely only on KILL_ON_JOB_CLOSE. A provider or one of # its native helpers can retain/duplicate a handle to the job, @@ -269,20 +270,15 @@ def close(self) -> None: # Terminate the private kernel job itself: unlike a PID/PPID # walk this cannot race PID reuse and cannot select an # unrelated process. Detached tasks never enter this job. - error = None if not _TerminateJobObject(job, 1): - error = _windows_error("TerminateJobObject during close failed") - if not _CloseHandle(job) and error is None: - error = _windows_error("CloseHandle(job) failed") - if error is not None: - raise error + # Keep the live handle: a later cleanup retry must not see + # a consumed boundary and falsely report success. + raise _windows_error("TerminateJobObject during close failed") + if not _CloseHandle(job): + raise _windows_error("CloseHandle(job) failed") + self._job = None return - group_id, self._group_id = self._group_id, None - if group_id is not None: - try: - os.killpg(group_id, signal.SIGKILL) - except ProcessLookupError: - pass + self.terminate() def __enter__(self) -> "OwnedProcess": return self diff --git a/promptpilot/project_pipeline.py b/promptpilot/project_pipeline.py index 012e267..a2fcc98 100644 --- a/promptpilot/project_pipeline.py +++ b/promptpilot/project_pipeline.py @@ -216,6 +216,7 @@ def validate_review_lease(lease: dict, config: dict) -> None: raise PipelineError("signed review lease validity exceeds configured limit") if not isinstance(nonce, str) or not re.fullmatch(r"[0-9a-f]{32}", nonce): raise PipelineError("signed review lease has an invalid nonce") + _validate_lease_reservation(lease, config) class GitHub: @@ -276,6 +277,7 @@ def load_config(path: str) -> dict: data.setdefault("merge_method", "merge") data.setdefault("review_completion_gate", "health") data.setdefault("review_lease_seconds", 7200) + data.setdefault("target_reservation_ttl_seconds", data["review_lease_seconds"]) data.setdefault("fallback_handoff", "legacy") if not isinstance(data["fallback_handoff"], str) or data["fallback_handoff"] not in {"legacy", "target-v1"}: raise PipelineError("fallback_handoff must be legacy or target-v1") @@ -304,6 +306,11 @@ def load_config(path: str) -> dict: isinstance(data["review_lease_seconds"], bool) or not 300 <= data["review_lease_seconds"] <= 28800): raise PipelineError("review_lease_seconds must be an integer from 300 to 28800") + if (not isinstance(data["target_reservation_ttl_seconds"], int) or + isinstance(data["target_reservation_ttl_seconds"], bool) or + not 300 <= data["target_reservation_ttl_seconds"] <= 28800): + raise PipelineError( + "target_reservation_ttl_seconds must be an integer from 300 to 28800") return data @@ -1208,10 +1215,233 @@ def capabilities(config: dict) -> dict: "merge": "clean-ordinary-with-cleanup-recovery"}, "review_completion_gate": config.get("review_completion_gate", "health"), "fallback_handoff": config.get("fallback_handoff", "legacy"), + "target_reservations": "sqlite-task-lease-v1", "fallback": "repository skill"} -def fallback_target(config: dict, health: dict, stage: str, target: dict, reason: str) -> dict: +def _configured_replica_count() -> int: + raw = os.environ.get("PP_PIPELINE_REPLICAS", "1") + try: + value = int(raw) + except (TypeError, ValueError) as exc: + raise PipelineError("PP_PIPELINE_REPLICAS must be an integer") from exc + if str(value) != str(raw).strip() or not 1 <= value <= 16: + raise PipelineError("PP_PIPELINE_REPLICAS must be an integer from 1 to 16") + return value + + +def _pipeline_task_id(*, required: bool) -> int | None: + raw = os.environ.get("PP_TASK_ID") + if raw is None and not required: + return None + try: + value = int(raw or "") + except (TypeError, ValueError) as exc: + raise PipelineError("replicated pipeline election requires PP_TASK_ID") from exc + if value <= 0 or str(value) != str(raw).strip(): + raise PipelineError("replicated pipeline election requires a positive PP_TASK_ID") + return value + + +def _pipeline_task_started_at(*, required: bool) -> str | None: + raw = os.environ.get("PP_TASK_STARTED_AT") + if raw is None and not required: + return None + try: + parsed = datetime.fromisoformat(str(raw or "").strip()) + except (TypeError, ValueError) as exc: + raise PipelineError( + "replicated pipeline election requires PP_TASK_STARTED_AT") from exc + if parsed.tzinfo is None: + raise PipelineError( + "replicated pipeline election requires an aware PP_TASK_STARTED_AT") + return str(raw).strip() + + +def _provider_ownership_kind(*, required: bool) -> str | None: + value = str(os.environ.get("PP_PROVIDER_OWNERSHIP_KIND") or "").strip().lower() + if not value and not required: + return None + if value not in {"headless", "herdr"}: + raise PipelineError( + "replicated pipeline election requires " + "PP_PROVIDER_OWNERSHIP_KIND=headless|herdr") + return value + + +def _reservation_identity(value: dict) -> tuple[str, str, int, str, int, str]: + if not isinstance(value, dict): + raise PipelineError("pipeline target reservation is missing") + repository = str(value.get("repository") or "").strip().lower() + stage = str(value.get("stage") or "").strip().lower() + number = value.get("number") + head = str(value.get("head") or "").strip().lower() + task_id = value.get("task_id") + token = str(value.get("token") or "") + if (not repository or "/" not in repository or not stage + or type(number) is not int or number <= 0 + or not re.fullmatch(r"[0-9a-f]{40}", head) + or type(task_id) is not int or task_id <= 0 + or not re.fullmatch(r"[0-9a-f]{32}", token)): + raise PipelineError("pipeline target reservation is invalid") + return repository, stage, number, head, task_id, token + + +def _validate_lease_reservation(lease: dict, config: dict) -> dict | None: + lease_replicas = lease.get("pipeline_replicas", 1) + if (type(lease_replicas) is not int or not 1 <= lease_replicas <= 16): + raise PipelineError("pipeline lease has an invalid replica count") + configured_replicas = _configured_replica_count() + if configured_replicas > 1 and lease_replicas != configured_replicas: + raise PipelineError("pipeline lease replica count changed; rerun next review") + reservation = lease.get("target_reservation") + if reservation is None: + if lease_replicas > 1 or configured_replicas > 1: + raise PipelineError( + "replicated REVIEW lease has no target reservation") + return None + repository, stage, number, head, task_id, _token = _reservation_identity( + reservation) + if (repository != str(config["repository"]).lower() + or stage != str(lease.get("target_stage") or lease.get("stage") or "").lower() + or number != lease.get("number") or head != lease.get("head")): + raise PipelineError("pipeline target reservation contradicts its lease") + current_task_id = _pipeline_task_id(required=True) + if current_task_id != task_id: + raise PipelineError("pipeline target reservation belongs to another task") + current_attempt = _pipeline_task_started_at(required=True) + if reservation.get("task_started_at") != current_attempt: + raise PipelineError("pipeline target reservation belongs to another task attempt") + if reservation.get("ownership_kind") != _provider_ownership_kind(required=True): + raise PipelineError("pipeline target reservation ownership kind changed") + return reservation + + +def renew_lease_target_reservation(lease: dict, config: dict) -> dict | None: + """Fence the first mutation with the still-live exact target lease.""" + reservation = _validate_lease_reservation(lease, config) + if reservation is None: + return None + # Keep the scheduler DB a lazy dependency. Plain pipelinectl capabilities, + # health and merge operations historically work without opening it. + from . import db as scheduler_db + + renewed = scheduler_db.renew_pipeline_target_reservation( + reservation, int(config.get( + "target_reservation_ttl_seconds", + config.get("review_lease_seconds", 7200), + ))) + if renewed is None: + raise PipelineError( + "pipeline target reservation expired or was released; rerun next review") + return renewed + + +def _target_key(value: dict) -> tuple[str, int, str]: + try: + stage = str(value["stage"]).lower() + raw_number = value["number"] + if type(raw_number) is not int: + raise ValueError("number must be an integer") + number = raw_number + head = str(value["head"]).lower() + except (KeyError, TypeError, ValueError) as exc: + raise PipelineError("pipeline health returned an invalid review candidate") from exc + if (not stage or number <= 0 or not re.fullmatch(r"[0-9a-f]{40}", head)): + raise PipelineError("pipeline health returned an invalid review candidate") + return stage, number, head + + +def _review_health_without_targets(health: dict, unavailable: set[tuple]) -> dict: + filtered = dict(health) + for field in ("review_candidates", "content_review_candidates"): + values = health.get(field) + if isinstance(values, list): + filtered[field] = [value for value in values + if _target_key(value) not in unavailable] + return filtered + + +def _elect_review_candidate(config: dict, health: dict, + candidates: list[dict]) -> tuple[dict | None, dict | None, dict]: + """Reserve the first free candidate and preserve queue order atomically.""" + replicas = _configured_replica_count() + if replicas == 1: + return candidates[0], None, health + if (config.get("review_completion_gate") != "target-v1" + or config.get("fallback_handoff") != "target-v1"): + raise PipelineError( + "replicated REVIEW requires review_completion_gate and " + "fallback_handoff to be target-v1") + task_id = _pipeline_task_id(required=True) + task_started_at = _pipeline_task_started_at(required=True) + ownership_kind = _provider_ownership_kind(required=True) + from . import db as scheduler_db + + unavailable = set() + selected = None + reservation = None + candidate_keys = [_target_key(candidate) for candidate in candidates] + own = next((item for item in scheduler_db.list_pipeline_target_reservations( + repository=config["repository"]) + if int(item["task_id"]) == task_id), None) + if own is not None: + own_key = (own["stage"], int(own["number"]), own["head"]) + if own_key not in candidate_keys: + # One task means one immutable election envelope. A repeated next + # after HEAD/eligibility changed must not delete the old fence and + # silently start reviewing a different PR. + raise PipelineError( + "existing pipeline target reservation is no longer eligible; " + "start a new task") + selected_index = candidate_keys.index(own_key) + selected = candidates[selected_index] + unavailable.update(candidate_keys[:selected_index]) + reservation = scheduler_db.reserve_pipeline_target( + config["repository"], own["stage"], int(own["number"]), own["head"], + task_id, int(config.get( + "target_reservation_ttl_seconds", + config.get("review_lease_seconds", 7200), + )), + task_started_at=task_started_at, + ownership_kind=ownership_kind, + ) + if reservation is not None: + return selected, reservation, _review_health_without_targets( + health, unavailable) + # The listed lease may have expired and been taken between the read + # and the renewal transaction. This task already observed an exact + # target, so it must not silently retarget within the same envelope. + raise PipelineError( + "existing pipeline target reservation was lost; start a new task") + for candidate in candidates: + target_stage, number, head = _target_key(candidate) + reservation = scheduler_db.reserve_pipeline_target( + config["repository"], target_stage, number, head, task_id, + int(config.get( + "target_reservation_ttl_seconds", + config.get("review_lease_seconds", 7200), + )), + task_started_at=task_started_at, + ownership_kind=ownership_kind, + ) + if reservation is not None: + selected = candidate + break + unavailable.add((target_stage, number, head)) + if selected is None: + return None, None, health + + # fallback_target's election proof expects its target at the head of the + # executable queue. Targets skipped solely because another local replica + # owns them are removed from this immutable health view. The later fallback + # gate uses the fresh global allowlist in membership mode, not this filter. + return selected, reservation, _review_health_without_targets( + health, unavailable) + + +def fallback_target(config: dict, health: dict, stage: str, target: dict, reason: str, + *, reservation: dict | None = None) -> dict: if config.get("fallback_handoff") != "target-v1": return {"action": "fallback", "reason": reason} from .fallback_handoff import MERGE_STAGES, REVIEW_STAGES, create @@ -1220,7 +1450,8 @@ def fallback_target(config: dict, health: dict, stage: str, target: dict, reason if not isinstance(target, dict) or target.get("stage") not in allowed: return {"action": "fallback", "reason": reason} - return create(config, health, stage, target, reason) + return create(config, health, stage, target, reason, + target_reservation=reservation) def review_empty_reason(health: dict) -> str: @@ -1246,26 +1477,43 @@ def next_review(gh: GitHub, config: dict, *, config_path: str | None = None) -> candidates = health.get("review_candidates") or [] if not candidates: return {"action": "empty", "verdict": "ПУСТО", "reason": review_empty_reason(health)} - item = candidates[0] + item, reservation, election_health = _elect_review_candidate( + config, health, candidates) + if item is None: + return { + "action": "wait", "verdict": "ПУСТО", + "reason": "all current REVIEW targets are reserved by other replicas", + } + replica_count = _configured_replica_count() + if replica_count > 1 and reservation is None: + raise PipelineError("replicated REVIEW election did not reserve its target") if item.get("stage") == "pre-review-validation": if config.get("fallback_handoff") != "target-v1": raise PipelineError( "pre-review validation requires the exact-target fallback protocol") return fallback_target( - config, health, "review", item, + config, election_health, "review", item, "pre-review sync provenance requires validation and a full content review", + reservation=reservation, ) if item.get("stage") != "review": - return fallback_target(config, health, "review", item, - "integration/base-sync state requires the full skill") + return fallback_target( + config, election_health, "review", item, + "integration/base-sync state requires the full skill", + reservation=reservation, + ) completion_gate = config.get("review_completion_gate", "health") - if completion_gate == "target-v1" and not content_review_elected(health, item): + if completion_gate == "target-v1" and not content_review_elected( + election_health, item): if config.get("fallback_handoff") == "target-v1": raise PipelineError("health election did not prove the exact content target") return {"action": "fallback", "reason": "health election did not prove the exact content target"} if int(item.get("review_depth", 0)) >= 2: - return fallback_target(config, health, "review", item, - "third review round requires human-escalation rules") + return fallback_target( + config, election_health, "review", item, + "third review round requires human-escalation rules", + reservation=reservation, + ) snapshot = stable_timeline(gh, config, int(item["number"])) validate_common(snapshot, config, item) info = epoch(snapshot, config["trusted_account"]) @@ -1279,6 +1527,10 @@ def next_review(gh: GitHub, config: dict, *, config_path: str | None = None) -> "snapshot": content_review_digest(snapshot), "epoch": info["hash"], "anchor": info["anchor_id"], "depth": depth, "completion_gate": completion_gate} + if reservation is not None: + lease["target_stage"] = item["stage"] + lease["target_reservation"] = reservation + lease["pipeline_replicas"] = replica_count if completion_gate == "target-v1": issued_at = int(time.time()) lease.update({ @@ -1290,7 +1542,10 @@ def next_review(gh: GitHub, config: dict, *, config_path: str | None = None) -> content_review_target_gate(snapshot, config, lease) except PipelineError as exc: if config.get("fallback_handoff") == "target-v1": - return fallback_target(config, health, "review", item, str(exc)) + return fallback_target( + config, election_health, "review", item, str(exc), + reservation=reservation, + ) return {"action": "fallback", "reason": str(exc), "target": item} lease_value = (encode_signed_lease(lease) if completion_gate == "target-v1" else encode_lease(lease)) @@ -1408,6 +1663,11 @@ def complete_review(gh: GitHub, config: dict, lease_value: str, report_path: str # Stable GraphQL reads can be slow. A lease that expired during them # must not authorize the first externally visible mutation. validate_review_lease(lease, config) + # The local election lease is independent from GitHub's state proof. It is + # renewed only after all read-only gates and immediately before the first + # comment/label mutation, preventing an expired replica from publishing a + # second review after another task took over the same HEAD. + renew_lease_target_reservation(lease, config) review = post_comment(gh, config, lease["number"], body) snapshot = stable_timeline(gh, config, lease["number"]) diff --git a/promptpilot/static/index.html b/promptpilot/static/index.html index 318f894..cb494af 100644 --- a/promptpilot/static/index.html +++ b/promptpilot/static/index.html @@ -1149,7 +1149,7 @@

Провайдеры ` : ''} ${t.status === 'running' ? `` : ''} - + ${t.status !== 'running' ? `` : ''} `; @@ -2762,9 +2762,17 @@

План этапов

const cadenceLine = cadence ? `интервал: ${cadence.mode} · ${cadence.effective_recurrence || 'нет серии'}${cadence.empty_runs_before_idle ? ` · ПУСТО ${cadence.empty_runs}/${cadence.empty_runs_before_idle}` : ''}${cadence.event_wake ? ' · event wake' : ''}` : null; + const replicaStatus = q.replica_status || {configured:1,present:q.series_id?1:0,active:q.series_id?1:0,valid:true,issues:[],series:[]}; + const replicaSummary = `реплики: ${replicaStatus.active}/${replicaStatus.configured} активны · найдено ${replicaStatus.present}`; + const replicaIssues = replicaStatus.valid ? '' : ` · ошибка: ${(replicaStatus.issues || []).join('; ')}`; + const replicaLines = (q.series_replicas || []).map(replica => + `#${replica.series_id} ${replica.task_status || (replica.paused ? 'paused' : replica.broken ? 'broken' : 'idle')} · ${replica.interval || 'без интервала'} · ${replica.working_dir || 'working_dir не задан'}` + ); + const replicaLine = `${replicaSummary}${replicaIssues}${replicaLines.length ? ` · ${replicaLines.join(' | ')}` : ''}`; const bottleneck = data.bottleneck===q.id ? 'bottleneck' : ''; - return `${data.bottleneck===q.id?'⚠ ':''}${esc(q.title)}${q.backlog == null ? '—' : q.backlog}${h5.complete?perHour(deltaRateByQueue[q.id] ?? 0):'—'}${q.capacity}${q.runs_needed == null ? '—' : q.runs_needed}${duration(q.avg_duration_seconds)}${q.eta_hours == null ? '—' : q.eta_hours+' ч'}${age(q.age.oldest_hours)}${esc(q.interval || 'нет серии')}${esc(q.recommendation || '—')} -
${esc(lastLine)}${esc(stageLine)}${esc(routeLine)}${wakeLine?`${esc(wakeLine)}`:''}${cadenceLine?`${esc(cadenceLine)}`:''}
`; + const capacity = `${q.capacity} × ${q.replicas_active || 0} = ${q.parallel_capacity}`; + return `${data.bottleneck===q.id?'⚠ ':''}${esc(q.title)}${q.replica_count>1?` ×${q.replica_count}`:''}${q.backlog == null ? '—' : q.backlog}${h5.complete?perHour(deltaRateByQueue[q.id] ?? 0):'—'}${capacity}${q.runs_needed == null ? '—' : q.runs_needed}${duration(q.avg_duration_seconds)}${q.eta_hours == null ? '—' : q.eta_hours+' ч'}${age(q.age.oldest_hours)}${esc(q.interval || 'нет серии')}${esc(q.recommendation || '—')} +
${esc(lastLine)}${esc(stageLine)}${esc(routeLine)}${esc(replicaLine)}${wakeLine?`${esc(wakeLine)}`:''}${cadenceLine?`${esc(cadenceLine)}`:''}
`; }).join(''); const priorityHtml = data.priority_control ? `
Приоритет элементов очереди · P0 выполняется первым, старые задачи повышаются каждые ${data.priority_control.aging_hours} ч до P1 @@ -2816,7 +2824,7 @@

План этапов

${budgetMetric} ${diagnosticHtml} - ${rows}
ЭтапОчередьΔ/чЗа прогонПрогоновСредний запускETAСамый старыйИнтервалРекомендация
+ ${rows}
ЭтапОчередьΔ/чЗа реплику × активныеПрогоновСредний запускETAСамый старыйИнтервалРекомендация
${priorityHtml} `; } catch(e) { box.innerHTML = `
Анализ очереди недоступен: ${esc(e.message)}
`; } @@ -2826,7 +2834,7 @@

План этапов

try { const path = `${API}/pipeline-insights/${encodeURIComponent(profile)}/queues/${encodeURIComponent(queue)}/items/${encodeURIComponent(kind)}/${number}/priority`; const result = await fetchJSON(path, {method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({level,run_now:runNow})}); - const wake = runNow ? (result.series_woken ? ' Серия поставлена на ближайший запуск.' : result.series_paused ? ' Серия на паузе.' : '') : ''; + const wake = runNow ? (result.series_woken ? ` Серии поставлены на ближайший запуск: ${result.series_woken_count || 1}.` : result.series_paused ? ' Серии на паузе.' : '') : ''; toast(level === 'auto' ? `#${number}: автоматический приоритет.${wake}` : `#${number}: ${level.toUpperCase()}.${wake}`, false); await loadPipelineInsights(profile, true); } catch(e) { toast('Не удалось изменить приоритет: ' + e.message); } diff --git a/promptpilot/worker.py b/promptpilot/worker.py index 41a8b10..e9505bc 100644 --- a/promptpilot/worker.py +++ b/promptpilot/worker.py @@ -191,6 +191,22 @@ def live_task_ids() -> set: try: pids = [p for p in os.listdir("/proc") if p.isdigit()] except OSError: + pids = None + if pids is None: + if os.name != "posix": + return ids + try: + result = subprocess.run( + ["ps", "eww", "-axo", "pid=,command="], + capture_output=True, text=True, timeout=15, + encoding="utf-8", errors="replace", + ) + except (OSError, subprocess.TimeoutExpired): + return ids + if result.returncode: + return ids + for match in re.finditer(r"(?:^|\s)PP_TASK_ID=(\d+)(?=\s|$)", result.stdout): + ids.add(int(match.group(1))) return ids for pid in pids: try: @@ -207,6 +223,218 @@ def live_task_ids() -> set: return ids +def _marked_task_process_groups(task_id: int, task_started_at: str, + reservation_token: str) -> tuple[set[int], str]: + """Find POSIX provider groups carrying one unguessable exact-run marker.""" + if os.name != "posix": + return set(), "" + if (not task_started_at + or not re.fullmatch(r"[0-9a-f]{32}", reservation_token or "")): + return set(), "recovered provider has no exact reservation marker" + pids = [] + try: + proc_entries = [value for value in os.listdir("/proc") if value.isdigit()] + except OSError: + proc_entries = None + if proc_entries is not None: + markers = { + f"PP_TASK_ID={task_id}".encode(), + f"PP_TASK_STARTED_AT={task_started_at}".encode(), + f"PP_PIPELINE_TARGET_TOKEN={reservation_token}".encode(), + } + for value in proc_entries: + try: + with open(f"/proc/{value}/environ", "rb") as stream: + if markers.issubset(set(stream.read().split(b"\0"))): + pids.append(int(value)) + except OSError: + continue + else: + try: + result = subprocess.run( + ["ps", "eww", "-axo", "pid=,command="], + capture_output=True, text=True, timeout=15, + encoding="utf-8", errors="replace", + ) + except (OSError, subprocess.TimeoutExpired) as exc: + return set(), f"could not inspect orphan provider processes: {exc}" + if result.returncode: + return set(), ( + "could not inspect orphan provider processes: " + + (result.stderr or f"ps exit {result.returncode}").strip()) + markers = [re.compile( + rf"(?:^|\s){re.escape(value)}(?=\s|$)") for value in ( + f"PP_TASK_ID={int(task_id)}", + f"PP_TASK_STARTED_AT={task_started_at}", + f"PP_PIPELINE_TARGET_TOKEN={reservation_token}", + )] + for line in result.stdout.splitlines(): + stripped = line.lstrip() + pid_text, separator, command = stripped.partition(" ") + if (separator and pid_text.isdigit() + and all(marker.search(command) for marker in markers)): + pids.append(int(pid_text)) + groups = set() + for pid in pids: + try: + groups.add(os.getpgid(pid)) + except ProcessLookupError: + continue + except OSError as exc: + return set(), f"could not inspect provider process {pid}: {exc}" + own_group = os.getpgrp() + if own_group in groups: + return set(), "refusing to stop a provider in the worker process group" + return groups, "" + + +def _cleanup_orphan_headless_provider(task_id: int, task_started_at: str, + reservation_token: str) -> str: + """Stop and verify a POSIX provider left by an earlier worker process.""" + if os.name == "nt": + # Headless Windows providers live in a private KILL_ON_JOB_CLOSE Job. + # Losing the worker's handle is already a kernel-enforced cleanup. + return "" + groups, error = _marked_task_process_groups( + task_id, task_started_at, reservation_token) + if error: + return error + for group_id in groups: + try: + os.killpg(group_id, signal.SIGKILL) + except ProcessLookupError: + pass + except OSError as exc: + return f"could not stop orphan provider group {group_id}: {exc}" + deadline = time.monotonic() + 10 + while groups and time.monotonic() < deadline: + time.sleep(0.1) + groups, error = _marked_task_process_groups( + task_id, task_started_at, reservation_token) + if error: + return error + return ("orphan provider processes are still alive: " + + ", ".join(str(value) for value in sorted(groups))) if groups else "" + + +def _recovered_provider_cleanup(task, reservation: dict): + """Build an idempotent cleanup for a reserved attempt after worker loss.""" + started_at = getattr(task, "started_at", None) + expected_attempt = ( + started_at.astimezone(timezone.utc).isoformat() + if isinstance(started_at, datetime) else "") + if (not expected_attempt + or reservation.get("task_started_at") != expected_attempt): + return lambda: "recovered reservation belongs to another task attempt" + ownership_kind = str(reservation.get("ownership_kind") or "") + if ownership_kind not in {"headless", "herdr"}: + return lambda: "recovered reservation has no trusted provider ownership kind" + if ownership_kind == "herdr": + state = str(reservation.get("herdr_session_state") or "") + if state == "reserved": + # Election was durable, but session creation never began. The + # ordered state machine proves no Herdr provider can exist yet. + return lambda: "" + if state not in {"creating", "owned"}: + return lambda: ( + "recovered Herdr reservation predates the durable " + "session descriptor") + host = None + if getattr(task, "machine", None): + from .config import load_machines, machine_remote + machine = load_machines().get(task.machine) + if not machine or not machine.get("host"): + return lambda: f"machine {task.machine!r} is unavailable for orphan cleanup" + host = machine_remote(machine) + + def cleanup_herdr(): + from .herdr_exec import ( + _close_owned_session, _close_stale_tabs, _ensure_server, + ) + try: + _ensure_server(host) + pane_id = str(reservation.get("herdr_pane_id") or "") + tab_id = str(reservation.get("herdr_tab_id") or "") + workspace_id = str( + reservation.get("herdr_workspace_id") or "") + if state == "owned": + if not pane_id or not tab_id: + return "recovered Herdr descriptor is incomplete" + close_args = ( + ["workspace", "close", workspace_id] + if workspace_id else ["tab", "close", tab_id]) + error = _close_owned_session( + f"recovered task #{task.id}", close_args, host, + pane_id=pane_id) + if error: + return error + elif state == "creating": + # The agent start is ordered strictly after descriptor CAS. + # A crash here can leave only an idle shell/tab; labels are + # hygiene, not the proof used for an owned provider. + _close_stale_tabs(task.id, host) + except Exception as exc: + return f"could not close recovered herdr provider: {type(exc).__name__}: {exc}" + return "" + + return cleanup_herdr + if getattr(task, "machine", None): + return lambda: "cannot prove a recovered remote headless provider stopped" + task_started_at = str(reservation.get("task_started_at") or "") + reservation_token = str(reservation.get("token") or "") + return lambda: _cleanup_orphan_headless_provider( + task.id, task_started_at, reservation_token) + + +def _reconcile_recovered_attempt(task) -> None: + """Repair projections after an orphan becomes pending or cancelled.""" + try: + # For a recovered cancellation this also consumes a latched pipeline + # wake, matching the normal execute_task finally path. + _recur_after_run(task) + from . import workflows + workflows.sync_task(task.id) + workflows.advance_linked_task(task.id) + except Exception as exc: + print( + f" !! could not reconcile recovered attempt #{task.id}: {exc}", + flush=True, + ) + + +def _reconcile_reserved_running_tasks(alive: set[int]): + """Clean provider orphans before expired target fences become reclaimable.""" + reservations = db.list_pipeline_target_reservations() + by_task = {int(item["task_id"]): item for item in reservations} + running = db.list_tasks(statuses=["running"], limit=100000) + quarantines = [] + keep = set(alive) + for task in running: + reservation = by_task.get(task.id) + if reservation is None: + continue + cleanup = _recovered_provider_cleanup(task, reservation) + error = cleanup() + if error: + keep.add(task.id) + quarantines.append((task, RecoveredProviderOwnershipError( + f"recovered provider ownership is uncertain: {error}", cleanup))) + else: + if db.recover_running_attempt(task.id, task.started_at): + _reconcile_recovered_attempt(task) + keep.discard(task.id) + else: + fresh = db.get_task(task.id) + if (fresh is not None and fresh.status.value == "running" + and db.task_has_live_pipeline_target_reservation(task.id)): + keep.add(task.id) + quarantines.append((task, RecoveredProviderOwnershipError( + "recovered provider stopped, but exact attempt requeue was rejected", + cleanup, + ))) + return keep, quarantines + + def env_failure(text: str) -> str: """The bit of text proving the environment failed, or "" if it did not.""" m = ENV_FAILURE_RE.search(text or "") @@ -439,6 +667,235 @@ def _stop_owned_process(tree: OwnedProcess) -> None: tree.close() +_active_provider_trees: dict[tuple[str, int], OwnedProcess] = {} +_active_provider_trees_lock = threading.Lock() + + +class ProviderOwnershipError(RuntimeError): + """A provider may still be live; retry cleanup before terminal release.""" + + def __init__(self, message: str, cleanup, heartbeat=None, on_cleaned=None): + super().__init__(message) + self.cleanup = cleanup + self.heartbeat = heartbeat + self.on_cleaned = on_cleaned + + def retry_cleanup(self) -> bool: + heartbeat_error = "" + if callable(self.heartbeat): + try: + self.heartbeat(force=True) + except Exception as exc: + heartbeat_error = f"{type(exc).__name__}: {exc}" + try: + error = self.cleanup() + except Exception as exc: + error = f"{type(exc).__name__}: {exc}" + if error: + detail = error + if heartbeat_error: + detail += f"; target heartbeat failed: {heartbeat_error}" + print(f" !! provider ownership still uncertain: {detail}", flush=True) + return False + return True + + def finish_after_cleanup(self) -> bool: + if not callable(self.on_cleaned): + return False + try: + return bool(self.on_cleaned()) + except Exception as exc: + print( + f" !! provider terminal intent could not be committed: " + f"{type(exc).__name__}: {exc}", flush=True, + ) + return False + + +class RecoveredProviderOwnershipError(ProviderOwnershipError): + """A pre-existing orphan should be requeued, not counted as a failed run.""" + + +class PipelineTargetLeaseLost(RuntimeError): + """The exact target fence disappeared while its provider was still live.""" + + +PIPELINE_TARGET_HEARTBEAT_TTL = 600 +PIPELINE_TARGET_HEARTBEAT_INTERVAL = 30.0 + + +class _PipelineTargetHeartbeat: + """Exact lease heartbeat spanning admission, provider, and quarantine.""" + + def __init__(self, reservation: dict, task_id: int, *, task_started_at: str, + ownership_kind: str, ttl_seconds: int, interval_seconds: float): + if not isinstance(reservation, dict): + raise PipelineTargetLeaseLost( + "replicated pipeline route has no exact target reservation") + if reservation.get("task_id") != task_id: + raise PipelineTargetLeaseLost( + "pipeline target reservation belongs to another task") + if reservation.get("task_started_at") != task_started_at: + raise PipelineTargetLeaseLost( + "pipeline target reservation belongs to another task attempt") + if reservation.get("ownership_kind") != ownership_kind: + raise PipelineTargetLeaseLost( + "pipeline target reservation ownership kind changed") + self.reservation = reservation + self.task_id = task_id + self.ttl_seconds = ttl_seconds + self.interval_seconds = interval_seconds + self.last_touch = None + self.failure = None + self.lock = threading.Lock() + self.stop_event = threading.Event() + self.thread = None + + def __call__(self, *, force: bool = False): + now = time.monotonic() + with self.lock: + if self.failure is not None: + raise PipelineTargetLeaseLost( + f"pipeline target heartbeat failed: {self.failure}") + if (not force and self.last_touch is not None + and now - self.last_touch < self.interval_seconds): + return self.reservation + renewed = _retry_sqlite_busy( + lambda: db.renew_pipeline_target_reservation( + self.reservation, self.ttl_seconds), + f"pipeline target heartbeat for task #{self.task_id}", + ) + if renewed is None: + raise PipelineTargetLeaseLost( + "pipeline target reservation expired or was released") + self.reservation = renewed + self.last_touch = now + return renewed + + def start(self): + """Fence immediately, then keep renewing independent of provider polls.""" + self(force=True) + if self.thread is None: + self.thread = threading.Thread( + target=self._run, + name=f"pp-target-heartbeat-{self.task_id}", + daemon=True, + ) + self.thread.start() + return self + + def begin_herdr_session(self): + """Record that no Herdr agent may start until exact IDs are bound.""" + with self.lock: + if self.failure is not None: + raise PipelineTargetLeaseLost( + f"pipeline target heartbeat failed: {self.failure}") + updated = db.begin_pipeline_target_herdr_session(self.reservation) + if updated is None: + raise PipelineTargetLeaseLost( + "pipeline target could not enter Herdr creation state") + self.reservation = updated + return updated + + def bind_herdr_session(self, pane_id: str, tab_id: str, + workspace_id: str = ""): + """Persist immutable Herdr IDs before agent start.""" + with self.lock: + if self.failure is not None: + raise PipelineTargetLeaseLost( + f"pipeline target heartbeat failed: {self.failure}") + updated = db.bind_pipeline_target_herdr_session( + self.reservation, pane_id, tab_id, workspace_id) + if updated is None: + raise PipelineTargetLeaseLost( + "pipeline target rejected its Herdr ownership descriptor") + self.reservation = updated + return updated + + def _run(self): + while not self.stop_event.wait(self.interval_seconds): + try: + task = db.get_task(self.task_id) + if task is None or task.status.value != "running": + return + self(force=True) + except Exception as exc: + with self.lock: + if self.failure is None: + self.failure = f"{type(exc).__name__}: {exc}" + return + + def stop(self): + self.stop_event.set() + + +def _pipeline_target_heartbeater(reservation: dict, task_id: int, + *, task_started_at: str, ownership_kind: str, + ttl_seconds: int = PIPELINE_TARGET_HEARTBEAT_TTL, + interval_seconds: float = PIPELINE_TARGET_HEARTBEAT_INTERVAL): + return _PipelineTargetHeartbeat( + reservation, task_id, task_started_at=task_started_at, + ownership_kind=ownership_kind, + ttl_seconds=ttl_seconds, + interval_seconds=interval_seconds) + + +def _renew_quarantined_target(task_id: int) -> int | None: + """Keep any target fenced while provider cleanup remains uncertain.""" + try: + return db.renew_task_pipeline_target_reservations( + task_id, PIPELINE_TARGET_HEARTBEAT_TTL) + except Exception as exc: + print( + f" !! pipeline target quarantine heartbeat #{task_id}: {exc}", + flush=True, + ) + return None + + +def _provider_tree_key(task_id: int) -> tuple[str, int]: + return os.path.realpath(str(db.DB_PATH)), int(task_id) + + +def _register_provider_tree(task_id: int, tree: OwnedProcess) -> None: + key = _provider_tree_key(task_id) + with _active_provider_trees_lock: + duplicate = key in _active_provider_trees + if not duplicate: + _active_provider_trees[key] = tree + if duplicate: + # The new child already exists but is not registered. End it before + # surfacing the duplicate; the old registered owner remains intact. + _stop_owned_process(tree) + raise RuntimeError(f"task #{task_id} already owns a provider tree") + + +def _forget_provider_tree(task_id: int, tree: OwnedProcess) -> None: + key = _provider_tree_key(task_id) + with _active_provider_trees_lock: + if _active_provider_trees.get(key) is tree: + _active_provider_trees.pop(key, None) + + +def _close_registered_provider_tree(task_id: int) -> bool: + """Stop a child before any caller can release its exact target lease.""" + key = _provider_tree_key(task_id) + with _active_provider_trees_lock: + tree = _active_provider_trees.get(key) + if tree is None: + return True + try: + _stop_owned_process(tree) + except Exception as exc: + print( + f" !! provider tree cleanup #{task_id} is not confirmed: {exc}", + flush=True, + ) + return False + _forget_provider_tree(task_id, tree) + return True + + def _effective_timeout(task): """Per-task timeout in seconds; None = no limit (0 disables the global one).""" if task.task_timeout == 0: @@ -544,17 +1001,25 @@ def _requeue_env_failure(task, marker: str, detail: str): broken for good ends up failing the task instead of retrying forever. """ if task.retry_count >= task.max_retries: - db.mark_failed(task.id, f"Срыв по вине среды ({marker}), " - f"попытки исчерпаны ({task.max_retries}).\n{detail}") - print(f" -> Env failure ({marker}), retries exhausted") - return + changed = _mark_failed( + task, + f"Срыв по вине среды ({marker}), " + f"попытки исчерпаны ({task.max_retries}).\n{detail}", + ) + if changed: + print(f" -> Env failure ({marker}), retries exhausted") + return changed next_run = compute_next_run(task.retry_count) - db.mark_rate_limited(task.id, next_run, - error=f"Срыв по вине среды ({marker}) — " - f"задача возвращена в очередь.\n{detail}") - _notify_requeued(task, next_run, f"срыв среды ({marker})") - print(f" -> Env failure ({marker}). Retry #{task.retry_count + 1} " - f"at {next_run.strftime('%H:%M:%S')}") + changed = _mark_rate_limited( + task, next_run, + error=f"Срыв по вине среды ({marker}) — " + f"задача возвращена в очередь.\n{detail}", + ) + if changed: + _notify_requeued(task, next_run, f"срыв среды ({marker})") + print(f" -> Env failure ({marker}). Retry #{task.retry_count + 1} " + f"at {next_run.strftime('%H:%M:%S')}") + return changed def _notify_requeued(task, next_run, reason: str): @@ -573,9 +1038,125 @@ def _notify_requeued(task, next_run, reason: str): print(f" -> notify requeue failed: {e}") +def _finalize_herdr_outcome(task, outcome: dict, *, + require_closing_verdict: bool, + allow_targeted_stale: bool) -> bool: + """Commit a verified-stopped Herdr result for this exact task attempt.""" + if outcome.get("cancelled"): + changed = _mark_cancelled( + task, + outcome.get("cancel_note") or "Отменена пользователем во время выполнения", + ) + if changed: + print(" -> Cancelled by user") + return changed + + if outcome["rate_limited"]: + reason = outcome.get("retry_reason") or RETRY_RATE_LIMIT + label = RETRY_REASON_ERR.get(reason, RETRY_REASON_ERR[RETRY_RATE_LIMIT]) + if task.retry_count >= task.max_retries: + return _mark_failed( + task, + f"{label}, max retries ({task.max_retries}) exceeded.\n" + f"{outcome['error']}", + ) + next_run = compute_next_run(task.retry_count) + changed = _mark_rate_limited( + task, next_run, error=outcome["error"] or label) + if changed: + _notify_requeued( + task, + next_run, + RETRY_REASON_RU.get(reason, RETRY_REASON_RU[RETRY_RATE_LIMIT]), + ) + print( + f" -> {label}. Retry #{task.retry_count + 1} " + f"at {next_run.strftime('%H:%M:%S')}") + return changed + + if outcome.get("env_failure"): + return _requeue_env_failure( + task, outcome["env_failure"], outcome["error"]) + + if not outcome["ok"]: + changed = _retry_sqlite_busy( + lambda: _mark_failed(task, outcome["error"], exit_code=1), + f"завершение herdr-задачи #{task.id} с ошибкой", + ) + if changed: + print(" -> Failed (herdr)") + return changed + + verdict = outcome.get("verdict") or parse_verdict(outcome["output"]) + if verdict == "УСТАРЕЛО" and not allow_targeted_stale: + verdict = "НЕ СМОГ" + if require_closing_verdict and not outcome.get("verdict"): + changed = _retry_sqlite_busy( + lambda: _mark_failed( + task, + "Pipeline provider returned success without a closing ИТОГ verdict", + exit_code=1, + ), + f"отклонение неполного результата herdr-задачи #{task.id}", + ) + if changed: + print(" -> Failed (pipeline result has no closing verdict)") + return changed + changed = _retry_sqlite_busy( + lambda: _mark_completed( + task, outcome["output"], exit_code=0, + verdict=verdict or None, + ), + f"завершение herdr-задачи #{task.id}", + ) + if changed: + text_preview = outcome["output"][:80].replace("\n", " ").strip() + print(f" -> Completed: {text_preview}") + return changed + + +def _commit_terminal_or_defer(task, label: str, commit) -> bool: + """Retry a known provider result instead of replacing it with failure.""" + try: + changed = commit() + except Exception as exc: + raise ProviderOwnershipError( + f"{label} is waiting for durable commit: " + f"{type(exc).__name__}: {exc}", + lambda: "", + on_cleaned=commit, + ) from exc + if changed is False: + fresh = db.get_task(task.id) + if (fresh is not None and fresh.status.value == "running" + and fresh.started_at == getattr(task, "started_at", None)): + raise ProviderOwnershipError( + f"{label} CAS was rejected for its live attempt", + lambda: "", + on_cleaned=commit, + ) + return changed is not False + + +def _commit_herdr_outcome_or_defer(task, outcome: dict, *, + require_closing_verdict: bool, + allow_targeted_stale: bool) -> None: + """Do not turn a durable Herdr result into a generic worker failure.""" + _commit_terminal_or_defer( + task, + "Herdr terminal outcome", + lambda: _finalize_herdr_outcome( + task, + outcome, + require_closing_verdict=require_closing_verdict, + allow_targeted_stale=allow_targeted_stale, + ), + ) + + def _execute_herdr_task(task, provider_cfg, host=None, machine=None, prompt_override=None, admission_complete=None, require_closing_verdict=False, - allow_targeted_stale=False): + allow_targeted_stale=False, target_heartbeat=None): """Run the task in a live herdr session (providers with executor=herdr). host is the ssh target of the machine the session lives on (None = local). @@ -629,73 +1210,62 @@ def on_started(_pane_id): # creating its owned tab and durably recording/starting the provider. _signal_admission_complete(admission_complete) + def on_session(pane_id, tab_id, workspace_id): + if target_heartbeat is None: + return + target_heartbeat.bind_herdr_session( + pane_id, tab_id, workspace_id or "") + + def cancel_or_heartbeat(): + if callable(target_heartbeat): + target_heartbeat() + return db.is_cancel_requested(task.id) + + if callable(target_heartbeat): + target_heartbeat(force=True) + if not hasattr(target_heartbeat, "begin_herdr_session"): + raise PipelineTargetLeaseLost( + "replicated Herdr route has no durable session binder") + target_heartbeat.begin_herdr_session() + outcome = run_in_herdr(task, provider_cfg, on_blocked=on_blocked, timeout=_effective_timeout(task), - cancel_check=lambda: db.is_cancel_requested(task.id), + cancel_check=cancel_or_heartbeat, keep_pane=task.keep_pane, host=host, on_worktree=on_worktree, on_pane=on_pane, + on_session=on_session, on_started=on_started, prompt_override=prompt_override, require_closing_verdict=require_closing_verdict, allow_targeted_stale=allow_targeted_stale) - if outcome.get("cancelled"): - db.clear_cancel_request(task.id) - db.mark_cancelled( - task.id, - outcome.get("cancel_note") or "Отменена пользователем во время выполнения", - ) - print(" -> Cancelled by user") - return - - if outcome["rate_limited"]: - reason = outcome.get("retry_reason") or RETRY_RATE_LIMIT - label = RETRY_REASON_ERR.get(reason, RETRY_REASON_ERR[RETRY_RATE_LIMIT]) - if task.retry_count >= task.max_retries: - db.mark_failed(task.id, f"{label}, max retries ({task.max_retries}) exceeded.\n{outcome['error']}") - return - next_run = compute_next_run(task.retry_count) - db.mark_rate_limited(task.id, next_run, error=outcome["error"] or label) - _notify_requeued(task, next_run, - RETRY_REASON_RU.get(reason, RETRY_REASON_RU[RETRY_RATE_LIMIT])) - print(f" -> {label}. Retry #{task.retry_count + 1} at {next_run.strftime('%H:%M:%S')}") - return - - if outcome.get("env_failure"): - _requeue_env_failure(task, outcome["env_failure"], outcome["error"]) - return - - if not outcome["ok"]: - _retry_sqlite_busy( - lambda: db.mark_failed(task.id, outcome["error"], exit_code=1), - f"завершение herdr-задачи #{task.id} с ошибкой", - ) - print(" -> Failed (herdr)") - return - - verdict = outcome.get("verdict") or parse_verdict(outcome["output"]) - if verdict == "УСТАРЕЛО" and not allow_targeted_stale: - verdict = "НЕ СМОГ" - if require_closing_verdict and not outcome.get("verdict"): - _retry_sqlite_busy( - lambda: db.mark_failed( - task.id, - "Pipeline provider returned success without a closing ИТОГ verdict", - exit_code=1, + ownership_cleanup = outcome.pop("_ownership_cleanup", None) + terminal_intent = outcome.pop("_terminal_intent", None) + if outcome.get("ownership_uncertain"): + if not callable(ownership_cleanup): + ownership_cleanup = lambda: "missing herdr ownership cleanup" + settled_outcome = dict(outcome) + settled_outcome.pop("ownership_uncertain", None) + if isinstance(terminal_intent, dict): + settled_outcome.update(terminal_intent) + raise ProviderOwnershipError( + outcome.get("error") or "herdr provider ownership is uncertain", + ownership_cleanup, + heartbeat=target_heartbeat, + on_cleaned=lambda: _finalize_herdr_outcome( + task, + settled_outcome, + require_closing_verdict=require_closing_verdict, + allow_targeted_stale=allow_targeted_stale, ), - f"отклонение неполного результата herdr-задачи #{task.id}", ) - print(" -> Failed (pipeline result has no closing verdict)") - return - _retry_sqlite_busy( - lambda: db.mark_completed( - task.id, outcome["output"], exit_code=0, - verdict=verdict or None, - ), - f"завершение herdr-задачи #{task.id}", + + _commit_herdr_outcome_or_defer( + task, + outcome, + require_closing_verdict=require_closing_verdict, + allow_targeted_stale=allow_targeted_stale, ) - text_preview = outcome["output"][:80].replace("\n", " ").strip() - print(f" -> Completed: {text_preview}") def _wrap_ssh(remote, cmd, env_extra): @@ -736,6 +1306,31 @@ def _retry_sqlite_busy(operation, label: str): time.sleep(delay) +def _mark_completed(task, *args, **kwargs): + kwargs["expected_started_at"] = getattr(task, "started_at", None) + return db.mark_completed(task.id, *args, **kwargs) + + +def _mark_failed(task, *args, **kwargs): + kwargs["expected_started_at"] = getattr(task, "started_at", None) + return db.mark_failed(task.id, *args, **kwargs) + + +def _mark_rate_limited(task, *args, **kwargs): + kwargs["expected_started_at"] = getattr(task, "started_at", None) + return db.mark_rate_limited(task.id, *args, **kwargs) + + +def _defer_task(task, *args, **kwargs): + kwargs["expected_started_at"] = getattr(task, "started_at", None) + return db.defer_task(task.id, *args, **kwargs) + + +def _mark_cancelled(task, *args, **kwargs): + kwargs["expected_started_at"] = getattr(task, "started_at", None) + return db.mark_cancelled(task.id, *args, **kwargs) + + def _execute_task_body(task, admission_complete=None): """Run CLI with the task's prompt.""" if task.series_id: @@ -754,7 +1349,7 @@ def _execute_task_body(task, admission_complete=None): next_run = _pipeline_defer_time(gate) if next_run: _retry_sqlite_busy( - lambda: db.defer_task(task.id, next_run, reason), + lambda: _defer_task(task, next_run, reason), f"отложить pipeline-задачу #{task.id}", ) print(f" -> Deferred without agent: {reason}") @@ -762,8 +1357,8 @@ def _execute_task_body(task, admission_complete=None): print(" !! invalid pipeline defer time", flush=True) elif gate["action"] == "complete_empty": _retry_sqlite_busy( - lambda: db.mark_completed( - task.id, + lambda: _mark_completed( + task, f"Предварительная проверка PromptPilot: {reason}\n" "Провайдер не запускался, токены не потрачены.\n\n" f"ИТОГ: ПУСТО ({reason})", @@ -777,6 +1372,13 @@ def _execute_task_body(task, admission_complete=None): agent_prompt = effective_prompt(task) require_closing_verdict = False allow_targeted_stale = False + pipeline_replicas = None + pipeline_data_dir = None + pipeline_lease_key_file = None + pipeline_task_started_at = None + pipeline_provider_ownership_kind = None + pipeline_target_token = None + target_heartbeat = None if task.series_id: try: from . import pipeline_insights @@ -792,8 +1394,8 @@ def _execute_task_body(task, admission_complete=None): if route["action"] == "block": reason = route["reason"] _retry_sqlite_busy( - lambda: db.mark_completed( - task.id, + lambda: _mark_completed( + task, f"Предварительная проверка PromptPilot: {reason}\n" "Провайдер не запускался, токены не потрачены.\n\n" f"ИТОГ: НУЖЕН ЧЕЛОВЕК ({reason})", @@ -808,8 +1410,8 @@ def _execute_task_body(task, admission_complete=None): next_run = _pipeline_defer_time(route) if next_run: _retry_sqlite_busy( - lambda: db.defer_task( - task.id, next_run, reason, + lambda: _defer_task( + task, next_run, reason, hard_not_before=( route.get("defer_policy") == "hard_not_before"), ), @@ -818,8 +1420,8 @@ def _execute_task_body(task, admission_complete=None): print(f" -> Pipeline preflight deferred without agent: {reason}") return _retry_sqlite_busy( - lambda: db.mark_completed( - task.id, + lambda: _mark_completed( + task, f"Pipeline preflight PromptPilot: {reason}\n" "Некорректное время повтора pipeline preflight\n\n" f"ИТОГ: НУЖЕН ЧЕЛОВЕК ({reason})", @@ -837,8 +1439,8 @@ def _execute_task_body(task, admission_complete=None): if verdict.upper() == "УСТАРЕЛО": verdict = "НЕ СМОГ" _retry_sqlite_busy( - lambda: db.mark_completed( - task.id, + lambda: _mark_completed( + task, f"Pipeline preflight PromptPilot: {reason}\n" "Провайдер не запускался, токены не потрачены.\n\n" f"ИТОГ: {verdict} ({reason})", @@ -849,6 +1451,50 @@ def _execute_task_body(task, admission_complete=None): print(f" -> Pipeline preflight completed without agent: {reason}") return agent_prompt = route["prompt"] + raw_replicas = route.get("pipeline_replicas") + if type(raw_replicas) is int and 2 <= raw_replicas <= 16: + pipeline_replicas = str(raw_replicas) + raw_data_dir = route.get("pipeline_data_dir") + if (pipeline_replicas is not None and isinstance(raw_data_dir, str) + and os.path.isabs(raw_data_dir)): + pipeline_data_dir = os.path.realpath(raw_data_dir) + if pipeline_replicas is not None and pipeline_data_dir is None: + raise RuntimeError( + "replicated pipeline route has no absolute scheduler data directory") + raw_lease_key = route.get("pipeline_lease_key_file") + if (pipeline_replicas is not None and isinstance(raw_lease_key, str) + and os.path.isabs(raw_lease_key)): + pipeline_lease_key_file = os.path.realpath(raw_lease_key) + if pipeline_replicas is not None and pipeline_lease_key_file is None: + raise RuntimeError( + "replicated pipeline route has no absolute scheduler lease key") + if pipeline_replicas is not None: + raw_started_at = route.get("pipeline_task_started_at") + task_started_at = getattr(task, "started_at", None) + expected_started_at = ( + task_started_at.astimezone(timezone.utc).isoformat() + if isinstance(task_started_at, datetime) else None) + if (not isinstance(raw_started_at, str) + or raw_started_at != expected_started_at): + raise RuntimeError( + "replicated pipeline route belongs to another task attempt") + pipeline_task_started_at = raw_started_at + raw_ownership_kind = route.get("pipeline_provider_ownership_kind") + if raw_ownership_kind not in {"headless", "herdr"}: + raise RuntimeError( + "replicated pipeline route has no provider ownership kind") + pipeline_provider_ownership_kind = raw_ownership_kind + target_heartbeat = _pipeline_target_heartbeater( + route.get("pipeline_target_reservation"), task.id, + task_started_at=pipeline_task_started_at, + ownership_kind=pipeline_provider_ownership_kind) + target_heartbeat.start() + raw_target_token = target_heartbeat.reservation.get("token") + if (not isinstance(raw_target_token, str) + or not re.fullmatch(r"[0-9a-f]{32}", raw_target_token)): + raise RuntimeError( + "replicated pipeline route has no exact target token") + pipeline_target_token = raw_target_token require_closing_verdict = bool( route.get("profile_id") and route.get("queue_id") ) @@ -876,6 +1522,27 @@ def _execute_task_body(task, admission_complete=None): provider = task.provider or DEFAULT_CLI provider_cfg = load_providers().get(provider, {}) + if pipeline_replicas is not None: + actual_ownership_kind = ( + "herdr" if provider_cfg.get("executor") == "herdr" else "headless") + if actual_ownership_kind != pipeline_provider_ownership_kind: + raise RuntimeError( + "replicated pipeline provider ownership changed after election") + # The provider may invoke pipelinectl again for the signed completion + # gate. Keep replica mode in its environment too: otherwise an + # accidental repeated `next` would silently fall back to the legacy, + # unreserved election path. + provider_cfg = dict(provider_cfg) + provider_cfg["strict_owned_session"] = True + provider_cfg["env"] = { + **(provider_cfg.get("env") or {}), + "PP_PIPELINE_REPLICAS": pipeline_replicas, + "PP_TASK_STARTED_AT": pipeline_task_started_at, + "PP_PROVIDER_OWNERSHIP_KIND": pipeline_provider_ownership_kind, + "PP_PIPELINE_TARGET_TOKEN": pipeline_target_token, + "PP_DATA_DIR": pipeline_data_dir, + "PP_PIPELINE_LEASE_KEY_FILE": pipeline_lease_key_file, + } machine = getattr(task, "machine", None) host = None @@ -883,7 +1550,7 @@ def _execute_task_body(task, admission_complete=None): from .config import load_machines, machine_remote m = load_machines().get(machine) if not m or not m.get("host"): - db.mark_failed(task.id, f"Машина «{machine}» не найдена в реестре") + _mark_failed(task, f"Машина «{machine}» не найдена в реестре") return host = machine_remote(m) @@ -897,6 +1564,7 @@ def _execute_task_body(task, admission_complete=None): admission_complete=admission_complete, require_closing_verdict=require_closing_verdict, allow_targeted_stale=allow_targeted_stale, + target_heartbeat=target_heartbeat, ) finally: # Startup failures have no on_started callback. Once their durable @@ -917,13 +1585,13 @@ def _execute_task_body(task, admission_complete=None): # A headless remote command runs in the ssh login directory, so a # worktree over there would simply never be entered. herdr-based # providers place the agent in the checkout and do support this. - db.mark_failed(task.id, f"worktree на машине «{machine}» поддерживается только " - f"через herdr-провайдер (executor: herdr)") + _mark_failed(task, f"worktree на машине «{machine}» поддерживается только " + f"через herdr-провайдер (executor: herdr)") return try: wt = worktree.prepare(task.working_dir, task.id) except worktree.WorktreeError as e: - db.mark_failed(task.id, f"worktree: {e}") + _mark_failed(task, f"worktree: {e}") print(f" -> Failed (worktree): {e}") return run_dir = wt["path"] @@ -939,11 +1607,22 @@ def _execute_task_body(task, admission_complete=None): env = get_provider_env(provider) # Marks the run in its own environment, inherited by the agent process. That # is what lets a live run be found by process rather than by our bookkeeping. + env.pop("PP_PIPELINE_REPLICAS", None) + env.pop("PP_TASK_STARTED_AT", None) + env.pop("PP_PROVIDER_OWNERSHIP_KIND", None) + env.pop("PP_PIPELINE_TARGET_TOKEN", None) env["PP_TASK_ID"] = str(task.id) + if pipeline_replicas is not None: + env["PP_PIPELINE_REPLICAS"] = pipeline_replicas + env["PP_TASK_STARTED_AT"] = pipeline_task_started_at + env["PP_PROVIDER_OWNERSHIP_KIND"] = pipeline_provider_ownership_kind + env["PP_PIPELINE_TARGET_TOKEN"] = pipeline_target_token + env["PP_DATA_DIR"] = pipeline_data_dir + env["PP_PIPELINE_LEASE_KEY_FILE"] = pipeline_lease_key_file if machine: if task.detached: - db.mark_failed(task.id, "Фоновый запуск (detached) на удалённой машине пока не поддерживается") + _mark_failed(task, "Фоновый запуск (detached) на удалённой машине пока не поддерживается") return cmd = _wrap_ssh(host, cmd, provider_cfg.get("env")) env = os.environ.copy() @@ -969,14 +1648,16 @@ def _execute_task_body(task, admission_complete=None): kwargs["start_new_session"] = True try: proc = subprocess.Popen(cmd, **kwargs) - db.mark_completed(task.id, f"Запущен в фоне (PID {proc.pid}){wt_note}", exit_code=0) + _mark_completed(task, f"Запущен в фоне (PID {proc.pid}){wt_note}", exit_code=0) print(f" -> Detached (PID {proc.pid})") except FileNotFoundError: - db.mark_failed(task.id, f"Command not found: {cmd[0]}", exit_code=-1) + _mark_failed(task, f"Command not found: {cmd[0]}", exit_code=-1) return effective_timeout = _effective_timeout(task) + if callable(target_heartbeat): + target_heartbeat(force=True) try: tree = OwnedProcess.start( cmd, @@ -990,13 +1671,14 @@ def _execute_task_body(task, admission_complete=None): env=env, ) except FileNotFoundError: - db.mark_failed(task.id, f"CLI '{provider}' not found. Is it installed and in PATH?", exit_code=-1) + _mark_failed(task, f"CLI '{provider}' not found. Is it installed and in PATH?", exit_code=-1) return except ProcessTreeError as exc: # Fail closed: without a lifetime boundary a timed-out agent can keep # changing the checkout after the queue has already moved on. - db.mark_failed(task.id, f"Could not isolate CLI process tree: {exc}", exit_code=-1) + _mark_failed(task, f"Could not isolate CLI process tree: {exc}", exit_code=-1) return + _register_provider_tree(task.id, tree) proc = tree.process if prompt_stdin is not None: @@ -1032,33 +1714,65 @@ def _execute_task_body(task, admission_complete=None): proc.wait(timeout=2) break except subprocess.TimeoutExpired: + if callable(target_heartbeat): + target_heartbeat() if db.is_cancel_requested(task.id): - _stop_owned_process(tree) + commit_cancel = lambda: _mark_cancelled( + task, "Отменена пользователем во время выполнения") + try: + _stop_owned_process(tree) + except Exception as exc: + raise ProviderOwnershipError( + "headless cancellation is waiting for process cleanup: " + f"{type(exc).__name__}: {exc}", + lambda: "", + on_cleaned=commit_cancel, + ) from exc try: proc.wait(timeout=10) except subprocess.TimeoutExpired: pass stdout_thread.join(timeout=10) stderr_thread.join(timeout=10) - db.clear_cancel_request(task.id) - db.mark_cancelled(task.id, "Отменена пользователем во время выполнения") - print(" -> Cancelled by user") + changed = _commit_terminal_or_defer( + task, + "headless cancellation", + commit_cancel, + ) + if changed: + print(" -> Cancelled by user") return if effective_timeout and time.monotonic() - started > effective_timeout: - _stop_owned_process(tree) + commit_timeout = lambda: _mark_failed( + task, f"Execution timed out after {effective_timeout}s", + exit_code=-1) + try: + _stop_owned_process(tree) + except Exception as exc: + raise ProviderOwnershipError( + "headless timeout is waiting for process cleanup: " + f"{type(exc).__name__}: {exc}", + lambda: "", + on_cleaned=commit_timeout, + ) from exc try: proc.wait(timeout=10) except subprocess.TimeoutExpired: pass stdout_thread.join(timeout=10) stderr_thread.join(timeout=10) - db.mark_failed(task.id, f"Execution timed out after {effective_timeout}s", exit_code=-1) + _commit_terminal_or_defer( + task, + "headless timeout", + commit_timeout, + ) return # A successful wrapper may exit while a child still owns the pipes. Close # the task boundary before joining readers so such a child cannot survive # (or hold this worker forever). tree.close() + _forget_provider_tree(task.id, tree) stdout_thread.join(timeout=10) stderr_thread.join(timeout=10) stdout = "".join(stdout_parts) @@ -1085,12 +1799,27 @@ def _execute_task_body(task, admission_complete=None): rl_error = format_result(parsed) or rl_error label = RETRY_REASON_ERR[reason] if task.retry_count >= task.max_retries: - db.mark_failed(task.id, f"{label}, max retries ({task.max_retries}) exceeded.\n{rl_error}") + _commit_terminal_or_defer( + task, + "headless exhausted retry outcome", + lambda: _mark_failed( + task, + f"{label}, max retries ({task.max_retries}) exceeded.\n" + f"{rl_error}"), + ) return next_run = compute_next_run(task.retry_count) - db.mark_rate_limited(task.id, next_run, error=rl_error or label) - _notify_requeued(task, next_run, RETRY_REASON_RU[reason]) - print(f" -> {label}. Retry #{task.retry_count + 1} at {next_run.strftime('%H:%M:%S')}") + def commit_rate_limit(): + changed = _mark_rate_limited( + task, next_run, error=rl_error or label) + if changed: + _notify_requeued(task, next_run, RETRY_REASON_RU[reason]) + print( + f" -> {label}. Retry #{task.retry_count + 1} " + f"at {next_run.strftime('%H:%M:%S')}") + return changed + _commit_terminal_or_defer( + task, "headless rate-limit outcome", commit_rate_limit) return if result.returncode != 0: @@ -1104,10 +1833,20 @@ def _execute_task_body(task, admission_complete=None): error_text = format_result(parsed) or error_text marker = env_failure(result.stderr) or env_failure(result.stdout) if marker: - _requeue_env_failure(task, marker, error_text) + _commit_terminal_or_defer( + task, + "headless environment-failure outcome", + lambda: _requeue_env_failure(task, marker, error_text), + ) return - db.mark_failed(task.id, error_text, exit_code=result.returncode) - print(f" -> Failed (exit {result.returncode})") + changed = _commit_terminal_or_defer( + task, + "headless failure outcome", + lambda: _mark_failed( + task, error_text, exit_code=result.returncode), + ) + if changed: + print(f" -> Failed (exit {result.returncode})") return # Parse output @@ -1123,13 +1862,28 @@ def _execute_task_body(task, admission_complete=None): rl = parsed.get("rate_limit_info") if rl and not parsed["text"]: if task.retry_count >= task.max_retries: - db.mark_failed(task.id, f"{RETRY_REASON_ERR[RETRY_RATE_LIMIT]}.\n{output}") + _commit_terminal_or_defer( + task, + "headless exhausted stream retry outcome", + lambda: _mark_failed( + task, f"{RETRY_REASON_ERR[RETRY_RATE_LIMIT]}.\n{output}"), + ) return next_run = compute_next_run(task.retry_count) - db.mark_rate_limited(task.id, next_run, - error=output or RETRY_REASON_ERR[RETRY_RATE_LIMIT]) - _notify_requeued(task, next_run, RETRY_REASON_RU[RETRY_RATE_LIMIT]) - print(f" -> Rate limited (stream event). Retry at {next_run.strftime('%H:%M:%S')}") + def commit_stream_rate_limit(): + changed = _mark_rate_limited( + task, next_run, + error=output or RETRY_REASON_ERR[RETRY_RATE_LIMIT]) + if changed: + _notify_requeued( + task, next_run, RETRY_REASON_RU[RETRY_RATE_LIMIT]) + print( + " -> Rate limited (stream event). Retry at " + f"{next_run.strftime('%H:%M:%S')}") + return changed + _commit_terminal_or_defer( + task, "headless stream rate-limit outcome", + commit_stream_rate_limit) return else: # Plain text output (non-Claude CLIs) @@ -1141,16 +1895,18 @@ def _execute_task_body(task, admission_complete=None): verdict = _closing_workflow_verdict( verdict_source, allow_targeted_stale=allow_targeted_stale) if not verdict: - _retry_sqlite_busy( - lambda: db.mark_failed( - task.id, + changed = _commit_terminal_or_defer( + task, + "headless missing-verdict outcome", + lambda: _mark_failed( + task, "Pipeline provider returned success without a closing ИТОГ verdict\n" + output[-4000:], exit_code=1, ), - f"отклонение неполного результата задачи #{task.id}", ) - print(" -> Failed (pipeline result has no closing verdict)") + if changed: + print(" -> Failed (pipeline result has no closing verdict)") return else: verdict = parse_verdict(output) @@ -1161,15 +1917,17 @@ def _execute_task_body(task, admission_complete=None): if changes and changes != worktree.NO_CHANGES: output += f"\nИзменения: {changes}" - _retry_sqlite_busy( - lambda: db.mark_completed( - task.id, output, exit_code=0, model_used=model_used, + changed = _commit_terminal_or_defer( + task, + "headless successful outcome", + lambda: _mark_completed( + task, output, exit_code=0, model_used=model_used, session_id=session_id, verdict=verdict or None, ), - f"завершение задачи #{task.id}", ) - text_preview = output[:80].replace("\n", " ").strip() - print(f" -> Completed: {text_preview}") + if changed: + text_preview = output[:80].replace("\n", " ").strip() + print(f" -> Completed: {text_preview}") # An empty checkout (no commits, no dirty files) is just clutter — remove it # so .pp-worktrees doesn't grow without bound. The branch stays regardless; @@ -1213,6 +1971,24 @@ def execute_task(task, admission_complete=None): return _execute_task_inner(task) return _execute_task_inner(task, admission_complete) finally: + # Project-pipeline election reserves an exact target before any + # provider can start. Every attempt exit (success, failure, defer, + # cancellation, startup error, or unexpected exception) releases that + # task's reservation; a hard process crash is recovered by the lease + # TTL in the database. + provider_stopped = _close_registered_provider_tree(task.id) + if not provider_stopped: + _renew_quarantined_target(task.id) + try: + fresh_for_cleanup = db.get_task(task.id) + if (provider_stopped and fresh_for_cleanup is not None + and fresh_for_cleanup.status.value != "running"): + db.release_pipeline_target_reservations(task.id) + except Exception as exc: + print( + f" !! pipeline target reservation cleanup #{task.id}: {exc}", + flush=True, + ) try: _recur_after_run(task) except Exception as exc: @@ -1308,6 +2084,56 @@ def _fail_stuck(task, exc) -> bool: the worker loop can retry it with backoff. A fenced no-op is terminal: the task already moved on and must not be changed by this old recovery. """ + if not _close_registered_provider_tree(task.id): + _renew_quarantined_target(task.id) + return False + if isinstance(exc, ProviderOwnershipError) and not exc.retry_cleanup(): + _renew_quarantined_target(task.id) + return False + if isinstance(exc, RecoveredProviderOwnershipError): + try: + changed = db.recover_running_attempt(task.id, task.started_at) + except Exception as error: + print( + f" !! could not requeue recovered provider #{task.id}: {error}", + flush=True, + ) + return False + if changed: + _reconcile_recovered_attempt(task) + return True + fresh = db.get_task(task.id) + return fresh is None or fresh.status.value != "running" + if isinstance(exc, ProviderOwnershipError) and callable(exc.on_cleaned): + if not exc.finish_after_cleanup(): + fresh = db.get_task(task.id) + if (fresh is None or fresh.status.value != "running" + or fresh.started_at != task.started_at): + # The exact attempt moved on concurrently. Its stale + # continuation is a fenced no-op, not a reason to retain a + # quarantine around a newer attempt. + return True + _renew_quarantined_target(task.id) + return False + try: + # execute_task's finally ran while the exact attempt was still + # fenced as running. Complete the projections only after cleanup + # has made the delayed terminal transition safe. + _recur_after_run(task) + fresh = db.get_task(task.id) + if fresh and fresh.status.value == "completed": + _notify_pipeline_completion(fresh, fresh.verdict) + from . import workflows + workflows.sync_task(task.id) + workflows.advance_linked_task(task.id) + except Exception as error: + # The terminal row is already durable. Startup reconciliation and + # the periodic pipeline sampler repair these idempotent projections. + print( + f" !! could not finish deferred terminal reconciliation " + f"#{task.id}: {error}", flush=True, + ) + return True try: changed = db.fail_running_attempt( task.id, @@ -1346,12 +2172,22 @@ def _stuck_recovery_delay(failures: int) -> float: def _queue_stuck_recovery(recoveries: dict, task, lock: str, exc) -> None: """Remember one failed execution attempt until its state is durable.""" key = (task.id, task.started_at) + try: + target_fenced = db.task_has_live_pipeline_target_reservation(task.id) + except Exception as fence_exc: + print( + f" !! could not inspect pipeline target fence #{task.id}: " + f"{fence_exc}", flush=True, + ) + target_fenced = isinstance(exc, ProviderOwnershipError) recoveries.setdefault(key, { "task": task, "lock": lock, "exc": exc, "failures": 0, "retry_at": 0.0, + "target_fenced": target_fenced, + "heartbeat_at": 0.0, }) @@ -1359,6 +2195,19 @@ def _drain_stuck_recoveries(recoveries: dict, now: float | None = None) -> None: """Run due recoveries and retain failed writes for a later loop pass.""" now = time.monotonic() if now is None else now for key, item in list(recoveries.items()): + if item.get("target_fenced") and item.get("heartbeat_at", 0.0) <= now: + renewed = _renew_quarantined_target(item["task"].id) + item["heartbeat_at"] = now + PIPELINE_TARGET_HEARTBEAT_INTERVAL + if renewed != 1: + # The fence is already lost or SQLite could not confirm its + # renewal. Do not wait through exponential backoff while a + # provider may still be mutating the old target. + print( + f" !! pipeline target fence lost during cleanup " + f"#{item['task'].id}; retrying ownership cleanup now", + flush=True, + ) + item["retry_at"] = 0.0 if item["retry_at"] > now: continue if _fail_stuck(item["task"], item["exc"]): @@ -1384,6 +2233,17 @@ def _reap_futures(in_flight: dict, recoveries: dict, _drain_stuck_recoveries(recoveries, now=now) +def _active_worker_task_ids(in_flight: dict, recoveries: dict) -> set[int]: + """Tasks that still own either execution or cleanup capacity.""" + return ({task.id for _lock, task in in_flight.values()} + | {item["task"].id for item in recoveries.values()}) + + +def _occupied_worker_slots(in_flight: dict, recoveries: dict) -> int: + """A quarantined provider keeps its slot until ownership is resolved.""" + return len(in_flight) + len(recoveries) + + def lock_key(task) -> str: """What a task must not share with another task running at the same time. @@ -1401,7 +2261,14 @@ def lock_key(task) -> str: return "" path = task.working_dir or os.getcwd() if not getattr(task, "machine", None): - path = os.path.abspath(path) + path = os.path.realpath(path) + try: + stat_result = os.stat(path) + return ( + f"{where}:dir-id:{int(stat_result.st_dev)}:" + f"{int(stat_result.st_ino)}") + except OSError: + path = os.path.abspath(path) path = os.path.normcase(path.rstrip("/\\")) return f"{where}:dir:{path}" @@ -1509,6 +2376,7 @@ def publish_heartbeat(): # Recover tasks stuck in 'running' from a previous crash — but leave alone # any whose agent is still working: the worker dying doesn't kill the agent. alive = live_task_ids() + alive, recovered_quarantines = _reconcile_reserved_running_tasks(alive) if alive: print(f"Живые прогоны найдены по метке в окружении, не трогаю: {sorted(alive)}") db.recover_running(keep_ids=alive) @@ -1544,6 +2412,9 @@ def publish_heartbeat(): in_flight = {} # Future -> (lock key, exact claimed task attempt) task_lanes = {} # task id -> configured scheduler lane stuck_recoveries = {} # exact attempt -> retry state; keeps its lock key + for orphan_task, orphan_error in recovered_quarantines: + _queue_stuck_recovery( + stuck_recoveries, orphan_task, lock_key(orphan_task), orphan_error) admission_fence = _AdmissionFence() short_on_memory = False if CONCURRENCY > 1: @@ -1551,11 +2422,11 @@ def publish_heartbeat(): pool = ThreadPoolExecutor(max_workers=CONCURRENCY, thread_name_prefix="pp-task") def reap(): - before = {task.id for _lock, task in in_flight.values()} _reap_futures(in_flight, stuck_recoveries) - after = {task.id for _lock, task in in_flight.values()} - for task_id in before - after: - task_lanes.pop(task_id, None) + active = _active_worker_task_ids(in_flight, stuck_recoveries) + for task_id in list(task_lanes): + if task_id not in active: + task_lanes.pop(task_id, None) while running: reap() @@ -1563,14 +2434,15 @@ def reap(): # Auto-reload: pick up code updates between tasks (dev-friendly — # a stale worker silently ignoring new features is worse than a restart) if code_snapshot is not None and _code_snapshot() != code_snapshot: - if in_flight: + if in_flight or stuck_recoveries: # Restarting now would orphan live agents — let them finish. time.sleep(POLL_INTERVAL) continue print("Код обновился — перезапускаю worker...", flush=True) os.execv(sys.executable, [sys.executable, "-m", "promptpilot", "worker"]) - if db.is_paused() or len(in_flight) >= CONCURRENCY: + if (db.is_paused() + or _occupied_worker_slots(in_flight, stuck_recoveries) >= CONCURRENCY): time.sleep(POLL_INTERVAL) continue diff --git a/tests/test_herdr_workflow_completion.py b/tests/test_herdr_workflow_completion.py index 1c1a0df..4fb2e6d 100644 --- a/tests/test_herdr_workflow_completion.py +++ b/tests/test_herdr_workflow_completion.py @@ -262,7 +262,9 @@ def fake_run(args, host=None, timeout=None): monkeypatch.setattr(herdr_exec, "_close_stale_tabs", lambda *_args: None) monkeypatch.setattr(herdr_exec, "_run", fake_run) monkeypatch.setattr(herdr_exec.time, "sleep", lambda _seconds: None) - monkeypatch.setattr(herdr_exec, "_close_owned_session", lambda *_args: "") + monkeypatch.setattr( + herdr_exec, "_close_owned_session", + lambda *_args, **_kwargs: "") outcome = herdr_exec.run_in_herdr( task, {"kind": "agy"}, prompt_override="OneBase - REVIEW", @@ -346,7 +348,7 @@ def fake_run(args, host=None, timeout=None): lambda *_args, **_kwargs: ("done", ""), ) monkeypatch.setattr( - herdr_exec, "_close_owned_session", lambda *_args: "") + herdr_exec, "_close_owned_session", lambda *_args, **_kwargs: "") outcome = herdr_exec.run_in_herdr( task, @@ -402,7 +404,7 @@ def fake_run(args, host=None, timeout=None): lambda *_args, **_kwargs: ("done", ""), ) monkeypatch.setattr( - herdr_exec, "_close_owned_session", lambda *_args: "") + herdr_exec, "_close_owned_session", lambda *_args, **_kwargs: "") outcome = herdr_exec.run_in_herdr( task, {"kind": "codex", "env": {"REMOTE_ONLY": "/srv/data"}}, @@ -566,7 +568,8 @@ def fake_run(args, host=None, timeout=None): monkeypatch.setattr(herdr_exec.time, "sleep", lambda _seconds: None) monkeypatch.setattr( herdr_exec, "_close_owned_session", - lambda name, args, host: closed.append((name, args, host)) or "", + lambda name, args, host, **_kwargs: + closed.append((name, args, host)) or "", ) outcome = herdr_exec.run_in_herdr( diff --git a/tests/test_pipeline_replicas.py b/tests/test_pipeline_replicas.py new file mode 100644 index 0000000..1bb1abd --- /dev/null +++ b/tests/test_pipeline_replicas.py @@ -0,0 +1,1655 @@ +import json +import os +import sqlite3 +import subprocess +import sys +import time +from concurrent.futures import ThreadPoolExecutor +from contextlib import contextmanager +from datetime import datetime, timedelta, timezone +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from promptpilot import fallback_handoff, herdr_exec, pipeline_insights, worker, workflows +from promptpilot import project_pipeline as pipelinectl +from promptpilot.models import TaskCreate + + +HEAD_A = "a" * 40 +HEAD_B = "b" * 40 + + +@pytest.fixture(autouse=True) +def _enough_worker_slots(monkeypatch): + monkeypatch.setattr(pipeline_insights, "CONCURRENCY", 2) + + +def _candidate(number, head): + return {"number": number, "head": head, "stage": "review", "review_depth": 0} + + +def _health(*candidates): + values = list(candidates) + return { + "state": "green", "findings": [], "integration_owner": None, + "review_candidates": values, + "content_review_candidates": values, + "merge_executable": [], + } + + +def _set_replica_attempt_env(monkeypatch, task, *, ownership_kind="headless"): + monkeypatch.setenv("PP_TASK_ID", str(task.id)) + monkeypatch.setenv( + "PP_TASK_STARTED_AT", task.started_at.astimezone(timezone.utc).isoformat()) + monkeypatch.setenv("PP_PROVIDER_OWNERSHIP_KIND", ownership_kind) + + +def _reserve_attempt(db_module, task, *, number=10, head=HEAD_A, + ownership_kind="headless"): + return db_module.reserve_pipeline_target( + "owner/repo", "review", number, head, task.id, 300, + task_started_at=task.started_at.astimezone(timezone.utc).isoformat(), + ownership_kind=ownership_kind, + ) + + +def _replica_series(series_id, directory, *, title="Project - REVIEW", + status="pending", paused=False, broken=False): + return { + "id": series_id, "title": title, "working_dir": str(directory), + "ended": False, "paused": paused, "broken": broken, + "next_task_id": series_id + 100, "next_status": status, + "effective_recurrence": "15m", "failure_rate": 0, + "empty_rate": 0, "avg_duration_seconds": 300, + "temporary_recurrence": None, "temporary_empty_count": 0, + } + + +def test_target_reservation_is_atomic_idempotent_and_skips_to_next(isolated_db): + barrier = __import__("threading").Barrier(2) + + def reserve(task_id): + barrier.wait() + return isolated_db.reserve_pipeline_target( + "Owner/Repo", "review", 10, HEAD_A, task_id, 300) + + with ThreadPoolExecutor(max_workers=2) as pool: + first, second = list(pool.map(reserve, (101, 102))) + + winners = [item for item in (first, second) if item is not None] + assert len(winners) == 1 + winner = winners[0] + assert isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, winner["task_id"], 300 + )["token"] == winner["token"] + loser = 102 if winner["task_id"] == 101 else 101 + assert isolated_db.reserve_pipeline_target( + "owner/repo", "review", 11, HEAD_B, loser, 300 + )["number"] == 11 + + +@pytest.mark.parametrize(("stage", "head"), [ + ("pre-review-validation", HEAD_A), + ("review", HEAD_B), +]) +def test_same_pr_cannot_be_reserved_across_stage_or_head_transition( + isolated_db, stage, head): + first = isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, 101, 300) + + assert first is not None + assert isolated_db.reserve_pipeline_target( + "owner/repo", stage, 10, head, 102, 300) is None + + +def test_one_task_cannot_reserve_targets_in_two_repositories(isolated_db): + first = isolated_db.reserve_pipeline_target( + "owner/first", "review", 10, HEAD_A, 101, 300) + + assert first is not None + assert isolated_db.reserve_pipeline_target( + "owner/second", "review", 20, HEAD_B, 101, 300) is None + assert len(isolated_db.list_pipeline_target_reservations()) == 1 + + +def test_herdr_descriptor_is_bound_exactly_before_provider_start(isolated_db): + isolated_db.create_task(TaskCreate(prompt="bind Herdr")) + task = isolated_db.get_next_runnable() + reservation = _reserve_attempt( + isolated_db, task, ownership_kind="herdr") + assert reservation["herdr_session_state"] == "reserved" + + creating = isolated_db.begin_pipeline_target_herdr_session(reservation) + owned = isolated_db.bind_pipeline_target_herdr_session( + creating, "pane-1", "tab-1", "workspace-1") + repeated = isolated_db.bind_pipeline_target_herdr_session( + owned, "pane-1", "tab-1", "workspace-1") + + assert owned["herdr_session_state"] == "owned" + assert ( + owned["herdr_pane_id"], owned["herdr_tab_id"], + owned["herdr_workspace_id"], + ) == ("pane-1", "tab-1", "workspace-1") + assert repeated["token"] == owned["token"] + assert isolated_db.bind_pipeline_target_herdr_session( + owned, "renamed-pane", "tab-1", "workspace-1") is None + + +def test_target_reservation_ttl_and_terminal_release(isolated_db): + now = datetime(2026, 9, 17, tzinfo=timezone.utc) + first = isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, 101, 60, now=now) + assert isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, 102, 60, + now=now + timedelta(seconds=59)) is None + replacement = isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, 102, 60, + now=now + timedelta(seconds=61)) + assert replacement["task_id"] == 102 + assert isolated_db.renew_pipeline_target_reservation( + first, 60, now=now + timedelta(seconds=61)) is None + isolated_db.release_pipeline_target_reservations(102) + + task = isolated_db.create_task(TaskCreate(prompt="terminal")) + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 12, "c" * 40, task.id, 300) + isolated_db.mark_completed(task.id, "done") + assert isolated_db.list_pipeline_target_reservations( + repository="owner/repo", stage="review") == [] + + +def test_running_owner_keeps_expired_target_fenced_across_system_sleep( + isolated_db): + now = datetime(2026, 9, 17, tzinfo=timezone.utc) + created = isolated_db.create_task(TaskCreate(prompt="sleeping provider")) + running = isolated_db.get_next_runnable() + assert running.id == created.id + reservation = isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, running.id, 60, now=now) + + assert isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, running.id + 1, 60, + now=now + timedelta(hours=1)) is None + renewed = isolated_db.renew_pipeline_target_reservation( + reservation, 60, now=now + timedelta(hours=1)) + assert renewed is not None + assert renewed["task_id"] == running.id + + isolated_db.mark_failed(running.id, "provider stopped") + replacement = isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, running.id + 1, 60, + now=now + timedelta(hours=1, seconds=61)) + assert replacement["task_id"] == running.id + 1 + + +def test_deleting_task_releases_target_reservation(isolated_db): + task = isolated_db.create_task(TaskCreate(prompt="delete reserved task")) + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, task.id, 300) + + assert isolated_db.delete_task(task.id) is True + assert isolated_db.list_pipeline_target_reservations( + repository="owner/repo", stage="review") == [] + + +def test_running_task_cannot_be_deleted_or_release_reservation(isolated_db): + task = isolated_db.create_task(TaskCreate(prompt="running reserved task")) + running = isolated_db.get_next_runnable() + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, running.id, 300) + + assert isolated_db.delete_task(running.id) is False + assert isolated_db.get_task(running.id).status.value == "running" + assert isolated_db.list_pipeline_target_reservations( + repository="owner/repo", stage="review")[0]["task_id"] == running.id + + +def test_running_reserved_task_cannot_be_reset_until_provider_stops(isolated_db): + isolated_db.create_task(TaskCreate(prompt="protected reset")) + running = isolated_db.get_next_runnable() + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, running.id, 300) + + assert isolated_db.reset_task(running.id) is False + assert isolated_db.get_task(running.id).status.value == "running" + isolated_db.release_pipeline_target_reservations(running.id) + assert isolated_db.reset_task(running.id) is True + + +def test_reset_before_preflight_fences_old_attempt_from_new_claim(isolated_db): + isolated_db.create_task(TaskCreate(prompt="slow preflight")) + old_attempt = isolated_db.get_next_runnable() + old_started_at = old_attempt.started_at + + # Reset is still allowed before election owns a target, but the delayed + # preflight Future must not reserve or finalize the replacement attempt. + assert isolated_db.reset_task(old_attempt.id) is True + new_attempt = isolated_db.get_next_runnable() + assert new_attempt.started_at != old_started_at + + assert isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, old_attempt.id, 300, + task_started_at=old_started_at.astimezone(timezone.utc).isoformat(), + ownership_kind="headless", + ) is None + new_reservation = _reserve_attempt(isolated_db, new_attempt) + assert new_reservation is not None + + assert isolated_db.mark_completed( + old_attempt.id, "stale future", + expected_started_at=old_started_at, + ) is False + fresh = isolated_db.get_task(new_attempt.id) + assert fresh.status.value == "running" + assert fresh.started_at == new_attempt.started_at + assert isolated_db.list_pipeline_target_reservations()[0]["token"] == \ + new_reservation["token"] + + +@pytest.mark.parametrize("transition", [ + "failed", "rate_limited", "deferred", "cancelled", "attempt_failed", +]) +def test_every_attempt_exit_releases_target_reservation(isolated_db, transition): + task = isolated_db.create_task(TaskCreate(prompt=f"exit {transition}")) + running = isolated_db.get_next_runnable() + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, running.id, 300) + retry_at = datetime.now(timezone.utc) + timedelta(minutes=5) + + if transition == "failed": + isolated_db.mark_failed(running.id, "failed") + elif transition == "rate_limited": + isolated_db.mark_rate_limited(running.id, retry_at, "limited") + elif transition == "deferred": + isolated_db.defer_task(running.id, retry_at, "deferred") + elif transition == "cancelled": + isolated_db.mark_cancelled(running.id, "cancelled") + else: + assert isolated_db.fail_running_attempt( + running.id, running.started_at, "crashed") is True + + assert isolated_db.list_pipeline_target_reservations( + repository="owner/repo", stage="review") == [] + + +def test_worker_outer_exception_path_releases_target( + isolated_db, monkeypatch): + task = isolated_db.create_task(TaskCreate(prompt="attempt")) + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, task.id, 300) + monkeypatch.setattr( + worker, "_execute_task_inner", + lambda *_args, **_kwargs: (_ for _ in ()).throw(RuntimeError("boom"))) + monkeypatch.setattr(worker, "_recur_after_run", lambda _task: None) + monkeypatch.setattr(workflows, "sync_task", lambda _task_id: None) + monkeypatch.setattr(workflows, "advance_linked_task", lambda _task_id: None) + + with pytest.raises(RuntimeError, match="boom"): + worker.execute_task(task) + + assert isolated_db.list_pipeline_target_reservations( + repository="owner/repo", stage="review") == [] + + +def test_poll_exception_stops_provider_before_terminal_reservation_release( + isolated_db, monkeypatch): + task = isolated_db.create_task(TaskCreate(prompt="owned provider")) + task = isolated_db.get_next_runnable() + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, task.id, 300) + events = [] + + class Pipe: + def readline(self): + return "" + + def close(self): + return None + + class Process: + stdin = None + stdout = Pipe() + stderr = Pipe() + returncode = None + + def wait(self, timeout=None): + raise subprocess.TimeoutExpired(["provider"], timeout) + + def kill(self): + events.append("kill") + + class Tree: + process = Process() + + def terminate(self): + events.append("terminate") + + def close(self): + events.append("close") + + monkeypatch.setattr(worker.OwnedProcess, "start", lambda *_args, **_kwargs: Tree()) + monkeypatch.setattr(worker, "build_cmd", lambda *_args, **_kwargs: [sys.executable]) + monkeypatch.setattr(worker, "load_providers", lambda: {}) + monkeypatch.setattr(worker, "get_provider_env", lambda _provider: os.environ.copy()) + monkeypatch.setattr( + isolated_db, "is_cancel_requested", + lambda _task_id: (_ for _ in ()).throw(RuntimeError("poll failed"))) + monkeypatch.setattr(workflows, "sync_task", lambda _task_id: None) + monkeypatch.setattr(workflows, "advance_linked_task", lambda _task_id: None) + + with pytest.raises(RuntimeError, match="poll failed") as caught: + worker.execute_task(task) + + assert events[:2] == ["terminate", "close"] + assert isolated_db.list_pipeline_target_reservations( + repository="owner/repo", stage="review") + assert worker._fail_stuck(task, caught.value) is True + assert isolated_db.list_pipeline_target_reservations( + repository="owner/repo", stage="review") == [] + + +def test_target_heartbeat_runs_independently_of_provider_poll( + isolated_db, monkeypatch): + isolated_db.create_task(TaskCreate(prompt="long admission")) + running = isolated_db.get_next_runnable() + reservation = isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, running.id, 300, + task_started_at=running.started_at.astimezone(timezone.utc).isoformat(), + ownership_kind="headless") + real_renew = isolated_db.renew_pipeline_target_reservation + renewals = [] + + def observed_renew(*args, **kwargs): + renewals.append(time.monotonic()) + return real_renew(*args, **kwargs) + + monkeypatch.setattr( + isolated_db, "renew_pipeline_target_reservation", observed_renew) + heartbeat = worker._pipeline_target_heartbeater( + reservation, running.id, + task_started_at=running.started_at.astimezone(timezone.utc).isoformat(), + ownership_kind="headless", ttl_seconds=300, interval_seconds=0.01) + heartbeat.start() + time.sleep(0.045) + heartbeat.stop() + isolated_db.mark_failed(running.id, "done") + + assert len(renewals) >= 2 + + +def test_quarantine_heartbeat_does_not_wait_for_cleanup_backoff( + isolated_db, monkeypatch): + isolated_db.create_task(TaskCreate(prompt="quarantined")) + running = isolated_db.get_next_runnable() + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, running.id, 300) + recoveries = {} + error = worker.ProviderOwnershipError("uncertain", lambda: "still live") + worker._queue_stuck_recovery(recoveries, running, "lane", error) + item = next(iter(recoveries.values())) + item["retry_at"] = 10_000 + renewals = [] + monkeypatch.setattr( + worker, "_renew_quarantined_target", + lambda task_id: renewals.append(task_id) or 1, + ) + + worker._drain_stuck_recoveries(recoveries, now=100) + + assert renewals == [running.id] + assert recoveries + assert worker._occupied_worker_slots({}, recoveries) == 1 + + +def test_reservation_schema_has_versioned_migration(isolated_db): + with sqlite3.connect(isolated_db.DB_PATH) as conn: + table = conn.execute( + "SELECT name FROM sqlite_master WHERE type='table' " + "AND name='pipeline_target_reservations'" + ).fetchone() + versions = {row[0] for row in conn.execute( + "SELECT version FROM schema_migrations WHERE version IN (?, ?, ?, ?)", + ( + isolated_db.PIPELINE_TARGET_RESERVATION_SCHEMA_VERSION, + isolated_db.PIPELINE_TARGET_ATTEMPT_SCHEMA_VERSION, + isolated_db.PIPELINE_TARGET_OWNERSHIP_SCHEMA_VERSION, + isolated_db.PIPELINE_TARGET_HERDR_SCHEMA_VERSION, + ), + )} + columns = {row[1] for row in conn.execute( + "PRAGMA table_info(pipeline_target_reservations)")} + assert table == ("pipeline_target_reservations",) + assert versions == { + isolated_db.PIPELINE_TARGET_RESERVATION_SCHEMA_VERSION, + isolated_db.PIPELINE_TARGET_ATTEMPT_SCHEMA_VERSION, + isolated_db.PIPELINE_TARGET_OWNERSHIP_SCHEMA_VERSION, + isolated_db.PIPELINE_TARGET_HERDR_SCHEMA_VERSION, + } + assert { + "task_started_at", "ownership_kind", "herdr_session_state", + "herdr_pane_id", "herdr_tab_id", "herdr_workspace_id", + } <= columns + + +def test_review_election_reserves_then_selects_next_candidate( + isolated_db, monkeypatch): + config = { + "repository": "owner/repo", "review_completion_gate": "target-v1", + "fallback_handoff": "target-v1", "review_lease_seconds": 300, + "target_reservation_ttl_seconds": 300, + } + health = _health(_candidate(10, HEAD_A), _candidate(11, HEAD_B)) + monkeypatch.setenv("PP_PIPELINE_REPLICAS", "2") + attempts = [] + for index in range(4): + isolated_db.create_task(TaskCreate(prompt=f"replica {index}")) + attempts.append(isolated_db.get_next_runnable()) + + _set_replica_attempt_env(monkeypatch, attempts[0]) + first, first_reservation, first_health = pipelinectl._elect_review_candidate( + config, health, health["review_candidates"]) + _set_replica_attempt_env(monkeypatch, attempts[1]) + second, second_reservation, second_health = pipelinectl._elect_review_candidate( + config, health, health["review_candidates"]) + isolated_db.release_pipeline_target_reservations(attempts[0].id) + repeated, repeated_reservation, _ = pipelinectl._elect_review_candidate( + config, health, health["review_candidates"]) + _set_replica_attempt_env(monkeypatch, attempts[2]) + replacement_first, _, _ = pipelinectl._elect_review_candidate( + config, health, health["review_candidates"]) + _set_replica_attempt_env(monkeypatch, attempts[3]) + exhausted, _, _ = pipelinectl._elect_review_candidate( + config, health, health["review_candidates"]) + + assert first["number"] == 10 + assert first_reservation["task_id"] == attempts[0].id + assert first_health["review_candidates"][0]["number"] == 10 + assert second["number"] == 11 + assert second_reservation["task_id"] == attempts[1].id + assert second_health["review_candidates"][0]["number"] == 11 + assert repeated["number"] == 11 + assert repeated_reservation["token"] == second_reservation["token"] + assert replacement_first["number"] == 10 + assert exhausted is None + + +def test_expired_owned_reservation_fails_without_retargeting( + isolated_db, monkeypatch): + config = { + "repository": "owner/repo", "review_completion_gate": "target-v1", + "fallback_handoff": "target-v1", "review_lease_seconds": 300, + "target_reservation_ttl_seconds": 300, + } + health = _health(_candidate(10, HEAD_A), _candidate(11, HEAD_B)) + monkeypatch.setenv("PP_PIPELINE_REPLICAS", "2") + isolated_db.create_task(TaskCreate(prompt="owner")) + owner = isolated_db.get_next_runnable() + _set_replica_attempt_env(monkeypatch, owner) + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, owner.id, 300, + task_started_at=owner.started_at.astimezone(timezone.utc).isoformat(), + ownership_kind="headless") + real_reserve = isolated_db.reserve_pipeline_target + first_call = True + + def steal_before_renew(*args, **kwargs): + nonlocal first_call + if first_call: + first_call = False + isolated_db.release_pipeline_target_reservations(owner.id) + real_reserve("owner/repo", "review", 10, HEAD_A, 999, 300) + return None + return real_reserve(*args, **kwargs) + + monkeypatch.setattr(isolated_db, "reserve_pipeline_target", steal_before_renew) + + with pytest.raises( + pipelinectl.PipelineError, match="reservation was lost"): + pipelinectl._elect_review_candidate( + config, health, health["review_candidates"]) + assert isolated_db.list_pipeline_target_reservations( + repository="owner/repo")[0]["task_id"] == 999 + + +def test_repeated_next_cannot_retarget_when_owned_target_left_queue( + isolated_db, monkeypatch): + config = { + "repository": "owner/repo", "review_completion_gate": "target-v1", + "fallback_handoff": "target-v1", "review_lease_seconds": 300, + "target_reservation_ttl_seconds": 300, + } + monkeypatch.setenv("PP_PIPELINE_REPLICAS", "2") + isolated_db.create_task(TaskCreate(prompt="owner")) + owner = isolated_db.get_next_runnable() + _set_replica_attempt_env(monkeypatch, owner) + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, owner.id, 300, + task_started_at=owner.started_at.astimezone(timezone.utc).isoformat(), + ownership_kind="headless") + changed = _health(_candidate(11, HEAD_B)) + + with pytest.raises(pipelinectl.PipelineError, match="no longer eligible"): + pipelinectl._elect_review_candidate( + config, changed, changed["review_candidates"]) + + owned = isolated_db.list_pipeline_target_reservations( + repository="owner/repo") + assert [(item["number"], item["head"]) for item in owned] == [(10, HEAD_A)] + + +def test_replicated_lease_without_reservation_is_rejected(monkeypatch): + monkeypatch.setenv("PP_PIPELINE_REPLICAS", "2") + monkeypatch.setenv("PP_TASK_ID", "101") + + with pytest.raises(pipelinectl.PipelineError, match="no target reservation"): + pipelinectl._validate_lease_reservation({ + "stage": "review", "repository": "owner/repo", "number": 10, + "head": HEAD_A, "pipeline_replicas": 2, + }, {"repository": "owner/repo"}) + + +def test_fallback_gate_requires_live_exact_task_reservation( + isolated_db, monkeypatch, tmp_path): + monkeypatch.setenv("PP_PIPELINE_REPLICAS", "2") + monkeypatch.setenv("PP_PIPELINE_LEASE_KEY_FILE", str(tmp_path / "lease.key")) + config = { + "repository": "owner/repo", "trusted_account": "owner", + "health_command": ["health"], "fallback_handoff": "target-v1", + "review_completion_gate": "target-v1", "review_lease_seconds": 300, + "target_reservation_ttl_seconds": 300, + } + target = _candidate(10, HEAD_A) + health = _health(target) + isolated_db.create_task(TaskCreate(prompt="gate owner")) + owner = isolated_db.get_next_runnable() + _set_replica_attempt_env(monkeypatch, owner) + reservation = isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, owner.id, 300, + task_started_at=owner.started_at.astimezone(timezone.utc).isoformat(), + ownership_kind="headless") + envelope = fallback_handoff.create( + config, health, "review", target, "full review", + target_reservation=reservation) + monkeypatch.setattr(pipelinectl, "run_health", lambda *_args, **_kwargs: health) + monkeypatch.setattr(pipelinectl, "ensure_identity", lambda *_args: None) + + validated = fallback_handoff.gate( + object(), config, "review", envelope["handoff"]["lease"]) + assert validated["action"] == "validated" + + isolated_db.release_pipeline_target_reservations(owner.id) + with pytest.raises(pipelinectl.PipelineError, match="expired or was released"): + fallback_handoff.gate( + object(), config, "review", envelope["handoff"]["lease"]) + + +def test_replica_working_dirs_are_required_existing_and_distinct(tmp_path): + first = tmp_path / "review-1" + second = tmp_path / "review-2" + first.mkdir() + second.mkdir() + queue = {"series_contains": "Project - REVIEW", "replicas": 2} + valid = pipeline_insights._queue_replica_status(queue, [ + _replica_series(1, first), _replica_series(2, second), + ]) + duplicate = pipeline_insights._queue_replica_status(queue, [ + _replica_series(1, first), _replica_series(2, first), + ]) + missing = pipeline_insights._queue_replica_status(queue, [ + _replica_series(1, first), _replica_series(2, tmp_path / "missing"), + ]) + + assert valid["valid"] is True + assert duplicate["valid"] is False + assert any("один working_dir" in issue for issue in duplicate["issues"]) + assert missing["valid"] is False + assert any("не существует" in issue for issue in missing["issues"]) + + +def test_replica_working_dirs_use_filesystem_identity_not_path_spelling( + monkeypatch, tmp_path): + first = tmp_path / "Review" + second = tmp_path / "review-alias" + first.mkdir() + second.mkdir() + monkeypatch.setattr( + pipeline_insights, "_local_directory_identity", + lambda _path: (7, 42), + ) + + status = pipeline_insights._queue_replica_status( + {"series_contains": "Project - REVIEW", "replicas": 2}, + [_replica_series(1, first), _replica_series(2, second)], + ) + + assert status["valid"] is False + assert any("один working_dir" in issue for issue in status["issues"]) + + +@pytest.mark.parametrize(("field", "value", "message"), [ + ("worktree", True, "worktree"), + ("detached", True, "detached"), + ("herdr_target", "shared-user-tab", "herdr_target"), +]) +def test_replica_status_rejects_unowned_active_occurrence_lifetime( + tmp_path, field, value, message): + first = tmp_path / "review-1" + second = tmp_path / "review-2" + first.mkdir() + second.mkdir() + replicas = [_replica_series(1, first), _replica_series(2, second)] + replicas[0][field] = value + queue = {"series_contains": "Project - REVIEW", "replicas": 2} + + status = pipeline_insights._queue_replica_status(queue, replicas) + projection = pipeline_insights._queue_replica_projection( + queue, replicas, capacity=1) + + assert status["valid"] is False + assert any(message in issue for issue in status["issues"]) + assert projection["parallel_capacity"] == 0 + + +def test_series_projection_exposes_active_occurrence_lifetime_flags( + isolated_db, tmp_path): + created = isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW unsafe", working_dir=str(tmp_path), + recurrence="15m", worktree=True, detached=True, + herdr_target="shared-user-tab", + )) + + series = isolated_db.get_series(created.series_id) + + assert series["worktree"] is True + assert series["detached"] is True + assert series["herdr_target"] == "shared-user-tab" + + +def test_replica_projection_keeps_per_run_capacity_separate(tmp_path): + first = tmp_path / "review-1" + second = tmp_path / "review-2" + first.mkdir() + second.mkdir() + queue = {"series_contains": "Project - REVIEW", "replicas": 2} + projection = pipeline_insights._queue_replica_projection(queue, [ + _replica_series(1, first), _replica_series(2, second), + ], capacity=2) + + assert projection["replica_count"] == 2 + assert projection["replicas_active"] == 2 + assert projection["parallel_capacity"] == 4 + assert len(projection["series_replicas"]) == 2 + + +def test_replica_projection_has_no_capacity_without_active_series(tmp_path): + first = tmp_path / "review-1" + second = tmp_path / "review-2" + first.mkdir() + second.mkdir() + queue = {"series_contains": "Project - REVIEW", "replicas": 2} + projection = pipeline_insights._queue_replica_projection(queue, [ + _replica_series(1, first, paused=True), + _replica_series(2, second, broken=True), + ], capacity=2) + + recommendation = pipeline_insights._replica_recommendation( + queue, 8, projection, 8) + assert projection["replicas_active"] == 0 + assert projection["parallel_capacity"] == 0 + assert pipeline_insights._replica_runs_needed(8, projection) is None + assert recommendation["eta_hours"] is None + assert recommendation["throughput_per_hour"] == 0 + + +@pytest.mark.parametrize("series", [ + [], + [{"paused": True}], + [{"broken": True}], +]) +def test_legacy_single_series_has_zero_capacity_when_not_active( + tmp_path, series): + values = [] + for index, overrides in enumerate(series, start=1): + item = _replica_series(index, tmp_path, **{ + key: value for key, value in overrides.items() + if key in {"paused", "broken"} + }) + values.append(item) + queue = {"series_contains": "Project - REVIEW"} + + projection = pipeline_insights._queue_replica_projection( + queue, values, capacity=1) + recommendation = pipeline_insights._replica_recommendation( + queue, 5, projection, 8) + + assert projection["replica_count"] == 1 + assert projection["replicas_active"] == 0 + assert projection["parallel_capacity"] == 0 + assert pipeline_insights._replica_runs_needed(5, projection) is None + assert recommendation["eta_hours"] is None + assert "нет активных" in recommendation["recommendation"] + + +def test_invalid_replica_set_has_zero_capacity_and_is_hard_bottleneck(tmp_path): + first = tmp_path / "review-1" + first.mkdir() + queue = {"series_contains": "Project - REVIEW", "replicas": 2} + projection = pipeline_insights._queue_replica_projection( + queue, [_replica_series(1, first)], capacity=2) + recommendation = pipeline_insights._replica_recommendation( + queue, 8, projection, 8) + stalled = {"backlog": 8, "eta_hours": None, "runs_needed": None} + healthy = {"backlog": 1, "eta_hours": 1, "runs_needed": 1} + + assert projection["replica_status"]["valid"] is False + assert projection["parallel_capacity"] == 0 + assert "конфигурация" in recommendation["recommendation"] + assert pipeline_insights._bottleneck_rank(stalled) == float("inf") + assert max([healthy, stalled], key=pipeline_insights._bottleneck_rank) is stalled + + +def test_cached_dashboard_health_is_red_for_invalid_replica_set( + isolated_db, monkeypatch, tmp_path): + first = tmp_path / "review-1" + first.mkdir() + profile = { + "target_clear_hours": 8, + "queues": [{ + "id": "review", "title": "Review", + "series_contains": "Project - REVIEW", "replicas": 2, + }], + } + saved = { + "profile_id": "p", "backlog_total": 4, + "queues": [{"id": "review", "title": "Review", "backlog": 4}], + "history": {"5h": { + "complete": True, "runs": {}, "backlog_delta": -1, + "exited": 1, "churn_items": 0, + }}, + } + series = [_replica_series(1, first)] + monkeypatch.setattr( + pipeline_insights, "_pipeline_runtime", + lambda *_args, **_kwargs: {"required": True, "state": "online", "stalled": []}, + ) + + result = pipeline_insights._refresh_local_state( + saved, profile, series, source="durable", generated_at=time.time(), + entry_epoch=0, entry_revision=1, current_epoch=0) + + assert result["health"]["state"] == "red" + assert result["health"]["label"] == "невалидная конфигурация реплик" + assert "review" in result["health"]["reason"] + + +def test_replica_set_is_invalid_when_worker_has_too_few_slots( + monkeypatch, tmp_path): + first = tmp_path / "review-1" + second = tmp_path / "review-2" + first.mkdir() + second.mkdir() + monkeypatch.setattr(pipeline_insights, "CONCURRENCY", 1) + queue = {"series_contains": "Project - REVIEW", "replicas": 2} + + projection = pipeline_insights._queue_replica_projection(queue, [ + _replica_series(1, first), _replica_series(2, second), + ], capacity=1) + + assert projection["parallel_capacity"] == 0 + assert any("PP_CONCURRENCY" in issue + for issue in projection["replica_status"]["issues"]) + + +def test_legacy_queue_still_uses_only_first_matching_series(tmp_path): + series = [ + _replica_series(1, tmp_path, title="Project - REVIEW first"), + _replica_series(2, tmp_path, title="Project - REVIEW accidental duplicate"), + ] + + assert [item["id"] for item in pipeline_insights._series_replicas_for_queue( + {"series_contains": "Project - REVIEW"}, series)] == [1] + + +def test_group_wake_latches_successors_for_running_replicas( + isolated_db, monkeypatch, tmp_path): + first = isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW 1", working_dir=str(tmp_path), recurrence="4h")) + second = isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW 2", working_dir=str(tmp_path), recurrence="4h")) + assert isolated_db.get_next_runnable().id == first.id + assert isolated_db.get_next_runnable().id == second.id + monkeypatch.setattr( + isolated_db, "_pipeline_cache_guard_matches", lambda _conn, _guard: True) + + woken = isolated_db.wake_series_group_once( + [first.series_id, second.series_id], "wake:review", "snapshot-a", + cache_guard={}) + + assert woken == [first.series_id, second.series_id] + assert isolated_db.get_setting( + f"pipeline_series_wake_intent:v1:{first.series_id}") == "1" + assert isolated_db.get_setting( + f"pipeline_series_wake_intent:v1:{second.series_id}") == "1" + assert isolated_db.wake_series_group_once( + [first.series_id, second.series_id], "wake:review", "snapshot-a", + cache_guard={}) == [] + + +def test_group_wake_recreates_replica_in_terminal_recurrence_gap( + isolated_db, monkeypatch, tmp_path): + first = isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW 1", working_dir=str(tmp_path), recurrence="4h")) + second = isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW 2", working_dir=str(tmp_path), recurrence="4h")) + running = isolated_db.get_next_runnable() + assert running.id == first.id + assert isolated_db.mark_completed(first.id, "done") + monkeypatch.setattr( + isolated_db, "_pipeline_cache_guard_matches", lambda _conn, _guard: True) + + woken = isolated_db.wake_series_group_once( + [first.series_id, second.series_id], "wake:review", "snapshot-gap", + cache_guard={}) + + assert woken == [first.series_id, second.series_id] + first_live = [ + task for task in isolated_db.list_tasks() + if task.series_id == first.series_id and task.status.value == "pending" + ] + assert len(first_live) == 1 + assert first_live[0].scheduled_at <= datetime.now(timezone.utc) + assert isolated_db.wake_series_group_once( + [first.series_id, second.series_id], "wake:review", "snapshot-gap", + cache_guard={}) == [] + + +def test_group_wake_does_not_restart_cancelled_replica( + isolated_db, monkeypatch, tmp_path): + cancelled = isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW 1", working_dir=str(tmp_path), recurrence="4h")) + pending = isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW 2", working_dir=str(tmp_path), recurrence="4h")) + assert isolated_db.cancel_task(cancelled.id) + monkeypatch.setattr( + isolated_db, "_pipeline_cache_guard_matches", lambda _conn, _guard: True) + + woken = isolated_db.wake_series_group_once( + [cancelled.series_id, pending.series_id], "wake:review", "snapshot-cancel", + cache_guard={}) + + assert woken == [pending.series_id] + cancelled_tasks = [ + task for task in isolated_db.list_tasks() + if task.series_id == cancelled.series_id + ] + assert len(cancelled_tasks) == 1 + assert cancelled_tasks[0].id == cancelled.id + assert cancelled_tasks[0].status.value == "cancelled" + assert isolated_db.get_setting("wake:review") == "snapshot-cancel" + + +def test_adaptive_cadence_updates_every_replica(isolated_db, tmp_path): + first = isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW 1", working_dir=str(tmp_path), recurrence="30m")) + second = isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW 2", working_dir=str(tmp_path), recurrence="30m")) + matching = [isolated_db.get_series(first.series_id), + isolated_db.get_series(second.series_id)] + queue = {"adaptive_cadence": { + "idle_recurrence": "30m", "busy_recurrence": "10m", + "backlog_above": 0, + }} + + status = pipeline_insights._reconcile_adaptive_cadence(queue, matching, 5) + + assert status["series_count"] == 2 + assert status["effective_recurrence"] == "10m" + assert isolated_db.get_series(first.series_id)["effective_recurrence"] == "10m" + assert isolated_db.get_series(second.series_id)["effective_recurrence"] == "10m" + + +def test_run_now_and_event_wake_address_every_replica(monkeypatch, tmp_path): + first = tmp_path / "review-1" + second = tmp_path / "review-2" + first.mkdir() + second.mkdir() + queue = { + "id": "review", "series_contains": "Project - REVIEW", "replicas": 2, + "wake_when": {"field": "review_candidates"}, + } + profile = { + "repository": "owner/repo", + "priority_control": {"trusted_account": "owner"}, + "queues": [queue], + } + series = [_replica_series(1, first), _replica_series(2, second)] + monkeypatch.setattr(pipeline_insights, "_profiles", lambda: {"p": profile}) + monkeypatch.setattr( + pipeline_insights, "_gh_api_json", + lambda args, _input=None: {"login": "owner"} if args == ["user"] else + {"state": "open", "pull_request": {"url": "x"}, "labels": []}) + run_now = [] + monkeypatch.setattr( + pipeline_insights.db, "series_action", + lambda series_id, action: run_now.append((series_id, action)) or True) + + result = pipeline_insights.set_item_priority( + "p", "review", "pr", 10, "auto", True, series) + + assert result["series_woken_ids"] == [1, 2] + assert run_now == [(1, "run_now"), (2, "run_now")] + + cache = { + "complete": True, "stale": False, + "token": { + "profile_hash": pipeline_insights._profile_fingerprint(profile), + "epoch": 0, "revision": 1, "generated_at": time.time(), + }, + } + group_wakes = [] + monkeypatch.setattr(pipeline_insights.db, "is_paused", lambda: False) + monkeypatch.setattr( + pipeline_insights.db, "wake_series_group_once", + lambda ids, key, fingerprint, **_kwargs: + group_wakes.append((ids, key, fingerprint)) or ids) + + assert pipeline_insights._wake_ready_queues( + "p", profile, {"cache": cache, "diagnostics": { + "review_candidates": [{"number": 10}], + }}, series) == ["review"] + assert group_wakes[0][0] == [1, 2] + + +def test_execution_route_passes_replica_identity_to_preflight( + isolated_db, monkeypatch, tmp_path): + first_dir = tmp_path / "review-1" + second_dir = tmp_path / "review-2" + first_dir.mkdir() + second_dir.mkdir() + first = isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW 1", working_dir=str(first_dir), recurrence="15m")) + isolated_db.create_task(TaskCreate( + prompt="Project - REVIEW 2", working_dir=str(second_dir), recurrence="15m")) + task = isolated_db.get_next_runnable() + assert task.id == first.id + queue = { + "id": "review", "series_contains": "Project - REVIEW", "replicas": 2, + "execution": {"mode": "auto", "command": ["pipelinectl", "next", "review"]}, + } + monkeypatch.setattr( + pipeline_insights, "_matching_queue", lambda _task: ("p", {}, queue)) + monkeypatch.setattr(pipeline_insights, "_tool_available", lambda *_args: (True, "")) + monkeypatch.setattr(pipeline_insights, "load_providers", lambda: {}) + captured = {} + + def preflight(_execution, _command, _working_dir, *, env_extra=None): + captured.update(env_extra or {}) + return {"action": "wait", "reason": "all targets reserved"} + + monkeypatch.setattr(pipeline_insights, "_tool_preflight", preflight) + route = pipeline_insights.execution_route(task, "skill", str(first_dir)) + + assert route["action"] == "complete_empty" + assert captured == { + "PP_TASK_ID": str(task.id), + "PP_TASK_STARTED_AT": task.started_at.astimezone(timezone.utc).isoformat(), + "PP_PIPELINE_REPLICAS": "2", + "PP_PROVIDER_OWNERSHIP_KIND": "headless", + "PP_DATA_DIR": str(isolated_db.DB_PATH.resolve().parent), + "PP_PIPELINE_LEASE_KEY_FILE": str( + (isolated_db.DB_PATH.resolve().parent / "pipeline-lease.key").resolve()), + } + + +def test_worker_propagates_replica_mode_into_herdr_provider(monkeypatch, tmp_path): + data_dir = tmp_path / "scheduler-data" + started_at = datetime(2026, 9, 17, 12, 0, tzinfo=timezone.utc) + started_iso = started_at.isoformat() + task = SimpleNamespace( + id=41, series_id=7, prompt="Project - REVIEW", + provider="test-provider", working_dir=None, machine=None, + keep_pane=False, started_at=started_at, + ) + monkeypatch.setattr(pipeline_insights, "dispatch_gate", lambda _task: None) + monkeypatch.setattr( + pipeline_insights, "execution_route", + lambda *_args, **_kwargs: { + "action": "prompt", "mode": "tool", "prompt": "run", + "profile_id": "project", "queue_id": "review", + "pipeline_replicas": 2, + "pipeline_data_dir": str(data_dir), + "pipeline_lease_key_file": str(data_dir / "pipeline-lease.key"), + "pipeline_task_started_at": started_iso, + "pipeline_provider_ownership_kind": "herdr", + "pipeline_target_reservation": { + "task_id": 41, "task_started_at": started_iso, + "ownership_kind": "herdr", "token": "1" * 32, + }, + }, + ) + heartbeat_calls = [] + + class Heartbeat: + reservation = {"token": "1" * 32} + + def start(self): + heartbeat_calls.append("start") + return self + + def __call__(self, *, force=False): + heartbeat_calls.append(("touch", force)) + + monkeypatch.setattr( + worker, "_pipeline_target_heartbeater", + lambda *_args, **_kwargs: Heartbeat(), + ) + monkeypatch.setattr( + worker, "load_providers", + lambda: {"test-provider": {"executor": "herdr", "env": {"KEEP": "1"}}}, + ) + captured = {} + + def provider(_task, provider_cfg, **kwargs): + captured.update(provider_cfg["env"]) + captured["strict_owned_session"] = provider_cfg["strict_owned_session"] + kwargs["target_heartbeat"]() + + monkeypatch.setattr(worker, "_execute_herdr_task", provider) + + worker._execute_task_body(task) + + assert captured == { + "KEEP": "1", "PP_PIPELINE_REPLICAS": "2", + "PP_TASK_STARTED_AT": started_iso, + "PP_PROVIDER_OWNERSHIP_KIND": "herdr", + "PP_PIPELINE_TARGET_TOKEN": "1" * 32, + "PP_DATA_DIR": os.path.realpath(str(data_dir)), + "PP_PIPELINE_LEASE_KEY_FILE": os.path.realpath( + str(data_dir / "pipeline-lease.key")), + "strict_owned_session": True, + } + assert heartbeat_calls == ["start", ("touch", False)] + + +@pytest.mark.parametrize("unsafe_field, unsafe_value", [ + ("detached", True), ("herdr_target", "user-session"), +]) +def test_execution_route_blocks_unowned_replica_lifetimes( + monkeypatch, tmp_path, unsafe_field, unsafe_value): + first = tmp_path / "review-1" + second = tmp_path / "review-2" + first.mkdir() + second.mkdir() + queue = { + "id": "review", "series_contains": "Project - REVIEW", "replicas": 2, + "execution": {"mode": "auto", "stage": "review"}, + } + task = SimpleNamespace( + id=101, series_id=1, machine=None, worktree=False, detached=False, + herdr_target=None, + ) + setattr(task, unsafe_field, unsafe_value) + monkeypatch.setattr( + pipeline_insights, "_matching_queue", lambda _task: ("p", {}, queue)) + monkeypatch.setattr( + pipeline_insights.db, "list_series", lambda: [ + _replica_series(1, first), _replica_series(2, second), + ]) + + route = pipeline_insights.execution_route(task, "skill", str(first)) + + assert route["action"] == "block" + assert unsafe_field in route["reason"] + + +def test_strict_herdr_wait_error_closes_owned_session(monkeypatch): + calls = [] + + def fake_run(args, host=None, timeout=None): + calls.append(args) + if args[:2] == ["tab", "create"]: + return 0, {"result": { + "root_pane": {"pane_id": "pane-1"}, + "tab": {"tab_id": "tab-1"}, + }}, "" + if args[:2] == ["agent", "start"]: + return 0, {"result": {"agent": {"agent_status": "idle"}}}, "" + if args[:2] == ["agent", "prompt"]: + return 1, {"error": {"code": "timeout"}}, "still working" + if args[:2] == ["agent", "wait"]: + return 1, {"error": {"code": "rpc_error"}}, "rpc failed" + if args[:2] == ["tab", "close"]: + return 0, {}, "" + if args[:2] == ["tab", "list"]: + return 0, {"result": {"tabs": []}}, "" + if args[:2] == ["agent", "list"]: + return 0, {"result": {"agents": []}}, "" + raise AssertionError(args) + + task = SimpleNamespace( + id=1, herdr_target=None, model=None, effort=None, session_id=None, + skip_permissions=False, detached=False, working_dir=".", worktree=False, + ) + monkeypatch.setattr(herdr_exec, "_ensure_server", lambda _host: None) + monkeypatch.setattr(herdr_exec, "_close_stale_tabs", lambda *_args: None) + monkeypatch.setattr(herdr_exec, "_run", fake_run) + monkeypatch.setattr(herdr_exec, "guard_enabled", lambda *_args: False) + + outcome = herdr_exec.run_in_herdr( + task, {"kind": "codex", "strict_owned_session": True}, + prompt_override="review", + ) + + assert outcome["ok"] is False + assert outcome.get("ownership_uncertain") is not True + assert any(args[:2] == ["tab", "close"] for args in calls) + + +def test_strict_herdr_stale_cleanup_failure_is_ownership_uncertain(monkeypatch): + task = SimpleNamespace( + id=1, herdr_target=None, model=None, effort=None, session_id=None, + skip_permissions=False, detached=False, working_dir=".", worktree=False, + ) + attempts = [] + monkeypatch.setattr(herdr_exec, "_ensure_server", lambda _host: None) + + def fail_stale(*_args): + attempts.append("close") + raise herdr_exec.HerdrError("stale provider still live") + + monkeypatch.setattr(herdr_exec, "_close_stale_tabs", fail_stale) + + outcome = herdr_exec.run_in_herdr( + task, {"kind": "codex", "strict_owned_session": True}, + prompt_override="review", + ) + + assert outcome["ownership_uncertain"] is True + assert callable(outcome["_ownership_cleanup"]) + assert "stale provider still live" in outcome["error"] + assert outcome["_ownership_cleanup"]() + assert len(attempts) == 2 + + +def test_strict_herdr_cleanup_exception_is_ownership_uncertain(monkeypatch): + task = SimpleNamespace( + id=1, herdr_target=None, model=None, effort=None, session_id=None, + skip_permissions=False, detached=False, working_dir=".", worktree=False, + ) + + def fake_run(args, host=None, timeout=None): + if args[:2] == ["tab", "create"]: + return 0, {"result": { + "root_pane": {"pane_id": "pane-1"}, + "tab": {"tab_id": "tab-1"}, + }}, "" + if args[:2] == ["agent", "start"]: + return 1, {"error": {"code": "start_failed"}}, "start failed" + raise AssertionError(args) + + monkeypatch.setattr(herdr_exec, "_ensure_server", lambda _host: None) + monkeypatch.setattr(herdr_exec, "_close_stale_tabs", lambda *_args: None) + monkeypatch.setattr(herdr_exec, "_run", fake_run) + monkeypatch.setattr( + herdr_exec, "_close_owned_session", + lambda *_args, **_kwargs: (_ for _ in ()).throw(OSError("close exploded")), + ) + + outcome = herdr_exec.run_in_herdr( + task, {"kind": "codex", "strict_owned_session": True}, + prompt_override="review", + ) + + assert outcome["ownership_uncertain"] is True + assert callable(outcome["_ownership_cleanup"]) + assert "close exploded" in outcome["error"] + + +def test_uncertain_provider_cleanup_keeps_reservation_until_confirmed( + isolated_db): + task = isolated_db.create_task(TaskCreate(prompt="uncertain provider")) + task = isolated_db.get_next_runnable() + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, task.id, 300) + attempts = iter(["still live", ""]) + error = worker.ProviderOwnershipError( + "herdr close not confirmed", lambda: next(attempts)) + + assert worker._fail_stuck(task, error) is False + assert isolated_db.get_task(task.id).status.value == "running" + assert isolated_db.list_pipeline_target_reservations( + repository="owner/repo", stage="review") + + assert worker._fail_stuck(task, error) is True + assert isolated_db.get_task(task.id).status.value == "failed" + assert isolated_db.list_pipeline_target_reservations( + repository="owner/repo", stage="review") == [] + + +@pytest.mark.parametrize(("terminal_intent", "expected_status"), [ + ({"ok": True, "rate_limited": False, "error": "", "output": "done"}, + "completed"), + ({"ok": False, "rate_limited": False, "cancelled": True, + "cancel_note": "cancelled", "error": ""}, "cancelled"), + ({"ok": False, "rate_limited": True, "retry_reason": "rate_limit", + "error": "429"}, "rate_limited"), + ({"ok": False, "rate_limited": False, "env_failure": "HTTP 503", + "error": "503"}, "rate_limited"), +]) +def test_terminal_intent_survives_two_close_failures_until_quarantine_cleanup( + isolated_db, monkeypatch, terminal_intent, expected_status): + isolated_db.create_task(TaskCreate(prompt="strict herdr outcome")) + task = isolated_db.get_next_runnable() + reservation = _reserve_attempt( + isolated_db, task, ownership_kind="herdr") + reservation = isolated_db.begin_pipeline_target_herdr_session(reservation) + cleanup_attempts = ["immediate close failed", "strict finally failed"] + + def cleanup(): + cleanup_attempts.append("quarantine cleanup succeeded") + return "" + + monkeypatch.setattr( + herdr_exec, "run_in_herdr", + lambda *_args, **_kwargs: { + "ok": False, "rate_limited": False, + "error": "owned session close failed twice", + "ownership_uncertain": True, + "_terminal_intent": dict(terminal_intent), + "_ownership_cleanup": cleanup, + }, + ) + + with pytest.raises(worker.ProviderOwnershipError) as caught: + worker._execute_herdr_task(task, {"strict_owned_session": True}) + + assert isolated_db.get_task(task.id).status.value == "running" + assert isolated_db.list_pipeline_target_reservations()[0]["token"] == \ + reservation["token"] + assert worker._fail_stuck(task, caught.value) is True + assert cleanup_attempts == [ + "immediate close failed", "strict finally failed", + "quarantine cleanup succeeded", + ] + assert isolated_db.get_task(task.id).status.value == expected_status + assert isolated_db.list_pipeline_target_reservations() == [] + + +def test_terminal_continuation_cas_false_keeps_exact_attempt_quarantined( + isolated_db): + isolated_db.create_task(TaskCreate(prompt="delayed CAS")) + task = isolated_db.get_next_runnable() + reservation = _reserve_attempt( + isolated_db, task, ownership_kind="herdr") + allow_commit = False + + def commit_terminal(): + if not allow_commit: + return False + return isolated_db.mark_completed( + task.id, "done", expected_started_at=task.started_at) + + error = worker.ProviderOwnershipError( + "cleanup settled", lambda: "", on_cleaned=commit_terminal) + + assert worker._fail_stuck(task, error) is False + assert isolated_db.get_task(task.id).status.value == "running" + assert isolated_db.list_pipeline_target_reservations()[0]["token"] == \ + reservation["token"] + + allow_commit = True + assert worker._fail_stuck(task, error) is True + assert isolated_db.get_task(task.id).status.value == "completed" + assert isolated_db.list_pipeline_target_reservations() == [] + + +def test_headless_success_survives_first_terminal_database_failure(isolated_db): + isolated_db.create_task(TaskCreate(prompt="headless success")) + task = isolated_db.get_next_runnable() + _reserve_attempt(isolated_db, task, ownership_kind="headless") + attempts = 0 + + def commit_success(): + nonlocal attempts + attempts += 1 + if attempts == 1: + raise sqlite3.OperationalError("database is locked") + return isolated_db.mark_completed( + task.id, "done", expected_started_at=task.started_at) + + with pytest.raises(worker.ProviderOwnershipError) as caught: + worker._commit_terminal_or_defer( + task, "headless successful outcome", commit_success) + + assert isolated_db.get_task(task.id).status.value == "running" + assert worker._fail_stuck(task, caught.value) is True + assert isolated_db.get_task(task.id).status.value == "completed" + assert attempts == 2 + + +def test_headless_cancel_survives_transient_tree_cleanup_failure(isolated_db): + isolated_db.create_task(TaskCreate(prompt="headless cancel")) + task = isolated_db.get_next_runnable() + _reserve_attempt(isolated_db, task, ownership_kind="headless") + + class Tree: + def __init__(self): + self.close_calls = 0 + self.process = SimpleNamespace(kill=lambda: None) + + def terminate(self): + return None + + def close(self): + self.close_calls += 1 + if self.close_calls == 1: + raise OSError("temporary close failure") + + tree = Tree() + worker._register_provider_tree(task.id, tree) + error = worker.ProviderOwnershipError( + "cancel cleanup pending", lambda: "", + on_cleaned=lambda: isolated_db.mark_cancelled( + task.id, "cancelled", expected_started_at=task.started_at), + ) + + assert worker._fail_stuck(task, error) is False + assert isolated_db.get_task(task.id).status.value == "running" + assert worker._fail_stuck(task, error) is True + assert isolated_db.get_task(task.id).status.value == "cancelled" + + +def test_restart_cleans_reserved_provider_before_requeue_and_release( + isolated_db, monkeypatch): + isolated_db.create_task(TaskCreate(prompt="restart cleanup")) + task = isolated_db.get_next_runnable() + _reserve_attempt(isolated_db, task, ownership_kind="herdr") + events = [] + + def cleanup_factory(recovered_task, reservation): + assert recovered_task.id == task.id + assert reservation["ownership_kind"] == "herdr" + + def cleanup(): + assert isolated_db.get_task(task.id).status.value == "running" + assert isolated_db.task_has_live_pipeline_target_reservation(task.id) + events.append("provider stopped") + return "" + + return cleanup + + monkeypatch.setattr(worker, "_recovered_provider_cleanup", cleanup_factory) + + keep, quarantines = worker._reconcile_reserved_running_tasks(set()) + + assert events == ["provider stopped"] + assert keep == set() + assert quarantines == [] + assert isolated_db.get_task(task.id).status.value == "pending" + assert isolated_db.list_pipeline_target_reservations() == [] + + +def test_restart_before_herdr_creation_requeues_without_false_quarantine( + isolated_db): + isolated_db.create_task(TaskCreate(prompt="crash before Herdr begin")) + task = isolated_db.get_next_runnable() + reservation = _reserve_attempt( + isolated_db, task, ownership_kind="herdr") + assert reservation["herdr_session_state"] == "reserved" + + keep, quarantines = worker._reconcile_reserved_running_tasks(set()) + + assert keep == set() + assert quarantines == [] + assert isolated_db.get_task(task.id).status.value == "pending" + assert isolated_db.list_pipeline_target_reservations() == [] + + +def test_restart_preserves_cancel_requested_before_worker_crash(isolated_db): + created = isolated_db.create_task(TaskCreate( + prompt="cancel before crash", recurrence="15m")) + task = isolated_db.get_next_runnable() + _reserve_attempt(isolated_db, task, ownership_kind="herdr") + wake_key = f"pipeline_series_wake_intent:v1:{created.series_id}" + isolated_db.set_setting(wake_key, "1") + assert isolated_db.request_cancel(task.id) is True + + keep, quarantines = worker._reconcile_reserved_running_tasks(set()) + + assert keep == set() + assert quarantines == [] + assert isolated_db.get_task(task.id).status.value == "cancelled" + assert isolated_db.is_cancel_requested(task.id) is False + assert isolated_db.get_setting(wake_key) is None + assert isolated_db.list_pipeline_target_reservations() == [] + + +def test_restart_closes_exact_herdr_ids_even_after_labels_are_renamed( + isolated_db, monkeypatch): + isolated_db.create_task(TaskCreate(prompt="renamed Herdr")) + task = isolated_db.get_next_runnable() + reservation = _reserve_attempt( + isolated_db, task, ownership_kind="herdr") + reservation = isolated_db.begin_pipeline_target_herdr_session(reservation) + reservation = isolated_db.bind_pipeline_target_herdr_session( + reservation, "pane-stable", "tab-stable", "workspace-stable") + events = [] + monkeypatch.setattr( + herdr_exec, "_ensure_server", lambda host: events.append(("ensure", host))) + monkeypatch.setattr( + herdr_exec, "_close_owned_session", + lambda name, args, host, **kwargs: + events.append((name, args, host, kwargs)) or "", + ) + monkeypatch.setattr( + herdr_exec, "_close_stale_tabs", + lambda *_args: pytest.fail("mutable label sweep was used"), + ) + + cleanup = worker._recovered_provider_cleanup(task, reservation) + + assert cleanup() == "" + assert events == [ + ("ensure", None), + (f"recovered task #{task.id}", + ["workspace", "close", "workspace-stable"], + None, {"pane_id": "pane-stable"}), + ] + + +def test_restart_cleanup_failure_quarantines_then_requeues_without_failure( + isolated_db, monkeypatch): + isolated_db.create_task(TaskCreate(prompt="restart retry")) + task = isolated_db.get_next_runnable() + reservation = _reserve_attempt( + isolated_db, task, ownership_kind="headless") + outcomes = iter(["provider still alive", ""]) + projections = [] + monkeypatch.setattr( + worker, "_recovered_provider_cleanup", + lambda _task, _reservation: lambda: next(outcomes), + ) + monkeypatch.setattr( + workflows, "sync_task", + lambda task_id: projections.append(("sync", task_id)), + ) + monkeypatch.setattr( + workflows, "advance_linked_task", + lambda task_id: projections.append(("advance", task_id)), + ) + + keep, quarantines = worker._reconcile_reserved_running_tasks(set()) + + assert keep == {task.id} + assert len(quarantines) == 1 + assert isolated_db.get_task(task.id).status.value == "running" + assert isolated_db.list_pipeline_target_reservations()[0]["token"] == \ + reservation["token"] + + recovered_task, error = quarantines[0] + assert worker._fail_stuck(recovered_task, error) is True + assert isolated_db.get_task(task.id).status.value == "pending" + assert isolated_db.list_pipeline_target_reservations() == [] + assert projections == [("sync", task.id), ("advance", task.id)] + + +@pytest.mark.parametrize("ownership_kind", ["headless", "herdr"]) +def test_restart_uses_persisted_ownership_kind_despite_provider_config_drift( + isolated_db, monkeypatch, ownership_kind): + provider = "now-headless" if ownership_kind == "herdr" else "now-herdr" + isolated_db.create_task(TaskCreate(prompt="config drift", provider=provider)) + task = isolated_db.get_next_runnable() + _reserve_attempt(isolated_db, task, ownership_kind=ownership_kind) + observed = [] + monkeypatch.setattr( + worker, "_recovered_provider_cleanup", + lambda _task, reservation: ( + lambda: observed.append(reservation["ownership_kind"]) or ""), + ) + + worker._reconcile_reserved_running_tasks(set()) + + assert observed == [ownership_kind] + + +def test_recovered_herdr_cleanup_starts_server_before_closing_tabs( + isolated_db, monkeypatch): + isolated_db.create_task(TaskCreate(prompt="herdr restart")) + task = isolated_db.get_next_runnable() + reservation = _reserve_attempt( + isolated_db, task, ownership_kind="herdr") + reservation = isolated_db.begin_pipeline_target_herdr_session(reservation) + events = [] + monkeypatch.setattr( + herdr_exec, "_ensure_server", + lambda host: events.append(("ensure", host)), + ) + monkeypatch.setattr( + herdr_exec, "_close_stale_tabs", + lambda task_id, host: events.append(("close", task_id, host)), + ) + + cleanup = worker._recovered_provider_cleanup(task, reservation) + + assert cleanup() == "" + assert events == [("ensure", None), ("close", task.id, None)] + + +def test_macos_orphan_scan_requires_exact_unguessable_reservation_marker( + monkeypatch): + started_at = "2026-09-17T12:00:00+00:00" + token = "a" * 32 + + class FakeOS: + name = "posix" + + @staticmethod + def listdir(_path): + raise OSError("no procfs") + + @staticmethod + def getpgid(pid): + return pid + 1000 + + @staticmethod + def getpgrp(): + return 9999 + + output = "\n".join([ + # Untrusted argv can contain the public task id and attempt, but it + # cannot predict the random reservation token. + f"101 provider --prompt PP_TASK_ID=7 PP_TASK_STARTED_AT={started_at}", + f"102 provider PP_TASK_ID=7 PP_TASK_STARTED_AT={started_at} " + f"PP_PIPELINE_TARGET_TOKEN={token}", + ]) + monkeypatch.setattr(worker, "os", FakeOS) + monkeypatch.setattr( + worker.subprocess, "run", + lambda *_args, **_kwargs: SimpleNamespace( + returncode=0, stdout=output, stderr=""), + ) + + groups, error = worker._marked_task_process_groups(7, started_at, token) + + assert error == "" + assert groups == {1102} + + +def test_registered_provider_cleanup_retry_keeps_boundary_and_reservation( + isolated_db): + isolated_db.create_task(TaskCreate(prompt="tree retry")) + task = isolated_db.get_next_runnable() + isolated_db.reserve_pipeline_target( + "owner/repo", "review", 10, HEAD_A, task.id, 300) + + class Tree: + def __init__(self): + self.terminate_calls = 0 + self.close_calls = 0 + self.process = SimpleNamespace(kill=lambda: None) + + def terminate(self): + self.terminate_calls += 1 + if self.terminate_calls == 1: + raise OSError("temporary terminate failure") + + def close(self): + self.close_calls += 1 + if self.close_calls == 1: + raise OSError("temporary close failure") + + tree = Tree() + worker._register_provider_tree(task.id, tree) + + assert worker._fail_stuck(task, RuntimeError("boom")) is False + assert isolated_db.get_task(task.id).status.value == "running" + assert isolated_db.task_has_live_pipeline_target_reservation(task.id) + assert worker._fail_stuck(task, RuntimeError("boom")) is True + assert tree.terminate_calls == 2 + assert tree.close_calls == 2 + assert isolated_db.get_task(task.id).status.value == "failed" + + +def test_duplicate_provider_registration_closes_new_tree(isolated_db): + first_events = [] + second_events = [] + + class Tree: + def __init__(self, events): + self.events = events + self.process = SimpleNamespace(kill=lambda: events.append("kill")) + + def terminate(self): + self.events.append("terminate") + + def close(self): + self.events.append("close") + + first = Tree(first_events) + second = Tree(second_events) + worker._register_provider_tree(777, first) + try: + with pytest.raises(RuntimeError, match="already owns"): + worker._register_provider_tree(777, second) + assert second_events == ["terminate", "close"] + assert first_events == [] + finally: + assert worker._close_registered_provider_tree(777) is True + + +def test_scheduler_requires_one_lane_per_replica(monkeypatch): + profile = { + "queues": [{ + "id": "review", "series_contains": "Project - REVIEW", "replicas": 2, + }], + "scheduler": {"lanes": [{"id": "one", "queues": ["review"]}]}, + } + monkeypatch.setattr(pipeline_insights, "_profiles", lambda: {"p": profile}) + with pytest.raises(ValueError, match="2 реплики"): + pipeline_insights.worker_lane_policy() + + profile["scheduler"]["lanes"].append({"id": "two", "queues": ["review"]}) + assert len(pipeline_insights.worker_lane_policy()["lanes"]) == 2 diff --git a/tests/test_process_tree.py b/tests/test_process_tree.py index 0ab4748..348f2f9 100644 --- a/tests/test_process_tree.py +++ b/tests/test_process_tree.py @@ -116,6 +116,48 @@ def test_closing_after_root_exit_kills_lingering_descendant(): tree.close() +def test_failed_boundary_close_can_be_retried_without_losing_ownership(monkeypatch): + process = SimpleNamespace(kill=lambda: None) + if os.name == "nt": + calls = [] + + def terminate(job, code): + calls.append((job, code)) + return len(calls) > 1 + + monkeypatch.setattr(process_tree, "_TerminateJobObject", terminate) + monkeypatch.setattr( + process_tree, "_windows_error", + lambda message: process_tree.ProcessTreeError(message), + ) + monkeypatch.setattr(process_tree, "_CloseHandle", lambda _job: True) + tree = OwnedProcess(process, job=99) + + with pytest.raises(process_tree.ProcessTreeError): + tree.close() + assert tree._job == 99 + tree.close() + assert tree._job is None + assert calls == [(99, 1), (99, 1)] + else: + calls = [] + + def killpg(group_id, sig): + calls.append((group_id, sig)) + if len(calls) == 1: + raise OSError("temporary kill failure") + + monkeypatch.setattr(process_tree.os, "killpg", killpg) + tree = OwnedProcess(process, group_id=4242) + + with pytest.raises(OSError, match="temporary"): + tree.close() + assert tree._group_id == 4242 + tree.close() + assert tree._group_id is None + assert calls == [(4242, process_tree.signal.SIGKILL)] * 2 + + @pytest.mark.skipif(os.name != "nt", reason="Windows Job Object setup") def test_windows_close_terminates_job_with_retained_handle_not_unrelated_process(): """Normal completion must not depend on PromptPilot owning the last handle.""" @@ -338,15 +380,18 @@ def still_running(args, host=None, timeout=None): ] -def test_owned_herdr_close_accepts_verified_missing_agent(monkeypatch): +def test_owned_herdr_close_accepts_verified_missing_exact_ids_after_rename( + monkeypatch): responses = iter([ (0, {"result": {}}, "closed"), - (1, {"error": {"code": "agent_not_found"}}, "not found"), + (0, {"result": {"tabs": []}}, "absent"), + (0, {"result": {"agents": []}}, "absent"), ]) monkeypatch.setattr(herdr_exec, "_run", lambda *_args, **_kwargs: next(responses)) assert herdr_exec._close_owned_session( - "pp-t42", ["tab", "close", "owned-tab"], host=None, + "renamed-agent", ["tab", "close", "owned-tab"], host=None, + pane_id="owned-pane", ) == "" @@ -432,7 +477,7 @@ def fake_run(args, host=None, timeout=None): monkeypatch.setattr(herdr_exec, "_run", fake_run) monkeypatch.setattr( herdr_exec, "_close_owned_session", - lambda name, close_args, host: closed.append( + lambda name, close_args, host, **_kwargs: closed.append( (name, close_args, host)) or "", ) diff --git a/tests/test_project_pipeline.py b/tests/test_project_pipeline.py index c32efd8..ea064d6 100644 --- a/tests/test_project_pipeline.py +++ b/tests/test_project_pipeline.py @@ -12,6 +12,7 @@ import pytest from promptpilot import project_pipeline as pp +from promptpilot.models import TaskCreate HEAD = "a" * 40 @@ -735,6 +736,47 @@ def test_complete_review_rechecks_expiry_before_first_mutation(tmp_path, monkeyp pp.complete_review(object(), config, pp.encode_signed_lease(lease), str(report)) +def test_complete_review_requires_live_replica_reservation_before_mutation( + isolated_db, tmp_path, monkeypatch): + config = { + "repository": "owner/repo", "trusted_account": "owner", + "base_branch": "main", "review_completion_gate": "target-v1", + "target_reservation_ttl_seconds": 300, + } + current = snapshot() + isolated_db.create_task(TaskCreate(prompt="complete review")) + owner = isolated_db.get_next_runnable() + started_at = owner.started_at.astimezone(timezone.utc).isoformat() + reservation = isolated_db.reserve_pipeline_target( + "owner/repo", "review", 42, HEAD, owner.id, 300, + task_started_at=started_at, ownership_kind="headless") + lease = target_review_lease(current) + lease.update({ + "target_stage": "review", "target_reservation": reservation, + }) + report = tmp_path / "report.json" + report.write_text(json.dumps({ + "change": "safe change", "checks": ["pytest"], + "blocking": [], "tail": [], + }), encoding="utf-8") + monkeypatch.setenv("PP_TASK_ID", str(owner.id)) + monkeypatch.setenv("PP_TASK_STARTED_AT", started_at) + monkeypatch.setenv("PP_PROVIDER_OWNERSHIP_KIND", "headless") + monkeypatch.setenv("PP_DATA_DIR", str(tmp_path / "data")) + monkeypatch.setattr(pp, "ensure_identity", lambda _gh, _config: None) + monkeypatch.setattr( + pp, "stable_timeline", + lambda _gh, _config, _number: copy.deepcopy(current), + ) + monkeypatch.setattr( + pp, "post_comment", lambda *_args: pytest.fail("mutation started")) + encoded = pp.encode_signed_lease(lease) + isolated_db.release_pipeline_target_reservations(owner.id) + + with pytest.raises(pp.PipelineError, match="expired or was released"): + pp.complete_review(object(), config, encoded, str(report)) + + @pytest.mark.parametrize("label", ["hold", "needs-decision", "changes-requested"]) def test_content_review_target_gate_rejects_routing_labels(label): value = snapshot() diff --git a/tests/test_schedule_series.py b/tests/test_schedule_series.py index 91241ef..0872a23 100644 --- a/tests/test_schedule_series.py +++ b/tests/test_schedule_series.py @@ -1031,7 +1031,16 @@ def test_pipeline_insights_finds_capacity_bottleneck(isolated_db, monkeypatch): lambda: {"example": PIPELINE_PROFILE}) pipeline_insights._cache.clear() - result = pipeline_insights.analyze("example", [], use_cache=False) + active_series = [ + { + "id": index, "title": f"ExampleProject - {queue['series_contains']}", + "ended": False, "paused": False, "broken": False, + "next_task_id": index + 100, "next_status": "pending", + } + for index, queue in enumerate(PIPELINE_PROFILE["queues"], start=1) + ] + result = pipeline_insights.analyze( + "example", active_series, use_cache=False) assert result["bottleneck"] == "review" review = next(q for q in result["queues"] if q["id"] == "review") @@ -1455,7 +1464,11 @@ def forbidden(*_args, **_kwargs): monkeypatch.setattr(pipeline_insights, "_run_profile_health_check", forbidden) pipeline_insights._cache.clear() - result = pipeline_insights.read_cached("legacy", []) + result = pipeline_insights.read_cached("legacy", [{ + "id": 1, "title": "Example - REVIEW", "ended": False, + "paused": False, "broken": False, + "next_task_id": 101, "next_status": "pending", + }]) assert result["cache"]["source"] == "snapshot" assert result["cache"]["complete"] is False diff --git a/tests/test_worker_admission.py b/tests/test_worker_admission.py index ee801c7..29738a5 100644 --- a/tests/test_worker_admission.py +++ b/tests/test_worker_admission.py @@ -308,8 +308,8 @@ def test_pipeline_success_without_closing_verdict_is_rejected(monkeypatch): monkeypatch.setattr(worker.db, "is_cancel_requested", lambda _task_id: False) monkeypatch.setattr( worker.db, "mark_failed", - lambda task_id, error, exit_code=None: failed.append( - (task_id, error, exit_code)), + lambda task_id, error, exit_code=None, **_kwargs: failed.append( + (task_id, error, exit_code)) or True, ) monkeypatch.setattr( worker.db, "mark_completed", @@ -354,8 +354,8 @@ def notification(*_args, **_kwargs): monkeypatch.setattr(worker.db, "add_notification", notification) monkeypatch.setattr( worker.db, "mark_failed", - lambda task_id, error, exit_code=None: failures.append( - (task_id, error, exit_code)), + lambda task_id, error, exit_code=None, **_kwargs: failures.append( + (task_id, error, exit_code)) or True, ) worker._execute_herdr_task(task, {"kind": "agy"})