Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 4 additions & 3 deletions openshift/Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -85,14 +85,15 @@ show-error-list:
@oc exec deployment/valkey -- valkey-cli --no-raw LRANGE error_list 0 -1 | python3 scripts/error_list.py --file -

# Requeue a specific error_list entry (by the [id=...] shown in show-error-list)
# back onto its original trigger queue, with attempts reset to 0.
# onto its priority trigger queue, with attempts reset to 0. By default it
# avoids user-triggered maintainer feedback; set USER_TRIGGERED=true to opt in.
# Required: ERROR_ID
requeue-error:
@if [ -z "$(ERROR_ID)" ]; then \
echo "Usage: make requeue-error ERROR_ID=42"; \
echo "Usage: make requeue-error ERROR_ID=42 [USER_TRIGGERED=true]"; \
exit 1; \
fi
python3 scripts/requeue_error.py $(ERROR_ID)
python3 scripts/requeue_error.py $(ERROR_ID) $(if $(filter true TRUE 1 yes YES,$(USER_TRIGGERED)),--user-triggered)

logs-triage:
oc logs -f deployment/triage-agent
Expand Down
41 changes: 25 additions & 16 deletions openshift/scripts/requeue_error.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,15 +7,12 @@
or entries recorded for a payload that never parsed into a `Task` in the first
place, have no `queue`/`task` to requeue and are reported as non-requeueable.

The task's `attempts` counter is reset to 0 and `user_triggered` is forced to
True before requeuing, on the assumption that whoever runs this has already
fixed the underlying issue and wants a fresh retry budget. Forcing
`user_triggered` also matters functionally: triage and reproducer skip
processing outright when a terminal `ymir_*_errored`/similar label is still
on the issue and the task isn't user-triggered — exactly the state a
just-requeued task is in until it gets a chance to run and clear that label
itself. It also routes the task onto the priority (`_todo`) twin of its
queue, same as any other maintainer-triggered run.
The task's `attempts` counter is reset to 0 before requeuing. By default,
`user_triggered` is set to False to keep acknowledgement and result comments
for maintainers low. Requeued tasks still go to the priority (`_todo`) twin
of their queue. A separate `requeued_from_error_list` marker lets triage and
reproducer process them even while the issue has a terminal errored label.
Pass `--user-triggered` to opt into maintainer-triggered comments and labels.

The entry is located and its replacement task computed here, client-side (a
plain LRANGE + local JSON parsing) rather than inside Redis: scanning and
Expand All @@ -34,8 +31,10 @@
size limits than a Redis value.

Usage:
make requeue-error ERROR_ID=42 # from openshift/ (preferred)
make requeue-error ERROR_ID=42 # priority requeue, fewer comments
make requeue-error ERROR_ID=42 USER_TRIGGERED=true # priority/user-triggered requeue
python3 scripts/requeue_error.py 42 # same, run directly
python3 scripts/requeue_error.py 42 --user-triggered # priority/user-triggered
python3 scripts/requeue_error.py 42 --dry-run # show the plan, don't mutate anything
"""

Expand Down Expand Up @@ -168,6 +167,15 @@ def atomic_requeue(
return int(out.strip())


def prepare_requeue_task(task: dict, source_queue: str, user_triggered: bool) -> tuple[str, str]:
"""Reset a task and send it to the priority queue, with optional maintainer feedback."""
task["attempts"] = 0
task["user_triggered"] = user_triggered
task["requeued_from_error_list"] = True
Comment thread
qodo-for-packit[bot] marked this conversation as resolved.
target_queue = source_queue if source_queue.endswith("_todo") else f"{source_queue}_todo"
return target_queue, json.dumps(task)


def main() -> None:
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument(
Expand All @@ -180,6 +188,11 @@ def main() -> None:
ap.add_argument(
"--dry-run", action="store_true", help="Print the plan without pushing or removing anything"
)
ap.add_argument(
"--user-triggered",
action="store_true",
help="Set user_triggered=true for maintainer comments and labels (priority queue either way)",
)
args = ap.parse_args()

blob = fetch_via_oc(args.queue, args.deployment)
Expand All @@ -198,15 +211,11 @@ def main() -> None:
)

old_attempts = task.get("attempts", 0)
task["attempts"] = 0
task["user_triggered"] = True
if not target_queue.endswith("_todo"):
target_queue = f"{target_queue}_todo"
task_json = json.dumps(task)
target_queue, task_json = prepare_requeue_task(task, target_queue, args.user_triggered)

print(
f"Requeuing error_id={args.error_id} ({jira_issue}) onto '{target_queue}' "
f"(attempts {old_attempts} -> 0, user_triggered -> true)"
f"(attempts {old_attempts} -> 0, user_triggered -> {str(args.user_triggered).lower()})"
)
if args.dry_run:
print(f"[dry-run] Would atomically remove from {args.queue} and LPUSH {target_queue}: {task_json}")
Expand Down
37 changes: 37 additions & 0 deletions openshift/scripts/tests/unit/test_requeue_error.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,37 @@
import json

import pytest

from openshift.scripts.requeue_error import prepare_requeue_task
from ymir.common.models import Task


@pytest.mark.parametrize("source_queue", ["backport_queue_c9s", "backport_queue_c9s_todo"])
def test_prepare_requeue_task_defaults_to_priority_queue_without_maintainer_feedback(source_queue) -> None:
task = {"metadata": {"issue": "RHEL-1"}, "attempts": 3, "user_triggered": True}

queue, task_json = prepare_requeue_task(task, source_queue, user_triggered=False)

assert queue == "backport_queue_c9s_todo"
assert json.loads(task_json) == {
"metadata": {"issue": "RHEL-1"},
"attempts": 0,
"user_triggered": False,
"requeued_from_error_list": True,
}
restored = Task.model_validate_json(task_json)
assert restored.requeued_from_error_list is True
assert restored.user_triggered is False


def test_prepare_requeue_task_uses_priority_queue_when_requested() -> None:
task = {"attempts": 2, "user_triggered": False}

queue, task_json = prepare_requeue_task(task, "backport_queue_c10s", user_triggered=True)

assert queue == "backport_queue_c10s_todo"
assert json.loads(task_json) == {
"attempts": 0,
"user_triggered": True,
"requeued_from_error_list": True,
}
8 changes: 8 additions & 0 deletions ymir/agents/backport_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -90,6 +90,7 @@
run_task_loop,
)
from ymir.common.constants import JiraLabels, RedisQueues
from ymir.common.error_list import clear_resolved_errors
from ymir.common.issue_lock import issue_lock
from ymir.common.logging_setup import configure_logging, current_jira_issue, get_trajectory_writeable
from ymir.common.mock_repos import get_mock_local_tool_env
Expand Down Expand Up @@ -2201,6 +2202,13 @@ async def retry(
state.backport_result.model_dump_json(),
)
)
await clear_resolved_errors(
redis,
backport_data.jira_issue,
backport_queue,
target_branch=dist_git_branch,
dry_run=dry_run,
)
else:
logger.warning(
f"Backport failed for {backport_data.jira_issue}: {state.backport_result.error}"
Expand Down
27 changes: 27 additions & 0 deletions ymir/agents/rebase_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,7 @@
)
from ymir.common.base_utils import fix_await, install_shutdown_handler, redis_client, run_task_loop
from ymir.common.constants import JiraLabels, RedisQueues
from ymir.common.error_list import clear_resolved_errors
from ymir.common.issue_lock import issue_lock
from ymir.common.logging_setup import configure_logging, current_jira_issue, get_trajectory_writeable
from ymir.common.mock_repos import get_mock_local_tool_env
Expand Down Expand Up @@ -99,6 +100,25 @@ def _consolidated_issue_keys(
return list(dict.fromkeys([primary_issue] + [item.issue_key for item in consolidated_issues]))


async def _clear_rebase_resolved_errors(
redis_conn,
rebase_data: RebaseData,
queue: str,
*,
target_branch: str,
dry_run: bool,
) -> None:
"""Clear prior rebase errors for the primary issue and each resolved sibling."""
for issue_key in _consolidated_issue_keys(rebase_data.jira_issue, rebase_data.consolidated_issues):
await clear_resolved_errors(
redis_conn,
issue_key,
queue,
target_branch=target_branch,
dry_run=dry_run,
)


def get_instructions() -> str:
return render_template("rebase/instructions.j2")

Expand Down Expand Up @@ -920,6 +940,13 @@ async def retry(
state.rebase_result.model_dump_json(),
)
)
await _clear_rebase_resolved_errors(
redis,
rebase_data,
rebase_queue,
target_branch=dist_git_branch,
dry_run=dry_run,
)
else:
logger.warning(f"Rebase failed for {rebase_data.jira_issue}: {state.rebase_result.error}")
# Label all consolidated issues with failure status
Expand Down
27 changes: 27 additions & 0 deletions ymir/agents/rebuild_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,7 @@
)
from ymir.common.base_utils import fix_await, install_shutdown_handler, redis_client, run_task_loop
from ymir.common.constants import JiraLabels, RedisQueues
from ymir.common.error_list import clear_resolved_errors
from ymir.common.issue_lock import issue_lock
from ymir.common.logging_setup import configure_logging, current_jira_issue
from ymir.common.mock_repos import get_mock_local_tool_env
Expand All @@ -56,6 +57,25 @@
redis_logger = logging.getLogger("agent.redis")


async def _clear_rebuild_resolved_errors(
redis_conn,
rebuild_data: RebuildData,
queue: str,
*,
target_branch: str,
dry_run: bool,
) -> None:
"""Clear prior rebuild errors for the primary issue and each resolved sibling."""
for issue_key in dict.fromkeys(rebuild_data.all_jira_issues):
await clear_resolved_errors(
redis_conn,
issue_key,
queue,
target_branch=target_branch,
dry_run=dry_run,
)


async def main() -> None:
init_sentry()

Expand Down Expand Up @@ -669,6 +689,13 @@ async def retry(
).model_dump_json(),
)
)
await _clear_rebuild_resolved_errors(
redis,
rebuild_data,
rebuild_queue,
target_branch=dist_git_branch,
dry_run=dry_run,
)
else:
logger.warning(f"Rebuild failed for {rebuild_data.jira_issue}: {state.rebuild_error}")
for issue_key in dict.fromkeys(rebuild_data.all_jira_issues):
Expand Down
28 changes: 28 additions & 0 deletions ymir/agents/reproducer_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,7 @@
from ymir.common.base_utils import fix_await, redis_client, run_task_loop
from ymir.common.constants import JiraLabels, RedisQueues
from ymir.common.delayed_queue import promote_due_tasks, schedule_task
from ymir.common.error_list import clear_resolved_errors
from ymir.common.logging_setup import configure_logging, current_jira_issue
from ymir.common.mock_repos import get_mock_local_tool_env
from ymir.common.models import (
Expand Down Expand Up @@ -265,6 +266,31 @@ def _should_finalize_jira(result: OutputSchema) -> bool:
return not result.retryable_error and not result.lock_deferred


async def _clear_finalized_reproducer_errors(
redis_conn, input_data: InputSchema, result: OutputSchema, *, dry_run: bool
) -> int:
"""Clear earlier failures only when the final Jira label resolves this work."""
if not _should_finalize_jira(result):
return 0
# An adapted test still needs its MR update. If that step failed, the
# original test's presence must not make this run count as resolved.
if result.adapted_existing and not result.success:
return 0
if _determine_result_label(result) not in {
JiraLabels.REPRODUCER_CREATED,
JiraLabels.REPRODUCER_ALREADY_EXISTS,
JiraLabels.REPRODUCER_NOT_REPRODUCIBLE,
}:
Comment thread
qodo-for-packit[bot] marked this conversation as resolved.
return 0
return await clear_resolved_errors(
redis_conn,
input_data.jira_issue,
RedisQueues.REPRODUCER_QUEUE.value,
target_branch=input_data.target_branch,
dry_run=dry_run,
)


def _needs_merge_request(result: OutputSchema) -> bool:
"""Whether orchestration should commit/push an MR for this result."""
if result.lock_deferred or result.retryable_error:
Expand Down Expand Up @@ -1235,6 +1261,7 @@ async def process_task(payload):
terminal_ymir_labels
and JiraLabels.REPRODUCER_IN_PROGRESS.value not in current_labels
and not user_triggered
and not task.requeued_from_error_list
):
logger.info(
f"Skipping duplicate reproducer for {input_data.jira_issue} — "
Expand Down Expand Up @@ -1439,6 +1466,7 @@ async def retry(
logger.info(
f"Pushed {input_data.jira_issue} to {RedisQueues.COMPLETED_REPRODUCER_LIST.value}"
)
await _clear_finalized_reproducer_errors(redis, input_data, output, dry_run=dry_run)
finally:
try:
await release_reproducer_lock(
Expand Down
36 changes: 34 additions & 2 deletions ymir/agents/tests/unit/test_rebase_consolidation.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
import pytest

from ymir.agents.rebase_agent import _consolidated_issue_keys
from ymir.agents.rebase_agent import _clear_rebase_resolved_errors, _consolidated_issue_keys
from ymir.agents.rebase_consolidation import (
add_jira_tickets_to_latest_changelog_entry,
build_rebase_siblings_jql,
Expand All @@ -9,7 +9,7 @@
has_new_latest_changelog_entry,
uses_autochangelog,
)
from ymir.common.models import ConsolidatedIssue
from ymir.common.models import ConsolidatedIssue, RebaseData
from ymir.common.utils import extract_text_from_adf


Expand Down Expand Up @@ -151,6 +151,38 @@ def test_consolidated_issue_keys_deduplicates_siblings_and_primary():
) == ["RHEL-100", "RHEL-200"]


@pytest.mark.asyncio
@pytest.mark.parametrize("dry_run", [False, True])
async def test_successful_consolidated_rebase_clears_primary_and_distinct_siblings(monkeypatch, dry_run):
calls = []

async def mock_clear(redis_conn, issue, queue, *, target_branch, dry_run):
calls.append((redis_conn, issue, queue, target_branch, dry_run))

monkeypatch.setattr("ymir.agents.rebase_agent.clear_resolved_errors", mock_clear)
redis = object()
data = RebaseData(
package="expat",
version="2.7.0",
jira_issue="RHEL-100",
consolidated_issues=[
ConsolidatedIssue(issue_key="RHEL-200"),
ConsolidatedIssue(issue_key="RHEL-100"),
ConsolidatedIssue(issue_key="RHEL-200"),
ConsolidatedIssue(issue_key="RHEL-300"),
],
)

await _clear_rebase_resolved_errors(
redis, data, "rebase_queue_c10s", target_branch="rhel-10.3", dry_run=dry_run
)

assert calls == [
(redis, issue, "rebase_queue_c10s", "rhel-10.3", dry_run)
for issue in ("RHEL-100", "RHEL-200", "RHEL-300")
]


def test_build_rebase_siblings_jql():
jql = build_rebase_siblings_jql("RHEL-100", "dotnet10.0", "rhel-10.2")
assert 'component = "dotnet10.0"' in jql
Expand Down
Loading
Loading