Skip to content

Skill Runner

skill

Generic skill runner framework.

Provides SkillConfig and run_skill() — a reusable pipeline for running AI agent skills against tickets/issues in CI. All domain-specific behavior is injected via callable hooks on SkillConfig, making this framework agnostic to the issue tracker, git forge, and skill content.

Usage::

from agentic_ci.skill import SkillConfig, run_skill

config = SkillConfig(
    skill_name="my-resolve",
    prompt_builder=my_prompt_fn,
    verdict_loader=my_verdict_fn,
    label_applier=my_label_fn,
)
rc = run_skill(config, ticket_key="PROJ-123", work_dir=Path("/tmp/work"), ...)

run_routed_skill() wraps the same pipeline with difficulty-based model routing: a classifier run on the default model rates the task, and the skill then runs on the matching ModelTier (see agentic_ci.routing).

SkillHook

Bases: _SkillHookRequired

Structured definition of an extra skill to run at a pipeline hook point.

SkillConfig(skill_name, skill_source='', skill_ref='main', prompt_builder=lambda **kw: '', context_writer=_noop, verdict_loader=_noop_verdict, verdict_path_fn=lambda wd: wd / 'verdict.json', label_applier=_noop, cost_formatter=lambda d: None, extension_config_writer=_noop, pre_gates=list(), post_gates=list(), extra_skills=list(), context_dir='.context', artifacts=list(), max_retries=1, retryable_modes=frozenset({'resolve'}), backend_name='podman', harness_name='claude-code', container_image=None, container_env=dict(), container_runner=None, model_tiers=dict(), sandbox_profile=None) dataclass

Configuration for a skill run. All domain-specific behavior via hooks.

model_tiers = field(default_factory=dict) class-attribute instance-attribute

Per-tier overrides of the harness default routing tiers (run_routed_skill only).

sandbox_profile = None class-attribute instance-attribute

What the target repo needs in the sandbox (see agentic_ci.sandbox_profile).

Reaches the backend through the container runner; only the OpenShell backend uses it, and in this release only its resources. A custom container_runner receives it as the sandbox_profile keyword, and only when it is set.

RoutedSkillResult(rc, route) dataclass

Result of :func:run_routed_skill.

route is None when no agent ran (dry run or a pre-gate blocked the run), so no routing decision was made.

run_skill(config, ticket_key, work_dir, config_dir, *, mode='resolve', ticket=None, dry_run=False, dry_run_verdict_path=None, **extra_kwargs)

Run a skill pipeline for a single ticket. Returns exit code.

Flow: 1. Run pre-gates (skip container if any gate returns a non-None message) 2. Write context via context_writer hook 3. Write extension config via extension_config_writer hook 4. Build prompt via prompt_builder hook 5. Launch container (or dry-run) 6. Read cost data (OTEL) 7. Run post-gates 8. Load verdict via verdict_loader hook 9. Format comment and apply labels via label_applier hook

Source code in src/agentic_ci/skill.py
def run_skill(
    config: SkillConfig,
    ticket_key: str,
    work_dir: Path,
    config_dir: Path,
    *,
    mode: str = "resolve",
    ticket: dict | None = None,
    dry_run: bool = False,
    dry_run_verdict_path: Path | None = None,
    **extra_kwargs,
) -> int:
    """Run a skill pipeline for a single ticket. Returns exit code.

    Flow:
    1. Run pre-gates (skip container if any gate returns a non-None message)
    2. Write context via context_writer hook
    3. Write extension config via extension_config_writer hook
    4. Build prompt via prompt_builder hook
    5. Launch container (or dry-run)
    6. Read cost data (OTEL)
    7. Run post-gates
    8. Load verdict via verdict_loader hook
    9. Format comment and apply labels via label_applier hook
    """
    log.info("[%s] Starting %s in %s mode", ticket_key, config.skill_name, mode)

    for gate in config.pre_gates:
        result = gate(
            ticket_key=ticket_key,
            ticket=ticket,
            mode=mode,
            work_dir=work_dir,
            **extra_kwargs,
        )
        if result is not None:
            log.info("[%s] Pre-gate blocked: %s", ticket_key, result)
            return 0

    config.context_writer(
        ticket_key=ticket_key,
        ticket=ticket,
        mode=mode,
        work_dir=work_dir,
        **extra_kwargs,
    )

    if config.extra_skills:
        raw_ctx_dir = work_dir / config.context_dir
        if raw_ctx_dir.is_symlink():
            raise ValueError(f"context_dir is a symlink: {raw_ctx_dir}")
        ctx_dir = raw_ctx_dir.resolve()
        try:
            ctx_dir.relative_to(work_dir.resolve())
        except ValueError as exc:
            raise ValueError(f"context_dir escapes work_dir: {config.context_dir!r}") from exc
        ctx_dir.mkdir(parents=True, exist_ok=True)
        config_path = ctx_dir / "config.json"
        if config_path.is_symlink():
            raise ValueError(f"config.json is a symlink: {config_path}")
        config_path.write_text(
            json.dumps(
                {"extra_skills": config.extra_skills},
                indent=2,
                ensure_ascii=False,
            ),
            encoding="utf-8",
        )

    config.extension_config_writer(
        ticket_key=ticket_key,
        ticket=ticket,
        config=config,
        work_dir=work_dir,
        **extra_kwargs,
    )

    prompt = config.prompt_builder(
        ticket_key=ticket_key,
        mode=mode,
        skill_name=config.skill_name,
        **extra_kwargs,
    )
    output_file = work_dir / "agent-output.txt"

    runner = config.container_runner or _default_run_container
    runner_kwargs: dict = {"image": config.container_image}
    if config.container_env:
        runner_kwargs["container_env"] = config.container_env
    if config.sandbox_profile is not None:
        runner_kwargs["sandbox_profile"] = config.sandbox_profile
    if config.container_runner is None:
        runner_kwargs["verdict_path"] = config.verdict_path_fn(work_dir)
        runner_kwargs["backend_name"] = config.backend_name
        runner_kwargs["harness_name"] = config.harness_name

    if dry_run:
        if dry_run_verdict_path:
            verdict_dest = config.verdict_path_fn(work_dir)
            verdict_dest.parent.mkdir(parents=True, exist_ok=True)
            shutil.copy2(dry_run_verdict_path, verdict_dest)
        rc = 0
    else:
        rc = runner(work_dir, prompt, output_file, **runner_kwargs)

        attempt = 0
        while (
            rc != 0
            and mode in config.retryable_modes
            and rc in TRANSIENT_EXIT_CODES
            and attempt < config.max_retries
        ):
            attempt += 1
            log.warning(
                "[%s] Transient failure (exit %d), retry %d/%d",
                ticket_key,
                rc,
                attempt,
                config.max_retries,
            )
            rc = runner(work_dir, prompt, output_file, **runner_kwargs)

    if rc != 0:
        log.error("[%s] Container exited with code %d", ticket_key, rc)
        config.label_applier(
            ticket_key=ticket_key,
            verdict=None,
            rc=rc,
            mode=mode,
            work_dir=work_dir,
            **extra_kwargs,
        )
        return rc

    cost_data = _load_otel_cost(work_dir)

    gate_errors: list[str] = []
    verdict = None
    for gate in config.post_gates:
        v, errors = gate(work_dir=work_dir, ticket_key=ticket_key, **extra_kwargs)
        if v is not None:
            verdict = v
        gate_errors.extend(errors)

    if gate_errors:
        log.error("[%s] Post-gate failures: %s", ticket_key, gate_errors)
        config.label_applier(
            ticket_key=ticket_key,
            verdict=None,
            gate_errors=gate_errors,
            mode=mode,
            work_dir=work_dir,
            **extra_kwargs,
        )
        return 1

    if verdict is None:
        verdict_error: Exception | None = None
        try:
            verdict = config.verdict_loader(work_dir)
        except Exception as exc:
            verdict_error = exc
            if not dry_run and mode in config.retryable_modes and config.max_retries > 0:
                log.warning("[%s] Verdict missing (%s), retrying once", ticket_key, exc)
                rc = runner(work_dir, prompt, output_file, **runner_kwargs)
                if rc != 0:
                    log.error("[%s] Retry container also failed (exit %d)", ticket_key, rc)
                    config.label_applier(
                        ticket_key=ticket_key,
                        verdict=None,
                        rc=rc,
                        mode=mode,
                        work_dir=work_dir,
                        **extra_kwargs,
                    )
                    return rc
                try:
                    verdict = config.verdict_loader(work_dir)
                    verdict_error = None
                except Exception as retry_exc:
                    log.error("[%s] Verdict still missing after retry: %s", ticket_key, retry_exc)
                    verdict_error = retry_exc

            if verdict is None:
                log.error(
                    "[%s] Failed to load verdict: %s: %s",
                    ticket_key,
                    type(verdict_error).__name__,
                    verdict_error,
                )
                config.label_applier(
                    ticket_key=ticket_key,
                    verdict=None,
                    gate_errors=[_verdict_error_text(verdict_error)],
                    mode=mode,
                    work_dir=work_dir,
                    **extra_kwargs,
                )
                return 1

    cost_summary = config.cost_formatter(cost_data)
    if cost_summary:
        verdict["_cost_summary"] = cost_summary

    label_rc = config.label_applier(
        ticket_key=ticket_key,
        verdict=verdict,
        mode=mode,
        work_dir=work_dir,
        **extra_kwargs,
    )

    if label_rc:
        log.error(
            "[%s] label_applier returned non-zero exit code: %d",
            ticket_key,
            label_rc,
        )
        return label_rc

    log.info(
        "[%s] %s complete: verdict=%s",
        ticket_key,
        config.skill_name,
        verdict.get("verdict", "unknown"),
    )
    return 0

run_routed_skill(config, ticket_key, work_dir, config_dir, *, mode='resolve', ticket=None, dry_run=False, dry_run_verdict_path=None, classifier_max_turns=DEFAULT_CLASSIFIER_MAX_TURNS, classifier_prompt_builder=None, force_tier=None, **extra_kwargs)

Run a skill with difficulty-based model routing.

Behaves like :func:run_skill with one addition: before the skill runs, a classifier invocation on the harness default model (CLAUDE_MODEL, OPENCODE_MODEL or CODEX_MODEL, else Harness.default_model()) rates the task as low, medium or high and writes _run/route.json. The skill then runs on the matching :class:~agentic_ci.routing.ModelTier from the harness defaults, overridden per key by config.model_tiers.

The classifier runs inside the same sandbox as the skill (same container, credentials and network policy). Its raw stream is written to _run/classifier-output.txt. Any classifier failure (non-zero exit, exception, missing or invalid route file) logs a warning and falls back to the default model at the harness's effective default effort (env var or registry default_effort), which is exactly what :func:run_skill would do. Only configuration errors raise, and they raise before any container starts.

The decision is made once per call and reused by every retry that :func:run_skill performs, so all attempts use the same model. A skill.routed event is appended to the run's _run/claude-otel.jsonl.

Parameters:

Name Type Description Default
classifier_max_turns int

turn cap for the classifier where the CLI supports one (Claude Code --max-turns).

DEFAULT_CLASSIFIER_MAX_TURNS
classifier_prompt_builder Callable[[str], str] | None

replaces the default classifier prompt; receives the skill prompt and returns the classifier prompt.

None
force_tier str | None

skip the classifier and pin this tier.

None

config.container_runner must be None; custom runners have no model surface. extension_config_writer receives a copy of config whose container_runner is the routing runner.

Returns:

Type Description
RoutedSkillResult

class:RoutedSkillResult with the exit code and the decision

RoutedSkillResult

(None when no agent ran, e.g. dry_run or a pre-gate block).

Source code in src/agentic_ci/skill.py
def run_routed_skill(
    config: SkillConfig,
    ticket_key: str,
    work_dir: Path,
    config_dir: Path,
    *,
    mode: str = "resolve",
    ticket: dict | None = None,
    dry_run: bool = False,
    dry_run_verdict_path: Path | None = None,
    classifier_max_turns: int = DEFAULT_CLASSIFIER_MAX_TURNS,
    classifier_prompt_builder: Callable[[str], str] | None = None,
    force_tier: str | None = None,
    **extra_kwargs,
) -> RoutedSkillResult:
    """Run a skill with difficulty-based model routing.

    Behaves like :func:`run_skill` with one addition: before the skill runs,
    a classifier invocation on the harness default model (``CLAUDE_MODEL``,
    ``OPENCODE_MODEL`` or ``CODEX_MODEL``, else ``Harness.default_model()``)
    rates the task as ``low``, ``medium`` or ``high`` and writes
    ``_run/route.json``. The skill then runs on the matching
    :class:`~agentic_ci.routing.ModelTier` from the harness defaults,
    overridden per key by ``config.model_tiers``.

    The classifier runs inside the same sandbox as the skill (same container,
    credentials and network policy). Its raw stream is written to
    ``_run/classifier-output.txt``. Any classifier failure (non-zero exit,
    exception, missing or invalid route file) logs a warning and falls back to
    the default model at the harness's effective default effort (env var or
    registry ``default_effort``), which is exactly what :func:`run_skill`
    would do. Only configuration errors raise, and they
    raise before any container starts.

    The decision is made once per call and reused by every retry that
    :func:`run_skill` performs, so all attempts use the same model. A
    ``skill.routed`` event is appended to the run's ``_run/claude-otel.jsonl``.

    Args:
        classifier_max_turns: turn cap for the classifier where the CLI
            supports one (Claude Code ``--max-turns``).
        classifier_prompt_builder: replaces the default classifier prompt;
            receives the skill prompt and returns the classifier prompt.
        force_tier: skip the classifier and pin this tier.

    ``config.container_runner`` must be ``None``; custom runners have no
    model surface. ``extension_config_writer`` receives a copy of *config*
    whose ``container_runner`` is the routing runner.

    Returns:
        :class:`RoutedSkillResult` with the exit code and the decision
        (``None`` when no agent ran, e.g. ``dry_run`` or a pre-gate block).
    """
    if config.container_runner is not None:
        raise ValueError("run_routed_skill requires the default container runner")
    harness = create_harness(config.harness_name)
    tiers = resolve_model_tiers(harness, config.model_tiers)
    if force_tier is not None and force_tier not in tiers:
        raise ValueError(f"Unknown force_tier {force_tier!r}; expected one of {sorted(tiers)}")

    state: dict[str, RouteDecision] = {}

    def _router(session, prompt):
        if "decision" not in state:
            if force_tier is not None:
                decision = forced_route(force_tier, tiers)
            else:
                decision = classify(
                    # The classifier's evidence of completion is its route file,
                    # not the skill verdict.
                    functools.partial(session.run, completion_file=route_path(work_dir)),
                    work_dir=work_dir,
                    task_prompt=prompt,
                    tiers=tiers,
                    classifier_model=session.default_model,
                    classifier_effort=session.harness.classifier_effort(),
                    classifier_args=session.harness.build_classifier_args(classifier_max_turns),
                    fallback=ModelTier(session.default_model, session.harness.resolve_efforts()[0]),
                    prompt_builder=classifier_prompt_builder,
                )
            state["decision"] = decision
            _emit_route_event(session, config, ticket_key, decision)
        return state["decision"]

    runner = functools.partial(
        _default_run_container,
        verdict_path=config.verdict_path_fn(work_dir),
        backend_name=config.backend_name,
        harness_name=config.harness_name,
        router=_router,
    )
    rc = run_skill(
        dataclasses.replace(config, container_runner=runner),
        ticket_key,
        work_dir,
        config_dir,
        mode=mode,
        ticket=ticket,
        dry_run=dry_run,
        dry_run_verdict_path=dry_run_verdict_path,
        **extra_kwargs,
    )
    return RoutedSkillResult(rc=rc, route=state.get("decision"))