Compare commits

..

8 Commits

Author SHA1 Message Date
Yuge Zhang eb3c7ca461 update 2025-10-15 22:20:29 +08:00
Yuge Zhang a31381d1fd debug 2025-10-15 11:19:42 +00:00
Yuge Zhang d4b5cbfdfd fix async issue in spider 2025-10-15 10:11:57 +00:00
Yuge Zhang 6dbd96ee27 . 2025-10-15 17:52:34 +08:00
Yuge Zhang ffd965b368 update sql agent training script 2025-10-15 17:42:20 +08:00
Yuge Zhang 773e4d372f partial upgrade to sql agent 2025-10-15 15:24:03 +08:00
Yuge Zhang cf69f5499a Merge branch 'main' of github.com:microsoft/agent-lightning into upgrade-sql-agent-example 2025-10-15 15:16:54 +08:00
Yuge Zhang e7044bb917 . 2025-10-15 15:16:48 +08:00
339 changed files with 6235 additions and 68497 deletions
-32
View File
@@ -1,32 +0,0 @@
name: Backport Merged Pull Request
on:
pull_request_target:
types: [closed]
permissions:
contents: write
issues: write
pull-requests: write
# NOTE:
# Microsoft requires rotating BOT_PAT every 3 months.
# Log onto agent-lightning-bot account and rotate the PAT if needed.
jobs:
backport:
name: Backport pull request
runs-on: ubuntu-latest
# Don't run on closed unmerged pull requests
if: github.event.pull_request.merged
steps:
- uses: actions/checkout@v4
- name: Create backport pull requests
uses: korthout/backport-action@v3
with:
branch_name: 'backport/${pull_number}/${target_branch}'
label_pattern: ^(stable/[^ ]+)$
github_token: ${{ secrets.BOT_PAT }}
add_labels: backport
add_author_as_assignee: true
git_committer_name: agent-lightning-bot
# This email address is not monitored.
git_committer_email: agl.msft@outlook.com
-29
View File
@@ -1,29 +0,0 @@
name: Badge - APO
on:
workflow_run:
workflows:
- Examples - APO
types: [completed]
workflow_dispatch:
permissions:
actions: read
contents: read
jobs:
badge:
if: ${{ github.event_name == 'workflow_dispatch' || (github.event_name == 'workflow_run' && github.event.workflow_run.head_branch == 'main') }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/github-script@v8
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
script: |
const badgeAggregation = require('./scripts/badge_aggregation.js');
const dependencies = [
{ workflow: 'examples-apo.yml', label: 'apo', variants: ['legacy', 'stable'] },
];
await badgeAggregation({ github, context, core, dependencies });
-29
View File
@@ -1,29 +0,0 @@
name: Badge - Calc-X
on:
workflow_run:
workflows:
- Examples - Calc-X
types: [completed]
workflow_dispatch:
permissions:
actions: read
contents: read
jobs:
badge:
if: ${{ github.event_name == 'workflow_dispatch' || (github.event_name == 'workflow_run' && github.event.workflow_run.head_branch == 'main') }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/github-script@v8
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
script: |
const badgeAggregation = require('./scripts/badge_aggregation.js');
const dependencies = [
{ workflow: 'examples-calc-x.yml', label: 'calc-x', variants: ['legacy', 'stable'] },
];
await badgeAggregation({ github, context, core, dependencies });
-29
View File
@@ -1,29 +0,0 @@
name: Badge - Compatibility
on:
workflow_run:
workflows:
- Examples - Backward Compatibility
types: [completed]
workflow_dispatch:
permissions:
actions: read
contents: read
jobs:
badge:
if: ${{ github.event_name == 'workflow_dispatch' || (github.event_name == 'workflow_run' && github.event.workflow_run.head_branch == 'main') }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/github-script@v8
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
script: |
const badgeAggregation = require('./scripts/badge_aggregation.js');
const dependencies = [
{ workflow: 'examples-compat.yml', label: 'examples-compat', variants: ['legacy', 'stable'] },
];
await badgeAggregation({ github, context, core, dependencies });
-35
View File
@@ -1,35 +0,0 @@
name: Badge - Examples
on:
workflow_run:
workflows:
- Examples - Calc-X
- Examples - Spider
- Examples - APO
- Examples - Unsloth
types: [completed]
workflow_dispatch:
permissions:
actions: read
contents: read
jobs:
badge:
if: ${{ github.event_name == 'workflow_dispatch' || (github.event_name == 'workflow_run' && github.event.workflow_run.head_branch == 'main') }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/github-script@v8
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
script: |
const badgeAggregation = require('./scripts/badge_aggregation.js');
const dependencies = [
{ workflow: 'examples-calc-x.yml', label: 'examples-calc-x.stable', variants: ['stable'] },
{ workflow: 'examples-spider.yml', label: 'examples-spider.stable', variants: ['stable'] },
{ workflow: 'examples-apo.yml', label: 'examples-apo.stable', variants: ['stable'] },
{ workflow: 'examples-unsloth.yml', label: 'examples-unsloth.stable', variants: ['stable'] },
];
await badgeAggregation({ github, context, core, dependencies });
-37
View File
@@ -1,37 +0,0 @@
name: Badge - Latest
on:
workflow_run:
workflows:
- Examples - Calc-X
- Examples - Spider
- Examples - APO
- Examples - Unsloth
- GPU Test
types: [completed]
workflow_dispatch:
permissions:
actions: read
contents: read
jobs:
badge:
if: ${{ github.event_name == 'workflow_dispatch' || (github.event_name == 'workflow_run' && github.event.workflow_run.head_branch == 'main') }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/github-script@v8
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
script: |
const badgeAggregation = require('./scripts/badge_aggregation.js');
const dependencies = [
{ workflow: 'examples-calc-x.yml', label: 'calc-x.latest', variants: ['latest'] },
{ workflow: 'examples-spider.yml', label: 'spider.latest', variants: ['latest'] },
{ workflow: 'examples-apo.yml', label: 'apo.latest', variants: ['latest'] },
{ workflow: 'examples-unsloth.yml', label: 'unsloth.latest', variants: ['latest'] },
{ workflow: 'tests-full.yml', label: 'tests-full.latest', variants: ['latest'] },
];
await badgeAggregation({ github, context, core, dependencies });
-29
View File
@@ -1,29 +0,0 @@
name: Badge - Spider
on:
workflow_run:
workflows:
- Examples - Spider
types: [completed]
workflow_dispatch:
permissions:
actions: read
contents: read
jobs:
badge:
if: ${{ github.event_name == 'workflow_dispatch' || (github.event_name == 'workflow_run' && github.event.workflow_run.head_branch == 'main') }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/github-script@v8
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
script: |
const badgeAggregation = require('./scripts/badge_aggregation.js');
const dependencies = [
{ workflow: 'examples-spider.yml', label: 'spider', variants: ['stable', 'legacy'] },
];
await badgeAggregation({ github, context, core, dependencies });
-31
View File
@@ -1,31 +0,0 @@
name: Badge - Unit Test
on:
workflow_run:
workflows:
- CPU Test
- GPU Test
types: [completed]
workflow_dispatch:
permissions:
actions: read
contents: read
jobs:
badge:
if: ${{ github.event_name == 'workflow_dispatch' || (github.event_name == 'workflow_run' && github.event.workflow_run.head_branch == 'main') }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/github-script@v8
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
script: |
const badgeAggregation = require('./scripts/badge_aggregation.js');
const dependencies = [
{ workflow: 'tests-full.yml', label: 'tests-full', variants: ['legacy', 'stable'] },
{ workflow: 'tests.yml', label: 'tests', variants: ['legacy', 'stable', 'Lint', 'documentation', 'JavaScript'] },
];
await badgeAggregation({ github, context, core, dependencies });
-29
View File
@@ -1,29 +0,0 @@
name: Badge - Unsloth
on:
workflow_run:
workflows:
- Examples - Unsloth
types: [completed]
workflow_dispatch:
permissions:
actions: read
contents: read
jobs:
badge:
if: ${{ github.event_name == 'workflow_dispatch' || (github.event_name == 'workflow_run' && github.event.workflow_run.head_branch == 'main') }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/github-script@v8
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
script: |
const badgeAggregation = require('./scripts/badge_aggregation.js');
const dependencies = [
{ workflow: 'examples-unsloth.yml', label: 'examples-unsloth.stable', variants: ['stable'] },
];
await badgeAggregation({ github, context, core, dependencies });
-33
View File
@@ -1,33 +0,0 @@
name: Dashboard
permissions:
contents: read
on:
schedule:
# Every day at 5 AM UTC+8
- cron: '0 21 * * *'
workflow_dispatch:
push:
branches: [ main, stable/**/* ]
jobs:
dashboard:
name: Chromatic
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: actions/setup-node@v6
with:
node-version: '22'
- name: Install JavaScript dependencies
run: cd dashboard && npm ci
- name: Run Chromatic
uses: chromaui/action@v13
with:
projectToken: ${{ secrets.CHROMATIC_PROJECT_TOKEN }}
workingDir: dashboard
exitZeroOnChanges: false
+10 -13
View File
@@ -8,10 +8,6 @@ on:
- 'v*'
workflow_dispatch:
concurrency:
group: docs-deploy
cancel-in-progress: false
permissions:
contents: write
pages: write
@@ -24,14 +20,15 @@ jobs:
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: actions/setup-python@v6
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.12'
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
- name: Sync dependencies
run: uv sync --frozen --no-default-groups --group dev
- name: Install dependencies
run: |
./scripts/setup_stable.sh
- name: Configure Git
run: |
@@ -54,11 +51,11 @@ jobs:
- name: Deploy versioned docs
if: startsWith(github.ref, 'refs/tags/')
run: |
uv run --locked --no-sync mike deploy --push --update-aliases ${{ steps.version.outputs.version }} stable
mike deploy --push --update-aliases ${{ steps.version.outputs.version }} stable
- name: Deploy dev docs
if: github.ref == 'refs/heads/main'
run: |
uv run --locked --no-sync mike deploy --push latest
mike deploy --push latest
# Always set stable to default
uv run --locked --no-sync mike set-default --push stable
mike set-default --push stable
-116
View File
@@ -1,116 +0,0 @@
name: Examples - APO
permissions:
contents: read
on:
schedule:
# Every day at 3 AM UTC+8
- cron: '0 19 * * *'
workflow_dispatch:
repository_dispatch:
types: [ci-apo, ci-all]
run-name: >-
${{ github.event_name == 'repository_dispatch'
&& format(
'PR #{0} - Label {1} - {2}',
github.event.client_payload.pull_number,
github.event.client_payload.ci_label,
github.event.client_payload.correlation_id
)
|| format('APO - {0}', github.event_name) }}
jobs:
apo:
if: >
github.event_name != 'repository_dispatch' ||
github.event.action == 'ci-apo' ||
github.event.action == 'ci-all'
name: APO (Python ${{ matrix.python-version }}, ${{ matrix.setup-script }})
# This job is run on GitHub hosted runners rather than self-hosted runners because it needs no GPU.
runs-on: ubuntu-latest
timeout-minutes: 30
strategy:
matrix:
include:
- python-version: '3.10'
setup-script: 'legacy'
- python-version: '3.12'
setup-script: 'stable'
- python-version: '3.13'
setup-script: 'latest'
fail-fast: false
steps:
- uses: actions/checkout@v4
with:
ref: ${{ github.event_name == 'repository_dispatch' && github.event.client_payload.pr_ref || (github.event.pull_request.number && format('refs/pull/{0}/merge', github.event.pull_request.number)) || github.ref }}
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
python-version: ${{ matrix.python-version }}
- name: Upgrade dependencies (latest)
run: uv lock --upgrade
if: matrix.setup-script == 'latest'
- name: Sync dependencies (latest)
run: |
uv sync --frozen --no-default-groups --extra apo \
--group dev --group experiment --group agents --group core-stable
if: matrix.setup-script == 'latest'
- name: Sync dependencies (stable & legacy)
run: |
uv sync --frozen --no-default-groups --extra apo \
--group dev --group experiment --group agents --group core-${{ matrix.setup-script }}
if: matrix.setup-script != 'latest'
- name: Freeze dependencies
run: |
set -ex
uv pip freeze | tee requirements-freeze.txt
echo "UV_LOCKED=1" >> $GITHUB_ENV
echo "UV_NO_SYNC=1" >> $GITHUB_ENV
- name: Upload dependencies artifact
uses: actions/upload-artifact@v4
with:
name: dependencies-apo-${{ matrix.python-version }}-${{ matrix.setup-script }}
path: requirements-freeze.txt
compression-level: 0
- name: Launch LiteLLM Proxy
run: |
./scripts/litellm_run.sh
env:
AZURE_API_BASE: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_BASE }}
AZURE_API_KEY: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_KEY }}
- name: APO custom algorithm
run: |
set -ex
cd examples/apo
uv run apo_custom_algorithm_trainer.py | tee _ci_apo.log
# Check whether the log contains "Best prompt found:"
grep "Best prompt found:" _ci_apo.log
env:
# New versions follow OPENAI_BASE_URL instead of OPENAI_API_BASE
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: APO custom algorithm debugger
run: |
set -ex
cd examples/apo
uv run apo_debug.py --mode runner
uv run apo_debug.py --mode hook
uv run apo_debug.py --mode trainer
env:
# New versions follow OPENAI_BASE_URL instead of OPENAI_API_BASE
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: APO built-in algorithm
run: |
set -ex
cd examples/apo
uv run room_selector_apo.py
env:
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
if: matrix.setup-script != 'legacy'
-307
View File
@@ -1,307 +0,0 @@
name: Examples - Calc-X
permissions:
contents: read
on:
schedule:
# Every day at 3 AM UTC+8
- cron: '0 19 * * *'
workflow_dispatch:
repository_dispatch:
types: [ci-calc-x, ci-all]
run-name: >-
${{ github.event_name == 'repository_dispatch'
&& format(
'PR #{0} - Label {1} - {2}',
github.event.client_payload.pull_number,
github.event.client_payload.ci_label,
github.event.client_payload.correlation_id
)
|| format('Calc-X - {0}', github.event_name) }}
jobs:
calc-x-perf:
if: >
github.event_name != 'repository_dispatch' ||
github.event.action == 'ci-calc-x' ||
github.event.action == 'ci-all'
name: Calc-X Performance (Python ${{ matrix.python-version }}, ${{ matrix.setup-script }})
runs-on: [self-hosted, 1ES.Pool=agl-runner-gpu]
timeout-minutes: 90
strategy:
matrix:
include:
- python-version: '3.10'
setup-script: 'legacy'
- python-version: '3.12'
setup-script: 'stable'
- python-version: '3.13'
setup-script: 'latest'
fail-fast: false
steps:
- name: Check GPU status
run: nvidia-smi
- name: Check disk space
run: df -h
- uses: actions/checkout@v4
with:
ref: ${{ github.event_name == 'repository_dispatch' && github.event.client_payload.pr_ref || (github.event.pull_request.number && format('refs/pull/{0}/merge', github.event.pull_request.number)) || github.ref }}
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
python-version: ${{ matrix.python-version }}
- name: Upgrade dependencies (latest)
run: uv lock --upgrade
if: matrix.setup-script == 'latest'
- name: Sync dependencies (latest)
run: |
uv sync --frozen --no-default-groups --extra verl \
--group dev --group experiment --group agents --group torch-gpu-stable
if: matrix.setup-script == 'latest'
- name: Sync dependencies (stable & legacy)
run: |
uv sync --frozen --no-default-groups --extra verl \
--group dev --group experiment --group agents --group torch-gpu-${{ matrix.setup-script }}
if: matrix.setup-script != 'latest'
- name: Freeze dependencies
run: |
set -ex
uv pip freeze | tee requirements-freeze.txt
echo "UV_LOCKED=1" >> $GITHUB_ENV
echo "UV_NO_SYNC=1" >> $GITHUB_ENV
- name: Upload dependencies artifact
uses: actions/upload-artifact@v4
with:
name: dependencies-calc-x-performance-${{ matrix.python-version }}-${{ matrix.setup-script }}
path: requirements-freeze.txt
compression-level: 0
- name: Launch LiteLLM Proxy
run: |
./scripts/litellm_run.sh
env:
AZURE_API_BASE: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_BASE }}
AZURE_API_KEY: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_KEY }}
- name: Prepare Calc-X dataset
run: |
set -ex
cd examples/calc_x
uv run gdown --fuzzy https://drive.google.com/file/d/1FQMyKLLd6hP9dw9rfZn1EZOWNvKaDsqw/view
unzip calc-x-data.zip -d data
rm calc-x-data.zip
- name: Calc-X MCP sanity check
run: |
set -ex
cd examples/calc_x
uv run tests/test_mcp_calculator.py
env:
OPENAI_API_BASE: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: Calc-X sanity check
run: |
set -ex
cd examples/calc_x
uv run legacy_calc_agent_debug.py
env:
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
# Calc-X training suddenly works after running the sanity check.
# And it has to be run before Spider training.
# The client side used to hang in many of my attempts.
# Don't ask why. Don't touch this.
- name: Calc-X training
run: |
source .venv/bin/activate
cd examples/calc_x
../../scripts/restart_ray.sh
sleep 5
python train_calc_agent.py --val-file data/test_mini.parquet --ci
shell: bash
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
id: calc_x_train
- name: Validate Calc-X training
run: |
set -ex
uv run scripts/validate_example_wandb.py ${{ steps.calc_x_train.outputs.project_name }} ${{ steps.calc_x_train.outputs.run_name }}
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
calc-x-variants:
if: >
github.event_name != 'repository_dispatch' ||
github.event.action == 'ci-calc-x' ||
github.event.action == 'ci-all'
name: Calc-X Variants (Python ${{ matrix.python-version }}, ${{ matrix.setup-script }})
runs-on: [self-hosted, 1ES.Pool=agl-runner-gpu]
timeout-minutes: 90
strategy:
matrix:
include:
- python-version: '3.10'
setup-script: 'legacy'
- python-version: '3.12'
setup-script: 'stable'
- python-version: '3.13'
setup-script: 'latest'
fail-fast: false
steps:
- name: Check GPU status
run: nvidia-smi
- name: Check disk space
run: df -h
- uses: actions/checkout@v4
with:
ref: ${{ github.event_name == 'repository_dispatch' && github.event.client_payload.pr_ref || (github.event.pull_request.number && format('refs/pull/{0}/merge', github.event.pull_request.number)) || github.ref }}
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
python-version: ${{ matrix.python-version }}
- name: Upgrade dependencies (latest)
run: uv lock --upgrade
if: matrix.setup-script == 'latest'
- name: Sync dependencies (latest)
run: |
uv sync --frozen --no-default-groups --extra verl \
--group dev --group experiment --group agents --group torch-gpu-stable
if: matrix.setup-script == 'latest'
- name: Sync dependencies (stable & legacy)
run: |
uv sync --frozen --no-default-groups --extra verl \
--group dev --group experiment --group agents --group torch-gpu-${{ matrix.setup-script }}
if: matrix.setup-script != 'latest'
- name: Freeze dependencies
run: |
set -ex
uv pip freeze | tee requirements-freeze.txt
echo "UV_LOCKED=1" >> $GITHUB_ENV
echo "UV_NO_SYNC=1" >> $GITHUB_ENV
- name: Upload dependencies artifact
uses: actions/upload-artifact@v4
with:
name: dependencies-calc-x-variants-${{ matrix.python-version }}-${{ matrix.setup-script }}
path: requirements-freeze.txt
compression-level: 0
- name: Launch LiteLLM Proxy
run: |
./scripts/litellm_run.sh
env:
AZURE_API_BASE: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_BASE }}
AZURE_API_KEY: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_KEY }}
- name: Prepare Calc-X dataset
run: |
set -ex
cd examples/calc_x
uv run gdown --fuzzy https://drive.google.com/file/d/1FQMyKLLd6hP9dw9rfZn1EZOWNvKaDsqw/view
unzip calc-x-data.zip -d data
rm calc-x-data.zip
- name: Calc-X MCP sanity check
run: |
set -ex
cd examples/calc_x
uv run tests/test_mcp_calculator.py
env:
OPENAI_API_BASE: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: Calc-X sanity check
run: |
set -ex
cd examples/calc_x
uv run legacy_calc_agent_debug.py
env:
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: Training with local model
run: |
set -ex
source .venv/bin/activate
cd examples/calc_x
../../scripts/restart_ray.sh
sleep 5
hf download Qwen/Qwen2.5-0.5B-Instruct --local-dir data/qwen_model
PYTHONUNBUFFERED=1 python train_calc_agent.py --val-file data/test_mini.parquet --ci-fast --model $(realpath data/qwen_model)
sleep 10
shell: bash
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
id: calc_x_train_local_model
- name: Training with LLM Proxy
run: |
set -ex
source .venv/bin/activate
cd examples/calc_x
../../scripts/restart_ray.sh
sleep 5
PYTHONUNBUFFERED=1 python train_calc_agent.py --val-file data/test_mini.parquet --ci-fast --llm-proxy
sleep 10
shell: bash
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
id: calc_x_train_llm_proxy
- name: Training with external store
run: |
set -euo pipefail
source .venv/bin/activate
cd examples/calc_x
../../scripts/restart_ray.sh
agl store --port 4747 &
sleep 5
AGL_MANAGED_STORE=0 AGL_CURRENT_ROLE=runner python train_calc_agent.py --external-store-address http://localhost:4747 --val-file data/test_mini.parquet --ci-fast &
sleep 5
AGL_MANAGED_STORE=0 AGL_CURRENT_ROLE=algorithm python train_calc_agent.py --external-store-address http://localhost:4747 --val-file data/test_mini.parquet --ci-fast
pkill -f agl && echo "SIGTERM sent to agl" || echo "No agl process found"
while pgrep -f agl; do
echo "Waiting for agl to finish..."
sleep 5
done
pkill -f train_calc_agent.py && echo "SIGTERM sent to train_calc_agent.py" || echo "No train_calc_agent.py process found"
while pgrep -f train_calc_agent.py; do
echo "Waiting for train_calc_agent.py to finish..."
sleep 5
done
echo "train_calc_agent.py has finished."
shell: bash
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
id: calc_x_train_external_store
- name: Training with role-based environment variables
run: |
set -euo pipefail
source .venv/bin/activate
cd examples/calc_x
../../scripts/restart_ray.sh
PYTHONUNBUFFERED=1 AGL_SERVER_HOST=127.0.0.1 AGL_SERVER_PORT=5858 AGL_CURRENT_ROLE=runner python train_calc_agent.py --val-file data/test_mini.parquet --ci-fast &
sleep 5
PYTHONUNBUFFERED=1 AGL_SERVER_HOST=0.0.0.0 AGL_SERVER_PORT=5858 AGL_CURRENT_ROLE=algorithm python train_calc_agent.py --val-file data/test_mini.parquet --ci-fast
pkill -f train_calc_agent.py && echo "SIGTERM sent to train_calc_agent.py" || echo "No train_calc_agent.py process found"
while pgrep -f train_calc_agent.py; do
echo "Waiting for train_calc_agent.py to finish..."
sleep 5
done
echo "train_calc_agent.py has finished."
shell: bash
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
-151
View File
@@ -1,151 +0,0 @@
name: Examples - Backward Compatibility
permissions:
contents: read
on:
schedule:
# Every day at 6 AM UTC+8
- cron: '0 22 * * *'
workflow_dispatch:
repository_dispatch:
types: [ci-compat, ci-all]
run-name: >-
${{ github.event_name == 'repository_dispatch'
&& format(
'PR #{0} - Label {1} - {2}',
github.event.client_payload.pull_number,
github.event.client_payload.ci_label,
github.event.client_payload.correlation_id
)
|| format('Backward Compatibility - {0}', github.event_name) }}
jobs:
backward-compatibility:
if: >
github.event_name != 'repository_dispatch' ||
github.event.action == 'ci-compat' ||
github.event.action == 'ci-all'
name: Backward Compatibility (Python ${{ matrix.python-version }}, ${{ matrix.setup-script }})
runs-on: [self-hosted, 1ES.Pool=agl-runner-gpu]
timeout-minutes: 30
strategy:
matrix:
include:
- python-version: '3.10'
setup-script: 'legacy'
- python-version: '3.12'
setup-script: 'stable'
fail-fast: false
steps:
- name: Check GPU status
run: nvidia-smi
- name: Check disk space
run: df -h
- uses: actions/checkout@v4
with:
ref: ${{ github.event_name == 'repository_dispatch' && github.event.client_payload.pr_ref || (github.event.pull_request.number && format('refs/pull/{0}/merge', github.event.pull_request.number)) || github.ref }}
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
python-version: ${{ matrix.python-version }}
- name: Sync dependencies
run: |
uv sync --frozen --no-default-groups --extra apo --extra verl \
--group dev --group experiment --group agents --group torch-gpu-${{ matrix.setup-script }}
- name: Override VERL (stable)
run: |
uv pip install verl==0.5.0
if: matrix.setup-script == 'stable'
- name: Freeze dependencies
run: |
set -ex
uv pip freeze | tee requirements-freeze.txt
echo "UV_LOCKED=1" >> $GITHUB_ENV
echo "UV_NO_SYNC=1" >> $GITHUB_ENV
- name: Upload dependencies artifact
uses: actions/upload-artifact@v4
with:
name: dependencies-backward-compatibility-${{ matrix.python-version }}-${{ matrix.setup-script }}
path: requirements-freeze.txt
compression-level: 0
- name: Launch LiteLLM Proxy
run: |
./scripts/litellm_run.sh
env:
AZURE_API_BASE: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_BASE }}
AZURE_API_KEY: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_KEY }}
- name: Prepare Calc-X dataset
run: |
set -ex
cd examples/calc_x
uv run gdown --fuzzy https://drive.google.com/file/d/1FQMyKLLd6hP9dw9rfZn1EZOWNvKaDsqw/view
unzip calc-x-data.zip -d data
rm calc-x-data.zip
- name: APO example (legacy client-server style)
run: |
set -ex
cd examples/apo
uv run legacy_apo_client.py &
sleep 3 # Wait for the client to be up
uv run legacy_apo_server.py
pkill -f legacy_apo_client.py && echo "SIGTERM sent to legacy_apo_client.py" || echo "No legacy_apo_client.py process found"
while pgrep -f legacy_apo_client.py; do
echo "Waiting for legacy_apo_client.py to finish..."
sleep 5
done
echo "legacy_apo_client.py has finished."
sleep 10
env:
OPENAI_API_BASE: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: Calc-X MCP sanity check
run: |
set -ex
cd examples/calc_x
uv run tests/test_mcp_calculator.py
env:
OPENAI_API_BASE: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: Calc-X sanity check
run: |
set -ex
cd examples/calc_x
uv run legacy_calc_agent_debug.py
env:
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: Calc-X training (legacy client-server style)
run: |
set -ex
source .venv/bin/activate
cd examples/calc_x
../../scripts/restart_ray.sh
sleep 5
PYTHONUNBUFFERED=1 python legacy_calc_agent.py &
bash legacy_train.sh
pkill -f legacy_calc_agent.py && echo "SIGTERM sent to legacy_calc_agent.py" || echo "No legacy_calc_agent.py process found"
while pgrep -f legacy_calc_agent.py; do
echo "Waiting for legacy_calc_agent.py to finish..."
sleep 5
done
echo "legacy_calc_agent.py has finished."
sleep 10
shell: bash
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
id: calc_x_train
- name: Validate Calc-X training
run: |
set -ex
uv run scripts/validate_example_wandb.py ${{ steps.calc_x_train.outputs.project_name }} ${{ steps.calc_x_train.outputs.run_name }}
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
-127
View File
@@ -1,127 +0,0 @@
name: Examples - Spider
permissions:
contents: read
on:
schedule:
# Every day at 4 AM UTC+8
- cron: '0 20 * * *'
workflow_dispatch:
repository_dispatch:
types: [ci-spider, ci-all]
run-name: >-
${{ github.event_name == 'repository_dispatch'
&& format(
'PR #{0} - Label {1} - {2}',
github.event.client_payload.pull_number,
github.event.client_payload.ci_label,
github.event.client_payload.correlation_id
)
|| format('Spider - {0}', github.event_name) }}
jobs:
spider:
if: >
github.event_name != 'repository_dispatch' ||
github.event.action == 'ci-spider' ||
github.event.action == 'ci-all'
name: Spider (Python ${{ matrix.python-version }}, ${{ matrix.setup-script }})
runs-on: [self-hosted, 1ES.Pool=agl-runner-gpu]
timeout-minutes: 60
strategy:
matrix:
include:
- python-version: '3.10'
setup-script: 'legacy'
- python-version: '3.12'
setup-script: 'stable'
- python-version: '3.13'
setup-script: 'latest'
fail-fast: false
steps:
- name: Check GPU status
run: nvidia-smi
- name: Check disk space
run: df -h
- uses: actions/checkout@v4
with:
ref: ${{ github.event_name == 'repository_dispatch' && github.event.client_payload.pr_ref || (github.event.pull_request.number && format('refs/pull/{0}/merge', github.event.pull_request.number)) || github.ref }}
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
python-version: ${{ matrix.python-version }}
- name: Upgrade dependencies (latest)
run: uv lock --upgrade
if: matrix.setup-script == 'latest'
- name: Sync dependencies (latest)
run: |
uv sync --frozen --no-default-groups --extra verl \
--group dev --group experiment --group agents --group torch-gpu-stable
if: matrix.setup-script == 'latest'
- name: Sync dependencies (stable & legacy)
run: |
uv sync --frozen --no-default-groups --extra verl \
--group dev --group experiment --group agents --group torch-gpu-${{ matrix.setup-script }}
if: matrix.setup-script != 'latest'
- name: Freeze dependencies
run: |
set -ex
uv pip freeze | tee requirements-freeze.txt
echo "UV_LOCKED=1" >> $GITHUB_ENV
echo "UV_NO_SYNC=1" >> $GITHUB_ENV
- name: Upload dependencies artifact
uses: actions/upload-artifact@v4
with:
name: dependencies-spider-${{ matrix.python-version }}-${{ matrix.setup-script }}
path: requirements-freeze.txt
compression-level: 0
- name: Launch LiteLLM Proxy
run: |
./scripts/litellm_run.sh
env:
AZURE_API_BASE: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_BASE }}
AZURE_API_KEY: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_KEY }}
- name: Prepare Spider dataset
run: |
set -ex
cd examples/spider
uv run gdown --fuzzy https://drive.google.com/file/d/1oi9J1jZP9TyM35L85CL3qeGWl2jqlnL6/view
unzip -q spider-data.zip -d data
rm spider-data.zip
- name: Spider sanity check
run: |
set -ex
cd examples/spider
uv run sql_agent.py
env:
OPENAI_API_BASE: http://localhost:12306/
OPENAI_API_KEY: dummy
if: success() || failure()
- name: Spider training
run: |
set -ex
source .venv/bin/activate
cd examples/spider
../../scripts/restart_ray.sh
sleep 5
PYTHONUNBUFFERED=1 python train_sql_agent.py fast
sleep 10
shell: bash
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
id: spider_train
- name: Validate Spider training
run: |
set -ex
uv run scripts/validate_example_wandb.py ${{ steps.spider_train.outputs.project_name }} ${{ steps.spider_train.outputs.run_name }}
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
-129
View File
@@ -1,129 +0,0 @@
name: Examples - Unsloth
permissions:
contents: read
on:
schedule:
# Every day at 5 AM UTC+8
- cron: '0 21 * * *'
workflow_dispatch:
repository_dispatch:
types: [ci-unsloth, ci-all]
run-name: >-
${{ github.event_name == 'repository_dispatch'
&& format(
'PR #{0} - Label {1} - {2}',
github.event.client_payload.pull_number,
github.event.client_payload.ci_label,
github.event.client_payload.correlation_id
)
|| format('Unsloth - {0}', github.event_name) }}
jobs:
unsloth:
if: >
github.event_name != 'repository_dispatch' ||
github.event.action == 'ci-unsloth' ||
github.event.action == 'ci-all'
name: Unsloth (Python ${{ matrix.python-version }}, ${{ matrix.setup-script }})
runs-on: [self-hosted, 1ES.Pool=agl-runner-gpu]
timeout-minutes: 60
strategy:
matrix:
# Legacy versions are not supported for Unsloth examples.
include:
- python-version: '3.12'
setup-script: 'stable'
- python-version: '3.13'
setup-script: 'latest'
fail-fast: false
steps:
- name: Check GPU status
run: nvidia-smi
- name: Check disk space
run: df -h
- uses: actions/checkout@v4
with:
ref: ${{ github.event_name == 'repository_dispatch' && github.event.client_payload.pr_ref || (github.event.pull_request.number && format('refs/pull/{0}/merge', github.event.pull_request.number)) || github.ref }}
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
python-version: ${{ matrix.python-version }}
- name: Upgrade dependencies (latest)
run: uv lock --upgrade
if: matrix.setup-script == 'latest'
- name: Sync dependencies
run: |
uv sync --frozen --no-default-groups --extra verl \
--group dev --group experiment --group trl --group agents --group torch-gpu-stable
- name: Freeze dependencies
run: |
set -ex
uv pip freeze | tee requirements-freeze.txt
echo "UV_LOCKED=1" >> $GITHUB_ENV
echo "UV_NO_SYNC=1" >> $GITHUB_ENV
- name: Upload dependencies artifact
uses: actions/upload-artifact@v4
with:
name: dependencies-unsloth-${{ matrix.python-version }}-${{ matrix.setup-script }}
path: requirements-freeze.txt
compression-level: 0
- name: Prepare Unsloth model
run: |
set -ex
cd examples/unsloth
rm -rf models
uv run hf download unsloth/Qwen3-4B-Instruct-2507 --local-dir models/version_0
- name: Unsloth SFT example
run: |
set -ex
source .venv/bin/activate
cd examples/unsloth
agl store --port 4747 &
sleep 5
python sft_rollout_runners.py &
sleep 5
python sft_algorithm.py
pkill -f agl && echo "SIGTERM sent to agl" || echo "No agl process found"
while pgrep -f agl; do
echo "Waiting for agl to finish..."
sleep 5
done
pkill -f sft_rollout_runners.py && echo "SIGTERM sent to sft_rollout_runners.py" || echo "No sft_rollout_runners.py process found"
while pgrep -f sft_rollout_runners.py; do
echo "Waiting for sft_rollout_runners.py to finish..."
sleep 5
done
echo "sft_rollout_runners.py has finished."
sleep 10
# Check models/version_2 must exist
if [ ! -d "models/version_2" ]; then
echo "models/version_2 does not exist"
exit 1
fi
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
- name: Unsloth SFT example all-in-one
run: |
set -ex
source .venv/bin/activate
cd examples/unsloth
rm -rf models/version_1 models/version_2
python sft_allinone.py
if [ ! -d "models/version_2" ]; then
echo "models/version_2 does not exist"
exit 1
fi
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
+330
View File
@@ -0,0 +1,330 @@
name: Examples Test
permissions:
contents: read
on:
schedule:
# Every day at 3 AM UTC+8
- cron: '0 19 * * *'
workflow_dispatch:
jobs:
examples:
runs-on: [self-hosted, 1ES.Pool=agl-runner-gpu]
timeout-minutes: 90
strategy:
matrix:
setup: [stable, latest]
fail-fast: false
steps:
- name: Check GPU status
run: nvidia-smi
- name: Check disk space
run: df -h
- uses: actions/checkout@v4
- name: Create a virtual environment
run: python3 -m venv .venv
- name: Install dependencies (${{ matrix.setup }})
run: |
. .venv/bin/activate
./scripts/setup_${{ matrix.setup }}_gpu.sh
- name: Freeze dependencies
run: |
. .venv/bin/activate
which python
which pip
which uvx
pip list | tee requirements-freeze.txt
- name: Upload dependencies artifact
uses: actions/upload-artifact@v4
with:
name: dependencies-${{ matrix.setup }}
path: requirements-freeze.txt
compression-level: 0
- name: Launch LiteLLM Proxy
run: |
set -ex
. .venv/bin/activate
litellm --config scripts/litellm_ci.yaml --port 12306 &
sleep 10 # Wait for the proxy to be up
env:
AZURE_API_BASE: ${{ secrets.AZURE_API_BASE }}
AZURE_API_KEY: ${{ secrets.AZURE_API_KEY }}
- name: Verify LiteLLM Proxy
run: |
set -ex
. .venv/bin/activate
python scripts/litellm_sanity_check.py
env:
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: Prepare Unsloth model
run: |
set -ex
. .venv/bin/activate
cd examples/unsloth
rm -rf models
hf download unsloth/Qwen3-4B-Instruct-2507 --local-dir models/version_0
- name: Prepare Spider dataset
run: |
set -ex
. .venv/bin/activate
cd examples/spider
gdown --fuzzy https://drive.google.com/file/d/1oi9J1jZP9TyM35L85CL3qeGWl2jqlnL6/view
unzip -q spider-data.zip -d data
rm spider-data.zip
- name: Prepare Calc-X dataset
run: |
set -ex
. .venv/bin/activate
cd examples/calc_x
gdown --fuzzy https://drive.google.com/file/d/1FQMyKLLd6hP9dw9rfZn1EZOWNvKaDsqw/view
unzip calc-x-data.zip -d data
rm calc-x-data.zip
# APO Examples test
- name: APO example (legacy)
run: |
set -ex
. .venv/bin/activate
cd examples/apo
python legacy_apo_client.py &
sleep 3 # Wait for the client to be up
python legacy_apo_server.py
pkill -f legacy_apo_client.py && echo "SIGTERM sent to legacy_apo_client.py" || echo "No legacy_apo_client.py process found"
while pgrep -f legacy_apo_client.py; do
echo "Waiting for legacy_apo_client.py to finish..."
sleep 5
done
echo "legacy_apo_client.py has finished."
sleep 10
env:
OPENAI_API_BASE: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: APO example
run: |
set -ex
. .venv/bin/activate
cd examples/apo
python apo.py | tee _ci_apo.log
# Check whether the log contains "Best prompt found:"
grep "Best prompt found:" _ci_apo.log
env:
# New versions follow OPENAI_BASE_URL instead of OPENAI_API_BASE
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: APO example debug sanity check
run: |
set -ex
. .venv/bin/activate
cd examples/apo
python apo_debug.py --mode runner
python apo_debug.py --mode trainer
env:
# New versions follow OPENAI_BASE_URL instead of OPENAI_API_BASE
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: APO built-in algorithm
run: |
set -ex
. .venv/bin/activate
cd examples/apo
python room_selector_apo.py
env:
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
if: success() || failure()
- name: Spider sanity check
run: |
set -ex
. .venv/bin/activate
cd examples/spider
python sql_agent.py --trainer.n-workers 1 --trainer.dev true --trainer.max-tasks 2
env:
VERL_API_BASE: http://localhost:9999/
OPENAI_API_BASE: http://localhost:12306/
OPENAI_API_KEY: dummy
if: success() || failure()
- name: Calc-X MCP sanity check
run: |
set -ex
. .venv/bin/activate
cd examples/calc_x
python tests/test_mcp_calculator.py
env:
OPENAI_API_BASE: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: Calc-X sanity check
run: |
set -ex
. .venv/bin/activate
cd examples/calc_x
python calc_agent_dev.py
env:
OPENAI_API_BASE: http://localhost:12306/
OPENAI_API_KEY: dummy
# Calc-X training suddenly works after running the sanity check.
# And it has to be run before Spider training.
# The client side used to hang in many of my attempts.
# Don't ask why. Don't touch this.
- name: Calc-X training v0.1
run: |
set -ex
source .venv/bin/activate
cd examples/calc_x
../../scripts/restart_ray.sh
sleep 5
PYTHONUNBUFFERED=1 python calc_agent.py &
bash train_ci.sh
pkill -f calc_agent.py && echo "SIGTERM sent to calc_agent.py" || echo "No calc_agent.py process found"
while pgrep -f calc_agent.py; do
echo "Waiting for calc_agent.py to finish..."
sleep 5
done
echo "calc_agent.py has finished."
sleep 10
shell: bash
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
id: calc_x_train
if: success() || failure()
- name: Validate Calc-X training
run: |
set -ex
. .venv/bin/activate
python scripts/validate_example_wandb.py ${{ steps.calc_x_train.outputs.project_name }} ${{ steps.calc_x_train.outputs.run_name }}
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
- name: Calc-X training v0.2
run: |
set -ex
source .venv/bin/activate
cd examples/calc_x
../../scripts/restart_ray.sh
sleep 5
PYTHONUNBUFFERED=1 python calc_agent_v0_2.py
sleep 10
shell: bash
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
id: calc_x_train_v0_2
if: success() || failure()
- name: Calc-X training v0.2 LLM Proxy
run: |
set -ex
source .venv/bin/activate
cd examples/calc_x
../../scripts/restart_ray.sh
sleep 5
PYTHONUNBUFFERED=1 python calc_agent_v0_2_llm_proxy.py
sleep 10
shell: bash
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
id: calc_x_train_v0_2_llm_proxy
if: success() || failure()
- name: Spider training
run: |
set -ex
source .venv/bin/activate
cd examples/spider
../../scripts/restart_ray.sh
sleep 5
PYTHONUNBUFFERED=1 python sql_agent.py --trainer.n-workers 10 &
bash train_ci.sh
pkill -f sql_agent.py && echo "SIGTERM sent to sql_agent.py" || echo "No sql_agent.py process found"
while pgrep -f sql_agent.py; do
echo "Waiting for sql_agent.py to finish..."
sleep 5
done
echo "sql_agent.py has finished."
sleep 10
shell: bash
env:
VERL_API_BASE: http://localhost:9991/
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
id: spider_train
if: success() || failure()
- name: Validate Spider training
run: |
set -ex
. .venv/bin/activate
python scripts/validate_example_wandb.py ${{ steps.spider_train.outputs.project_name }} ${{ steps.spider_train.outputs.run_name }}
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
# Unsloth Examples test
- name: Unsloth SFT example
run: |
set -ex
. .venv/bin/activate
cd examples/unsloth
agl store --port 4747 &
sleep 5
python sft_rollout_runners.py &
sleep 5
python sft_algorithm.py
pkill -f agl && echo "SIGTERM sent to agl" || echo "No agl process found"
while pgrep -f agl; do
echo "Waiting for agl to finish..."
sleep 5
done
pkill -f sft_rollout_runners.py && echo "SIGTERM sent to sft_rollout_runners.py" || echo "No sft_rollout_runners.py process found"
while pgrep -f sft_rollout_runners.py; do
echo "Waiting for sft_rollout_runners.py to finish..."
sleep 5
done
echo "sft_rollout_runners.py has finished."
sleep 10
# Check models/version_2 must exist
if [ ! -d "models/version_2" ]; then
echo "models/version_2 does not exist"
exit 1
fi
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
if: ${{ (success() || failure()) && matrix.setup == 'latest' }}
- name: Unsloth SFT example all-in-one
run: |
set -ex
. .venv/bin/activate
cd examples/unsloth
rm -rf models/version_1 models/version_2
python sft_allinone.py
if [ ! -d "models/version_2" ]; then
echo "models/version_2 does not exist"
exit 1
fi
env:
WANDB_BASE_URL: ${{ secrets.MSR_WANDB_BASE_URL }}
WANDB_API_KEY: ${{ secrets.MSR_WANDB_API_KEY }}
if: matrix.setup == 'latest'
# Cleanup
- name: Cleanup
run: ./scripts/cleanup.sh
if: success() || failure()
-309
View File
@@ -1,309 +0,0 @@
name: Issue Comment
on:
issue_comment:
types: [created]
permissions:
pull-requests: write
issues: write
contents: write
actions: read
jobs:
dispatch:
# Only run for comments on pull requests AND when the comment starts with "/ci"
if: >
github.event.issue.pull_request != null &&
startsWith(github.event.comment.body, '/ci')
runs-on: ubuntu-latest
outputs:
dispatched: ${{ steps.dispatch.outputs.dispatched }}
event_types: ${{ steps.dispatch.outputs.event_types }}
correlation_id: ${{ steps.dispatch.outputs.correlation_id }}
trigger_comment_id: ${{ steps.dispatch.outputs.trigger_comment_id }}
ack_comment_id: ${{ steps.ack.outputs.comment_id }}
steps:
- name: Guardrail — allow only members/collaborators
id: guard
uses: actions/github-script@v8
with:
script: |
const allowed = ['MEMBER','OWNER','COLLABORATOR'];
const assoc = context.payload.comment.author_association;
if (!allowed.includes(assoc)) {
core.notice(`Ignoring /ci from ${context.payload.comment.user.login} (author_association=${assoc}).`);
core.setOutput('skip', 'true');
}
- name: Trigger repository dispatch
id: dispatch
if: steps.guard.outputs.skip != 'true'
uses: actions/github-script@v8
with:
script: |
const owner = context.repo.owner;
const repo = context.repo.repo;
const pull_number = context.payload.issue.number;
const comment = context.payload.comment;
// Fetch current PR state
const { data: pr } = await github.rest.pulls.get({ owner, repo, pull_number });
// Add reaction so folks know we saw it
try {
await github.rest.reactions.createForIssueComment({
owner,
repo,
comment_id: comment.id,
content: 'rocket'
});
} catch (e) {
core.info('Could not add reaction (likely due to permissions). Continuing.');
}
const labels = (pr.labels ?? []).map(label => label.name);
const directCiLabels = labels.filter(label => label.startsWith('ci-'));
const hasCiAll = directCiLabels.includes('ci-all');
const dedupe = new Set(
directCiLabels.filter(label => label !== 'ci-all')
);
if (!hasCiAll && dedupe.size === 0) {
core.notice('No ci-* labels found on the pull request; nothing to dispatch.');
core.setOutput('dispatched', 'false');
core.setOutput('event_types', '');
return;
}
const correlation_id = `id-${comment.id}-${Date.now().toString(36)}`;
const clientPayload = {
correlation_id,
pull_number,
pr_ref: `refs/pull/${pull_number}/merge`,
pr_head_ref: pr.head.ref,
pr_head_sha: pr.head.sha,
pr_base_ref: pr.base.ref,
pr_base_sha: pr.base.sha,
trigger_comment_id: comment.id,
trigger_comment_user: comment.user.login,
};
const eventTypes = hasCiAll
? ['ci-all']
: Array.from(dedupe);
for (const eventType of eventTypes) {
await github.rest.repos.createDispatchEvent({
owner,
repo,
event_type: eventType,
client_payload: { ...clientPayload, ci_label: eventType }
});
core.notice(`Dispatched '${eventType}' event for PR #${pull_number}.`);
}
core.setOutput('dispatched', 'true');
core.setOutput('event_types', eventTypes.join(','));
core.setOutput('correlation_id', correlation_id);
core.setOutput('trigger_comment_id', String(comment.id));
- name: Acknowledge in thread (optional)
if: steps.guard.outputs.skip != 'true' && steps.dispatch.outputs.dispatched == 'true'
id: ack
uses: actions/github-script@v8
env:
EVENT_TYPES: ${{ steps.dispatch.outputs.event_types }}
CORRELATION_ID: ${{ steps.dispatch.outputs.correlation_id }}
with:
script: |
const eventTypes = (process.env.EVENT_TYPES || '')
.split(',')
.map(label => label.trim())
.filter(Boolean);
const formatted = eventTypes.map(label => `\`repository_dispatch:${label}\``).join(', ');
const { owner, repo } = context.repo;
const issue_number = context.payload.issue.number;
const body = [
`✅ CI trigger requested by @${context.payload.comment.user.login}.`,
`Fired ${formatted}.`,
'',
`_Collecting run links for correlation \`${process.env.CORRELATION_ID}\`…_`
].join('\n');
const { data: comment } = await github.rest.issues.createComment({
owner, repo, issue_number,
body
});
core.setOutput('comment_id', String(comment.id));
- name: Notify missing ci label
if: steps.guard.outputs.skip != 'true' && steps.dispatch.outputs.dispatched != 'true'
uses: actions/github-script@v8
with:
script: |
const { owner, repo } = context.repo;
const issue_number = context.payload.issue.number;
await github.rest.issues.createComment({
owner,
repo,
issue_number,
body: `⚠️ CI trigger ignored because the pull request has no \`ci-*\` labels (e.g. \`ci-apo\`, \`ci-calc-x\`). Add the desired labels and try \`/ci\` again.`
});
watch:
needs: dispatch
if: needs.dispatch.outputs.dispatched == 'true'
runs-on: ubuntu-latest
timeout-minutes: 180
steps:
- name: Track dispatched runs and update comment
uses: actions/github-script@v8
env:
CORRELATION_ID: ${{ needs.dispatch.outputs.correlation_id }}
ACK_COMMENT_ID: ${{ needs.dispatch.outputs.ack_comment_id }}
TRIGGER_COMMENT_ID: ${{ needs.dispatch.outputs.trigger_comment_id }}
with:
script: |
const owner = context.repo.owner;
const repo = context.repo.repo;
const correlationId = process.env.CORRELATION_ID;
if (!correlationId) {
core.warning('No correlation id supplied; nothing to watch.');
return;
}
const ackCommentId = Number(process.env.ACK_COMMENT_ID || 0);
if (!ackCommentId) {
core.warning('No comment id available for updates; skipping watch.');
return;
}
const triggerCommentId = Number(process.env.TRIGGER_COMMENT_ID || 0);
if (!triggerCommentId) {
core.warning('No trigger comment id available; skipping watch.');
return;
}
const prefix = `🚀 CI Watcher for correlation ${correlationId} triggered by comment ${triggerCommentId}`;
core.notice(`Watching workflow runs for correlation '${correlationId}' using comment ${ackCommentId}.`);
function fmt(run) {
const status = run.status;
const conclusion = run.conclusion;
const badge = status === 'completed'
? (conclusion === 'success' ? '🟢' : conclusion === 'failure' ? '🔴' : '🟡')
: (status === 'in_progress' ? '🟣' : '⚪️');
const title = run.display_title || run.name || `run ${run.id}`;
const statusText = status === 'completed' ? `${status}/${conclusion}` : status;
return `- ${badge} [${title}](${run.html_url}) — \`${statusText}\``;
}
const signatureOf = runs =>
runs
.map(run => `${run.id}:${run.status}/${run.conclusion || ''}`)
.sort()
.join('|');
const deadlineMs = Date.now() + 175 * 60 * 1000; // 175 minutes
let found = [];
async function searchOnce() {
const runs = await github.paginate(
github.rest.actions.listWorkflowRunsForRepo,
{ owner, repo, event: 'repository_dispatch', per_page: 100 }
);
const cutoff = new Date(Date.now() - 60 * 60 * 1000); // last hour
return runs.filter(run => {
const createdAt = new Date(run.created_at);
const title = String(run.display_title || run.name || '');
return createdAt >= cutoff && title.includes(correlationId);
});
}
while (Date.now() < deadlineMs) {
found = await searchOnce();
if (found.length > 0) {
core.notice(`Discovered ${found.length} workflow run(s) for correlation '${correlationId}'.`);
break;
}
core.notice(`No runs found yet for correlation '${correlationId}'; retrying shortly.`);
await new Promise(res => setTimeout(res, 10000));
}
if (found.length === 0) {
core.notice(`Watcher timed out with no runs for correlation '${correlationId}'; notifying thread.`);
await github.rest.issues.updateComment({
owner,
repo,
comment_id: ackCommentId,
body: [
prefix,
`⚠️ I couldn't find any workflow runs for correlation \`${correlationId}\`.`,
`They may be delayed or misconfigured.`
].join('\n')
});
return;
}
const runIds = new Set(found.map(run => run.id));
let lastSignature = '';
async function refreshRuns() {
const ids = Array.from(runIds);
const refreshed = [];
for (const id of ids) {
const { data } = await github.rest.actions.getWorkflowRun({
owner,
repo,
run_id: id
});
refreshed.push(data);
}
return refreshed;
}
async function updateCommentIfChanged(runs, allDone) {
const signature = signatureOf(runs);
if (signature === lastSignature) {
// Run statuses unchanged; skipping comment update.
return;
}
lastSignature = signature;
core.notice(`Updating comment ${ackCommentId} with ${runs.length} run status entries (allDone=${allDone}).`);
await github.rest.issues.updateComment({
owner,
repo,
comment_id: ackCommentId,
body: [
prefix,
`🏃‍♀️ Tracking ${runs.length} workflow run(s):`,
'',
...runs.map(fmt),
'',
allDone ? '✅ All runs completed.' : '_Still running…_'
].join('\n')
});
}
await updateCommentIfChanged(found, found.every(run => run.status === 'completed'));
while (Date.now() < deadlineMs) {
const latest = await searchOnce();
for (const run of latest) {
if (!runIds.has(run.id)) {
runIds.add(run.id);
core.notice(`Detected additional run ${run.id} (${run.name || run.display_title || 'unnamed'}) for correlation '${correlationId}'.`);
}
}
const current = await refreshRuns();
const allDone = current.every(run => run.status === 'completed');
await updateCommentIfChanged(current, allDone);
if (allDone) {
core.notice(`All runs for correlation '${correlationId}' completed; stopping watcher.`);
break;
}
await new Promise(res => setTimeout(res, 60000));
}
if (Date.now() >= deadlineMs) {
core.warning(`Watcher hit the deadline while monitoring correlation '${correlationId}'.`);
}
+19 -19
View File
@@ -2,8 +2,8 @@ name: PyPI Nightly Build
on:
schedule:
# Run daily at 6:00 AM UTC+8
- cron: '0 22 * * *'
# Run daily at 6:00 AM UTC
- cron: '0 6 * * *'
workflow_dispatch: # Allow manual trigger
jobs:
@@ -14,25 +14,18 @@ jobs:
contents: read
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: actions/setup-python@v6
- name: Checkout code
uses: actions/checkout@v4
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.12'
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
- name: Sync dependencies
run: uv sync --frozen --no-default-groups --group dev
- uses: actions/setup-node@v6
with:
node-version: '22'
- name: Install JavaScript dependencies
run: cd dashboard && npm ci
- name: Build dashboard
run: cd dashboard && npm run build
- name: Install build dependencies
run: |
python -m pip install --upgrade pip
pip install -e .[dev]
- name: Get current version
id: get_version
@@ -51,9 +44,16 @@ jobs:
- name: Build package
run: |
uv build
hatch build
- name: Publish to Test PyPI
uses: pypa/gh-action-pypi-publish@release/v1
with:
repository-url: https://test.pypi.org/legacy/
- name: Test installation from Test PyPI
run: |
# Wait a bit for the package to be available
sleep 30
pip install --index-url https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple/ agentlightning
python -c "import agentlightning; print('Package installed successfully')"
+19 -19
View File
@@ -48,34 +48,34 @@ jobs:
contents: read
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: actions/setup-python@v6
- name: Checkout code
uses: actions/checkout@v4
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.12'
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
- name: Sync dependencies
run: uv sync --frozen --no-default-groups --group dev
- uses: actions/setup-node@v6
with:
node-version: '22'
- name: Install JavaScript dependencies
run: cd dashboard && npm ci
- name: Build dashboard
run: cd dashboard && npm run build
- name: Install build dependencies
run: |
python -m pip install --upgrade pip
pip install -e .[dev]
- name: Build package
run: |
uv build
hatch build
- name: Verify package contents
run: |
uv run --locked --no-sync python -m tarfile -l dist/*.tar.gz
uv run --locked --no-sync python -m zipfile -l dist/*.whl
python -m tarfile -l dist/*.tar.gz
python -m zipfile -l dist/*.whl
- name: Publish to PyPI
uses: pypa/gh-action-pypi-publish@release/v1
- name: Test installation from PyPI
run: |
# Wait a bit for the package to be available
sleep 30
pip install --index-url https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple/ agentlightning
python -c "import agentlightning; print('Package installed successfully')"
+30 -58
View File
@@ -8,89 +8,61 @@ on:
workflow_dispatch:
repository_dispatch:
types: [ci-gpu, ci-all]
run-name: >-
${{ github.event_name == 'repository_dispatch'
&& format(
'PR #{0} - Label {1} - {2}',
github.event.client_payload.pull_number,
github.event.client_payload.ci_label,
github.event.client_payload.correlation_id
)
|| format('GPU Test - {0}', github.event_name) }}
jobs:
tests-full:
if: >
github.event_name != 'repository_dispatch' ||
github.event.action == 'ci-gpu' ||
github.event.action == 'ci-all'
name: GPU Test with Python ${{ matrix.python-version }} (${{ matrix.setup-script }})
runs-on: [self-hosted, 1ES.Pool=agl-runner-gpu]
timeout-minutes: 30
strategy:
matrix:
include:
- python-version: '3.10'
setup-script: 'legacy'
- python-version: '3.12'
setup-script: 'stable'
- python-version: '3.13'
setup-script: 'latest'
setup: [stable, latest]
fail-fast: false
steps:
- name: Check GPU status
run: nvidia-smi
- uses: actions/checkout@v4
with:
ref: ${{ github.event_name == 'repository_dispatch' && github.event.client_payload.pr_ref || (github.event.pull_request.number && format('refs/pull/{0}/merge', github.event.pull_request.number)) || github.ref }}
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
python-version: ${{ matrix.python-version }}
- name: Upgrade dependencies (latest)
run: uv lock --upgrade
if: matrix.setup-script == 'latest'
- name: Sync dependencies (latest)
run: uv sync --frozen --no-default-groups --extra apo --group dev --group agents --group torch-gpu-stable
if: matrix.setup-script == 'latest'
- name: Sync dependencies (stable & legacy)
run: uv sync --frozen --no-default-groups --extra apo --group dev --group agents --group torch-gpu-${{ matrix.setup-script }}
if: matrix.setup-script != 'latest'
- name: Create a virtual environment
run: python3 -m venv .venv
- name: Install dependencies (${{ matrix.setup }})
run: |
. .venv/bin/activate
./scripts/setup_${{ matrix.setup }}_gpu.sh
- name: Freeze dependencies
run: |
set -ex
uv pip freeze | tee requirements-freeze.txt
echo "UV_LOCKED=1" >> $GITHUB_ENV
echo "UV_NO_SYNC=1" >> $GITHUB_ENV
. .venv/bin/activate
which python
which pip
which uvx
pip list | tee requirements-freeze.txt
- name: Upload dependencies artifact
uses: actions/upload-artifact@v4
with:
name: dependencies-${{ matrix.python-version }}-${{ matrix.setup-script }}
name: dependencies-${{ matrix.setup }}
path: requirements-freeze.txt
compression-level: 0
- uses: actions/setup-node@v6
with:
node-version: '22'
- name: Install JavaScript dependencies
run: cd dashboard && npm ci
- name: Build dashboard
run: cd dashboard && npm run build
- name: Launch LiteLLM Proxy
run: |
./scripts/litellm_run.sh
set -ex
. .venv/bin/activate
litellm --config scripts/litellm_ci.yaml --port 12306 &
sleep 10 # Wait for the proxy to be up
env:
AZURE_API_BASE: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_BASE }}
AZURE_API_KEY: ${{ secrets.AZURE_GROUP_SUBSCRIPTION_API_KEY }}
AZURE_API_BASE: ${{ secrets.AZURE_API_BASE }}
AZURE_API_KEY: ${{ secrets.AZURE_API_KEY }}
- name: Verify LiteLLM Proxy
run: |
set -ex
. .venv/bin/activate
python scripts/litellm_sanity_check.py
env:
OPENAI_BASE_URL: http://localhost:12306/
OPENAI_API_KEY: dummy
- name: Run tests
run: |
uv run pytest -v --durations=0 tests
set -ex
. .venv/bin/activate
pytest -v --durations=0 tests
env:
PYTEST_ADDOPTS: "--color=yes"
OPENAI_BASE_URL: http://localhost:12306/
+46 -113
View File
@@ -5,9 +5,9 @@ permissions:
on:
push:
branches: [ main, stable/**/* ]
branches: [ main ]
pull_request:
branches: [ main, stable/**/* ]
branches: [ main ]
workflow_dispatch:
schedule:
@@ -16,94 +16,70 @@ on:
jobs:
lint:
strategy:
matrix:
setup: [fast, slow]
name: Lint - ${{ matrix.setup }}
lint-fast:
name: Lint - Fast
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@v4
- uses: astral-sh/setup-uv@v7
- uses: actions/checkout@v3
- uses: actions/setup-python@v4
with:
enable-cache: true
python-version: '3.12'
- name: Sync dependencies (fast)
run: uv sync --frozen --group dev --no-default-groups
if: matrix.setup == 'fast'
- name: Sync dependencies (slow)
- name: Install dependencies
run: |
uv sync --frozen \
--extra apo \
--extra verl \
--group dev \
--group torch-cpu \
--group torch-stable \
--group trl \
--group tinker \
--group agents \
--no-default-groups
if: matrix.setup == 'slow'
# This pre-commit skips JavaScript on purpose.
python -m pip install --upgrade pip
pip install -e .[dev]
- name: Run pre-commit
uses: pre-commit/action@v3.0.1
- name: Check Python headers
run: uv run --locked --no-sync scripts/check_headers.py
run: |
python scripts/check_python_headers.py
- name: Run Black
run: uv run --locked --no-sync black --check .
run: black --check .
- name: Run isort
run: uv run --locked --no-sync isort --check-only .
- name: Run pyright (fast)
run: uv run --locked --no-sync pyright -p pyrightconfig.fast.json
if: matrix.setup == 'fast'
- name: Run pyright (slow)
run: uv run --locked --no-sync pyright -p pyrightconfig.json
if: matrix.setup == 'slow'
run: isort --check-only .
- name: Run pyright
run: pyright -p pyrightconfig.fast.json
lint-js:
name: Lint - JavaScript
lint-slow:
name: Lint - Slow
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@v4
- uses: actions/setup-node@v6
- uses: actions/checkout@v3
- uses: actions/setup-python@v4
with:
node-version: '22'
python-version: '3.12'
- name: Install dependencies
run: cd dashboard && npm ci
- name: Run ESLint
run: cd dashboard && npm run eslint
- name: Run Prettier
run: cd dashboard && npm run prettier
- name: Run Stylelint
run: cd dashboard && npm run stylelint
- name: Run Typecheck
run: cd dashboard && npm run typecheck
- name: Verify build
run: cd dashboard && npm run build
run: |
./scripts/setup_type_checking.sh
- name: Run Black
run: black --check .
- name: Run isort
run: isort --check-only .
- name: Run pyright
run: pyright -p pyrightconfig.json
docs:
name: Build documentation
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v3
with:
fetch-depth: 0
- uses: actions/setup-python@v6
- uses: actions/setup-python@v4
with:
python-version: '3.12'
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
- name: Sync dependencies
run: uv sync --frozen --no-default-groups --group dev
- name: Install documentation dependencies
run: |
./scripts/setup_stable.sh
- name: Set source commit for docs
run: |
echo "SOURCE_COMMIT=${{ github.sha }}" >> $GITHUB_ENV
- name: Build documentation
run: uv run --locked --no-sync mkdocs build --strict
run: |
mkdocs build --strict
- name: Upload docs artifact
uses: actions/upload-artifact@v4
with:
@@ -116,78 +92,35 @@ jobs:
matrix:
include:
- python-version: '3.10'
setup-script: 'legacy'
- python-version: '3.11'
setup-script: 'stable'
- python-version: '3.12'
setup-script: 'stable'
- python-version: '3.13'
setup-script: 'latest'
- python-version: '3.12'
setup-script: 'stable'
fail-fast: false
name: Test with Python ${{ matrix.python-version }} (${{ matrix.setup-script }})
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- uses: actions/checkout@v4
- uses: astral-sh/setup-uv@v7
- uses: actions/checkout@v3
- uses: actions/setup-python@v4
with:
enable-cache: true
python-version: ${{ matrix.python-version }}
- name: Upgrade dependencies (latest)
run: uv lock --upgrade
if: matrix.setup-script == 'latest'
- name: Sync dependencies (latest)
run: uv sync --frozen --no-default-groups --extra apo --group dev --group agents --group core-stable
if: matrix.setup-script == 'latest'
- name: Sync dependencies (stable & legacy)
run: uv sync --frozen --no-default-groups --extra apo --group dev --group agents --group core-${{ matrix.setup-script }}
if: matrix.setup-script != 'latest'
- name: Install dependencies
run: |
./scripts/setup_${{ matrix.setup-script }}.sh
- name: Freeze dependencies
run: |
set -ex
uv pip freeze | tee requirements-freeze.txt
echo "UV_LOCKED=1" >> $GITHUB_ENV
echo "UV_NO_SYNC=1" >> $GITHUB_ENV
pip list | tee requirements-freeze-${{ matrix.python-version }}-${{ matrix.setup-script }}.txt
- name: Upload dependencies artifact
uses: actions/upload-artifact@v4
with:
name: dependencies-${{ matrix.python-version }}-${{ matrix.setup-script }}
path: requirements-freeze.txt
name: dependencies-python-${{ matrix.python-version }}-${{ matrix.setup-script }}
path: requirements-freeze-${{ matrix.python-version }}-${{ matrix.setup-script }}.txt
compression-level: 0
- uses: actions/setup-node@v6
with:
node-version: '22'
- name: Install JavaScript dependencies
run: cd dashboard && npm ci
- name: Build dashboard
run: cd dashboard && npm run build
- name: Run tests
run: |
uv run pytest -v --durations=0 tests
pytest -v --durations=0 tests
env:
PYTEST_ADDOPTS: "--color=yes"
test-js:
name: Test - JavaScript
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: actions/setup-node@v6
with:
node-version: '22'
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
python-version: '3.12'
- name: Sync Python dependencies
run: uv sync --frozen --no-default-groups --extra apo --group dev --group agents --group core-stable
- name: Install JavaScript dependencies
run: cd dashboard && npm ci
- name: Run vitest
run: cd dashboard && npm run vitest
-9
View File
@@ -189,9 +189,6 @@ cython_debug/
# you could uncomment the following to ignore the enitre vscode folder
.vscode/
# Emacs backup files
*~
# Ruff stuff:
.ruff_cache/
@@ -207,9 +204,3 @@ cython_debug/
# Claude
.claude/*.local.json
# Dashboard generated files
agentlightning/dashboard/**/*.css
agentlightning/dashboard/**/*.js
agentlightning/dashboard/**/*.html
agentlightning/dashboard/**/*.svg
-52
View File
@@ -8,8 +8,6 @@ repos:
exclude: ^mkdocs\.yml$
- id: check-toml
- id: check-added-large-files
args: ["--maxkb=1024"]
exclude: (^uv\.lock$)|(^docs/assets/.*\.svg$)
- id: check-shebang-scripts-are-executable
- id: detect-private-key
- repo: https://github.com/pycqa/isort
@@ -24,53 +22,3 @@ repos:
pass_filenames: false
always_run: true
args: ["."]
- repo: local
hooks:
- id: prettier
name: prettier (dashboard)
language: system
pass_filenames: false
always_run: true
entry: >
bash -c '
cd dashboard || exit 1
if [ -d node_modules ]; then
echo "✅ node_modules already exists"
npx prettier --cache --write "**/*.{ts,tsx,mjs,cjs}"
else
echo "⚠️ node_modules not found — npx is not reliable. Skipping."
fi
'
- id: eslint
name: eslint (dashboard)
language: system
pass_filenames: false
always_run: true
entry: >
bash -c '
cd dashboard || exit 1
if [ -d node_modules ]; then
echo "✅ node_modules already exists"
npx eslint --cache --fix .
else
echo "⚠️ node_modules not found — npx is not reliable. Skipping."
fi
'
- id: stylelint
name: stylelint (dashboard)
language: system
pass_filenames: false
always_run: true
entry: >
bash -c '
cd dashboard || exit 1
if [ -d node_modules ]; then
echo "✅ node_modules already exists"
npx stylelint --cache --fix "**/*.css"
else
echo "⚠️ node_modules not found — npx is not reliable. Skipping."
fi
'
-1
View File
@@ -1 +0,0 @@
3.12
+111 -48
View File
@@ -1,14 +1,13 @@
<p align="center">
<img src="docs/assets/readme-banner.svg" alt="Agent-lightning-banner" style="width:600px"/>
</p>
<div style="text-align:center; margin-bottom:20px;">
<img src="docs/assets/readme-banner.png" alt="Agent-lightning-banner" style="max-width:600px"/>
</div>
# Agent Lightning⚡
[![Unit Tests](https://github.com/microsoft/agent-lightning/actions/workflows/badge-unit.yml/badge.svg)](https://github.com/microsoft/agent-lightning/actions/workflows/badge-unit.yml)
[![Documentation](https://img.shields.io/badge/GitHub%20Pages-Documentation-blue)](https://microsoft.github.io/agent-lightning/)
[![CPU Test](https://github.com/microsoft/agent-lightning/actions/workflows/tests.yml/badge.svg)](https://github.com/microsoft/agent-lightning/actions/workflows/tests.yml)
[![GPU Test](https://github.com/microsoft/agent-lightning/actions/workflows/examples.yml/badge.svg)](https://github.com/microsoft/agent-lightning/actions/workflows/examples.yml)
[![PyPI version](https://badge.fury.io/py/agentlightning.svg)](https://badge.fury.io/py/agentlightning)
[![License](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
[![Ask DeepWiki](https://deepwiki.com/badge.svg)](https://deepwiki.com/microsoft/agent-lightning)
[![Discord](https://img.shields.io/badge/Discord-Join-5865F2?logo=discord&logoColor=white)](https://discord.gg/RYk7CdvDR7)
**The absolute trainer to light up AI agents.**
@@ -18,36 +17,14 @@ Join our [Discord community](https://discord.gg/RYk7CdvDR7) to connect with othe
## ⚡ Core Features
- Turn your agent into an optimizable beast with **ZERO CODE CHANGE** (almost)! 💤
- Build with **ANY** agent framework (LangChain, OpenAI Agent SDK, AutoGen, CrewAI, Microsoft Agent Framework...); or even WITHOUT agent framework (Python OpenAI). You name it! 🤖
- Build with **ANY** agent framework (LangChain, OpenAI Agent SDK, AutoGen, CrewAI, ...); or even WITHOUT agent framework (Python OpenAI). You name it! 🤖
- **Selectively** optimize one or more agents in a multi-agent system. 🎯
- Embraces **Algorithms** like Reinforcement Learning, Automatic Prompt Optimization, Supervised Fine-tuning and more. 🤗
- Embraces Reinforcement Learning, Automatic Prompt Optimization and more **algorithms**. 🤗
Read more on our [documentation website](https://microsoft.github.io/agent-lightning/).
![Agent-Lightning-code-diff](docs/assets/readme-diff.png)
<p align="center">
<img src="docs/assets/readme-diff.svg" alt="Agent-Lightning Core Quickstart" style="width:100%"/>
</p>
## ⚡ Resources
## ⚡ Installation
```bash
pip install agentlightning
```
For the latest nightly build (cutting-edge features), you can install from Test PyPI:
```bash
pip install --upgrade --index-url https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple/ agentlightning
```
Please refer to our [installation guide](https://microsoft.github.io/agent-lightning/stable/tutorials/installation/) for more details.
To start using Agent-lightning, check out our [documentation](https://microsoft.github.io/agent-lightning/) and [examples](./examples).
## ⚡ Articles
- 11/4/2025 [Tuning ANY AI agent with Tinker ✕ Agent-lightning](https://medium.com/@yugez/tuning-any-ai-agent-with-tinker-agent-lightning-part-1-1d8c9a397f0e) Medium. See also [Part 2](https://medium.com/@yugez/tuning-any-ai-agent-with-tinker-agent-lightning-part-2-332c5437f0dc).
- 10/22/2025 [No More Retokenization Drift: Returning Token IDs via the OpenAI Compatible API Matters in Agent RL](https://blog.vllm.ai/2025/10/22/agent-lightning.html) vLLM blog. See also [Zhihu writeup](https://zhuanlan.zhihu.com/p/1965067274642785725).
- 8/11/2025 [Training AI Agents to Write and Self-correct SQL with Reinforcement Learning](https://medium.com/@yugez/training-ai-agents-to-write-and-self-correct-sql-with-reinforcement-learning-571ed31281ad) Medium.
- 8/5/2025 [Agent Lightning: Train ANY AI Agents with Reinforcement Learning](https://arxiv.org/abs/2508.03680) arXiv paper.
- 7/26/2025 [We discovered an approach to train any AI agent with RL, with (almost) zero code changes.](https://www.reddit.com/r/LocalLLaMA/comments/1m9m670/we_discovered_an_approach_to_train_any_ai_agent/) Reddit.
@@ -58,28 +35,114 @@ To start using Agent-lightning, check out our [documentation](https://microsoft.
- [DeepWerewolf](https://github.com/af-74413592/DeepWerewolf) — A case study of agent RL training for the Chinese Werewolf game built with AgentScope and Agent Lightning.
- [AgentFlow](https://agentflow.stanford.edu/) — A modular multi-agent framework that combines planner, executor, verifier, and generator agents with the Flow-GRPO algorithm to tackle long-horizon, sparse-reward tasks.
## ⚡ Installation
First, let's get your environment set up. We'll be using `/path/to/agentlightning` to refer to the directory containing this README file.
### 1. Set Up Your Environment
We strongly recommend creating a new virtual environment to avoid conflicts with other packages. You can use either `conda` or `venv`. **Python 3.10 or later** is recommended.
### 2. Install Core Training Dependencies (Optional)
If you are running RL with Agent-Lightning, the next step is to install the essential packages: `PyTorch`, `FlashAttention`, `vLLM` and `VERL`. The following versions and installation order have been tested and are confirmed to work.
```bash
pip install torch==2.7.0 torchvision==0.22.0 torchaudio==2.7.0 --index-url https://download.pytorch.org/whl/cu128
pip install flash-attn --no-build-isolation
pip install vllm==0.9.2
pip install verl==0.5.0
```
See `scripts/setup_stable_gpu.sh` for a full installation script.
### 3. Install Agent Lightning
Now, you're ready to install Agent Lightning itself.
```bash
pip install agentlightning
```
### 4. Install Agent Frameworks (Optional)
If you plan to use other agent frameworks, you can install them with the following commands. If you don't need these, feel free to skip this step.
We recommend doing this as the final step to avoid dependency versions being overwritten by mistake.
```bash
# AutoGen (Recommended to install first)
pip install "autogen-agentchat" "autogen-ext[openai]"
# LiteLLM
pip install "litellm[proxy]"
# MCP
pip install mcp
# UV
pip install uv
# OpenAI Agents
pip install openai-agents
# LangChain
pip install langgraph "langchain[openai]" langchain-community langchain-text-splitters
# SQL-related dependencies
pip install sqlparse nltk
```
Don't worry if dependency conflicts arise during this step. Follow the installation order above and the conflicts generally do not matter.
## ⚡ Examples
For more detailed examples, please see the `examples` folder:
1. [calc_x](examples/calc_x): An agent built with AutoGen with calculator tool use, trained on Calc-X dataset with Reinforcement Learning.
2. [spider](examples/spider): A write-check-rewrite looped agent with LangGraph with SQL execution; selectively optimize write and rewrite on Spider dataset with Reinforcement Learning.
3. [apo](examples/apo): An example to customize an optimization algorithm: Automatic Prompt Optimization.
## ⚡ Important Caveats
1. **AgentOps Integration**: Agent Lightning uses [AgentOps](https://github.com/AgentOps-AI/agentops) for agent tracking by default. If you're already using AgentOps in your own code, you'll need to disable our managed AgentOps client by modifying the `tracer` parameter of trainer.
2. **Debugging Traces**: If you encounter issues with tracing, you can visualize the trace tree using `tracer.last_trace().visualize("tree_graph")`. Please note that this API is experimental and may change in future releases.
3. **Launching the Server and Agents**: Currently, the training server and agent clients must be launched in separate processes. You can open two terminal windows or run one of them in the background. The launching order generally doesn't matter.
4. **Environment Variables**: The environment variables and working directory at the time of `ray init` are important. If you run into "file not found" errors, try restarting Ray from your current working directory.
5. **Handling Timeouts**: The training server may hang if samples fail or time out on the agent side. To prevent this, we recommend setting limits on the prompt and response lengths, as this is the most common cause of failures.
6. **VERL Failures**: Save checkpoints frequently, as VERL with vLLM may sometimes experience out-of-memory issues. If you encounter a VERL failure, you can resume training from the last checkpoint.
## ⚡ Architecture
Agent Lightning keeps the moving parts to a minimum so you can focus on your idea, not the plumbing. Your agent continues to run as usual; you can still use any agent framework you like; you drop in the lightweight `agl.emit_xxx()` helper, or let the tracer collect every prompt, tool call, and reward. Those events become structured spans that flow into the LightningStore, a central hub that keeps tasks, resources, and traces in sync.
Currently, Agent Lightning is built around a **training server** and one or multiple **agents**.
On the other side of the store sits the algorithm you choose, or write yourself. The algorithm reads spans, learns from them, and posts updated resources such as refined prompt templates or new policy weights. The Trainer ties it all together: it streams datasets to runners, ferries resources between the store and the algorithm, and updates the inference engine when improvements land. You can either stop there, or simply let the same loop keep turning.
* The **server** manages the training data, prepares samples for the agents, and provides the LLM endpoint.
* **Agents** retrieve samples from the server, process them (which may involve interacting with the LLM), and send the results back. These results, or "trajectories," are lists of prompts and responses from the LLM.
* The **server** then collects these trajectories and computes the losses to optimize the language models.
No rewrites, no lock-in, just a clear path from first rollout to steady improvement.
![Agent-Lightning-architecture](docs/assets/readme-architecture.png)
<p align="center">
<img src="docs/assets/readme-architecture.svg" alt="Agent-lightning Architecture" style="width:100%"/>
</p>
## ⚡ Development Instructions
## ⚡ CI Status
Install with development dependencies:
| Workflow | Status |
|----------|--------|
| CPU Tests | [![tests workflow status](https://github.com/microsoft/agent-lightning/actions/workflows/tests.yml/badge.svg)](https://github.com/microsoft/agent-lightning/actions/workflows/tests.yml) |
| Full Tests | [![tests summary workflow status](https://github.com/microsoft/agent-lightning/actions/workflows/badge-unit.yml/badge.svg)](https://github.com/microsoft/agent-lightning/actions/workflows/badge-unit.yml) |
| UI Tests | [![UI Tests](https://github.com/microsoft/agent-lightning/actions/workflows/dashboard.yml/badge.svg)](https://github.com/microsoft/agent-lightning/actions/workflows/dashboard.yml) |
| Examples Integration | [![examples summary workflow status](https://github.com/microsoft/agent-lightning/actions/workflows/badge-examples.yml/badge.svg)](https://github.com/microsoft/agent-lightning/actions/workflows/badge-examples.yml) |
| Latest Dependency Compatibility | [![latest summary workflow status](https://github.com/microsoft/agent-lightning/actions/workflows/badge-latest.yml/badge.svg)](https://github.com/microsoft/agent-lightning/actions/workflows/badge-latest.yml) |
| Legacy Examples Compatibility | [![compat summary workflow status](https://github.com/microsoft/agent-lightning/actions/workflows/badge-compat.yml/badge.svg)](https://github.com/microsoft/agent-lightning/actions/workflows/badge-compat.yml) |
```
git clone https://github.com/microsoft/agent-lightning
cd agent-lightning
pip install -e .[dev]
```
Please run pre-commit hooks before checking in code:
```
pre-commit install
pre-commit run --all-files --show-diff-on-failure --color=always
```
Serve documentation locally:
```bash
mkdocs serve
```
## ⚡ Citation
@@ -99,7 +162,7 @@ If you find Agent Lightning useful in your research or projects, please cite our
## ⚡ Contributing
This project welcomes contributions and suggestions. Start by reading the [Contributing Guide](docs/community/contributing.md) for environment setup, branching conventions, and pull request expectations. Most contributions require you to agree to a Contributor License Agreement (CLA) declaring that you have the right to, and actually do, grant us the rights to use your contribution. For details, visit https://cla.opensource.microsoft.com.
This project welcomes contributions and suggestions. Most contributions require you to agree to a Contributor License Agreement (CLA) declaring that you have the right to, and actually do, grant us the rights to use your contribution. For details, visit https://cla.opensource.microsoft.com.
When you submit a pull request, a CLA bot will automatically determine whether you need to provide a CLA and decorate the PR appropriately (e.g., status check, comment). Simply follow the instructions provided by the bot. You will only need to do this once across all repos using our CLA.
+22
View File
@@ -0,0 +1,22 @@
import asyncio
async def a():
print("a")
b()
print("finish")
def b():
print("b")
loop = asyncio.get_running_loop()
fut = asyncio.run_coroutine_threadsafe(c(), loop)
fut.result(timeout=5.0)
async def c():
print("c")
await asyncio.sleep(0.1)
asyncio.run(a())
+2 -4
View File
@@ -1,6 +1,6 @@
# Copyright (c) Microsoft. All rights reserved.
__version__ = "0.2.2"
__version__ = "0.2.0"
from .adapter import *
from .algorithm import *
@@ -10,9 +10,7 @@ from .emitter import *
from .execution import *
from .litagent import *
from .llm_proxy import *
from .logging import configure_logger # deprecated # type: ignore
from .logging import setup as setup_logging # type: ignore
from .logging import setup_module as setup_module_logging # type: ignore
from .logging import *
from .runner import *
from .server import AgentLightningServer # deprecated # type: ignore
from .store import *
+1 -2
View File
@@ -1,12 +1,11 @@
# Copyright (c) Microsoft. All rights reserved.
from .base import Adapter, OtelTraceAdapter, TraceAdapter
from .base import Adapter, TraceAdapter
from .messages import TraceToMessages
from .triplet import LlmProxyTraceToTriplet, TracerTraceToTriplet, TraceToTripletBase
__all__ = [
"TraceAdapter",
"OtelTraceAdapter",
"Adapter",
"TraceToTripletBase",
"TracerTraceToTriplet",
+24 -23
View File
@@ -13,20 +13,18 @@ T_to = TypeVar("T_to")
class Adapter(Generic[T_from, T_to]):
"""Base class for synchronous adapters that convert data from one format to another.
The class defines a minimal protocol so that adapters can be treated like callables while
still allowing subclasses to supply the concrete transformation logic.
This class defines a simple protocol for transformation:
!!! note
Subclasses must override [`adapt()`][agentlightning.Adapter.adapt] to provide
the actual conversion.
- The `__call__` method makes adapters callable, so they can be used like functions.
- Subclasses must implement the `adapt` method to define the actual conversion logic.
Type Variables:
Type parameters:
T_from: Source data type supplied to the adapter.
- T_from: The source data type (input).
- T_to: The target data type (output).
T_to: Target data type produced by the adapter.
Example:
Examples:
>>> class IntToStrAdapter(Adapter[int, str]):
... def adapt(self, source: int) -> str:
... return str(source)
@@ -39,9 +37,8 @@ class Adapter(Generic[T_from, T_to]):
def __call__(self, source: T_from, /) -> T_to:
"""Convert the data to the target format.
This method delegates to [`adapt()`][agentlightning.Adapter.adapt] so that an
instance of [`Adapter`][agentlightning.Adapter] can be used like a standard
function.
This method delegates to `adapt` and allows the adapter
to be invoked as a function.
Args:
source: Input data in the source format.
@@ -54,8 +51,8 @@ class Adapter(Generic[T_from, T_to]):
def adapt(self, source: T_from, /) -> T_to:
"""Convert the data to the target format.
Subclasses must override this method with the concrete transformation logic. The base
implementation raises `NotImplementedError` to make the requirement explicit.
Subclasses should override this method with the concrete
transformation logic.
Args:
source: Input data in the source format.
@@ -69,12 +66,17 @@ class Adapter(Generic[T_from, T_to]):
class OtelTraceAdapter(Adapter[List[ReadableSpan], T_to], Generic[T_to]):
"""Base class for adapters that convert OpenTelemetry trace spans into other formats.
This specialization of [`Adapter`][agentlightning.Adapter] expects a list of
`opentelemetry.sdk.trace.ReadableSpan` instances and produces any target format, such as
reinforcement learning trajectories, structured logs, or analytics-ready payloads.
This class specializes `Adapter` for working with OpenTelemetry `ReadableSpan`
objects. It expects a list of spans as input and produces a custom target format
(e.g., reinforcement learning training data, SFT datasets, logs, metrics).
Examples:
>>> class TraceToDictAdapter(OtelTraceAdapter[dict]):
Subclasses should override `adapt` to define the desired conversion.
Type parameters:
T_to: The target data type that spans should be converted into.
Example:
>>> class TraceToDictAdapter(TraceAdapter[dict]):
... def adapt(self, spans: List[ReadableSpan]) -> dict:
... return {"count": len(spans)}
...
@@ -87,8 +89,7 @@ class OtelTraceAdapter(Adapter[List[ReadableSpan], T_to], Generic[T_to]):
class TraceAdapter(Adapter[List[Span], T_to], Generic[T_to]):
"""Base class for adapters that convert trace spans into other formats.
This class specializes [`Adapter`][agentlightning.Adapter] for working with
[`Span`][agentlightning.Span] instances emitted by Agent Lightning instrumentation.
Subclasses receive entire trace slices and return a format suited for the downstream consumer,
for example reinforcement learning training data or observability metrics.
This class specializes `Adapter` for working with trace spans. It expects a list of
Agent-lightning spans as input and produces a custom target format
(e.g., reinforcement learning training data, SFT datasets, logs, metrics).
"""
+30 -83
View File
@@ -1,48 +1,28 @@
# Copyright (c) Microsoft. All rights reserved.
from __future__ import annotations
import json
from collections import defaultdict
from typing import TYPE_CHECKING, Any, Dict, Generator, Iterable, List, Optional, TypedDict, Union, cast
from typing import Any, Dict, Generator, Iterable, List, Optional, TypedDict, Union, cast
from openai.types.chat import (
ChatCompletionAssistantMessageParam,
ChatCompletionFunctionToolParam,
ChatCompletionMessageFunctionToolCallParam,
ChatCompletionMessageParam,
)
from pydantic import TypeAdapter
from agentlightning.types import Span
from .base import TraceAdapter
if TYPE_CHECKING:
from openai.types.chat import (
ChatCompletionFunctionToolParam,
ChatCompletionMessageFunctionToolCallParam,
ChatCompletionMessageParam,
)
class OpenAIMessages(TypedDict):
"""OpenAI-style chat messages with optional tool definitions.
Attributes:
messages: Ordered chat messages that describe the conversation.
tools: Tool specifications available to the assistant, if any.
"""
messages: List[ChatCompletionMessageParam]
tools: Optional[List[ChatCompletionFunctionToolParam]]
class _RawSpanInfo(TypedDict):
"""Intermediate representation parsed from a span.
Attributes:
prompt: Prompt messages reconstructed from span attributes.
completion: Assistant completions following tool invocations.
request: Request payload recorded in the trace.
response: Response payload recorded in the trace.
tools: Tool call metadata extracted from child spans.
"""
prompt: List[Dict[str, Any]]
completion: List[Dict[str, Any]]
request: Dict[str, Any]
@@ -51,20 +31,16 @@ class _RawSpanInfo(TypedDict):
def group_genai_dict(data: Dict[str, Any], prefix: str) -> Union[Dict[str, Any], List[Any]]:
"""Convert flattened trace attributes into nested structures.
Attributes emitted by the tracing pipeline often arrive as dotted paths (for example
`gen_ai.prompt.0.role`). This helper groups those keys into nested dictionaries or lists so that
downstream processing can operate on structured data.
"""
Convert a flat dict with keys like 'gen_ai.prompt.0.role'
into structured nested dicts or lists under the given prefix.
Args:
data: Flat dictionary whose keys are dotted paths.
prefix: Top-level key (for example `gen_ai.prompt`) that determines which attributes are
grouped.
data: Flat dictionary (keys are dotted paths).
prefix: Top-level key to extract (e.g., 'gen_ai.prompt').
Returns:
A nested dictionary (no numeric index detected) or list (numeric indices detected) containing
the grouped values.
A nested dict (if no index detected) or list (if indexed).
"""
result: Union[Dict[str, Any], List[Any]] = {}
@@ -104,28 +80,12 @@ def group_genai_dict(data: Dict[str, Any], prefix: str) -> Union[Dict[str, Any],
def convert_to_openai_messages(prompt_completion_list: List[_RawSpanInfo]) -> Generator[OpenAIMessages, None, None]:
"""Convert raw trace payloads into OpenAI-style chat messages.
The function consumes an iterable produced by
[`TraceToMessages.adapt()`][agentlightning.TraceToMessages.adapt] and yields
structures that match the OpenAI fine-tuning JSONL schema, including tool definitions.
Args:
prompt_completion_list: Raw prompt/completion/tool payloads extracted from a trace.
Returns:
A generator that yields [`OpenAIMessages`][agentlightning.adapter.messages.OpenAIMessages]
entries compatible with the OpenAI Functions fine-tuning format.
"""
Convert raw tool call traces + prompt/completion list
into OpenAI fine-tuning JSONL format (tool calling style).
# Import locally to avoid legacy OpenAI version type import errors
from openai.types.chat import (
ChatCompletionAssistantMessageParam,
ChatCompletionFunctionToolParam,
ChatCompletionMessageFunctionToolCallParam,
ChatCompletionMessageParam,
)
https://learn.microsoft.com/en-us/azure/ai-foundry/openai/how-to/fine-tuning-functions
"""
for pc_entry in prompt_completion_list:
messages: List[ChatCompletionMessageParam] = []
@@ -197,29 +157,25 @@ def convert_to_openai_messages(prompt_completion_list: List[_RawSpanInfo]) -> Ge
class TraceToMessages(TraceAdapter[List[OpenAIMessages]]):
"""Convert trace spans into OpenAI-compatible conversation messages.
"""
Adapter that converts OpenTelemetry trace spans into OpenAI-compatible message format.
The adapter reconstructs prompts, completions, tool calls, and function definitions from
`gen_ai.*` span attributes. The resulting objects match the JSONL structure expected by the
OpenAI fine-tuning pipeline.
This adapter processes trace spans containing LLM conversation data and transforms them
into structured OpenAI message format suitable for fine-tuning or analysis. It extracts
prompts, completions, tool calls, and function definitions from trace attributes and
reconstructs the conversation flow.
!!! warning
The adapter assumes all spans share a common trace and that tool call spans are direct
children of the associated completion span.
The adapter handles:
- Converting flat trace attributes into structured message objects
- Extracting and matching tool calls with their corresponding requests
- Building proper OpenAI ChatCompletionMessage objects with roles, content, and tool calls
- Generating function definitions for tools used in conversations
"""
def get_tool_calls(self, completion: Span, all_spans: List[Span], /) -> Iterable[Dict[str, Any]]:
"""Yield tool call payloads for a completion span.
"""Find tool calls in the trace. Returns a dict with the tool call id, name, and arguments.
Args:
completion: The completion span whose descendants should be inspected.
all_spans: The complete span list belonging to the trace.
Yields:
Dictionaries describing tool calls with identifiers, names, and arguments.
Raises:
ValueError: If a candidate tool span cannot be converted into a dictionary.
The spans that are direct children of the completion span are the tool calls.
"""
# Get all the spans that are children of the completion span
children = [span for span in all_spans if span.parent_id == completion.span_id]
@@ -232,15 +188,6 @@ class TraceToMessages(TraceAdapter[List[OpenAIMessages]]):
yield tool_call
def adapt(self, source: List[Span], /) -> List[OpenAIMessages]:
"""Transform trace spans into OpenAI chat payloads.
Args:
source: Spans containing `gen_ai.*` attributes emitted by the tracing pipeline.
Returns:
A list of [`OpenAIMessages`][agentlightning.adapter.messages.OpenAIMessages] entries that
capture prompts, completions, tools, and metadata.
"""
raw_prompt_completions: List[_RawSpanInfo] = []
for span in source:
+109 -209
View File
@@ -3,7 +3,6 @@
from __future__ import annotations
import json
import logging
import re
from enum import Enum
from typing import Any, Dict, List, Optional, Tuple, Union, cast
@@ -11,22 +10,16 @@ from typing import Any, Dict, List, Optional, Tuple, Union, cast
from opentelemetry.sdk.trace import ReadableSpan
from pydantic import BaseModel
from agentlightning.types import Span, SpanNames, Triplet
from agentlightning.types import SpanNames, Triplet
from agentlightning.types.tracer import Span
from .base import TraceAdapter
logger = logging.getLogger(__name__)
class Transition(BaseModel):
"""A single transition within a reinforcement learning trajectory.
Attributes:
state: Token identifiers describing the model input state.
action: Token identifiers representing the model output.
response_id: Identifier of the LLM response used to deduplicate spans.
agent_name: Human-readable agent name captured from the trace.
reward: Scalar reward associated with the transition, if available.
"""
Transition class representing one transition in a trajectory.
State and action are a list of token IDs.
"""
state: List[int]
@@ -38,27 +31,22 @@ class Transition(BaseModel):
class RewardMatchPolicy(str, Enum):
"""Strategies for matching rewards to LLM call spans.
!!! note
Each reward span must expose a payload shaped like `{"type": "reward", "value": <float>|None}`
as described in `reward.py`.
"""How to find the reward for each transition from the trace.
In all cases, the reward must have data `{"type": "reward", "value": <float>|None}`,
as defined in `reward.py`.
"""
FIRST_SIBLING = "first_sibling"
"""Use the first sibling in the current trace subtree as the reward unless another LLM call match is found."""
"""Use the first sibling in the current trace subtree as the reward, except another LLM call match is found."""
FIRST_OCCURRENCE = "first_occurrence"
"""Use the first reward encountered in chronological order after the current LLM call match."""
"""Use the first occurrence of the reward (in start time order) that occur after the current LLM call match.
"""
class TraceTree:
"""Tree representation of a trace span and its descendants.
Attributes:
id: Unique identifier for the span node.
span: [`Span`][agentlightning.Span] backing this node.
children: Child nodes connected to the current span.
"""
A trace item, along with its span and children.
"""
def __init__(
@@ -92,16 +80,10 @@ class TraceTree:
self.children.append(child)
def visualize(self, filename: str, interested_span_match: str | None = None) -> None:
"""Render the trace tree with Graphviz for debugging purposes.
Args:
filename: Base filename for the generated `.png` diagram.
interested_span_match: Optional regular expression used to keep only matching spans
(and their ancestors) in the output.
!!! note
The method requires the optional `graphviz` dependency to be available in the runtime
environment.
"""
Visualize the trace tree using graphviz.
For debugging purposes only.
Use `interested_span_match` to filter the spans (and its ancesters) to be visualized.
"""
import graphviz
@@ -143,11 +125,9 @@ class TraceTree:
dot.render(filename, format="png", cleanup=True) # type: ignore
def names_tuple(self) -> Tuple[str, List[Any]]:
"""Return the span name alongside nested child names.
Returns:
A tuple of the current span name and a list of tuples for each child containing the
child name and its descendants.
"""Return the span name, and a list of children.
Each child is also a tuple of span name and a list of children.
Useful for debugging and testing.
"""
name = self.span.name
agent_name = self.agent_name()
@@ -160,14 +140,15 @@ class TraceTree:
return name, children_names
def traverse(self) -> List["TraceTree"]:
"""Traverse the tree depth first and return every node."""
"""
Traverse the trace tree and return a list of all spans.
"""
spans: List["TraceTree"] = [self]
for child in self.children:
spans.extend(child.traverse())
return spans
def to_json(self) -> dict[str, Any]:
"""Convert the tree node into a JSON-serialisable structure."""
if isinstance(self.span, ReadableSpan):
span_data = json.loads(self.span.to_json())
else:
@@ -180,17 +161,10 @@ class TraceTree:
@classmethod
def from_spans(cls, spans: List[Span]) -> "TraceTree":
"""Construct a tree from a flat list of spans.
Args:
spans: Spans that collectively form a single trace segment.
Returns:
A [`TraceTree`][agentlightning.adapter.triplet.TraceTree] rooted at either the
discovered root span or a synthetic root when multiple roots are present.
Raises:
ValueError: If the span list is empty or no root span can be inferred.
"""
Create a TraceTree from a list of spans.
All spans without parents found will be considered as candidate root spans.
If multiple root spans are found, a virtual root span will be created as the parent of all root spans.
"""
if not spans:
@@ -271,11 +245,8 @@ class TraceTree:
return root_span
def agent_name(self) -> Optional[str]:
"""Return the agent name associated with the span, if any.
Returns:
Agent name extracted from known attributes, otherwise `None`.
"""
"""Return the name of agent span. Return the agent or None (not an agent at all).
Extend this function to support more agent frameworks."""
attributes = self.span.attributes
if attributes is None: # type: ignore
return None
@@ -308,11 +279,6 @@ class TraceTree:
return agent_name
def maybe_reward_dict(self) -> dict[str, Any]:
"""Return a reward payload if the span encodes one.
Returns:
Dictionary containing reward metadata, or an empty dictionary when no reward is found.
"""
for key in [
"agentops.task.output", # newer versions of agentops
"agentops.entity.output",
@@ -333,11 +299,6 @@ class TraceTree:
return {}
def is_reward_span(self) -> bool:
"""Return whether the span explicitly encodes a reward.
Returns:
`True` when the span payload describes a reward, otherwise `False`.
"""
maybe_reward = self.maybe_reward_dict()
return maybe_reward and maybe_reward.get("type") == "reward" # type: ignore
@@ -351,19 +312,12 @@ class TraceTree:
within_llm_call: Optional[bool] = None,
existing_llm_call_response_ids: Optional[set[str]] = None,
) -> List[Tuple["TraceTree", str]]:
"""Find LLM call spans matching the supplied filters.
"""Find all LLM calls in the trace tree.
Args:
llm_call_match: Regular expression used to match span names that qualify as LLM calls.
agent_match: Optional regular expression that must match the enclosing agent span name.
within_matching_subtree: Marker propagated through recursive calls to record matching agents.
within_reward: When `True`, suppresses LLM matches under reward spans.
within_llm_call: When `True`, prevents duplicate matches for nested LLM calls.
existing_llm_call_response_ids: Known response identifiers used to deduplicate spans.
The LLM call is defined as a span with type = request and name matching `llm_call_match`.
If `agent_match` is not None, it must also reside in an agent span (type = agent) with name matched.
Returns:
A list of tuples pairing the matching node with the agent subtree label that triggered the
match.
Return a list of traces and the agent names (why it's selected).
"""
llm_calls: List[Tuple[TraceTree, str]] = []
@@ -419,26 +373,19 @@ class TraceTree:
return llm_calls
def repair_hierarchy(self) -> None:
"""Repair missing parent-child relationships introduced by mixed tracing systems.
Some agent frameworks emit spans via multiple subsystems, which can cause LLM completion
spans to float directly under the root span instead of being nested under the correct agent.
The method re-parents those spans to the closest ancestor that fully envelopes the child in
time.
If we don't, when we want to select the LLM completion span with agent as filter.
We will never get the correct span underneath.
"""
# If the current node has only one child, recursively repair its hierarchy directly.
# This special-case handling is needed because when a trace is manually ended
# (via agentops.end_trace), the AgentOps provider automatically wraps all spans
# under an extra synthetic root node (e.g., "run_one.session").
if len(self.children) == 1:
self.children[0].repair_hierarchy()
return
We find that sometimes the hierarchy is not correct, due to the way the spans are created.
The spans within the agent frameworks (e.g., OpenAI Agent SDK) and spans within the LLM frameworks
(e.g., Anthropic) are created in two systems.
So the inner LLM completion span does not necessarily have an agent span as a parent.
Rather they sometimes directly become children of the root span.
This becomes a problem when we want to select the LLM completion span with agent as filter.
To repair the hierarchy, for each children of the root span, we find a span over the whole tree,
with duration covering the current span and being closest to the current span.
This function modifies the tree in place.
"""
nodes_to_repair = list(self.children)
for repair_node in nodes_to_repair:
if len(self.children) == 1:
# If there is only one child, we don't need to repair the hierarchy.
@@ -463,16 +410,7 @@ class TraceTree:
closest_parent.children.append(repair_node)
def match_rewards(self, reward_match: str, llm_calls: List["TraceTree"]) -> dict[str, Optional[float]]:
"""Assign rewards to previously matched LLM calls.
Args:
reward_match: Strategy identifier from
[`RewardMatchPolicy`][agentlightning.adapter.triplet.RewardMatchPolicy].
llm_calls: Trace nodes representing LLM call spans.
Returns:
Mapping from span identifier to reward value or `None` when no reward is available.
"""
"""Match the rewards to the LLM calls."""
llm_call_ids = set([llm_call.id for llm_call in llm_calls])
rewards: dict[str, Optional[float]] = {}
@@ -517,30 +455,6 @@ class TraceTree:
return rewards
def span_to_triplet(self, span: Span, agent_name: str) -> Triplet:
"""Convert a span to a triplet.
Subclass can override this method to add more fields to the triplet,
such as chat messages and tool calls.
"""
prompt_token_ids = span.attributes.get("prompt_token_ids", []) # type: ignore
response_token_ids = span.attributes.get("response_token_ids", []) # type: ignore
response_id = span.attributes.get("gen_ai.response.id", None) # type: ignore
logprobs_content = span.attributes.get("logprobs.content", None) # type: ignore
if isinstance(logprobs_content, str):
logprobs_content = json.loads(logprobs_content)
response: Dict[str, Any] = {"token_ids": response_token_ids, "logprobs": logprobs_content}
else:
response = {"token_ids": response_token_ids}
return Triplet(
prompt={"token_ids": prompt_token_ids},
response=response,
reward=None,
metadata=dict(response_id=response_id, agent_name=agent_name),
)
def to_trajectory(
self,
llm_call_match: str = r"openai\.chat\.completion",
@@ -549,21 +463,20 @@ class TraceTree:
dedup_llm_call: bool = True,
reward_match: RewardMatchPolicy = RewardMatchPolicy.FIRST_OCCURRENCE,
final_reward: Optional[float] = None,
_skip_empty_token_spans: bool = False,
) -> List[Triplet]:
"""Convert the trace tree into a trajectory of [`Triplet`][agentlightning.Triplet] items.
"""Convert the trace tree to a trajectory.
Args:
llm_call_match: Regular expression for LLM call span names.
agent_match: Optional regular expression for agent span names.
exclude_llm_call_in_reward: When `True`, prevents searching for rewards under the LLM
call subtree.
dedup_llm_call: When `True`, deduplicates spans using the LLM response identifier.
reward_match: Reward matching policy used to associate reward spans with LLM calls.
final_reward: Optional reward appended to the final transition when provided.
First, we find all the LLM calls (span type = request, `llm_call_match` matching the span name).
If the agent match is set, we check, for each LLM call,
if it resides in an agent (span type = agent, `agent_match` matching the span name).
The above sets the basis for the trajectory, as we use the prompt token IDs and response token IDs for each LLM call,
as the state and action of each transition.
Returns:
A list of [`Triplet`][agentlightning.Triplet] objects ordered by call sequence.
Then, we find the reward for each transition.
The reward is searched on the trace tree, after the LLM call,
until the next LLM call or the end of the tree depending on the policy.
It can be enforced to a sibling or the first occurrence in the time order, depending on the policy.
If a reward is never found for a transition, it is set to None.
"""
# Find all LLM calls
llm_calls = self.find_llm_calls(
@@ -574,23 +487,25 @@ class TraceTree:
within_llm_call=False if dedup_llm_call else None,
existing_llm_call_response_ids=set(),
)
id_transitions = [
(
llm_call.id,
Triplet(
prompt={"token_ids": llm_call.span.attributes.get("prompt_token_ids", [])}, # type: ignore
response={"token_ids": llm_call.span.attributes.get("response_token_ids", [])}, # type: ignore
reward=None,
metadata=dict(
response_id=llm_call.span.attributes.get( # type: ignore
"gen_ai.response.id", None
), # it works at least for OpenAI
agent_name=agent_name,
),
),
)
for llm_call, agent_name in llm_calls
]
id_transitions: List[Tuple[str, Triplet]] = []
# We need to filter out the LLM calls with unrecorded token IDs
filtered_llm_calls: List[Tuple[TraceTree, str]] = []
for llm_call, agent_name in llm_calls:
triplet = self.span_to_triplet(llm_call.span, agent_name)
# This is a hot-fix for Tinker+CrewAI, which has some anonymous requests outside the trained agent.
# TODO: We might need to reconsider this.
if _skip_empty_token_spans and (
not triplet.prompt.get("token_ids") or not triplet.response.get("token_ids")
):
logger.warning(f"Skipping LLM call with unrecorded token IDs: {triplet}")
continue
filtered_llm_calls.append((llm_call, agent_name))
id_transitions.append((llm_call.id, triplet))
rewards = self.match_rewards(reward_match, [call for call, _ in filtered_llm_calls])
rewards = self.match_rewards(reward_match, [call for call, _ in llm_calls])
transitions = [
transition.model_copy(update={"reward": rewards.get(id, None)}) for id, transition in id_transitions
]
@@ -607,22 +522,22 @@ class TraceTree:
class TraceToTripletBase(TraceAdapter[List[Triplet]]):
"""Base class for adapters that emit [`Triplet`][agentlightning.Triplet] trajectories."""
"""
Base class for trace triplet adapters.
"""
class TracerTraceToTriplet(TraceToTripletBase):
"""Convert tracer-emitted spans into triplet trajectories.
"""
An adapter to convert OpenTelemetry spans to triplet data.
Attributes:
repair_hierarchy: When `True`, repair the span tree using
[`TraceTree.repair_hierarchy()`][agentlightning.adapter.triplet.TraceTree.repair_hierarchy]
before matching calls and rewards.
llm_call_match: Regular expression pattern that selects LLM call span names.
agent_match: Optional regular expression pattern for agent span names. When omitted, spans
from any agent are considered.
exclude_llm_call_in_reward: When `True`, ignore matches under reward spans while searching
for rewards.
reward_match: Strategy used to associate rewards with LLM calls.
repair_hierarchy: When `repair_hierarchy` is set to True, the trace will be repaired with the time information.
See `TraceTree.repair_hierarchy` for more details.
llm_call_match: Regular expression pattern to match LLM call span names.
agent_match: Optional regular expression pattern to match agent span names. If None, all agents are matched.
exclude_llm_call_in_reward: Whether to exclude LLM calls that occur within reward spans.
reward_match: Policy for matching rewards to LLM calls.
"""
def __init__(
@@ -632,14 +547,12 @@ class TracerTraceToTriplet(TraceToTripletBase):
agent_match: Optional[str] = None,
exclude_llm_call_in_reward: bool = True,
reward_match: RewardMatchPolicy = RewardMatchPolicy.FIRST_OCCURRENCE,
_skip_empty_token_spans: bool = False,
):
self.repair_hierarchy = repair_hierarchy
self.llm_call_match = llm_call_match
self.agent_match = agent_match
self.exclude_llm_call_in_reward = exclude_llm_call_in_reward
self.reward_match = reward_match
self._skip_empty_token_spans = _skip_empty_token_spans
def visualize(
self,
@@ -648,17 +561,16 @@ class TracerTraceToTriplet(TraceToTripletBase):
filename: str = "trace_tree",
interested_span_match: str | None = None,
) -> TraceTree:
"""Visualize the trace tree built from the supplied spans.
"""
Visualize the trace tree.
Args:
source: Collection of Agent Lightning [`Span`][agentlightning.Span] objects
or raw `opentelemetry.sdk.trace.ReadableSpan` instances.
filename: Base filename for the generated image; `.png` is appended automatically.
interested_span_match: Optional regular expression used to highlight a subset of spans.
source (List[Span]): The list of OpenTelemetry spans to visualize.
filename (str): The base filename for the output visualization (default: "trace_tree").
interested_span_match (str | None): Optional regular expression pattern to highlight or focus on specific spans in the visualization.
Returns:
The [`TraceTree`][agentlightning.adapter.triplet.TraceTree] built from the provided
spans.
TraceTree: The constructed trace tree object.
"""
source_normalized = [
Span.from_opentelemetry(span, "dummy", "dummy", 0) if isinstance(span, ReadableSpan) else span
@@ -671,14 +583,7 @@ class TracerTraceToTriplet(TraceToTripletBase):
return trace_tree
def adapt(self, source: Union[List[Span], List[ReadableSpan]], /) -> List[Triplet]: # type: ignore
"""Convert tracer spans into [`Triplet`][agentlightning.Triplet] trajectories.
Args:
source: Agent Lightning spans or raw OpenTelemetry spans that form a trace.
Returns:
Ordered list of trajectory transitions with prompt, response, and reward information.
"""
"""Convert OpenTelemetry spans to a list of Triplet objects."""
source_normalized = [
Span.from_opentelemetry(span, "dummy", "dummy", 0) if isinstance(span, ReadableSpan) else span
for span in source
@@ -691,29 +596,30 @@ class TracerTraceToTriplet(TraceToTripletBase):
agent_match=self.agent_match,
exclude_llm_call_in_reward=self.exclude_llm_call_in_reward,
reward_match=self.reward_match,
_skip_empty_token_spans=self._skip_empty_token_spans,
)
return trajectory
class LlmProxyTraceToTriplet(TraceToTripletBase):
"""Convert telemetry emitted by the LLM Proxy into triplet trajectories.
"""
Converting telemetry data emitted by the LLM Proxy to triplet data.
This adapter is very experimental. Should only be used when the TracerTraceToTriplet does not work at all.
!!! warning
This adapter is experimental and might be merged with
[`TracerTraceToTriplet`][agentlightning.TracerTraceToTriplet] in the future.
!!! danger
Do not rely on timestamps when using this adapter. Proxy spans can originate on different
machines with unsynchronised clocks, so `sequence_id` is treated as the sole source of
ordering.
IMPORTANT: Do NOT rely on timestamps here. Proxy spans can be emitted from different
machines with unsynchronized clocks. We therefore treat `sequence_id` as the only
reliable ordering primitive and perform "first occurrence" reward matching using
sequence order only.
Strategy:
1. Sort spans by `(sequence_id, start_time)` for deterministic processing.
2. Extract token identifiers from `litellm_request` or `raw_gen_ai_request` spans.
3. Extract rewards from spans exposing AgentOps-style payloads or explicit reward spans.
4. Match each reward to the most recent unmatched LLM call whose sequence is smaller.
1) Sort spans by (sequence_id, start_time).
2) Extract LLM calls that expose prompt/response token IDs from either:
- litellm_request (sometimes only metadata, ignore if no token ids)
- raw_gen_ai_request (llm.hosted_vllm.* stringified fields)
3) Extract rewards from spans whose attributes contain an AgentOps-style
reward payload or explicit REWARD span.
4) For each reward with sequence R, assign it to the most recent *unmatched* LLM call
with sequence < R. Ignore timestamps completely.
"""
def _literal_eval_maybe(self, v: Any) -> Any:
@@ -775,7 +681,9 @@ class LlmProxyTraceToTriplet(TraceToTripletBase):
return cast(List[int], prompt_ids), cast(List[int], resp_ids)
def _maybe_reward_value(self, span: Span) -> Optional[float]:
"""Parse reward from typical AgentOps payloads or explicit reward spans."""
"""
Parse reward from typical AgentOps payload or explicit REWARD span.
"""
attrs = span.attributes or {}
# AgentOps new/old keys
@@ -801,14 +709,6 @@ class LlmProxyTraceToTriplet(TraceToTripletBase):
return str(rid) if isinstance(rid, str) and rid else None
def adapt(self, source: List[Span], /) -> List[Triplet]: # type: ignore
"""Convert LLM Proxy spans into [`Triplet`][agentlightning.Triplet] trajectories.
Args:
source: Spans emitted by the LLM Proxy containing prompt, response, and reward data.
Returns:
Ordered trajectory transitions matched purely by `sequence_id`.
"""
# 1) Sort deterministically by (sequence_id, start_time).
spans = sorted(
source,
+2 -2
View File
@@ -4,7 +4,7 @@ from __future__ import annotations
from typing import TYPE_CHECKING, Any
from .base import Algorithm
from .base import BaseAlgorithm
from .decorator import algo
from .fast import Baseline, FastAlgorithm
@@ -12,7 +12,7 @@ if TYPE_CHECKING:
from .apo import APO as APOType
from .verl import VERL as VERLType
__all__ = ["Algorithm", "algo", "FastAlgorithm", "Baseline", "APO", "VERL"]
__all__ = ["BaseAlgorithm", "algo", "FastAlgorithm", "Baseline", "APO", "VERL"]
# Shortcuts for usages like algo.APO(...)
+39 -11
View File
@@ -19,8 +19,7 @@ import poml
from openai import AsyncOpenAI
from agentlightning.adapter.messages import TraceToMessages
from agentlightning.algorithm.base import Algorithm
from agentlightning.algorithm.utils import batch_iter_over_dataset
from agentlightning.algorithm.base import BaseAlgorithm
from agentlightning.reward import find_final_reward
from agentlightning.types import Dataset, NamedResources, PromptTemplate, Rollout, RolloutMode, RolloutStatus
@@ -57,7 +56,42 @@ APPLY_EDIT_PROMPT_FILES = [
]
class APO(Algorithm, Generic[T_task]):
def batch_iter_over_dataset(dataset: Dataset[T_task], batch_size: int) -> Iterator[Sequence[T_task]]:
"""
Create an infinite iterator that yields batches from the dataset.
When batch_size >= dataset size, yields the entire shuffled dataset repeatedly.
When batch_size < dataset size, yields batches of the specified size, reshuffling
after each complete pass through the dataset.
Args:
dataset: The dataset to iterate over.
batch_size: The desired batch size.
Yields:
Sequences of tasks from the dataset. Each task appears at most once per epoch.
"""
if batch_size >= len(dataset):
while True:
dataset_copy = [dataset[i] for i in range(len(dataset))]
random.shuffle(dataset_copy)
yield dataset_copy
else:
current_batch: List[int] = []
while True:
indices = list(range(len(dataset)))
random.shuffle(indices)
for index in indices:
if index in current_batch:
continue
current_batch.append(index)
if len(current_batch) == batch_size:
yield [dataset[index] for index in current_batch]
current_batch = []
class APO(BaseAlgorithm, Generic[T_task]):
"""Automatic Prompt Optimization (APO) algorithm using textual gradients and beam search.
APO is an iterative prompt optimization algorithm that uses LLM-generated textual gradients
@@ -65,16 +99,14 @@ class APO(Algorithm, Generic[T_task]):
computes critiques based on the results, and applies edits to generate improved prompts.
The algorithm operates in rounds, where each round:
1. Samples parent prompts from the current beam
2. Generates new prompts by computing textual gradients and applying edits
3. Evaluates all candidates on a validation set
4. Selects the top-k prompts for the next round
Based on the ideas from:
- [ProTeGi](https://aclanthology.org/2023.emnlp-main.494.pdf)
- [TextGrad](https://github.com/zou-group/textgrad)
- ProTeGi: https://aclanthology.org/2023.emnlp-main.494.pdf
- TextGrad: https://github.com/zou-group/textgrad
"""
def __init__(
@@ -305,7 +337,6 @@ class APO(Algorithm, Generic[T_task]):
Generate an improved prompt by computing a textual gradient and applying an edit.
This is the main optimization step that:
1. Computes a critique (textual gradient) based on rollout performance
2. Uses another LLM to apply the critique and generate an improved prompt
@@ -412,7 +443,6 @@ class APO(Algorithm, Generic[T_task]):
Evaluate a prompt on a batch of tasks by running rollouts and computing average reward.
This method:
1. Adds the prompt as a named resource to the store
2. Enqueues rollouts for each task in the dataset
3. Waits for rollouts to complete (with timeout)
@@ -557,7 +587,6 @@ class APO(Algorithm, Generic[T_task]):
Generate new candidate prompts from parents using textual gradients.
For each parent prompt, generates branch_factor new candidates by:
1. Evaluating the parent on a training batch
2. Computing textual gradient
3. Applying edit to generate improved prompt
@@ -785,7 +814,6 @@ class APO(Algorithm, Generic[T_task]):
Execute the APO algorithm to optimize prompts through beam search with textual gradients.
The algorithm performs iterative prompt optimization over multiple rounds:
- Each round: samples parent prompts, generates new candidates via textual gradients,
evaluates all candidates on validation data, and keeps the top performers
- Tracks the historically best prompt across all rounds
+1 -1
View File
@@ -22,7 +22,7 @@ if TYPE_CHECKING:
from agentlightning.trainer import Trainer
class Algorithm:
class BaseAlgorithm:
"""Algorithm is the strategy, or tuner to train the agent."""
_trainer_ref: weakref.ReferenceType[Trainer] | None = None
+41 -49
View File
@@ -26,7 +26,7 @@ from agentlightning.types import Dataset, NamedResources
if TYPE_CHECKING:
from agentlightning.llm_proxy import LLMProxy
from .base import Algorithm
from .base import BaseAlgorithm
# Algorithm function signature types
# We've missed a lot of combinations here.
@@ -100,13 +100,12 @@ AsyncFlag = Literal[True, False]
AF = TypeVar("AF", bound=AsyncFlag)
class FunctionalAlgorithm(Algorithm, Generic[AF]):
"""An algorithm wrapper built from a callable implementation.
class FunctionalAlgorithm(BaseAlgorithm, Generic[AF]):
"""A BaseAlgorithm that wraps a function-based algorithm implementation.
Functional algorithms let you provide an ordinary function instead of
subclassing [`Algorithm`][agentlightning.Algorithm]. The wrapper inspects
the callable signature to supply optional dependencies
such as the store, adapter, and LLM proxy.
This class allows users to define algorithm behavior using a simple function
that takes train_dataset and val_dataset parameters, rather than implementing
a full BaseAlgorithm subclass.
"""
@overload
@@ -116,12 +115,13 @@ class FunctionalAlgorithm(Algorithm, Generic[AF]):
def __init__(self: "FunctionalAlgorithm[Literal[True]]", algorithm_func: AlgorithmFuncAsyncLike) -> None: ...
def __init__(self, algorithm_func: Union[AlgorithmFuncSyncLike, AlgorithmFuncAsyncLike]) -> None:
"""Wrap a function that implements algorithm behaviour.
"""
Initialize the FunctionalAlgorithm with an algorithm function.
Args:
algorithm_func: Sync or async callable implementing the algorithm
contract. Arguments are detected automatically based on the
function signature.
algorithm_func: A function that defines the algorithm's behavior.
Can be sync or async with signature:
(train_dataset, val_dataset) -> None
"""
super().__init__()
self._algorithm_func = algorithm_func
@@ -156,20 +156,14 @@ class FunctionalAlgorithm(Algorithm, Generic[AF]):
train_dataset: Optional[Dataset[Any]] = None,
val_dataset: Optional[Dataset[Any]] = None,
) -> Union[None, Awaitable[None]]:
"""Execute the wrapped function with injected dependencies.
"""Execute the algorithm using the wrapped function.
Args:
train_dataset: Optional training dataset passed through when the
callable declares a `train_dataset` parameter.
val_dataset: Optional validation dataset passed through when the
callable declares a `val_dataset` parameter.
train_dataset: The dataset to train on.
val_dataset: The dataset to validate on.
Returns:
None for sync callables or an awaitable when the callable is async.
Raises:
TypeError: If a dataset is provided but the function signature does
not accept the corresponding argument.
None or Awaitable[None] if the function is async.
"""
kwargs: Dict[str, Any] = {}
if "store" in self._sig.parameters:
@@ -223,42 +217,40 @@ def algo(
AlgorithmFuncAsyncFallback,
],
) -> Union[FunctionalAlgorithm[Literal[False]], FunctionalAlgorithm[Literal[True]]]:
"""Convert a callable into a [`FunctionalAlgorithm`][agentlightning.algorithm.decorator.FunctionalAlgorithm].
"""Create a BaseAlgorithm from a function.
The decorator inspects the callable signature to decide which dependencies
to inject at runtime, enabling concise algorithm definitions that still
leverage the full training runtime.
This decorator allows you to define an algorithm using a simple function
instead of creating a full BaseAlgorithm subclass. The returned FunctionalAlgorithm
instance is callable, preserving the original function's behavior.
Args:
func: Function implementing the algorithm logic. May be synchronous or
asynchronous. The function can expect all of, or a subset of the following parameters:
- `store`: [`LightningStore`][agentlightning.store.base.LightningStore],
- `train_dataset`: [`Dataset`][agentlightning.Dataset],
- `val_dataset`: [`Dataset`][agentlightning.Dataset],
- `llm_proxy`: [`LLMProxy`][agentlightning.LLMProxy],
- `adapter`: [`TraceAdapter`][agentlightning.TraceAdapter],
- `initial_resources`: [`NamedResources`][agentlightning.NamedResources],
If the function does not expect a parameter, the wrapper will not inject it into the call.
Using `*args` and `**kwargs` will not work and no parameters will be injected.
func: A function that defines the algorithm's behavior with signature:
(train_dataset, val_dataset) -> None
Can be sync or async.
Returns:
FunctionalAlgorithm that proxies the callable while exposing the
`Algorithm` interface.
A callable FunctionalAlgorithm instance that preserves the original function's
type hints and behavior while providing all algorithm functionality.
Examples:
```python
from agentlightning.algorithm.decorator import algo
Example:
@algo
def my_algorithm(train_dataset, val_dataset):
# Algorithm logic here
for task in train_dataset:
# Process training tasks
pass
@algo
def batching_algorithm(*, store, train_dataset, val_dataset):
for sample in train_dataset:
store.enqueue_rollout(input=sample, mode="train")
async def my_async_algorithm(train_dataset, val_dataset):
# Async algorithm logic here
async for task in train_dataset:
# Process training tasks asynchronously
pass
@algo
async def async_algorithm(*, store, train_dataset=None, val_dataset=None):
await store.enqueue_rollout(input={"prompt": "hello"}, mode="train")
```
# Function is still callable with original behavior
my_algorithm(train_data, val_data)
# Algorithm methods are also available
my_algorithm.run(train_data, val_data)
"""
return FunctionalAlgorithm(func)
+32 -65
View File
@@ -7,21 +7,21 @@ import logging
from datetime import datetime
from typing import Any, List, Literal, Optional
from agentlightning.llm_proxy import ModelConfig
from agentlightning.types import Attempt, Dataset, Rollout, RolloutStatus, Span
from .base import Algorithm
from .base import BaseAlgorithm
logger = logging.getLogger(__name__)
__all__ = ["FastAlgorithm", "Baseline"]
class FastAlgorithm(Algorithm):
"""Base class for lightweight algorithms optimised for developer workflows.
class FastAlgorithm(BaseAlgorithm):
"""Algorithm that can run fast and qualify for dev mode.
Fast algorithms prioritise short feedback loops so an agent developer can run
small-scale experiments without waiting for long-running training jobs to
finish.
Fast algorithms enable agent developers to quickly iterate on agent development
without waiting for a long training to complete.
"""
@@ -30,38 +30,24 @@ def _timestamp_to_iso_str(timestamp: float) -> str:
class Baseline(FastAlgorithm):
"""Reference implementation that streams the full dataset through the rollout queue.
"""A dummy implementation of algorithm interface that puts all dataset into the queue, and waits for all rollouts to complete.
The baseline algorithm batches task submissions, waits for each rollout to
finish, and logs every collected span and reward. It is primarily useful as
a smoke test for the platform plumbing rather than a performant trainer.
Logs all collected spans and rewards.
Args:
n_epochs: Number of dataset passes to execute for both the train and val
splits during developer experiments.
train_split: Fraction of the concatenated dataset to treat as training
data. Must be strictly between 0 and 1.
polling_interval: Interval, in seconds, to poll the store for queue
depth and rollout completion.
max_queue_length: Number of rollouts allowed to wait in the queue before
throttling additional submissions.
span_verbosity: Level of detail to include when logging span metadata.
Raises:
ValueError: If `train_split` falls outside the `(0, 1)` interval.
Examples:
```python
from agentlightning.algorithm.fast import Baseline
algorithm = Baseline(n_epochs=2, train_split=0.8, span_verbosity="key_values")
trainer.fit(algorithm, train_dataset=my_train, val_dataset=my_val)
```
model_list: Optional list of models to load into the llm proxy.
If both model_list and llm_proxy is provided, llm_proxy will be launched.
Not implemented yet.
n_epochs: Number of epochs to run through the dev dataset.
train_split: Fraction of dev dataset to use for training vs validation. Must be between 0 and 1.
polling_interval: Time interval (in seconds) to poll the store for queue length and for completed rollouts.
max_queue_length: Maximum number of rollouts to keep in the queue at any time.
"""
def __init__(
self,
*,
model_list: Optional[List[ModelConfig]] = None,
n_epochs: int = 1,
train_split: float = 0.5,
polling_interval: float = 5.0,
@@ -80,7 +66,6 @@ class Baseline(FastAlgorithm):
self._finished_rollout_count = 0
def _span_to_string(self, rollout_id: str, attempt: Attempt, span: Span) -> str:
"""Format a span for logging based on the configured verbosity."""
if self.span_verbosity == "none":
return ""
@@ -100,7 +85,6 @@ class Baseline(FastAlgorithm):
return msg
async def _handle_rollout_finish(self, rollout: Rollout) -> None:
"""Log attempt metadata and emit adapted traces when a rollout ends."""
store = self.get_store()
rollout_id = rollout.rollout_id
@@ -113,12 +97,7 @@ class Baseline(FastAlgorithm):
attempts = await store.query_attempts(rollout_id)
for attempt in attempts:
logger.info(
"[Rollout %s | Attempt %s] ID: %s. Status: %s. Worker: %s",
rollout_id,
attempt.sequence_id,
attempt.attempt_id,
attempt.status,
attempt.worker_id,
f"[Rollout {rollout_id} | Attempt {attempt.sequence_id}] ID: {attempt.attempt_id}. Status: {attempt.status}. Worker: {attempt.worker_id}"
)
spans = await store.query_spans(rollout_id=rollout_id)
for span in spans:
@@ -128,18 +107,15 @@ class Baseline(FastAlgorithm):
# Attempts to adapt the spans using the adapter if provided
try:
adapter = self.get_adapter()
except ValueError:
logger.warning("No adapter set for MockAlgorithm. Skipping trace adaptation.")
adapter = None
if adapter is not None:
spans = await store.query_spans(rollout_id=rollout_id, attempt_id="latest")
transformed_data = adapter.adapt(spans)
logger.info(f"[Rollout {rollout_id}] Adapted data: {transformed_data}")
except ValueError:
logger.warning("No adapter set for MockAlgorithm. Skipping trace adaptation.")
async def _enqueue_rollouts(
self, dataset: Dataset[Any], train_indices: List[int], val_indices: List[int], resources_id: str
) -> None:
"""Submit rollouts while respecting the maximum queue length."""
store = self.get_store()
for index in train_indices + val_indices:
@@ -153,7 +129,6 @@ class Baseline(FastAlgorithm):
await asyncio.sleep(self.polling_interval)
async def _harvest_rollout_spans(self, rollout_id: str):
"""Poll rollout status updates until completion and log transitions."""
store = self.get_store()
last_status: Optional[RolloutStatus] = None
@@ -185,12 +160,11 @@ class Baseline(FastAlgorithm):
train_dataset: Optional[Dataset[Any]] = None,
val_dataset: Optional[Dataset[Any]] = None,
) -> None:
"""Execute the baseline loop across the provided datasets."""
train_dataset_length = len(train_dataset) if train_dataset is not None else 0
val_dataset_length = len(val_dataset) if val_dataset is not None else 0
if train_dataset_length == 0 and val_dataset_length == 0:
logger.error(
"MockAlgorithm requires at least one dataset. Provide train_dataset or val_dataset before running."
"MockAlgorithm requires at least a train_dataset or val_dataset to run. No train_dataset or val_dataset is provided. Exiting."
)
return
@@ -199,8 +173,6 @@ class Baseline(FastAlgorithm):
]
train_indices = list(range(0, train_dataset_length))
val_indices = list(range(train_dataset_length, train_dataset_length + val_dataset_length))
logger.debug(f"Train indices: {train_indices}")
logger.debug(f"Val indices: {val_indices}")
store = self.get_store()
@@ -218,24 +190,19 @@ class Baseline(FastAlgorithm):
harvest_tasks: List[asyncio.Task[None]] = []
logger.info(f"Proceeding epoch {epoch + 1}/{self.n_epochs}.")
for index in train_indices + val_indices:
logger.info(
f"Processing index {index}. {len(train_indices)} train indices and {len(val_indices)} val indices in total."
)
while True:
queuing_rollouts = await store.query_rollouts(status=["queuing", "requeuing"])
if len(queuing_rollouts) <= self.max_queue_length:
# Only enqueue a new rollout when there is at most "max_queue_length" rollout in the queue.
sample = concatenated_dataset[index]
mode = "train" if index in train_indices else "val"
rollout = await store.enqueue_rollout(input=sample, mode=mode, resources_id=resources_id)
harvest_tasks.append(asyncio.create_task(self._harvest_rollout_spans(rollout.rollout_id)))
logger.info(f"Enqueued rollout {rollout.rollout_id} in {mode} mode with sample: {sample}")
break
else:
# Sleep a bit and try again later.
await asyncio.sleep(self.polling_interval)
queuing_rollouts = await store.query_rollouts(status=["queuing", "requeuing"])
if len(queuing_rollouts) <= self.max_queue_length:
# Only enqueue a new rollout when there is at most "max_queue_length" rollout in the queue.
sample = concatenated_dataset[index]
mode = "train" if index in train_indices else "val"
rollout = await store.enqueue_rollout(input=sample, mode=mode, resources_id=resources_id)
harvest_tasks.append(asyncio.create_task(self._harvest_rollout_spans(rollout.rollout_id)))
logger.info(f"Enqueued rollout {rollout.rollout_id} in {mode} mode with sample: {sample}")
else:
# Sleep a bit and try again later.
await asyncio.sleep(self.polling_interval)
# Wait for all harvest tasks to complete
logger.info(f"Waiting for {len(harvest_tasks)} harvest tasks to complete...")
print(f"Waiting for {len(harvest_tasks)} harvest tasks to complete...")
if len(harvest_tasks) > 0:
await asyncio.gather(*harvest_tasks)
-43
View File
@@ -1,43 +0,0 @@
# Copyright (c) Microsoft. All rights reserved.
import random
from typing import Iterator, List, Sequence, TypeVar
from agentlightning.types import Dataset
T_task = TypeVar("T_task")
def batch_iter_over_dataset(dataset: Dataset[T_task], batch_size: int) -> Iterator[Sequence[T_task]]:
"""
Create an infinite iterator that yields batches from the dataset.
When batch_size >= dataset size, yields the entire shuffled dataset repeatedly.
When batch_size < dataset size, yields batches of the specified size, reshuffling
after each complete pass through the dataset.
Args:
dataset: The dataset to iterate over.
batch_size: The desired batch size.
Yields:
Sequences of tasks from the dataset. Each task appears at most once per epoch.
"""
if batch_size >= len(dataset):
while True:
dataset_copy = [dataset[i] for i in range(len(dataset))]
random.shuffle(dataset_copy)
yield dataset_copy
else:
current_batch: List[int] = []
while True:
indices = list(range(len(dataset)))
random.shuffle(indices)
for index in indices:
if index in current_batch:
continue
current_batch.append(index)
if len(current_batch) == batch_size:
yield [dataset[index] for index in current_batch]
current_batch = []
+9 -93
View File
@@ -5,89 +5,23 @@ from typing import Any, Optional
from hydra import compose, initialize
from omegaconf import OmegaConf
from agentlightning.algorithm.base import Algorithm
from agentlightning.algorithm.base import BaseAlgorithm
from agentlightning.client import AgentLightningClient
from agentlightning.types import Dataset
from agentlightning.verl.entrypoint import run_ppo # type: ignore
class VERL(Algorithm):
"""VERL-powered algorithm that delegates training to the VERL PPO runner.
class VERL(BaseAlgorithm):
"""Algorithm leveraging VERL as the backend framework.
!!! warning
Advanced customisation currently requires copying the VERL source and
modifying it directly. Native hooks for overriding training behaviour
will land in a future release.
**Note on Customization:**
At present, we recommend copying the source code from VERL and modifying it as needed to suit your requirements.
Native support for customizing training logic will be provided in future releases.
Args:
config: Dictionary mirroring the overrides passed to the VERL CLI. The
overrides are merged with VERL's packaged defaults via Hydra before
launching training.
Examples:
```python
from agentlightning.algorithm.verl import VERL
algorithm = VERL(
config={
"algorithm": {
"adv_estimator": "grpo",
"use_kl_in_reward": False,
},
"data": {
"train_batch_size": 32,
"max_prompt_length": 4096,
"max_response_length": 2048,
},
"actor_rollout_ref": {
"rollout": {
"tensor_model_parallel_size": 1,
"n": 4,
"log_prob_micro_batch_size_per_gpu": 4,
"multi_turn": {"format": "hermes"},
"name": "vllm",
"gpu_memory_utilization": 0.6,
},
"actor": {
"ppo_mini_batch_size": 32,
"ppo_micro_batch_size_per_gpu": 4,
"optim": {"lr": 1e-6},
"use_kl_loss": False,
"kl_loss_coef": 0.0,
"entropy_coeff": 0,
"clip_ratio_low": 0.2,
"clip_ratio_high": 0.3,
"fsdp_config": {
"param_offload": True,
"optimizer_offload": True,
},
},
"ref": {
"log_prob_micro_batch_size_per_gpu": 8,
"fsdp_config": {"param_offload": True},
},
"model": {
"path": "Qwen/Qwen2.5-1.5B-Instruct",
"use_remove_padding": True,
"enable_gradient_checkpointing": True,
},
},
"trainer": {
"n_gpus_per_node": 1,
"val_before_train": True,
"critic_warmup": 0,
"logger": ["console", "wandb"],
"project_name": "AgentLightning",
"experiment_name": "calc_x",
"nnodes": 1,
"save_freq": 64,
"test_freq": 32,
"total_epochs": 2,
},
}
)
trainer.fit(algorithm, train_dataset=my_train_dataset)
```
config: The VERL configuration, matching what is typically provided when running VERL via the command line.
This config will be merged with VERL's base configuration and processed by Hydra.
"""
def __init__(self, config: dict[str, Any]):
@@ -99,8 +33,6 @@ class VERL(Algorithm):
# Merge your dict overrides
override_conf = OmegaConf.create(config)
# Allow adding new fields
OmegaConf.set_struct(base_cfg, False)
self.config = OmegaConf.merge(base_cfg, override_conf)
def run(
@@ -108,17 +40,6 @@ class VERL(Algorithm):
train_dataset: Optional[Dataset[Any]] = None,
val_dataset: Optional[Dataset[Any]] = None,
) -> None:
"""Launch the VERL PPO entrypoint with the configured runtime context.
Args:
train_dataset: Optional dataset forwarded to VERL for training.
val_dataset: Optional dataset forwarded to VERL for evaluation.
Raises:
ValueError: If required dependencies such as the store, LLM proxy, or
adapter have been garbage-collected when using the V1 execution
mode.
"""
try:
store = self.get_store()
except Exception:
@@ -145,10 +66,5 @@ class VERL(Algorithm):
)
def get_client(self) -> AgentLightningClient:
"""Create a client bound to the VERL-managed Agent Lightning server.
Deprecated:
Since v0.2.
"""
port = self.config.agentlightning.port
return AgentLightningClient(endpoint=f"http://localhost:{port}")
+30
View File
@@ -0,0 +1,30 @@
# Copyright (c) Microsoft. All rights reserved.
from __future__ import annotations
import argparse
import time
from typing import Iterable
from agentlightning.instrumentation.agentops import AgentOpsServerManager
def main(argv: Iterable[str] | None = None) -> int:
parser = argparse.ArgumentParser(description="Start AgentOps server")
parser.add_argument("--daemon", action="store_true", help="Run server as a daemon")
parser.add_argument("--port", type=int, default=8002, help="Port to run the server on")
args = parser.parse_args(list(argv) if argv is not None else None)
manager = AgentOpsServerManager(daemon=args.daemon, port=args.port)
try:
manager.start()
# Wait forever
while True:
time.sleep(1)
except KeyboardInterrupt:
manager.stop()
return 0
if __name__ == "__main__":
raise SystemExit(main())
+4 -23
View File
@@ -6,42 +6,23 @@ from __future__ import annotations
import argparse
import asyncio
import logging
from typing import Iterable
from agentlightning import setup_logging
from agentlightning.logging import configure_logger
from agentlightning.store.client_server import LightningStoreServer
from agentlightning.store.memory import InMemoryLightningStore
logger = logging.getLogger(__name__)
def main(argv: Iterable[str] | None = None) -> int:
parser = argparse.ArgumentParser(description="Run a LightningStore server")
parser.add_argument("--port", type=int, default=4747, help="Port to run the server on")
parser.add_argument(
"--cors-origin",
dest="cors_origins",
action="append",
help="Allowed CORS origin. Repeat for multiple origins. Use '*' to allow all origins.",
)
args = parser.parse_args(list(argv) if argv is not None else None)
setup_logging()
configure_logger()
store = InMemoryLightningStore()
server = LightningStoreServer(
store,
host="0.0.0.0",
port=args.port,
cors_allow_origins=args.cors_origins,
launch_mode="asyncio",
)
try:
asyncio.run(server.run_forever())
except RuntimeError as exc:
logger.error("LightningStore server failed to start: %s", exc, exc_info=True)
return 1
server = LightningStoreServer(store, host="0.0.0.0", port=args.port)
asyncio.run(server.run_forever())
return 0
+61 -96
View File
@@ -1,12 +1,6 @@
# Copyright (c) Microsoft. All rights reserved.
"""Utilities for interacting with legacy Agent Lightning servers.
This module contains compatibility shims that speak the deprecated HTTP
interface used by older Agent Lightning deployments. Modern code should prefer
the store-based APIs exposed by `agentlightning.store`, but keeping these
clients available makes it easier to migrate existing workflows incrementally.
"""
"""Legacy client for interacting with a legacy Agent Lightning server."""
import asyncio
import logging
@@ -24,24 +18,13 @@ logger = logging.getLogger(__name__)
class AgentLightningClient:
"""Client wrapper for the legacy version-aware Agent Lightning server.
"""
Client for interacting with a version-aware Agent Lightning Server.
The client exposes synchronous and asynchronous helpers for polling tasks,
retrieving resource bundles, and submitting rollouts. It also maintains a
simple in-memory cache keyed by the server-provided resource identifier to
avoid redundant network requests.
!!! warning "Deprecated"
[`AgentLightningClient`][agentlightning.client.AgentLightningClient] is part of
the legacy client/server stack. New code should rely on the store-based APIs
implemented in `agentlightning.store`.
Attributes:
endpoint: Base URL of the Agent Lightning server.
poll_interval: Delay in seconds between polling attempts when no task is
available.
timeout: Timeout in seconds applied to HTTP requests.
task_count: Number of tasks claimed during the lifetime of this client.
This client handles polling for tasks, fetching specific versions of resources
(like model configurations), and posting completed rollouts back to the server.
It provides both synchronous and asynchronous methods for these operations and
includes a cache for resources.
"""
_next_task_uri = "/task"
@@ -50,12 +33,12 @@ class AgentLightningClient:
_report_rollout_uri = "/rollout"
def __init__(self, endpoint: str, poll_interval: float = 5.0, timeout: float = 10.0):
"""Initialize the client.
"""Initializes the AgentLightningClient.
Args:
endpoint: Root URL of the Agent Lightning server.
poll_interval: Seconds to wait between polling attempts.
timeout: Seconds before a request to the server is considered timed out.
endpoint: The root URL of the Agent Lightning server.
poll_interval: The interval in seconds to wait between polling for new tasks.
timeout: The timeout in seconds for HTTP requests.
"""
warnings.warn(
"AgentLightningClient is deprecated. Please use LightningStoreClient instead.", DeprecationWarning
@@ -68,13 +51,13 @@ class AgentLightningClient:
self._default_headers = {"X-AgentLightning-Client": "true"}
async def _request_json_async(self, url: str) -> Optional[Dict[str, Any]]:
"""Perform an asynchronous ``GET`` request and parse the JSON payload.
"""Makes an async GET request to the specified URL and returns the JSON response.
Args:
url: Fully qualified URL to query.
url: The URL to request.
Returns:
Parsed JSON body as a dictionary if the request succeeds; otherwise ``None``.
The JSON response as a dictionary or None if the request fails.
"""
timeout = aiohttp.ClientTimeout(total=self.timeout)
async with aiohttp.ClientSession(timeout=timeout) as session:
@@ -87,14 +70,14 @@ class AgentLightningClient:
return None
async def _post_json_async(self, url: str, payload: Dict[str, Any]) -> Optional[Dict[str, Any]]:
"""Perform an asynchronous ``POST`` request with a JSON body.
"""Makes an async POST request with a JSON payload.
Args:
url: Fully qualified URL that accepts the payload.
payload: Dictionary that will be serialized and sent as JSON.
url: The URL to post to.
payload: The dictionary data to send as JSON.
Returns:
Parsed JSON body as a dictionary if the request succeeds; otherwise ``None``.
The JSON response as a dictionary or None if the request fails.
"""
timeout = aiohttp.ClientTimeout(total=self.timeout)
async with aiohttp.ClientSession(timeout=timeout) as session:
@@ -107,11 +90,10 @@ class AgentLightningClient:
return None
async def poll_next_task_async(self) -> Optional[Task]:
"""Poll the server asynchronously until a task becomes available.
"""Polls the server asynchronously for the next task until one is available.
Returns:
The next [`Task`][agentlightning.Task] exposed by the server,
or ``None`` if polling fails.
A Task object containing the task details.
"""
url = urllib.parse.urljoin(self.endpoint, self._next_task_uri)
while True:
@@ -126,15 +108,13 @@ class AgentLightningClient:
await asyncio.sleep(self.poll_interval)
async def get_resources_by_id_async(self, resource_id: str) -> Optional[ResourcesUpdate]:
"""Fetch a specific resource bundle by identifier.
"""Fetches a specific version of resources by its ID, using a cache.
Args:
resource_id: Identifier sourced from the task metadata.
resource_id: The ID of the resources to fetch, usually from a Task's metadata.
Returns:
Cached or freshly downloaded
[`ResourcesUpdate`][agentlightning.ResourcesUpdate], or
``None`` when the server returns an error.
A ResourcesUpdate object containing the versioned resources, or None if not found.
"""
if resource_id in self._resource_cache:
logger.debug(f"Found resources '{resource_id}' in cache.")
@@ -150,11 +130,10 @@ class AgentLightningClient:
return None
async def get_latest_resources_async(self) -> Optional[ResourcesUpdate]:
"""Fetch the most recent resource bundle advertised by the server.
"""Fetches the latest available resources from the server.
Returns:
[`ResourcesUpdate`][agentlightning.ResourcesUpdate] for the
newest version, or ``None`` when unavailable.
A ResourcesUpdate object containing the latest resources.
"""
url = urllib.parse.urljoin(self.endpoint, self._latest_resources_uri)
response = await self._request_json_async(url)
@@ -166,26 +145,26 @@ class AgentLightningClient:
return None
async def post_rollout_async(self, rollout: RolloutLegacy) -> Optional[Dict[str, Any]]:
"""Submit a completed rollout back to the server.
"""Posts a completed rollout to the server asynchronously.
Args:
rollout: Legacy rollout payload produced by the executor.
rollout: A Rollout object containing the results of a task.
Returns:
Parsed JSON response returned by the server, or ``None`` when the request fails.
The server's JSON response as a dictionary.
"""
url = urllib.parse.urljoin(self.endpoint, self._report_rollout_uri)
payload = rollout.model_dump(mode="json")
return await self._post_json_async(url, payload)
def _request_json(self, url: str) -> Optional[Dict[str, Any]]:
"""Perform a blocking ``GET`` request and parse the JSON payload.
"""Makes a sync GET request to the specified URL and returns the JSON response.
Args:
url: Fully qualified URL to query.
url: The URL to request.
Returns:
Parsed JSON body as a dictionary if the request succeeds; otherwise ``None``.
The JSON response as a dictionary or None if the request fails.
"""
try:
response = requests.get(url, timeout=self.timeout, headers=self._default_headers)
@@ -196,14 +175,14 @@ class AgentLightningClient:
return None
def _post_json(self, url: str, payload: Dict[str, Any]) -> Optional[Dict[str, Any]]:
"""Perform a blocking ``POST`` request with a JSON payload.
"""Makes a sync POST request with a JSON payload.
Args:
url: Fully qualified URL that accepts the payload.
payload: Dictionary that will be serialized and sent as JSON.
url: The URL to post to.
payload: The dictionary data to send as JSON.
Returns:
Parsed JSON body as a dictionary if the request succeeds; otherwise ``None``.
The JSON response as a dictionary or None if the request fails.
"""
try:
response = requests.post(url, json=payload, timeout=self.timeout, headers=self._default_headers)
@@ -214,11 +193,10 @@ class AgentLightningClient:
return None
def poll_next_task(self) -> Optional[Task]:
"""Poll the server synchronously until a task becomes available.
"""Polls the server synchronously for the next task until one is available.
Returns:
The next [`Task`][agentlightning.Task] available for execution, or
``None`` if polling fails.
A Task object containing the task details, including the required `resources_id`.
"""
url = urllib.parse.urljoin(self.endpoint, self._next_task_uri)
while True:
@@ -233,15 +211,13 @@ class AgentLightningClient:
time.sleep(self.poll_interval)
def get_resources_by_id(self, resource_id: str) -> Optional[ResourcesUpdate]:
"""Fetch a specific resource bundle by identifier.
"""Fetches a specific version of resources by its ID synchronously, using a cache.
Args:
resource_id: Identifier sourced from the task metadata.
resource_id: The ID of the resources to fetch, usually from a Task's metadata.
Returns:
Cached or freshly downloaded
[`ResourcesUpdate`][agentlightning.ResourcesUpdate], or
``None`` when the server returns an error.
A ResourcesUpdate object containing the versioned resources, or None if not found.
"""
if resource_id in self._resource_cache:
logger.debug(f"Found resources '{resource_id}' in cache.")
@@ -257,11 +233,10 @@ class AgentLightningClient:
return None
def get_latest_resources(self) -> Optional[ResourcesUpdate]:
"""Fetch the most recent resource bundle advertised by the server.
"""Fetches the latest available resources from the server synchronously.
Returns:
[`ResourcesUpdate`][agentlightning.ResourcesUpdate] for the
newest version, or ``None`` when unavailable.
A ResourcesUpdate object containing the latest resources.
"""
url = urllib.parse.urljoin(self.endpoint, self._latest_resources_uri)
response = self._request_json(url)
@@ -272,13 +247,13 @@ class AgentLightningClient:
return None
def post_rollout(self, rollout: RolloutLegacy) -> Optional[Dict[str, Any]]:
"""Submit a completed rollout back to the server.
"""Posts a completed rollout to the server synchronously.
Args:
rollout: Legacy rollout payload produced by the executor.
rollout: A Rollout object containing the results of a task.
Returns:
Parsed JSON response returned by the server, or ``None`` when the request fails.
The server's JSON response as a dictionary.
"""
url = urllib.parse.urljoin(self.endpoint, self._report_rollout_uri)
payload = rollout.model_dump(mode="json")
@@ -286,16 +261,14 @@ class AgentLightningClient:
class DevTaskLoader(AgentLightningClient):
"""In-memory task loader used for development and integration tests.
"""A local task manager for development that provides sample tasks and resources.
The loader mimics the behavior of the legacy HTTP server by storing tasks and
resources locally. Polling methods simply iterate over the provided collection,
allowing rapid iteration without provisioning any external infrastructure.
This client mocks the server APIs by maintaining a local queue of tasks and resources
within the same process. It's designed for development, testing, and scenarios where
a full Agent Lightning server is not needed.
!!! warning "Deprecated"
[`DevTaskLoader`][agentlightning.client.DevTaskLoader] is a compatibility shim.
Prefer [`Trainer.dev`][agentlightning.Trainer.dev] for new code.
The DevTaskLoader overrides the polling and resource fetching methods to return data
from local collections instead of making HTTP requests to a remote server.
"""
def __init__(
@@ -304,17 +277,12 @@ class DevTaskLoader(AgentLightningClient):
resources: Union[NamedResources, ResourcesUpdate],
**kwargs: Any,
):
"""Initialize the loader with predefined tasks and resources.
"""Initializes the DevTaskLoader with pre-defined tasks and resources.
Args:
tasks: Sequence of task inputs or preconstructed tasks that will be served in
order.
resources: Static resources returned for any `resources_id` query.
**kwargs: Additional keyword arguments forwarded to the parent client.
Raises:
ValueError: If no tasks are provided or both [`Task`][agentlightning.Task]
and [`TaskInput`][agentlightning.TaskInput] instances are mixed.
tasks: Either a List of TaskInput objects or a List of Task objects.
resources: Either NamedResources or ResourcesUpdate object.
**kwargs: Additional arguments passed to the parent AgentLightningClient.
"""
warnings.warn("DevTaskLoader is deprecated. Please use Trainer.dev instead.", DeprecationWarning)
super().__init__(endpoint="local://", **kwargs)
@@ -332,27 +300,24 @@ class DevTaskLoader(AgentLightningClient):
if isinstance(resources, ResourcesUpdate):
self._resources_update = resources
else:
self._resources_update = ResourcesUpdate(
resources_id="local", resources=resources, create_time=time.time(), update_time=time.time(), version=1
)
self._resources_update = ResourcesUpdate(resources_id="local", resources=resources)
# Store rollouts posted back to the loader for easy debugging of local runs
self._rollouts: List[RolloutLegacy] = []
@property
def rollouts(self) -> List[RolloutLegacy]:
"""Return the rollouts posted back to the loader during development runs."""
"""Return rollouts that have been posted back to the loader."""
return self._rollouts
def poll_next_task(self) -> Optional[Task]:
"""Return the next task from the local queue.
"""Returns the next task from the local queue.
If [`TaskInput`][agentlightning.TaskInput] instances were provided,
they are converted into [`Task`][agentlightning.Task] objects on the
fly. Otherwise, the preconstructed tasks are returned in sequence.
If tasks are TaskInput objects, assembles them into Task objects.
If tasks are already Task objects, returns them directly.
Returns:
Next task to execute.
The next Task object from the local task list.
"""
if self._task_index >= len(self._tasks):
self._task_index = 0
+6 -11
View File
@@ -83,17 +83,12 @@ def _str_to_bool(v: str) -> bool:
def _get_param_type_details(param_annotation: Any) -> Tuple[Any, bool, bool]:
"""Normalize an annotation into its core type, optionality, and list status.
Args:
param_annotation: The annotation to inspect.
Returns:
A tuple ``(core_type, is_optional, is_list)`` describing the normalized type.
- For ``Optional[T]`` → ``(T, True, is_list_status_of_T)``
- For ``List[T]`` → ``(List[T], is_optional_status_of_List, True)``
- For ``Optional[List[T]]`` → ``(List[T], True, True)``
"""
Determines the core type, if it's Optional, and if it's a List.
Returns: (core_type, is_optional, is_list)
- For Optional[T]: (T, True, is_list_status_of_T)
- For List[T]: (List[T], is_optional_status_of_List, True)
- For Optional[List[T]]: (List[T], True, True)
"""
is_optional = False
is_list = False
+1 -9
View File
@@ -13,15 +13,7 @@ logger = logging.getLogger(__name__)
def emit_exception(exception: BaseException) -> None:
"""Record an exception with OpenTelemetry metadata.
Args:
exception: Raised exception instance to serialize into telemetry attributes.
!!! note
The helper validates its input. Non-exception values are ignored to prevent
noisy telemetry and indicate programming mistakes via the logger.
"""
"""Emit an exception as a span."""
if not isinstance(exception, BaseException): # type: ignore
logger.error(f"Expected an BaseException instance, got: {type(exception)}. Skip emit_exception.")
return
+3 -7
View File
@@ -10,14 +10,10 @@ logger = logging.getLogger(__name__)
def emit_message(message: str) -> None:
"""Emit a textual message as an OpenTelemetry span.
"""Emit a string message as a span.
Args:
message: Human readable message to attach as a span attribute.
!!! note
OpenTelemetry distinguishes between logs and spans. Emitting the message as a
span keeps all Agent Lightning telemetry in a single data store for analysis.
OpenTelemetry has a dedicated design of logs by design, but we can also use spans to emit messages.
So that it can all be unified in the data store and analyzed together.
"""
if not isinstance(message, str): # type: ignore
logger.error(f"Message must be a string, got: {type(message)}. Skip emit_message.")
+1 -9
View File
@@ -12,15 +12,7 @@ logger = logging.getLogger(__name__)
def emit_object(object: Any) -> None:
"""Emit an object's serialized representation as an OpenTelemetry span.
Args:
object: Data structure to encode as JSON and attach to the span payload.
!!! note
The payload must be JSON serializable. Non-serializable objects are ignored and
an error is logged to aid debugging.
"""
"""Emit any object as a span. Make sure the object is JSON serializable."""
try:
serialized = json.dumps(object)
except (TypeError, ValueError):
+22 -45
View File
@@ -1,7 +1,5 @@
# Copyright (c) Microsoft. All rights reserved.
"""Helpers for emitting reward spans and integrating with AgentOps telemetry."""
import asyncio
import inspect
import json
@@ -49,29 +47,20 @@ FnType = TypeVar("FnType", bound=Callable[..., Any])
def _agentops_initialized() -> bool:
"""Return `True` when the AgentOps client has been configured."""
"""Check if AgentOps is initialized in the current context."""
return agentops.get_client().initialized
def reward(fn: FnType) -> FnType:
"""Decorate a reward function so its outputs are tracked as spans.
The decorator integrates with AgentOps when it is available and falls back to
the built-in telemetry otherwise. Both synchronous and asynchronous functions
are supported transparently.
Deprecated:
This decorator is deprecated. Use [`emit_reward`][agentlightning.emit_reward] instead.
Args:
fn: Callable that produces a numeric reward.
Returns:
Wrapped callable that preserves the original signature.
"""
A decorator to wrap a function that computes rewards.
It will automatically handle the input and output of the function.
"""
def wrap_result(result: Optional[float]) -> RewardSpanData:
"""Normalize the reward value into the span payload format."""
"""
Wrap the result of the function in a dict.
"""
if result is None:
return {"type": "reward", "value": None}
if not isinstance(result, (float, int)): # type: ignore
@@ -130,18 +119,8 @@ def reward(fn: FnType) -> FnType:
def emit_reward(reward: float) -> ReadableSpan:
"""Emit a reward value as an OpenTelemetry span.
Args:
reward: Numeric reward to record. Integers and booleans are converted to
floating point numbers for consistency.
Returns:
Readable span capturing the recorded reward.
Raises:
ValueError: If the provided reward cannot be interpreted as a float or the
resulting span is not a [`ReadableSpan`](https://opentelemetry.io/docs/concepts/signals/traces/) instance.
"""
Record a new reward as a new span.
"""
logger.debug(f"Emitting reward: {reward}")
if isinstance(reward, (int, bool)):
@@ -149,7 +128,6 @@ def emit_reward(reward: float) -> ReadableSpan:
if not isinstance(reward, float):
raise ValueError(f"Reward must be a number, got: {type(reward)}")
# TODO: This should use the tracer from current context by tracer
tracer = get_tracer()
span = tracer.start_span(SpanNames.REWARD.value, attributes={"reward": reward})
# Do nothing; it's just a number
@@ -161,13 +139,8 @@ def emit_reward(reward: float) -> ReadableSpan:
def get_reward_value(span: SpanLike) -> Optional[float]:
"""Extract the reward value from a span, if available.
Args:
span: Span object produced by AgentOps or Agent Lightning emitters.
Returns:
The reward encoded in the span or `None` when the span does not represent a reward.
"""
Get the reward value from a span.
"""
for key in [
"agentops.task.output", # newer versions of agentops
@@ -205,31 +178,35 @@ def get_reward_value(span: SpanLike) -> Optional[float]:
def is_reward_span(span: SpanLike) -> bool:
"""Return ``True`` when the provided span encodes a reward value."""
"""
Check if a span is a reward span.
"""
maybe_reward = get_reward_value(span)
return maybe_reward is not None
def find_reward_spans(spans: Sequence[SpanLike]) -> List[SpanLike]:
"""Return all reward spans in the provided sequence.
"""
Find all reward spans in the given list of spans.
Args:
spans: Sequence containing [`ReadableSpan`](https://opentelemetry.io/docs/concepts/signals/traces/) objects or mocked span-like values.
spans: A list of spans (either ReadableSpan or Span).
Returns:
List of spans that could be parsed as rewards.
A list of spans whose name matches the reward span name.
"""
return [span for span in spans if is_reward_span(span)]
def find_final_reward(spans: Sequence[SpanLike]) -> Optional[float]:
"""Return the last reward value present in the provided spans.
"""
Get the last reward value from a list of spans.
Args:
spans: Sequence containing [`ReadableSpan`](https://opentelemetry.io/docs/concepts/signals/traces/) objects or mocked span-like values.
spans: A list of spans (either ReadableSpan or Span).
Returns:
Reward value from the latest reward span, or `None` when none are found.
The reward value from the last reward span, or None if not found.
"""
for span in reversed(spans):
reward = get_reward_value(span)
+6 -6
View File
@@ -1,19 +1,19 @@
# Copyright (c) Microsoft. All rights reserved.
"""Utilities shared across emitter implementations."""
"""Common utilities for the emitter module."""
import opentelemetry.trace as trace_api
from opentelemetry.trace import get_tracer_provider
def get_tracer() -> trace_api.Tracer:
"""Resolve the OpenTelemetry tracer configured for Agent Lightning.
Returns:
OpenTelemetry tracer tagged with the `agentlightning` instrumentation name.
"""Return the tracer used for AgentLightning spans.
Raises:
RuntimeError: If OpenTelemetry was not initialized before calling this helper.
RuntimeError: If the tracer is not initialized.
Returns:
The AgentLightning tracer instance.
"""
if hasattr(trace_api, "_TRACER_PROVIDER") and trace_api._TRACER_PROVIDER is None: # type: ignore[attr-defined]
raise RuntimeError("Tracer is not initialized. Cannot emit a meaningful span.")
+10 -79
View File
@@ -1,9 +1,6 @@
# Copyright (c) Microsoft. All rights reserved.
from __future__ import annotations
import logging
import os
from typing import Protocol
from agentlightning.store.base import LightningStore
@@ -13,94 +10,28 @@ from .events import ExecutionEvent
logger = logging.getLogger(__name__)
_TRUTHY_VALUES = {"1", "true", "yes", "on"}
_FALSY_VALUES = {"0", "false", "no", "off"}
def resolve_managed_store_flag(value: bool | None) -> bool:
"""Determine whether execution helpers should wrap the provided store.
The helper first honours an explicit `value`. When `None` it falls back
to the `AGL_MANAGED_STORE` environment variable, accepting a variety
of truthy and falsy spellings. Missing environment configuration defaults to
`True` so that higher-level strategies create the appropriate client or
server wrappers automatically.
Args:
value: Optional override supplied by the caller.
Returns:
`True` when a managed store should be created around the provided
instance, otherwise `False`.
Raises:
ValueError: If `AGL_MANAGED_STORE` is set to an unsupported
value.
"""
if value is not None:
return value
env_value = os.getenv("AGL_MANAGED_STORE")
if env_value is None:
return True
normalized = env_value.strip().lower()
if normalized in _TRUTHY_VALUES:
return True
if normalized in _FALSY_VALUES:
return False
raise ValueError("AGL_MANAGED_STORE must be one of 1, 0, true, false, yes, no, on, or off")
class AlgorithmBundle(Protocol):
"""Callable bundle produced by [`Trainer`][agentlightning.Trainer].
Execution strategies treat the returned coroutine as opaque, only providing
the shared store instance and cooperative stop event. Bundles typically
encapsulate algorithm setup plus adapter and LLM proxy, etc.
"""
async def __call__(self, store: LightningStore, event: ExecutionEvent) -> None:
"""Execute algorithm logic using ``store`` until completion or stop."""
"""Initalization and execution logic."""
class RunnerBundle(Protocol):
"""Callable bundle wrapping runner setup and the worker loop, as opposed to the
[`AlgorithmBundle`][agentlightning.AlgorithmBundle]."""
async def __call__(self, store: LightningStore, worker_id: int, event: ExecutionEvent) -> None:
"""Execute runner logic for ``worker_id`` using ``store`` and ``event``."""
"""Initalization and execution logic."""
class ExecutionStrategy:
"""Coordinate algorithm and runner bundles within a single process abstraction.
"""When trainer has created the executable of algorithm and runner in two bundles,
the execution strategy defines how to run them together, and how many parallel runners to run.
Strategies decide how many worker bundles to launch, whether to communicate
through shared memory or an HTTP boundary, and how to react to shutdown
signals. They intentionally avoid inspecting the bundle internals; instead,
each bundle remains responsible for its own scheduling semantics.
The store is the centric place for the two bundles to communicate.
!!! note
Implementations must honor the [execute()][agentlightning.ExecutionStrategy.execute]
contract by propagating `KeyboardInterrupt` and ensuring resources are
released when an error occurs on either side of the algorithm/runner
pair.
The algorithm and runner's behavior (whether runner should perform one step or run forever,
whether the algo would send out the tasks or not) are defined inside the bundle,
and does not belong to the execution strategy.
The execute should support Ctrl+C to exit gracefully.
"""
def execute(self, algorithm: AlgorithmBundle, runner: RunnerBundle, store: LightningStore) -> None:
"""Run the provided bundles using the configured orchestration model.
Args:
algorithm: Callable bundle responsible for algorithm execution.
runner: Callable bundle for runner workers.
store: Concrete [`LightningStore`][agentlightning.LightningStore]
shared across bundles.
Raises:
NotImplementedError: Subclasses must provide the orchestration
implementation.
"""
raise NotImplementedError()
+67 -95
View File
@@ -12,51 +12,52 @@ from typing import Callable, Iterable, Literal, cast
from agentlightning.store.base import LightningStore
from agentlightning.store.client_server import LightningStoreClient, LightningStoreServer
from .base import AlgorithmBundle, ExecutionStrategy, RunnerBundle, resolve_managed_store_flag
from .base import AlgorithmBundle, ExecutionStrategy, RunnerBundle
from .events import ExecutionEvent, MultiprocessingEvent
logger = logging.getLogger(__name__)
class ClientServerExecutionStrategy(ExecutionStrategy):
"""Run algorithm and runner bundles as separate processes over HTTP.
"""Run algorithm (server) and runners (clients) as separate processes over HTTP.
Execution Roles:
**Execution Roles:**
- `"algorithm"`: Start [`LightningStoreServer`][agentlightning.LightningStoreServer]
in-process and execute the algorithm bundle against it.
- `"runner"`: Connect to an existing server with
[`LightningStoreClient`][agentlightning.LightningStoreClient] and run the
runner bundle locally (spawning multiple processes when requested).
- `"both"`: Spawn runner processes first, then execute the algorithm and
server on the same machine. This mode orchestrates the full loop locally.
- "algorithm": Start the HTTP server (`LightningStoreServer`) in-process and run the
algorithm bundle against it.
- "runner": Connect to an already running server via `LightningStoreClient` and
execute runner bundles (optionally in multiple processes).
- "both": Spawn the runner processes first, then launch the algorithm/server
bundle on the main process. This mode orchestrates the full loop locally.
When `role == "both"` you may choose which side runs on the main process
via `main_process`. The runner-on-main option is limited to
`n_runners == 1` because each additional runner requires its own event
loop and process.
When role == "both", you may choose which side runs on the main process via
`main_process` (debug helper). Running the runner bundle on the main process
is only supported with `n_runners == 1`.
!!! warning
When `main_process == "runner"` the algorithm and HTTP server execute
in a child process. Store mutations remain isolated inside that process,
so the original store instance passed to
[execute()][agentlightning.ExecutionStrategy.execute] is not updated.
Important: When `main_process == "runner"`, the algorithm runs in a subprocess
with the LightningStore server. This means any state modifications made during
execution remain in that subprocess and are NOT reflected in the original store
object passed to `execute()`. The main process runner accesses the store only
through the HTTP client interface.
Abort Model (four-step escalation):
**Abort / Stop Model (four-step escalation):**
1. Cooperative stop. Every bundle receives a shared
[`MultiprocessingEvent`][agentlightning.MultiprocessingEvent] (`stop_evt`).
Any failure flips the event so peers can exit cleanly. Ctrl+C on the main
process also sets the flag.
2. KeyboardInterrupt synthesis. Remaining subprocesses receive ``SIGINT`` to
trigger `KeyboardInterrupt` handlers.
3. Termination. Stubborn processes are asked to ``terminate()``
(`SIGTERM` on POSIX).
4. Kill. As a last resort `kill()` is invoked (`SIGKILL` on POSIX).
1. Cooperative stop:
A shared :class:`~agentlightning.execution.events.MultiprocessingEvent`
(`stop_evt`) is passed to *all* bundles. Bundles should check it to exit.
Any crash (algorithm or runner) sets `stop_evt` so the other side can
stop cooperatively. Ctrl+C on the main process also flips the event.
2. KeyboardInterrupt synth:
Remaining subprocesses receive `SIGINT` to trigger `KeyboardInterrupt`
handlers.
3. Termination:
Stubborn subprocesses get `terminate()` (SIGTERM on POSIX).
4. Kill:
As a last resort we call `kill()` (SIGKILL on POSIX).
This mirrors the semantics implemented in
[`SharedMemoryExecutionStrategy`][agentlightning.SharedMemoryExecutionStrategy]
but adapts them to multiple processes and the HTTP client/server boundary.
Notes:
This mirrors the semantics implemented in :mod:`shared_memory`, but adapted
to multiple processes and the HTTP client/server boundary.
"""
alias: str = "cs"
@@ -67,21 +68,20 @@ class ClientServerExecutionStrategy(ExecutionStrategy):
server_host: str | None = None,
server_port: int | None = None,
n_runners: int = 1,
graceful_timeout: float = 10.0,
terminate_timeout: float = 10.0,
graceful_timeout: float = 5.0,
terminate_timeout: float = 5.0,
main_process: Literal["algorithm", "runner"] = "algorithm",
managed_store: bool | None = None,
) -> None:
"""Configure the strategy.
Args:
role: Which side(s) to run in this process. When omitted, the
`AGL_CURRENT_ROLE` environment variable is used.
:envvar:`AGL_CURRENT_ROLE` environment variable is used.
server_host: Interface the HTTP server binds to when running the
algorithm bundle locally. Defaults to `AGL_SERVER_HOST`
or `"localhost"` if unset.
algorithm bundle locally. Defaults to :envvar:`AGL_SERVER_HOST`
or ``"localhost"`` if unset.
server_port: Port for the HTTP server in "algorithm"/"both" modes.
Defaults to `AGL_SERVER_PORT` or `4747` if unset.
Defaults to :envvar:`AGL_SERVER_PORT` or ``4747`` if unset.
n_runners: Number of runner processes to spawn in "runner"/"both".
graceful_timeout: How long to wait (seconds) after setting the stop
event before escalating to signals.
@@ -90,20 +90,14 @@ class ClientServerExecutionStrategy(ExecutionStrategy):
main_process: Which bundle runs on the main process when
`role == "both"`. `"runner"` requires `n_runners == 1` and is
primarily intended for debugging.
managed_store: When `True` (default) the strategy constructs
LightningStore client/server wrappers automatically. When
`False` the provided `store` is passed directly to the
bundles, allowing callers to manage store wrappers manually.
"""
if role is None:
role_env = os.getenv("AGL_CURRENT_ROLE")
if role_env is None:
# Use both if not specified via env var or argument
role = "both"
elif role_env not in ("algorithm", "runner", "both"):
raise ValueError("role must be provided via argument or AGL_CURRENT_ROLE env var")
if role_env not in ("algorithm", "runner", "both"):
raise ValueError("role must be one of 'algorithm', 'runner', or 'both'")
else:
role = role_env
role = role_env
if server_host is None:
server_host = os.getenv("AGL_SERVER_HOST", "localhost")
@@ -132,26 +126,19 @@ class ClientServerExecutionStrategy(ExecutionStrategy):
if n_runners != 1:
raise ValueError("main_process='runner' requires n_runners to be 1")
self.main_process = main_process
self.managed_store = resolve_managed_store_flag(managed_store)
async def _execute_algorithm(
self, algorithm: AlgorithmBundle, store: LightningStore, stop_evt: ExecutionEvent
) -> None:
wrapper_store: LightningStore | None = None
if self.managed_store:
logger.info("Starting LightningStore server on %s:%s", self.server_host, self.server_port)
wrapper_store = LightningStoreServer(store, host=self.server_host, port=self.server_port)
server_started = False
else:
wrapper_store = store
server_started = False
logger.info("Starting LightningStore server on %s:%s", self.server_host, self.server_port)
server_store = LightningStoreServer(store, host=self.server_host, port=self.server_port)
server_started = False
try:
if self.managed_store and isinstance(wrapper_store, LightningStoreServer):
await wrapper_store.start()
server_started = True
logger.debug("Algorithm bundle starting against endpoint %s", wrapper_store.endpoint)
await algorithm(wrapper_store, stop_evt)
await server_store.start()
server_started = True
logger.debug("Algorithm bundle starting against endpoint %s", server_store.endpoint)
await algorithm(server_store, stop_evt)
logger.debug("Algorithm bundle completed successfully")
except KeyboardInterrupt:
logger.warning("Algorithm received KeyboardInterrupt; signaling stop event")
@@ -162,31 +149,18 @@ class ClientServerExecutionStrategy(ExecutionStrategy):
stop_evt.set()
raise
finally:
if self.managed_store and isinstance(wrapper_store, LightningStoreServer) and server_started:
if server_started:
try:
await wrapper_store.stop()
await server_store.stop()
except Exception:
logger.exception("Error stopping LightningStore server")
else:
logger.debug("LightningStore server shutdown completed")
async def _execute_runner(
self,
runner: RunnerBundle,
worker_id: int,
store: LightningStore,
stop_evt: ExecutionEvent,
) -> None:
if self.managed_store:
# If managed, we actually do not use the provided store
client_store = LightningStoreClient(f"http://{self.server_host}:{self.server_port}")
else:
client_store = store
async def _execute_runner(self, runner: RunnerBundle, worker_id: int, stop_evt: ExecutionEvent) -> None:
client_store = LightningStoreClient(f"http://{self.server_host}:{self.server_port}")
try:
if self.managed_store:
logger.debug("Runner %s connecting to server at %s:%s", worker_id, self.server_host, self.server_port)
else:
logger.debug("Runner %s executing with provided store", worker_id)
logger.debug("Runner %s connecting to server at %s:%s", worker_id, self.server_host, self.server_port)
await runner(client_store, worker_id, stop_evt)
logger.debug("Runner %s completed successfully", worker_id)
except KeyboardInterrupt:
@@ -198,18 +172,16 @@ class ClientServerExecutionStrategy(ExecutionStrategy):
stop_evt.set()
raise
finally:
if self.managed_store and isinstance(client_store, LightningStoreClient):
try:
await client_store.close()
except Exception:
logger.exception("Error closing LightningStore client for runner %s", worker_id)
else:
logger.debug("Runner %s closed LightningStore client", worker_id)
try:
await client_store.close()
except Exception:
logger.exception("Error closing LightningStore client for runner %s", worker_id)
else:
logger.debug("Runner %s closed LightningStore client", worker_id)
def _spawn_runners(
self,
runner: RunnerBundle,
store: LightningStore,
stop_evt: ExecutionEvent,
*,
ctx: BaseContext,
@@ -217,15 +189,15 @@ class ClientServerExecutionStrategy(ExecutionStrategy):
"""Used when `role == "runner"` or `role == "both"` and `n_runners > 1`."""
processes: list[multiprocessing.Process] = []
def _runner_sync(runner: RunnerBundle, worker_id: int, store: LightningStore, stop_evt: ExecutionEvent) -> None:
def _runner_sync(runner: RunnerBundle, worker_id: int, stop_evt: ExecutionEvent) -> None:
# Runners are executed in child processes; each process owns its own
# event loop to keep the asyncio scheduler isolated.
asyncio.run(self._execute_runner(runner, worker_id, store, stop_evt))
asyncio.run(self._execute_runner(runner, worker_id, stop_evt))
for i in range(self.n_runners):
process = cast(
multiprocessing.Process,
ctx.Process(target=_runner_sync, args=(runner, i, store, stop_evt), name=f"runner-{i}"), # type: ignore
ctx.Process(target=_runner_sync, args=(runner, i, stop_evt), name=f"runner-{i}"), # type: ignore
)
process.start()
logger.debug("Spawned runner process %s (pid=%s)", process.name, process.pid)
@@ -369,10 +341,10 @@ class ClientServerExecutionStrategy(ExecutionStrategy):
elif self.role == "runner":
if self.n_runners == 1:
logger.info("Running runner solely...")
asyncio.run(self._execute_runner(runner, 0, store, stop_evt))
asyncio.run(self._execute_runner(runner, 0, stop_evt))
else:
logger.info("Spawning runner processes...")
processes = self._spawn_runners(runner, store, stop_evt, ctx=ctx)
processes = self._spawn_runners(runner, stop_evt, ctx=ctx)
# Wait for the processes to finish naturally.
for process in processes:
process.join()
@@ -380,7 +352,7 @@ class ClientServerExecutionStrategy(ExecutionStrategy):
elif self.role == "both":
if self.main_process == "algorithm":
logger.info("Spawning runner processes...")
processes = self._spawn_runners(runner, store, stop_evt, ctx=ctx)
processes = self._spawn_runners(runner, stop_evt, ctx=ctx)
try:
logger.info("Running algorithm...")
asyncio.run(self._execute_algorithm(algorithm, store, stop_evt))
@@ -401,7 +373,7 @@ class ClientServerExecutionStrategy(ExecutionStrategy):
# the background process spawned above (the provided
# store must therefore be picklable when using spawn).
logger.info("Running runner...")
asyncio.run(self._execute_runner(runner, 0, store, stop_evt))
asyncio.run(self._execute_runner(runner, 0, stop_evt))
# Wait for the algorithm process to finish.
algorithm_process.join()
+18 -12
View File
@@ -7,18 +7,15 @@ from typing import Optional, Protocol
class ExecutionEvent(Protocol):
"""Protocol capturing the cooperative stop contract shared by strategies.
Implementations mirror the API of ``threading.Event`` and
``multiprocessing.Event`` so the rest of the execution layer can remain
agnostic to the underlying concurrency primitive.
"""
A minimal protocol similar to threading.Event.
Methods:
set: Signal cancellation. The call must be idempotent.
clear: Reset the event to the unsignaled state.
is_set: Return ``True`` when cancellation has been requested.
wait: Block until the event is signaled or an optional timeout elapses.
set(): Signal event like a cancellation (idempotent).
clear(): Reset to the non-set state.
is_set() -> bool: True if event has been signaled.
wait(timeout: Optional[float] = None) -> bool:
Block until event is set or timeout. Returns True if event has signaled.
"""
def set(self) -> None: ...
@@ -28,7 +25,11 @@ class ExecutionEvent(Protocol):
class ThreadingEvent:
"""Thread-safe implementation of [`ExecutionEvent`][agentlightning.ExecutionEvent]."""
"""
An Event implementation using threading.Event.
Provides a thread-safe event object for signaling between threads.
"""
__slots__ = ("_evt",)
@@ -49,7 +50,12 @@ class ThreadingEvent:
class MultiprocessingEvent:
"""Process-safe implementation of [`ExecutionEvent`][agentlightning.ExecutionEvent]."""
"""
An Event implementation using multiprocessing.Event.
Provides a process-safe event object for signaling between processes.
Optionally accepts a multiprocessing context for custom process start methods.
"""
__slots__ = ("_evt",)
@@ -4,12 +4,6 @@ from .base import ExecutionStrategy
class InterProcessExecutionStrategy(ExecutionStrategy):
"""Placeholder strategy for future inter-process primitives.
The class exists to reserve the `ipc` alias and make the planned
implementation discoverable. Attempting to use it today will raise
`NotImplementedError` once the execution contract is finalized.
"""
alias: str = "ipc"
+24 -39
View File
@@ -10,33 +10,28 @@ from typing import Any, Awaitable, Callable, List, Literal, Optional, Tuple
from agentlightning.store.base import LightningStore
from agentlightning.store.threading import LightningStoreThreaded
from .base import AlgorithmBundle, ExecutionStrategy, RunnerBundle, resolve_managed_store_flag
from .base import AlgorithmBundle, ExecutionStrategy, RunnerBundle
from .events import ExecutionEvent, ThreadingEvent
logger = logging.getLogger(__name__)
class SharedMemoryExecutionStrategy(ExecutionStrategy):
"""Execute bundles in a single process with cooperative worker threads.
"""Run algorithm and runners in a single process with threads sharing memory.
Stop Model:
Termination & abort model:
- All bundles share one [`ThreadingEvent`][agentlightning.ThreadingEvent]
named `stop_evt`.
- Only the main thread receives `KeyboardInterrupt`. When Ctrl+C occurs we
set `stop_evt`.
- Any exception raised inside a bundle sets `stop_evt` so other threads can
unwind cooperatively.
- Once the bundle running on the main thread exits successfully the
treatment depends on `main_thread`:
- `"algorithm"`: the runners are asked to stop by setting `stop_evt`.
- `"runner"`: the algorithm keeps running until it exits naturally.
- Background threads are marked as daemons. We join them briefly and log any
stragglers before shutting down.
- One shared ThreadingEvent (`stop_evt`) is passed to *all* bundles.
- The main thread (only) receives KeyboardInterrupt on Ctrl+C; we set `stop_evt` there.
- If any bundle raises, we set `stop_evt` from that thread to stop the rest.
- After the main-thread bundle finishes normally:
- If main_thread is "algorithm", we also set `stop_evt` to stop the runners.
- If main_thread is "runner", we do not set `stop_evt` to stop the algorithm.
We instead wait for the algorithm to finish naturally.
- Background threads are daemons; we join briefly and log any stragglers.
!!! note
Signals other than `SIGINT` (such as `SIGTERM`) are not intercepted;
Python's default behavior for those signals is preserved.
Notes: Signals other than SIGINT (e.g., SIGTERM) are not intercepted; we respect
Python's default behavior for them.
"""
alias: str = "shm"
@@ -48,39 +43,32 @@ class SharedMemoryExecutionStrategy(ExecutionStrategy):
join_timeout: float = 15.0,
graceful_delay: float = 5.0,
poll_interval: float = 0.05,
managed_store: bool | None = None,
) -> None:
if main_thread not in ("algorithm", "runner"):
raise ValueError("main_thread must be 'algorithm' or 'runner'")
if main_thread == "runner" and n_runners != 1:
raise ValueError(
"When main_thread is 'runner', n_runners must be 1. "
"Either use 'algorithm' on the main thread or set n_runners to 1."
)
raise ValueError("When main_thread is 'runner', n_runners must be 1")
self.n_runners = n_runners
self.main_thread = main_thread
self.join_timeout = join_timeout
self.graceful_delay = graceful_delay
self.poll_interval = poll_interval
self.managed_store = resolve_managed_store_flag(managed_store)
async def _run_until_completed_or_canceled(self, coro: Awaitable[Any], stop_evt: ExecutionEvent) -> Any:
"""Run `coro` until it finishes or a cooperative stop is requested.
Control flow:
1. Start the bundle coroutine as `task`.
2. Launch a watcher that polls `stop_evt` without blocking the loop.
3. When the stop event flips:
a. Give the bundle `graceful_delay` seconds to finish on its own,
because well-behaved bundles will check the event and return.
b. Cancel the bundle task if it is still running after the grace
period.
4. Await both tasks and swallow `CancelledError` where appropriate.
1) Start the bundle coroutine as `task`.
2) Start a watcher task that waits for `stop_evt` *without blocking* the loop
by periodically polling the threading event.
3) When the stop event flips:
a) Give the bundle *graceful_delay* seconds to finish on its own,
because well-behaved bundles will check the event and return.
b) If still running after the grace period, cancel the bundle task.
4) Ensure both tasks are awaited; swallow `CancelledError` where appropriate.
This is a *backup* mechanism for bundles that might not poll the event
frequently; cooperative shutdown (checking `stop_evt` inside the
bundle) remains the preferred approach.
frequently; cooperative shutdown (checking `stop_evt` yourself) is still preferred.
"""
task: asyncio.Task[Any] = asyncio.create_task(coro) # type: ignore
task_exception: Optional[BaseException] = None
@@ -203,10 +191,7 @@ class SharedMemoryExecutionStrategy(ExecutionStrategy):
# Create stop event and thread-safe store.
stop_evt = ThreadingEvent()
if self.managed_store:
thread_safe_store = LightningStoreThreaded(store)
else:
thread_safe_store = store
thread_safe_store = LightningStoreThreaded(store)
thread_exceptions: SimpleQueue[BaseException] = SimpleQueue()
raised_from_thread: Optional[BaseException] = None
+132 -174
View File
@@ -2,76 +2,28 @@
from __future__ import annotations
import json
import logging
from typing import Any, Callable, no_type_check
import multiprocessing
import signal
import socket
import time
from typing import Any, Callable
import requests
from agentops.client.api import V3Client, V4Client
from agentops.client.api.types import AuthTokenResponse
from agentops.sdk.exporters import AuthenticatedOTLPExporter
from opentelemetry.exporter.otlp.proto.http.metric_exporter import OTLPMetricExporter
from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter
from opentelemetry.sdk.metrics.export import MetricExportResult
from opentelemetry.sdk.trace.export import SpanExportResult
import flask
import setproctitle
logger = logging.getLogger(__name__)
__all__ = [
"instrument_agentops",
"uninstrument_agentops",
"agentops_local_server",
"AgentOpsServerManager",
]
# Module-level storage for originals
_original_handle_chat_attributes: Callable[..., Any] | None = None
_original_handle_response: Callable[..., Any] | None = None
_agentops_service_enabled = False
def enable_agentops_service(enabled: bool = True) -> None:
"""
Enable or disable communication with the AgentOps service.
False (default): AgentOps exporters and clients will run in local mode
and will not attempt to communicate with the remote AgentOps service.
True: all exporters and clients will operate in normal mode and send data
to the AgentOps service as expected.
"""
global _agentops_service_enabled
_agentops_service_enabled = enabled
logger.info(f"Switch set to {enabled} for exporters and clients.")
def _patch_exporters():
import agentops.client.api
import agentops.sdk.core
import opentelemetry.exporter.otlp.proto.http.metric_exporter
import opentelemetry.exporter.otlp.proto.http.trace_exporter
agentops.sdk.core.AuthenticatedOTLPExporter = BypassableAuthenticatedOTLPExporter # type: ignore
opentelemetry.exporter.otlp.proto.http.metric_exporter.OTLPMetricExporter = BypassableOTLPMetricExporter
opentelemetry.exporter.otlp.proto.http.trace_exporter.OTLPSpanExporter = BypassableOTLPSpanExporter
agentops.client.api.V3Client = BypassableV3Client
agentops.client.api.V4Client = BypassableV4Client
def _unpatch_exporters():
import agentops.client.api
import agentops.sdk.core
import opentelemetry.exporter.otlp.proto.http.metric_exporter
import opentelemetry.exporter.otlp.proto.http.trace_exporter
agentops.sdk.core.AuthenticatedOTLPExporter = AuthenticatedOTLPExporter # type: ignore
opentelemetry.exporter.otlp.proto.http.metric_exporter.OTLPMetricExporter = OTLPMetricExporter
opentelemetry.exporter.otlp.proto.http.trace_exporter.OTLPSpanExporter = OTLPSpanExporter
agentops.client.api.V3Client = V3Client
agentops.client.api.V4Client = V4Client
def _unwrap_legacy_response(response: Any) -> Any:
if hasattr(response, "parse") and callable(response.parse):
return response.parse()
return response
def _patch_new_agentops():
@@ -87,58 +39,41 @@ def _patch_new_agentops():
_original_handle_chat_attributes = handle_chat_attributes # type: ignore
@no_type_check
def _handle_chat_attributes_with_tokens(args=None, kwargs=None, return_value=None, **kws): # type: ignore
attributes = _original_handle_chat_attributes(args=args, kwargs=kwargs, return_value=return_value, **kws)
# In some cases, response is a openai._legacy_response.LegacyAPIResponse (e.g., LiteLLM, or LangChain),
# This is created by client.with_raw_response.create()
return_value = _unwrap_legacy_response(return_value)
if (
return_value is not None
and hasattr(return_value, "prompt_token_ids")
and return_value.prompt_token_ids is not None
):
attributes["prompt_token_ids"] = list(return_value.prompt_token_ids)
if (
return_value is not None
and hasattr(return_value, "response_token_ids")
and return_value.response_token_ids is not None
):
attributes["response_token_ids"] = list(return_value.response_token_ids[0])
attributes = _original_handle_chat_attributes(args=args, kwargs=kwargs, return_value=return_value, **kws) # type: ignore
if return_value is not None and hasattr(return_value, "prompt_token_ids"): # type: ignore
attributes["prompt_token_ids"] = list(return_value.prompt_token_ids) # type: ignore
if return_value is not None and hasattr(return_value, "response_token_ids"): # type: ignore
attributes["response_token_ids"] = list(return_value.response_token_ids[0]) # type: ignore
# For LiteLLM Proxy (v0.2) with vLLM return_token_ids, response_token_ids now lives in choices
if (
return_value is not None
and hasattr(return_value, "choices")
and return_value.choices
and isinstance(return_value.choices, list)
and len(return_value.choices) > 0
not attributes.get("response_token_ids")
and return_value is not None
and hasattr(return_value, "choices") # type: ignore
and return_value.choices # type: ignore
and isinstance(return_value.choices, list) # type: ignore
):
first_choice = return_value.choices[0]
# Token IDs from "choices[0].token_ids"
if "response_token_ids" not in attributes:
if hasattr(first_choice, "token_ids") and first_choice.token_ids is not None:
attributes["response_token_ids"] = list(first_choice.token_ids)
# newer versions of OpenAI client SDK
elif (
hasattr(first_choice, "provider_specific_fields")
and first_choice.provider_specific_fields.get("token_ids") is not None
):
attributes["response_token_ids"] = list(first_choice.provider_specific_fields["token_ids"])
first_choice = return_value.choices[0] # type: ignore
if hasattr(first_choice, "token_ids"): # type: ignore
attributes["response_token_ids"] = list(first_choice.token_ids) # type: ignore
# newer versions of OpenAI client SDK
elif hasattr(first_choice, "provider_specific_fields") and "token_ids" in first_choice.provider_specific_fields: # type: ignore
attributes["response_token_ids"] = list(first_choice.provider_specific_fields["token_ids"]) # type: ignore
# log probability
# This is temporary. We need a unified convention for classifying and naming logprobs.
if hasattr(first_choice, "logprobs") and first_choice.logprobs is not None:
if hasattr(first_choice.logprobs, "content") and first_choice.logprobs.content is not None:
attributes["logprobs.content"] = json.dumps(
[logprob.model_dump() for logprob in first_choice.logprobs.content]
)
if hasattr(first_choice.logprobs, "refusal") and first_choice.logprobs.refusal is not None:
attributes["logprobs.refusal"] = json.dumps(
[logprob.model_dump() for logprob in first_choice.logprobs.refusal]
)
# For LiteLLM, response is a openai._legacy_response.LegacyAPIResponse
if (
return_value is not None
and hasattr(return_value, "http_response") # type: ignore
and return_value.http_response is not None # type: ignore
and hasattr(return_value.http_response, "json") # type: ignore
):
json_data = return_value.http_response.json() # type: ignore
if isinstance(json_data, dict):
if "prompt_token_ids" in json_data:
attributes["prompt_token_ids"] = list(json_data["prompt_token_ids"]) # type: ignore
if "response_token_ids" in json_data:
attributes["response_token_ids"] = list(json_data["response_token_ids"][0]) # type: ignore
return attributes
@@ -210,8 +145,6 @@ def instrument_agentops():
Instrument agentops to capture token IDs.
Automatically detects and uses the appropriate patching method based on the installed agentops version.
"""
_patch_exporters()
# Try newest version first (tested for 0.4.16)
try:
return _patch_new_agentops()
@@ -231,8 +164,6 @@ def instrument_agentops():
def uninstrument_agentops():
"""Uninstrument agentops to stop capturing token IDs."""
_unpatch_exporters()
try:
_unpatch_new_agentops()
except Exception:
@@ -243,75 +174,102 @@ def uninstrument_agentops():
pass
class BypassableAuthenticatedOTLPExporter(AuthenticatedOTLPExporter):
def agentops_local_server():
"""
AuthenticatedOTLPExporter with switchable service control.
When `_agentops_service_enabled` is False, skip export and return success.
Returns a Flask app that can be used to test agentops integration.
This server provides endpoints for token fetching and a catch-all endpoint.
"""
app = flask.Flask(__name__)
def export(self, *args: Any, **kwargs: Any) -> SpanExportResult:
if _agentops_service_enabled:
return super().export(*args, **kwargs)
@app.route("/v3/auth/token", methods=["POST"])
def fetch_token(): # type: ignore
return {"token": "dummy", "project_id": "dummy"}
@app.route("/", defaults={"path": ""}, methods=["GET", "POST"])
@app.route("/<path:path>", methods=["GET", "POST"])
def catch_all(path: str): # type: ignore
return {"path": path}
return app
def _run_server(**kwargs: Any): # type: ignore
"""
Internal function to run the Flask server.
This is used to avoid issues with multiprocessing and Flask's reloader.
"""
signal.signal(signal.SIGINT, signal.SIG_IGN) # Ignore SIGINT in worker processes
setproctitle.setproctitle(multiprocessing.current_process().name)
app = agentops_local_server()
app.run(**kwargs)
class AgentOpsServerManager:
"""Manages a AgentOps local server to bypass the online service of AgentOps."""
def __init__(self, daemon: bool = True, port: int | None = None):
self.server_process: multiprocessing.Process | None = None
self.server_port = port
self.daemon = daemon
logger.info("AgentOpsServerManager initialized.")
def _find_available_port(self) -> int:
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
s.bind(("", 0))
return s.getsockname()[1]
def start(self):
if self.server_process and self.server_process.is_alive():
logger.warning("AgentOps server process appears to be already running.")
return
if self.server_port is None:
self.server_port = self._find_available_port()
logger.info(f"Starting AgentOps local server on port {self.server_port}...")
self.server_process = multiprocessing.Process(
target=_run_server,
kwargs={"host": "127.0.0.1", "port": self.server_port, "use_reloader": False, "debug": False},
daemon=self.daemon,
name="AgentLightning-AgentOpsServer",
)
self.server_process.start()
logger.info(
f"AgentOps local server process (PID: {self.server_process.pid}) started, targeting port {self.server_port}."
)
time.sleep(0.5) # Brief wait for server to start up
if not self.server_process.is_alive():
logger.error(f"AgentOps local server failed to start or exited prematurely.")
def is_alive(self) -> bool:
if self.server_process and self.server_process.is_alive():
return True
return False
def stop(self):
if self.server_process is not None and self.server_process.is_alive():
logger.info(f"Stopping AgentOps local server (PID: {self.server_process.pid})...")
self.server_process.terminate() # Send SIGTERM
self.server_process.join(timeout=5) # Wait for clean exit
if self.server_process.is_alive():
logger.warning(
f"AgentOps server (PID: {self.server_process.pid}) did not terminate gracefully, killing..."
)
self.server_process.kill() # Force kill
self.server_process.join(timeout=10) # Wait for kill
self.server_process = None
logger.info(f"AgentOps local server stopped.")
else:
logger.debug("SwitchableAuthenticatedOTLPExporter is switched off, skipping export.")
return SpanExportResult.SUCCESS
logger.info("AgentOps local server was not running or already stopped.")
class BypassableOTLPMetricExporter(OTLPMetricExporter):
"""
OTLPMetricExporter with switchable service control.
When `_agentops_service_enabled` is False, skip export and return success.
"""
def export(self, *args: Any, **kwargs: Any) -> MetricExportResult:
if _agentops_service_enabled:
return super().export(*args, **kwargs) # type: ignore[reportUnknownMemberType]
else:
logger.debug("SwitchableOTLPMetricExporter is switched off, skipping export.")
return MetricExportResult.SUCCESS
class BypassableOTLPSpanExporter(OTLPSpanExporter):
"""
OTLPSpanExporter with switchable service control.
When `_agentops_service_enabled` is False, skip export and return success.
"""
def export(self, *args: Any, **kwargs: Any) -> SpanExportResult:
if _agentops_service_enabled:
return super().export(*args, **kwargs)
else:
logger.debug("SwitchableOTLPSpanExporter is switched off, skipping export.")
return SpanExportResult.SUCCESS
class BypassableV3Client(V3Client):
"""
V3Client with toggleable authentication calls.
Returns dummy auth response when `_agentops_service_enabled` is False.
"""
# Temporary synchronous override of fetch_auth_token for mock purposes.
def fetch_auth_token(self, *args: Any, **kwargs: Any) -> AuthTokenResponse: # type: ignore[override]
if _agentops_service_enabled:
return super().fetch_auth_token(*args, **kwargs) # type: ignore[override]
else:
logger.debug("SwitchableV3Client is switched off, skipping fetch_auth_token request.")
return AuthTokenResponse(token="dummy", project_id="dummy")
class BypassableV4Client(V4Client):
"""
V4Client with toggleable post requests.
Returns dummy response when `_agentops_service_enabled` is False.
"""
def post(self, *args: Any, **kwargs: Any) -> requests.Response:
if _agentops_service_enabled:
return super().post(*args, **kwargs)
else:
logger.debug("SwitchableV4Client is switched off, skipping post request.")
response = requests.Response()
response.status_code = 200
response._content = b"{}"
return response
def get_port(self) -> int | None:
# Check liveness again in case it died since start()
if self.is_alive() and self.server_port is not None:
return self.server_port
# If called after server stopped or failed, port might be stale or None
if self.server_port is not None and (self.server_process is None or not self.server_process.is_alive()):
logger.warning(
f"AgentOps server port {self.server_port} is stored, but server process is not alive. Returning stored port."
)
return self.server_port
+1 -2
View File
@@ -4,8 +4,7 @@
It's unclear whether or not this file is useful.
It seems that LiteLLM owns its own telemetry from their own entrance
[Related documentation](https://docs.litellm.ai/docs/observability/agentops_integration).
https://docs.litellm.ai/docs/observability/agentops_integration
"""
from typing import Any, Optional
+91 -106
View File
@@ -1,7 +1,5 @@
# Copyright (c) Microsoft. All rights reserved.
"""Convenience decorators for building lightweight `LitAgent` implementations."""
from __future__ import annotations
import functools
@@ -92,25 +90,24 @@ class FunctionalLitAgentFunc(Protocol[T_contra]):
class FunctionalLitAgent(LitAgent[T]):
"""Adapter that turns plain rollout functions into [`LitAgent`][agentlightning.LitAgent] instances.
"""A specialized LitAgent that wraps a function-based rollout that accepts
dynamically a task input and a configured resource (LLM / prompt template / ...).
The helper inspects the wrapped function to determine which resources to
inject, allowing both synchronous and asynchronous callables to participate
in the training loop without writing a dedicated subclass.
This class allows users to define agent behavior using a simple function
that takes task input and a resource, rather than implementing a full
LitAgent subclass.
"""
def __init__(self, rollout_func: FunctionalLitAgentFunc[T], *, strip_proxy: bool = True) -> None:
"""Initialize the wrapper around a rollout function.
"""
Initialize the FunctionalLitAgent with a functional rollout function.
Args:
rollout_func: Callable that implements the rollout. It may be synchronous
or asynchronous and can optionally receive a
[`Rollout`][agentlightning.Rollout] alongside resources such as
`llm` or `prompt_template`.
strip_proxy: When ``True``, convert
[`ProxyLLM`][agentlightning.ProxyLLM] inputs into
[`LLM`][agentlightning.LLM] instances before calling the
rollout function. Defaults to `True`.
rollout_func: A function that defines the agent's behavior.
Can be sync or async, and can optionally accept a Rollout parameter.
The function signature determines which resources are injected (llm, prompt_template, etc.).
strip_proxy: Whether to strip the ProxyLLM resource into a LLM resource when the function accepts an llm parameter.
Defaults to True.
"""
super().__init__()
self._rollout_func = rollout_func
@@ -141,15 +138,12 @@ class FunctionalLitAgent(LitAgent[T]):
"""Execute a synchronous rollout using the wrapped function.
Args:
task: Task input data.
resources: Mapping of named resources available to the agent.
rollout: Rollout metadata provided by the runtime.
task: The task input data.
resources: Dictionary of named resources including LLMs.
rollout: The rollout object with metadata.
Returns:
Result produced by the wrapped rollout function.
Raises:
RuntimeError: If the wrapped function is asynchronous.
The result from the wrapped rollout function.
"""
if self._is_async:
raise RuntimeError(f"{self._rollout_func} is asynchronous. Use rollout_async instead.")
@@ -161,15 +155,12 @@ class FunctionalLitAgent(LitAgent[T]):
"""Execute an asynchronous rollout using the wrapped function.
Args:
task: Task input data.
resources: Mapping of named resources available to the agent.
rollout: Rollout metadata provided by the runtime.
task: The task input data.
resources: Dictionary of named resources including LLMs.
rollout: The rollout object with metadata.
Returns:
Result produced by the wrapped rollout coroutine.
Raises:
RuntimeError: If the wrapped function is synchronous.
The result from the wrapped rollout function.
"""
if not self._is_async:
raise RuntimeError(f"{self._rollout_func} is synchronous. Use rollout instead.")
@@ -178,19 +169,18 @@ class FunctionalLitAgent(LitAgent[T]):
return await self._rollout_func(task, **kwargs) # type: ignore
def _get_kwargs(self, resources: NamedResources, rollout: Rollout) -> Dict[str, Any]:
"""Prepare keyword arguments expected by the wrapped rollout function.
"""Extract the kwargs needed for the rollout function based on its signature.
It dynamically builds the `kwargs` dictionary by inspecting the function signature and
Dynamically builds the kwargs dictionary by inspecting the function signature and
including only the parameters the function accepts. This allows flexible function
signatures that can request any combination of: rollout, llm, and/or prompt_template.
Args:
resources: Mapping of named resources available for the rollout.
rollout: Rollout metadata provided by the runtime.
resources: Dictionary of named resources available for the rollout.
rollout: The rollout object with metadata.
Returns:
Dictionary of keyword arguments to forward to the rollout function.
A dictionary of kwargs to pass to the rollout function.
"""
kwargs: Dict[str, Any] = {}
@@ -204,19 +194,19 @@ class FunctionalLitAgent(LitAgent[T]):
return kwargs
def _get_llm_resource(self, resources: NamedResources, rollout: Rollout) -> LLM:
"""Retrieve the first LLM resource from the available resources.
"""Extract the first LLM resource from the resources dictionary.
Strip the ProxyLLM resource into a LLM resource if needed.
Args:
resources: Mapping of named resources.
rollout: Rollout metadata used when stripping proxy endpoints.
resources: Dictionary of named resources.
rollout: The rollout object with metadata.
Returns:
First [`LLM`][agentlightning.LLM] resource encountered.
The first LLM resource found.
Raises:
ValueError: If no LLM resource is present.
ValueError: If no LLM resource is found.
"""
resource_found: LLM | None = None
for name, resource in resources.items():
@@ -235,17 +225,17 @@ class FunctionalLitAgent(LitAgent[T]):
return resource_found
def _get_prompt_template_resource(self, resources: NamedResources, rollout: Rollout) -> PromptTemplate:
"""Retrieve the first prompt template resource from the available resources.
"""Extract the first PromptTemplate resource from the resources dictionary.
Args:
resources: Mapping of named resources.
rollout: Rollout metadata (unused).
resources: Dictionary of named resources.
rollout: The rollout object with metadata. Not used in this method.
Returns:
First [`PromptTemplate`][agentlightning.PromptTemplate] resource encountered.
The first PromptTemplate resource found.
Raises:
ValueError: If no prompt template resource is present.
ValueError: If no PromptTemplate resource is found.
"""
resource_found: PromptTemplate | None = None
for name, resource in resources.items():
@@ -263,22 +253,21 @@ class FunctionalLitAgent(LitAgent[T]):
return resource_found
def _strip_proxy_helper(self, proxy_llm: LLM, rollout: Rollout) -> LLM:
"""Convert [`ProxyLLM`][agentlightning.ProxyLLM] instances into concrete LLMs.
"""Strip the ProxyLLM resource into a concrete LLM resource.
It resolves ProxyLLM instances to their concrete LLM implementation
This method resolves ProxyLLM instances to their concrete LLM implementation
by attaching the attempted rollout context. This is only used when the function
signature accepts an `llm` parameter and strip_proxy is True.
signature accepts an 'llm' parameter and strip_proxy is True.
Args:
proxy_llm: Candidate LLM resource.
rollout: Rollout metadata that provides rollout and attempt identifiers.
proxy_llm: The LLM resource, which may be a ProxyLLM.
rollout: The rollout object with metadata.
Returns:
[`LLM`][agentlightning.LLM] with rollout context baked into the endpoint.
The concrete LLM resource.
Raises:
ValueError: If the rollout is not an
[`AttemptedRollout`][agentlightning.AttemptedRollout].
ValueError: If the rollout is not an AttemptedRollout (required for stripping ProxyLLM).
"""
if not isinstance(proxy_llm, ProxyLLM):
@@ -304,37 +293,41 @@ def llm_rollout(*, strip_proxy: bool = True) -> Callable[[LlmRolloutFunc[T]], Fu
def llm_rollout(
func: LlmRolloutFunc[T] | None = None, *, strip_proxy: bool = True
) -> FunctionalLitAgent[T] | Callable[[LlmRolloutFunc[T]], FunctionalLitAgent[T]]:
"""Create a [`FunctionalLitAgent`][agentlightning.litagent.decorator.FunctionalLitAgent] for LLM-based rollouts.
"""Create a FunctionalLitAgent from a function that takes (task, llm[, rollout]).
This decorator allows you to define an agent using a simple function
instead of creating a full LitAgent subclass. The returned FunctionalLitAgent
instance is callable, preserving the original function's behavior.
Args:
func: Callable defining the agent's behaviour. Supported signatures include:
* `(task, llm) -> result`
* `(task, llm, rollout) -> result`
* `async (task, llm) -> result`
* `async (task, llm, rollout) -> result`
strip_proxy: When `True`, convert proxy resources into concrete
[`LLM`][agentlightning.LLM] instances before calling the
function. Defaults to `True`.
func: A function that defines the agent's behavior. Can be:
- sync: (task, llm) -> result
- sync with rollout: (task, llm, rollout) -> result
- async: async (task, llm) -> result
- async with rollout: async (task, llm, rollout) -> result
strip_proxy: Whether to strip the ProxyLLM resource into a LLM resource.
Defaults to True.
Returns:
[`FunctionalLitAgent`][agentlightning.litagent.decorator.FunctionalLitAgent] that
wraps the supplied function.
A callable FunctionalLitAgent instance that preserves the original function's
type hints and behavior while providing all agent functionality.
Examples:
```python
Example:
@llm_rollout
def my_agent(task, llm):
return llm.endpoint
# Agent logic here
return response
@llm_rollout(strip_proxy=False)
def my_agent_no_strip(task, llm):
return llm.model
# Agent logic here
return response
# Function is still callable with original behavior
result = my_agent(task, llm)
# Agent methods are also available
result = my_agent.rollout(task, resources, rollout)
```
"""
def decorator(f: LlmRolloutFunc[T]) -> FunctionalLitAgent[T]:
@@ -350,20 +343,19 @@ def llm_rollout(
def _validate_llm_rollout_func(func: Any) -> TypeGuard[LlmRolloutFunc[Any]]:
"""Validate the function signature of an LLM rollout function.
"""Validate the function signature of a LLM rollout function.
Ensures the function follows the expected pattern for LLM-based rollouts:
- Must have at least 2 parameters
- First parameter must be named 'task'
- Must have a parameter named 'llm'
- Optionally can have a 'rollout' parameter
Args:
func: Function to inspect.
func: The function to validate.
Returns:
`True` when the signature matches the supported patterns.
True if the function signature is valid.
Raises:
ValueError: If the function signature does not match the expected pattern.
@@ -391,34 +383,36 @@ def prompt_rollout() -> Callable[[PromptRolloutFunc[T]], FunctionalLitAgent[T]]:
def prompt_rollout(
func: PromptRolloutFunc[T] | None = None,
) -> FunctionalLitAgent[T] | Callable[[PromptRolloutFunc[T]], FunctionalLitAgent[T]]:
"""Create a [`FunctionalLitAgent`][agentlightning.litagent.decorator.FunctionalLitAgent] for prompt-based rollouts.
"""Create a FunctionalLitAgent from a function that takes (task, prompt_template[, rollout]).
This decorator is designed for agents that work with tunable prompt templates. It enables
a workflow where algorithms manage and optimize the prompt template, while agents consume
the template to perform rollouts. This is particularly useful for prompt optimization scenarios.
Args:
func: Callable defining the agent's behavior. Supported signatures include:
* `(task, prompt_template) -> result`
* `(task, prompt_template, rollout) -> result`
* `async (task, prompt_template) -> result`
* `async (task, prompt_template, rollout) -> result`
func: A function that defines the agent's behavior. Can be:
- sync: (task, prompt_template) -> result
- sync with rollout: (task, prompt_template, rollout) -> result
- async: async (task, prompt_template) -> result
- async with rollout: async (task, prompt_template, rollout) -> result
Returns:
[`FunctionalLitAgent`][agentlightning.litagent.decorator.FunctionalLitAgent] that
wraps the supplied function.
A callable FunctionalLitAgent instance that preserves the original function's
type hints and behavior while providing all agent functionality.
Examples:
```python
Example:
@prompt_rollout
def my_agent(task, prompt_template):
# Use the prompt template to generate a response
messages = prompt_template.format(task=task.input)
return messages
# ... perform rollout with the formatted prompt
return response
# Function is still callable with original behavior
result = my_agent(task, prompt_template)
# Agent methods are also available
result = my_agent.rollout(task, resources, rollout)
```
"""
def decorator(f: PromptRolloutFunc[T]) -> FunctionalLitAgent[T]:
@@ -435,17 +429,16 @@ def _validate_prompt_rollout_func(func: Any) -> TypeGuard[PromptRolloutFunc[Any]
"""Validate the function signature of a prompt rollout function.
Ensures the function follows the expected pattern for prompt-template-based rollouts:
- Must have at least 2 parameters
- First parameter must be named 'task'
- Must have a parameter named 'prompt_template'
- Optionally can have a 'rollout' parameter
Args:
func: Function to inspect.
func: The function to validate.
Returns:
`True` when the signature matches the supported patterns.
True if the function signature is valid.
Raises:
ValueError: If the function signature does not match the expected pattern.
@@ -463,30 +456,23 @@ def _validate_prompt_rollout_func(func: Any) -> TypeGuard[PromptRolloutFunc[Any]
def rollout(func: Union[LlmRolloutFunc[T], PromptRolloutFunc[T], Callable[..., Any]]) -> FunctionalLitAgent[T]:
"""Create a [`FunctionalLitAgent`][agentlightning.litagent.decorator.FunctionalLitAgent] from an arbitrary rollout function.
"""Create a LitAgent from a function, automatically detecting the appropriate type.
This function inspects the provided callable and creates the appropriate
agent type based on its signature. It supports both LLM-based and prompt-template-based
agents. The returned agent instance is callable, preserving the original function's
behavior and type hints.
See [`llm_rollout`][agentlightning.litagent.decorator.llm_rollout] and
[`prompt_rollout`][agentlightning.litagent.decorator.prompt_rollout] for more details.
Args:
func: Callable that implements the rollout. Supported signatures:
- `[async ](task, llm[, rollout])` for LLM-based agents
- `[async ](task, prompt_template[, rollout])` for prompt-template-based agents
The supported output types of `func` is same as the return type of [`rollout`][agentlightning.LitAgent.rollout].
func: A function that defines the agent's behavior. Supported signatures:
- (task, llm[, rollout]) for LLM-based agents
- (task, prompt_template[, rollout]) for prompt-template-based agents
Returns:
[`FunctionalLitAgent`][agentlightning.litagent.decorator.FunctionalLitAgent] that
wraps the supplied function.
A callable FunctionalLitAgent instance that preserves the original function's
type hints and behavior while providing all agent functionality.
Examples:
```python
Example:
# LLM-based agent
@rollout
def my_llm_agent(task, llm):
@@ -509,7 +495,6 @@ def rollout(func: Union[LlmRolloutFunc[T], PromptRolloutFunc[T], Callable[..., A
# Agent methods are also available
result = my_llm_agent.rollout(task, resources, rollout)
```
Raises:
NotImplementedError: If the function signature doesn't match any known patterns.
+154 -94
View File
@@ -1,7 +1,5 @@
# Copyright (c) Microsoft. All rights reserved.
"""Base abstractions for building agents that plug into Agent Lightning."""
from __future__ import annotations
import inspect
@@ -13,8 +11,8 @@ from typing import TYPE_CHECKING, Any, Callable, Generic, Optional, TypeVar
from agentlightning.types import NamedResources, Rollout, RolloutRawResult, Task
if TYPE_CHECKING:
from agentlightning.runner import Runner
from agentlightning.tracer import Tracer
from agentlightning.runner import BaseRunner
from agentlightning.tracer import BaseTracer
from agentlightning.trainer import Trainer
@@ -28,38 +26,32 @@ __all__ = [
def is_v0_1_rollout_api(func: Callable[..., Any]) -> bool:
"""Return `True` when the rollout function uses the deprecated v0.1 signature.
The helper inspects the callable's signature to detect whether a `rollout_id`
parameter is present, which indicates the legacy API.
"""Check if the rollout API is v0.1.
Inspect the function signature to see if it has a rollout_id parameter.
Args:
func: Function to analyze.
Returns:
`True` if the callable exposes a `rollout_id` parameter.
func: The function to check.
"""
return "rollout_id" in inspect.signature(func).parameters
class LitAgent(Generic[T]):
"""Base class for implementing agent rollouts.
"""Base class for the training and validation logic of an agent.
Subclasses override the rollout methods to process tasks while the trainer and
runner infrastructure manages orchestration, tracing, and persistence.
Developers should subclass this class and implement the rollout methods
to define the agent's behavior for a single task. The agent's logic
is completely decoupled from the server communication and training
infrastructure.
"""
def __init__(self, *, trained_agents: Optional[str] = None) -> None: # FIXME: str | None won't work for cli
"""Initialize the agent instance.
"""
Initialize the LitAgent.
Args:
trained_agents: Optional identifier used by legacy tooling to mark trained
agents.
!!! warning "Deprecated"
The `trained_agents` flag is deprecated. Configure `agent_match` in the adapter
layer instead. See [`TracerTraceToTriplet`][agentlightning.TracerTraceToTriplet]
for more details.
trained_agents: Optional string representing the trained agents.
This can be used to track which agents have been trained by this instance.
Deprecated. Configure `agent_match` in adapter instead.
"""
if trained_agents is not None:
warnings.warn(
@@ -70,12 +62,15 @@ class LitAgent(Generic[T]):
self.trained_agents = trained_agents
self._trainer_ref: weakref.ReferenceType[Trainer] | None = None
self._runner_ref: weakref.ReferenceType[Runner[T]] | None = None
self._runner_ref: weakref.ReferenceType[BaseRunner[T]] | None = None
def is_async(self) -> bool:
"""Return `True` when the agent overrides any asynchronous rollout methods.
"""
Check if the agent implements asynchronous rollout methods.
Override this property for customized async detection logic.
Override this method for customized async detection logic.
Returns:
True if the agent has custom async rollout methods, False otherwise.
"""
return (
(
@@ -90,15 +85,21 @@ class LitAgent(Generic[T]):
)
def set_trainer(self, trainer: Trainer) -> None:
"""Attach the trainer responsible for orchestration.
"""
Set the trainer for this agent.
Args:
trainer: [`Trainer`][agentlightning.Trainer] that manages the agent.
trainer: The Trainer instance that will handle training and validation.
"""
self._trainer_ref = weakref.ref(trainer)
def get_trainer(self) -> Trainer:
"""Return the trainer associated with this agent."""
"""
Get the trainer for this agent.
Returns:
The Trainer instance associated with this agent.
"""
if self._trainer_ref is None:
raise ValueError("Trainer has not been set for this agent.")
trainer = self._trainer_ref()
@@ -108,31 +109,42 @@ class LitAgent(Generic[T]):
@property
def trainer(self) -> Trainer:
"""Return the trainer associated with this agent."""
"""Convenient shortcut of self.get_trainer()."""
return self.get_trainer()
def get_tracer(self) -> Tracer:
"""Return the tracer configured for this agent."""
def get_tracer(self) -> BaseTracer:
"""
Get the tracer for this agent.
Returns:
The BaseTracer instance associated with this agent.
"""
if hasattr(self.runner, "tracer"):
return self.runner.tracer # type: ignore
else:
return self.trainer.tracer
@property
def tracer(self) -> Tracer:
"""Return the tracer configured for this agent."""
def tracer(self) -> BaseTracer:
"""Convenient shortcut of self.get_tracer()."""
return self.get_tracer()
def set_runner(self, runner: Runner[T]) -> None:
"""Attach the runner responsible for executing rollouts.
def set_runner(self, runner: BaseRunner[T]) -> None:
"""
Set the runner for this agent.
Args:
runner: [`Runner`][agentlightning.Runner] coordinating execution.
runner: The runner instance that will handle the execution of rollouts.
"""
self._runner_ref = weakref.ref(runner)
def get_runner(self) -> Runner[T]:
"""Return the runner responsible for executing rollouts."""
def get_runner(self) -> BaseRunner[T]:
"""
Get the runner for this agent.
Returns:
The runner instance associated with this agent.
"""
if self._runner_ref is None:
raise ValueError("Runner has not been set for this agent.")
runner = self._runner_ref()
@@ -141,111 +153,159 @@ class LitAgent(Generic[T]):
return runner
@property
def runner(self) -> Runner[T]:
"""Return the runner responsible for executing rollouts."""
def runner(self) -> BaseRunner[T]:
"""Convenient shortcut of self.get_runner()."""
return self.get_runner()
def on_rollout_start(self, task: Task, runner: Runner[T], tracer: Tracer) -> None:
"""Hook invoked immediately before a rollout begins.
def on_rollout_start(self, task: Task, runner: BaseRunner[T], tracer: BaseTracer) -> None:
"""Hook called immediately before a rollout begins.
Subclasses can override this method to implement custom logic such as logging,
metric collection, or resource setup. The default implementation is a no-op.
Deprecated in favor of `on_rollout_start` in the `Hook` interface.
Args:
task: [`Task`][agentlightning.Task] that will be processed.
runner: [`Runner`][agentlightning.Runner] managing the rollout.
tracer: [`Tracer`][agentlightning.Tracer] associated with the runner.
task: The :class:`Task` object that will be processed.
runner: The :class:`BaseRunner` managing the rollout.
tracer: The tracer instance associated with the runner.
!!! warning "Deprecated"
Override [`Hook.on_rollout_start`][agentlightning.Hook.on_rollout_start]
instead of this method when extending agents.
Subclasses can override this method to implement custom logic such as
logging, metric collection, or resource setup. By default, this is a
no-op.
"""
def on_rollout_end(self, task: Task, rollout: Rollout, runner: Runner[T], tracer: Tracer) -> None:
"""Hook invoked after a rollout completes.
def on_rollout_end(self, task: Task, rollout: Rollout, runner: BaseRunner[T], tracer: BaseTracer) -> None:
"""Hook called after a rollout completes.
Subclasses can override this method for cleanup or additional logging. The default
implementation is a no-op.
Deprecated in favor of `on_rollout_end` in the `Hook` interface.
Args:
task: [`Task`][agentlightning.Task] that was processed.
rollout: Resulting [`Rollout`][agentlightning.Rollout].
runner: [`Runner`][agentlightning.Runner] managing the rollout.
tracer: [`Tracer`][agentlightning.Tracer] associated with the runner.
task: The :class:`Task` object that was processed.
rollout: The resulting :class:`Rollout` object.
runner: The :class:`BaseRunner` managing the rollout.
tracer: The tracer instance associated with the runner.
!!! warning "Deprecated"
Override [`Hook.on_rollout_end`][agentlightning.Hook.on_rollout_end]
instead of this method when extending agents.
Subclasses can override this method for cleanup or additional
logging. By default, this is a no-op.
"""
def rollout(self, task: T, resources: NamedResources, rollout: Rollout) -> RolloutRawResult:
"""Execute a rollout synchronously.
"""Main entry point for executing a rollout.
This method determines whether to call the synchronous or
asynchronous rollout method based on the agent's implementation.
If you don't wish to implement both training rollout and validation
rollout separately, you can just implement `rollout` which will work for both.
Args:
task: Task payload provided by the scheduler.
resources: Mapping of named resources (for example LLMs or prompt templates).
rollout: Rollout metadata. Avoid mutating this object directly unless a
subclass needs to override defaults.
task: The task object received from the server, containing the
input data and metadata.
resources: A dictionary of named resources (e.g., LLMs, prompt
templates) for the agent to use.
rollout: The full rollout object, please avoid from directly modifying it.
Most agents should only use `task` and `resources`. Use `rollout`
only if you need to access metadata like `rollout_id`.
Returns:
One of the following values:
* `None` when tracing is handled by the runner.
* `float` representing the final reward.
* `List[ReadableSpan]` with OpenTelemetry spans.
* `List[Span]` with Agent Lightning spans.
The result of the rollout, which can be one of:
- None. The tracing should be handled by the agent runner.
- A float representing the final reward.
- A list of `Triplet` objects for detailed, step-by-step feedback.
- A list of `ReadableSpan` objects for OpenTelemetry tracing.
- A list of dictionaries for any trace spans.
- A complete `Rollout` object for full control over reporting.
"""
raise NotImplementedError("Agents must implement the `rollout` method.")
async def rollout_async(self, task: T, resources: NamedResources, rollout: Rollout) -> RolloutRawResult:
"""Execute a rollout asynchronously.
"""Asynchronous version of the main rollout method.
This method determines whether to call the synchronous or
asynchronous rollout method based on the agent's implementation.
Args:
task: Task payload provided by the scheduler.
resources: Mapping of named resources (for example LLMs or prompt templates).
rollout: Rollout metadata. Avoid mutating this object directly unless a
subclass needs to override defaults.
task: The task object received from the server, containing the
input data and metadata.
resources: A dictionary of named resources (e.g., LLMs, prompt
templates) for the agent to use.
rollout: The full rollout object, please avoid from directly modifying it.
Most agents should only use `task` and `resources`. Use `rollout`
only if you need to access metadata like `rollout_id`.
Returns:
Same possible return values as
[`rollout`][agentlightning.LitAgent.rollout].
The result of the rollout, which can be one of:
- None. The tracing should be handled by the agent runner.
- A float representing the final reward.
- A list of `Triplet` objects for detailed, step-by-step feedback.
- A list of `ReadableSpan` objects for OpenTelemetry tracing.
- A list of dictionaries for any trace spans.
- A complete `Rollout` object for full control over reporting.
"""
raise NotImplementedError("Agents must implement the `rollout_async` method for async operations.")
def training_rollout(self, task: T, resources: NamedResources, rollout: Rollout) -> RolloutRawResult:
"""Process a single training task synchronously.
"""Defines the agent's behavior for a single training task.
By default, this method delegates to
[`rollout`][agentlightning.LitAgent.rollout].
This method should contain the logic for how the agent processes an
input, uses the provided resources (like LLMs or prompts), and
produces a result.
Args:
task: The task object received from the server, containing the
input data and metadata.
resources: A dictionary of named resources (e.g., LLMs, prompt
templates) for the agent to use.
rollout: The full rollout object, please avoid from directly modifying it.
"""
return self.rollout(task, resources, rollout)
def validation_rollout(self, task: T, resources: NamedResources, rollout: Rollout) -> RolloutRawResult:
"""Process a single validation task synchronously.
"""Defines the agent's behavior for a single validation task.
Override this method when validation should differ from training. The default
implementation delegates to
[`training_rollout`][agentlightning.LitAgent.training_rollout].
By default, this method redirects to `training_rollout`. Override it
if the agent should behave differently during validation.
Args:
task: The task object received from the server, containing the
input data and metadata.
resources: A dictionary of named resources for the agent to use.
rollout: The full rollout object, avoid from modifying it.
Returns:
The result of the validation rollout. See `rollout` for
possible return types.
"""
return self.rollout(task, resources, rollout)
async def training_rollout_async(self, task: T, resources: NamedResources, rollout: Rollout) -> RolloutRawResult:
"""Process a single training task asynchronously.
"""Asynchronous version of `training_rollout`.
By default, this method delegates to
[`rollout_async`][agentlightning.LitAgent.rollout_async].
This method should be implemented by agents that perform asynchronous
operations (e.g., non-blocking I/O, concurrent API calls).
Args:
task: The task object received from the server.
resources: A dictionary of named resources for the agent to use.
rollout: The full rollout object, avoid from modifying it.
Returns:
The result of the asynchronous training rollout. See `rollout` for
possible return types.
"""
return await self.rollout_async(task, resources, rollout)
async def validation_rollout_async(self, task: T, resources: NamedResources, rollout: Rollout) -> RolloutRawResult:
"""Process a single validation task asynchronously.
"""Asynchronous version of `validation_rollout`.
Override this method when validation should differ from training. The default
implementation delegates to
[`training_rollout_async`][agentlightning.LitAgent.training_rollout_async].
By default, this method redirects to `training_rollout_async`.
Override it for different asynchronous validation behavior.
Args:
task: The task object received from the server.
resources: A dictionary of named resources for the agent to use.
rollout: The full rollout object, avoid from modifying it.
Returns:
The result of the asynchronous validation rollout. See `rollout` for
possible return types.
"""
return await self.rollout_async(task, resources, rollout)
File diff suppressed because it is too large Load Diff
+12 -362
View File
@@ -1,370 +1,20 @@
# Copyright (c) Microsoft. All rights reserved.
from __future__ import annotations
import logging
import os
import platform
import sys
import warnings
from logging.config import dictConfig
from typing import Any, Dict, Optional
from rich.console import Console
__all__ = ["setup", "configure_logger", "setup_module"]
__all__ = ["configure_logger"]
def configure_logger(level: int = logging.INFO, name: str = "agentlightning") -> logging.Logger:
"""Create or reset a namespaced logger with a consistent console format.
logger = logging.getLogger(name)
logger.handlers.clear() # clear existing handlers
This helper clears any previously attached handlers before binding a single
`StreamHandler` that writes to standard output. The resulting logger does
not propagate to the root logger, preventing duplicate log emission when
applications compose multiple logging configurations.
!!! danger
This function is deprecated in favor of [`setup_logging`][agentlightning.setup_logging].
Args:
level: Logging level applied both to the logger and the installed
handler. Defaults to `logging.INFO`.
name: Dotted path for the logger instance. Defaults to
`"agentlightning"`.
Returns:
Configured logger instance ready for immediate use.
Examples:
```python
from agentlightning import configure_logger
logger = configure_logger(level=logging.INFO)
logger.info("agent-lightning is ready!")
```
"""
warnings.warn("This function is deprecated in favor of `setup_logging`.", DeprecationWarning, stacklevel=2)
return setup_module(level=level, name=name, console=True, color=True, propagate=False)
DEFAULT_FORMAT = "%(asctime)s [%(levelname)s] (Process-%(process)d %(name)s) %(message)s"
DATE_FORMAT = "%H:%M:%S"
def _to_level_value(lvl: int | str) -> int:
if isinstance(lvl, int):
return lvl
val = getattr(logging, str(lvl).upper(), None)
if val is None:
raise ValueError(f"Invalid log level: {lvl}")
return val
def _ensure_file_handler(
logger: logging.Logger,
filename: str,
*,
level: int,
formatter: Optional[logging.Formatter],
) -> None:
"""Attach a FileHandler to `logger` for `filename` if it doesn't already exist."""
abspath = os.path.abspath(filename)
# Avoid duplicates
for h in logger.handlers:
if isinstance(h, logging.FileHandler) and getattr(h, "baseFilename", None) == abspath:
return
# Ensure directory exists
dirname = os.path.dirname(abspath)
if dirname:
os.makedirs(dirname, exist_ok=True)
fh = logging.FileHandler(abspath, encoding="utf-8")
fh.setLevel(level)
if formatter is not None:
fh.setFormatter(formatter)
else:
fh.setFormatter(logging.Formatter(DEFAULT_FORMAT, DATE_FORMAT))
logger.addHandler(fh)
def setup(
level: int | str = "INFO",
*,
console: bool = True,
color: bool | Dict[str, Any] = True,
propagate: bool = False,
disable_existing_loggers: bool = False,
capture_warnings: bool = False,
submodule_levels: Optional[dict[str, int | str]] = None,
extra_handlers: Optional[list[logging.Handler]] = None,
formatter: Optional[logging.Formatter] = None,
apply_to: Optional[list[str]] = None,
files: Optional[str | dict[str, str]] = None,
) -> None:
"""Configures logging for the `agentlightning` logger hierarchy.
This function provides a one-stop setup utility for configuring the
`agentlightning` root logger and optionally its submodules or external
loggers. It supports console logging, colored rich output, per-submodule
log levels, and optional handler/formatter injection.
The setup is intentionally isolated: it does not modify the global root
logger or loggers belonging to other libraries unless explicitly directed
via `apply_to`.
Args:
level:
Logging level for the base `agentlightning` logger. Accepts either
an integer (e.g., `logging.DEBUG`) or a string level name
(e.g., `"INFO"`). Defaults to `"INFO"`.
console:
Whether to attach a console handler to the logger. Defaults to
`True`.
color:
Enables rich-formatted output using `RichHandler` when `True`
or a configuration dict. If `False`, a plain text formatter is
used instead. Defaults to `True`.
propagate:
Whether `agentlightning` logs should propagate to ancestor
loggers. Defaults to `False`.
disable_existing_loggers:
Passed to `logging.config.dictConfig`. If `True`, disables all
existing configured loggers before applying this configuration.
Defaults to `False`.
capture_warnings:
If `True`, redirects Python `warnings` emitted via the `warnings`
module into the logging system. Defaults to `False`.
submodule_levels:
Mapping of submodule logger names to logging levels. If a specified
submodule level is more verbose than the base level, a warning is emitted.
extra_handlers:
A list of user-provided handlers to attach to the `agentlightning` logger.
Handlers are added idempotently; duplicates are not reattached.
formatter:
A formatter to apply to any handler under `agentlightning` that does not
already have one assigned. Useful for customizing output without overwriting
formatters on custom handlers.
apply_to:
A list of additional logger names to configure identically to
`agentlightning` base logger. Their handlers are replaced with copies of the base
handlers, and propagation is disabled to avoid duplicate log emission.
files:
If a string, attach a FileHandler to the base `agentlightning` logger.
If a dict, for each `(logger_name, filename)` pair, attach a FileHandler
directly to that logger.
Each file handler should use the logger's effective level at creation.
Notes:
* On Windows, this function forces UTF-8 mode in the console to prevent
issues with rich output or special characters.
* Submodule loggers can generate records below the handler's emission
threshold. Whether such records appear depends on both the logger's
level and the handler's level.
* `apply_to` loggers inherit the same handlers but do not propagate
upward, yielding isolated, consistent behavior.
Examples:
Basic setup:
>>> setup()
Enabling debug mode with no color:
>>> setup(level="DEBUG", color=False)
Overriding specific submodule levels:
>>> setup(submodule_levels={"agentlightning.io": "DEBUG"})
Attaching an additional file handler:
>>> fh = logging.FileHandler("app.log")
>>> setup(extra_handlers=[fh])
"""
# Ensure UTF-8 encoding on Windows consoles
# Note: This change does not fully represent support for execution under the windows system.
# It only fixes console printing issues caused by special characters.
# TODO: More comprehensive Windows support may be needed in the future.
if platform.system() == "Windows":
os.environ["PYTHONUTF8"] = "1"
base_logger = setup_module(
level,
name="agentlightning",
console=console,
color=color,
propagate=propagate,
disable_existing_loggers=disable_existing_loggers,
)
base_level_value = base_logger.level
# Apply user-provided formatter (only to handlers without one,
# so we don't clobber custom extra_handlers)
if formatter is not None:
for h in base_logger.handlers:
if h.formatter is None:
h.setFormatter(formatter)
# Attach user-provided handler(s) if any, idempotently
if extra_handlers:
for h in extra_handlers:
if h not in base_logger.handlers:
base_logger.addHandler(h)
# Per-submodule levels
if submodule_levels:
for name, lvl in submodule_levels.items():
sub_level = _to_level_value(lvl)
# Emit a warning if submodule level is lower (more verbose) than the global/base level
if sub_level < base_level_value:
base_logger.warning(
"Submodule logger '%s' level %s (%s) is more verbose than base "
"logger level %s (%s). Records below the base level may still be "
"filtered out by handlers depending on their own levels.",
name,
lvl,
sub_level,
logging.getLevelName(base_level_value),
base_level_value,
)
# The logger will *create* records down to the logger's level, but a handler
# with a higher level will still drop anything below its own threshold.
# Effective emission is gated by both: record.level >= logger.level AND handler.level.
logging.getLogger(name).setLevel(lvl)
# Attach file handlers if requested
if files is not None:
if isinstance(files, str):
# Single file for the entire `agentlightning` hierarchy.
_ensure_file_handler(
logger=base_logger,
filename=files,
level=base_level_value,
formatter=formatter,
)
else:
# Per-logger files
for logger_name, filename in files.items():
lg = logging.getLogger(logger_name)
# Use the logger's *effective* level at creation time
effective_level = lg.getEffectiveLevel()
_ensure_file_handler(
logger=lg,
filename=filename,
level=effective_level,
formatter=formatter,
)
# Optionally apply the same handler setup to other loggers outside this module
if apply_to:
for name in apply_to:
lg = logging.getLogger(name)
# This removes any existing handlers so we don't duplicate output
# and ensures these loggers share exactly the same handlers as base_logger.
lg.handlers.clear()
for h in base_logger.handlers:
lg.addHandler(h)
lg.setLevel(base_logger.level)
# We've attached handlers directly to these loggers; if propagate
# stayed True, records would bubble up to ancestor loggers and could be
# emitted twice (here and on the parent/root). Setting False isolates them.
lg.propagate = False
# Optionally capture warnings
if capture_warnings:
logging.captureWarnings(True)
def setup_module(
level: int | str = "INFO",
*,
name: str = "agentlightning",
console: bool = True,
color: bool | Dict[str, Any] = True,
propagate: bool = False,
disable_existing_loggers: bool = False,
) -> logging.Logger:
"""Initializes and returns the base logger for `agentlightning`.
This function constructs and applies a `dictConfig` configuration for the
logger hierarchy rooted at `name`. It supports either rich console
formatting (via `RichHandler`) or plain text formatting, based on the
`color` argument.
Unlike [`setup_logging`][agentlightning.setup_logging], this function configures only a single logger namespace
and does not attach extra handlers or submodule levels. It is primarily used
internally by [`setup_logging`][agentlightning.setup_logging] but is also suitable for direct integration in
custom logging workflows.
"""
root_cfg: Dict[str, Any] = {
"version": 1,
"disable_existing_loggers": disable_existing_loggers,
"loggers": {
name: {
"handlers": [],
"level": level,
"propagate": propagate,
}
},
"handlers": {},
"formatters": {},
}
# Choose formatter / handler definition
if color is not False and console:
# Console must be true to display colored outputs
if isinstance(color, dict):
rich_handler_config = color
else:
rich_handler_config: Dict[str, Any] = {
"rich_tracebacks": False,
"markup": False,
"show_time": True,
"show_path": True,
}
if not _has_width():
# e.g., in a CI environment.
rich_handler_config["console"] = Console(width=200)
root_cfg["handlers"]["console"] = {
"class": "rich.logging.RichHandler",
"level": level,
**rich_handler_config,
}
# RichHandler manages its own style; keep formatter None
else:
fmt_name = "plain"
root_cfg["formatters"][fmt_name] = {
"format": DEFAULT_FORMAT,
"datefmt": DATE_FORMAT,
}
if console:
root_cfg["handlers"]["console"] = {
"class": "logging.StreamHandler",
"level": level,
"formatter": fmt_name,
}
# Attach selected handlers to agentlightning
handler_names = list(root_cfg["handlers"].keys())
root_cfg["loggers"][name]["handlers"] = handler_names
# Apply dictConfig (this resets the logger handlers)
dictConfig(root_cfg)
return logging.getLogger(name)
def _has_width() -> bool:
"""Automatically determine whether the terminal has a width."""
return sys.stdout.isatty()
# log to stdout
handler = logging.StreamHandler()
handler.setLevel(level)
formatter = logging.Formatter("%(asctime)s [%(levelname)s] (Process-%(process)d %(name)s) %(message)s")
handler.setFormatter(formatter)
logger.addHandler(handler)
logger.setLevel(level)
logger.propagate = False # prevent double logging
return logger
+2 -2
View File
@@ -1,11 +1,11 @@
# Copyright (c) Microsoft. All rights reserved.
from .agent import LitAgentRunner
from .base import Runner
from .base import BaseRunner
from .legacy import LegacyAgentRunner
__all__ = [
"Runner",
"BaseRunner",
"LegacyAgentRunner",
"LitAgentRunner",
]
+54 -158
View File
@@ -11,21 +11,8 @@ from __future__ import annotations
import asyncio
import logging
import threading
import time
from contextlib import suppress
from typing import (
TYPE_CHECKING,
Any,
Awaitable,
Callable,
List,
Literal,
Optional,
Sequence,
TypeVar,
cast,
)
from typing import TYPE_CHECKING, Any, List, Literal, Optional, Sequence, TypeVar, cast
from opentelemetry.sdk.trace import ReadableSpan
@@ -33,7 +20,7 @@ from agentlightning.litagent import LitAgent
from agentlightning.reward import emit_reward, find_final_reward
from agentlightning.store.base import LightningStore
from agentlightning.tracer.agentops import AgentOpsTracer
from agentlightning.tracer.base import Tracer
from agentlightning.tracer.base import BaseTracer
from agentlightning.types import (
AttemptedRollout,
Hook,
@@ -43,54 +30,42 @@ from agentlightning.types import (
RolloutRawResult,
Span,
)
from agentlightning.utils.system_snapshot import system_snapshot
if TYPE_CHECKING:
from agentlightning.execution.events import ExecutionEvent
from .base import Runner
from .base import BaseRunner
T_task = TypeVar("T_task")
logger = logging.getLogger(__name__)
class LitAgentRunner(Runner[T_task]):
"""Execute [`LitAgent`][agentlightning.LitAgent] tasks with tracing support.
class LitAgentRunner(BaseRunner[T_task]):
"""Runner implementation for executing agent tasks with distributed support.
This runner manages the complete lifecycle of agent rollout execution,
including task polling, resource management, tracing, and hooks. It supports
both continuous iteration over tasks from the store and single-step execution.
Attributes:
worker_id: Identifier for the active worker process, if any.
worker_id: The unique identifier for this worker process.
"""
def __init__(
self,
tracer: Tracer,
max_rollouts: Optional[int] = None,
poll_interval: float = 5.0,
heartbeat_interval: float = 10.0,
heartbeat_launch_mode: Literal["asyncio", "thread"] = "asyncio",
) -> None:
def __init__(self, tracer: BaseTracer, max_rollouts: Optional[int] = None, poll_interval: float = 5.0) -> None:
"""Initialize the agent runner.
Args:
tracer: [`Tracer`][agentlightning.Tracer] used for rollout spans.
max_rollouts: Optional cap on iterations processed by
[`iter`][agentlightning.LitAgentRunner.iter].
poll_interval: Seconds to wait between store polls when no work is available.
heartbeat_interval: Seconds to wait between sending heartbeats to the store.
heartbeat_launch_mode: Launch mode for the heartbeat loop. Can be "asyncio" or "thread".
"asyncio" is the default and recommended mode. Use "thread" if you are experiencing blocking coroutines.
tracer: The tracer instance for recording execution traces and spans.
max_rollouts: Maximum number of tasks to process in iter() mode. If None,
the runner will continue indefinitely until interrupted.
poll_interval: Time in seconds to wait between polling attempts when
no tasks are available in the store.
"""
super().__init__()
self._tracer = tracer
self._max_rollouts = max_rollouts
self._poll_interval = poll_interval
self._heartbeat_interval = heartbeat_interval
self._heartbeat_launch_mode = heartbeat_launch_mode
# Set later
self._agent: Optional[LitAgent[T_task]] = None
@@ -105,9 +80,10 @@ class LitAgentRunner(Runner[T_task]):
initializes the tracer.
Args:
agent: [`LitAgent`][agentlightning.LitAgent] instance executed by the runner.
hooks: Optional sequence of [`Hook`][agentlightning.Hook]
callbacks invoked around tracing and rollout boundaries.
agent: The LitAgent instance to be managed by this runner.
hooks: Optional sequence of Hook objects to be called at various
lifecycle stages (on_trace_start, on_trace_end, on_rollout_start,
on_rollout_end).
**kwargs: Additional initialization arguments (currently unused).
"""
self._agent = agent
@@ -124,8 +100,7 @@ class LitAgentRunner(Runner[T_task]):
Args:
worker_id: Unique identifier for this worker process.
store: [`LightningStore`][agentlightning.LightningStore]
used for task coordination and persistence.
store: The LightningStore instance for task coordination and data persistence.
**kwargs: Additional worker-specific initialization arguments (currently unused).
"""
self._store = store
@@ -156,7 +131,7 @@ class LitAgentRunner(Runner[T_task]):
This method cleans up worker-specific resources and resets the worker ID.
Args:
worker_id: Unique identifier of the worker being torn down.
worker_id: The unique identifier of the worker being torn down.
*args: Additional teardown arguments (currently unused).
**kwargs: Additional teardown keyword arguments (currently unused).
"""
@@ -165,11 +140,11 @@ class LitAgentRunner(Runner[T_task]):
self._tracer.teardown_worker(worker_id)
@property
def tracer(self) -> Tracer:
def tracer(self) -> BaseTracer:
"""Get the tracer instance.
Returns:
The Tracer instance used by this runner.
The BaseTracer instance used by this runner.
"""
return self._tracer
@@ -180,7 +155,7 @@ class LitAgentRunner(Runner[T_task]):
The LitAgent instance managed by this runner.
Raises:
ValueError: If the agent has not been initialized via [`init`][agentlightning.LitAgentRunner.init].
ValueError: If the agent has not been initialized via init().
"""
if self._agent is None:
raise ValueError("Agent not initialized. Call init() first.")
@@ -193,7 +168,7 @@ class LitAgentRunner(Runner[T_task]):
The LightningStore instance for this worker.
Raises:
ValueError: If the store has not been initialized via [`init_worker`][agentlightning.LitAgentRunner.init_worker].
ValueError: If the store has not been initialized via init_worker().
"""
if self._store is None:
raise ValueError("Store not initialized. Call init_worker() first.")
@@ -330,67 +305,6 @@ class LitAgentRunner(Runner[T_task]):
return trace_spans
async def _emit_heartbeat(self, store: LightningStore) -> None:
"""Send a heartbeat tick to the store."""
worker_id = self.get_worker_id()
try:
await store.update_worker(worker_id, system_snapshot())
except asyncio.CancelledError:
# bypass the exception
raise
except Exception:
logger.exception("%s Unable to update worker heartbeat.", self._log_prefix())
def _start_heartbeat_loop(self, store: LightningStore) -> Optional[Callable[[], Awaitable[None]]]:
"""Start a background heartbeat loop and return an async stopper."""
if self._heartbeat_interval <= 0:
return None
if self.worker_id is None:
logger.warning("%s Cannot start heartbeat loop without worker_id.", self._log_prefix())
return None
if self._heartbeat_launch_mode == "asyncio":
stop_event = asyncio.Event()
async def heartbeat_loop() -> None:
while not stop_event.is_set():
await self._emit_heartbeat(store)
with suppress(asyncio.TimeoutError):
await asyncio.wait_for(stop_event.wait(), timeout=self._heartbeat_interval)
task = asyncio.create_task(heartbeat_loop(), name=f"{self.get_worker_id()}-heartbeat")
async def stop() -> None:
stop_event.set()
with suppress(asyncio.CancelledError):
await task
return stop
if self._heartbeat_launch_mode == "thread":
stop_evt = threading.Event()
def thread_worker() -> None:
loop = asyncio.new_event_loop()
asyncio.set_event_loop(loop)
while not stop_evt.is_set():
loop.run_until_complete(self._emit_heartbeat(store))
stop_evt.wait(self._heartbeat_interval)
thread = threading.Thread(target=thread_worker, name=f"{self.get_worker_id()}-heartbeat", daemon=True)
thread.start()
async def stop() -> None:
stop_evt.set()
await asyncio.to_thread(thread.join)
return stop
raise ValueError(f"Unsupported heartbeat launch mode: {self._heartbeat_launch_mode}")
async def _sleep_until_next_poll(self, event: Optional[ExecutionEvent] = None) -> None:
"""Sleep until the next poll interval, with optional event-based interruption.
@@ -398,7 +312,7 @@ class LitAgentRunner(Runner[T_task]):
and return early if the event is set.
Args:
event: Optional [`ExecutionEvent`][agentlightning.ExecutionEvent] object that can be used to interrupt the sleep.
event: Optional ExecutionEvent object that can be used to interrupt the sleep.
If set during the sleep period, the method returns immediately.
"""
if event is None:
@@ -450,7 +364,7 @@ class LitAgentRunner(Runner[T_task]):
await self._trigger_hooks(hook_type="on_rollout_start", agent=agent, runner=self, rollout=next_rollout)
start_time = time.time()
async with self._tracer.trace_context(
with self._tracer.trace_context(
name=rollout_id, store=store, rollout_id=rollout_id, attempt_id=next_rollout.attempt.attempt_id
):
await self._trigger_hooks(
@@ -521,7 +435,6 @@ class LitAgentRunner(Runner[T_task]):
"""Run the runner, continuously iterating over tasks in the store.
This method polls the store for new rollouts and executes them until:
- The event is set (if provided)
- The max_rollouts limit is reached (if configured)
- No more tasks are available
@@ -537,49 +450,39 @@ class LitAgentRunner(Runner[T_task]):
logger.info(f"{self._log_prefix()} Started async rollouts (max: {self._max_rollouts or 'unlimited'}).")
store = self.get_store()
stop_heartbeat = self._start_heartbeat_loop(store)
try:
while not (event is not None and event.is_set()) and (
self._max_rollouts is None or num_tasks_processed < self._max_rollouts
):
# Retrieve the next rollout
next_rollout: Optional[Rollout] = None
while not (event is not None and event.is_set()):
logger.debug(f"{self._log_prefix()} Try to poll for next rollout.")
next_rollout = await store.dequeue_rollout(worker_id=self.get_worker_id())
if next_rollout is None:
logger.debug(
f"{self._log_prefix()} No rollout to poll. Waiting for {self._poll_interval} seconds."
)
await self._sleep_until_next_poll(event)
else:
break
while not (event is not None and event.is_set()) and (
self._max_rollouts is None or num_tasks_processed < self._max_rollouts
):
# Retrieve the next rollout
next_rollout: Optional[Rollout] = None
while not (event is not None and event.is_set()):
logger.debug(f"{self._log_prefix()} Try to poll for next rollout.")
next_rollout = await store.dequeue_rollout()
if next_rollout is None:
return
logger.debug(f"{self._log_prefix()} No rollout to poll. Waiting for {self._poll_interval} seconds.")
await self._sleep_until_next_poll(event)
else:
break
try:
# Claim the rollout but updating the current worker id
await store.update_attempt(
next_rollout.rollout_id, next_rollout.attempt.attempt_id, worker_id=self.get_worker_id()
)
except Exception:
# This exception could happen if the rollout is dequeued and the other end died for some reason
logger.exception(f"{self._log_prefix()} Exception during update_attempt, giving up the rollout.")
continue
if next_rollout is None:
return
# Execute the step
await self._step_impl(next_rollout)
try:
# Claim the rollout but updating the current worker id
await store.update_attempt(
next_rollout.rollout_id, next_rollout.attempt.attempt_id, worker_id=self.get_worker_id()
)
except Exception:
# This exception could happen if the rollout is dequeued and the other end died for some reason
logger.exception(f"{self._log_prefix()} Exception during update_attempt, giving up the rollout.")
continue
num_tasks_processed += 1
if num_tasks_processed % 10 == 0 or num_tasks_processed == 1:
logger.info(
f"{self._log_prefix()} Progress: {num_tasks_processed}/{self._max_rollouts or 'unlimited'}"
)
finally:
if stop_heartbeat is not None:
await stop_heartbeat()
# Execute the step
await self._step_impl(next_rollout)
num_tasks_processed += 1
if num_tasks_processed % 10 == 0 or num_tasks_processed == 1:
logger.info(f"{self._log_prefix()} Progress: {num_tasks_processed}/{self._max_rollouts or 'unlimited'}")
logger.info(f"{self._log_prefix()} Finished async rollouts. Processed {num_tasks_processed} tasks.")
@@ -594,8 +497,7 @@ class LitAgentRunner(Runner[T_task]):
"""Execute a single task directly, bypassing the task queue.
This method creates a new rollout for the given input and executes it
immediately. Unlike [`iter()`][agentlightning.LitAgentRunner.iter],
exceptions are propagated to the caller.
immediately. Unlike iter(), exceptions are propagated to the caller.
Args:
input: The task input to be processed by the agent.
@@ -623,12 +525,6 @@ class LitAgentRunner(Runner[T_task]):
resources_id = None
attempted_rollout = await self.get_store().start_rollout(input=input, mode=mode, resources_id=resources_id)
# Register the attempt as running by the current worker
await self.get_store().update_attempt(
attempted_rollout.rollout_id,
attempted_rollout.attempt.attempt_id,
worker_id=self.get_worker_id(),
)
rollout_id = await self._step_impl(attempted_rollout, raise_on_exception=True)
completed_rollout = await store.get_rollout_by_id(rollout_id)
+74 -53
View File
@@ -1,6 +1,11 @@
# Copyright (c) Microsoft. All rights reserved.
"""Abstract runner interface for executing agent tasks."""
"""Base runner interface for executing agent tasks.
This module defines the abstract base class for all runner implementations
in the agent-lightning framework. Runners are responsible for managing the
execution lifecycle of agents and coordinating with the store.
"""
from __future__ import annotations
@@ -22,75 +27,90 @@ T_task = TypeVar("T_task")
logger = logging.getLogger(__name__)
class Runner(ParallelWorkerBase, Generic[T_task]):
"""Abstract base class for long-running agent executors.
class BaseRunner(ParallelWorkerBase, Generic[T_task]):
"""Base class for all runners.
Runner implementations coordinate [`LitAgent`][agentlightning.LitAgent]
instances, acquire work from a [`LightningStore`][agentlightning.LightningStore],
and emit [`Rollout`][agentlightning.Rollout] objects. Subclasses decide how
to schedule work (polling, streaming, etc.) while this base class provides a
minimal lifecycle contract.
This abstract base class defines the interface that all runner implementations
must follow. Runners are responsible for executing agent tasks, managing the
execution lifecycle, and coordinating with the store.
"""
def init(self, agent: LitAgent[T_task], **kwargs: Any) -> None:
"""Prepare the runner to execute tasks for `agent`.
"""Initialize the runner with the agent.
This method is called only once during the setup for all workers, not for each worker.
This method is called once during setup to configure the runner with
the agent it will execute.
Args:
agent: Agent instance providing task-specific logic.
**kwargs: Optional runner-specific configuration.
agent: The LitAgent instance to be managed by this runner.
**kwargs: Additional initialization arguments specific to the runner implementation.
Raises:
NotImplementedError: Subclasses must supply the initialization
routine.
NotImplementedError: Must be implemented by subclasses.
"""
raise NotImplementedError()
def init_worker(self, worker_id: int, store: LightningStore, **kwargs: Any) -> None:
"""Configure worker-local state before processing tasks.
"""Initialize the runner for each worker with worker_id and store.
This method is called for **each** worker during the setup.
This method is called once per worker process in a distributed setup.
It provides the worker with its unique ID and the store instance for
task coordination.
Args:
worker_id: Unique identifier for this worker process or thread.
store: Shared [`LightningStore`][agentlightning.LightningStore]
backing task coordination.
**kwargs: Optional worker-specific configuration.
worker_id: Unique identifier for this worker process.
store: The LightningStore instance for task coordination and data persistence.
**kwargs: Additional worker-specific initialization arguments.
Raises:
NotImplementedError: Subclasses must prepare per-worker resources.
NotImplementedError: Must be implemented by subclasses.
"""
raise NotImplementedError()
def run(self, *args: Any, **kwargs: Any) -> None:
"""Deprecated synchronous entry point.
"""Undefined method - use iter() or step() instead.
Use [`iter()`][agentlightning.Runner.iter] or [`step()`][agentlightning.Runner.step] instead.
This method is intentionally not implemented as the execution behavior
should be defined through iter() for continuous execution or step()
for single-task execution.
Args:
*args: Unused positional arguments.
**kwargs: Unused keyword arguments.
Raises:
RuntimeError: Always raised to direct callers to
[iter()][agentlightning.Runner.iter] or
[step()][agentlightning.Runner.step].
RuntimeError: Always raised to indicate this method should not be used.
"""
raise RuntimeError("The behavior of run() of Runner is undefined. Use iter() or step() instead.")
def teardown(self, *args: Any, **kwargs: Any) -> None:
"""Release resources acquired during [`init()`][agentlightning.Runner.init].
"""Clean up runner resources and reset state.
This method is called once during shutdown to clean up any resources
allocated during initialization and reset the runner state.
Args:
*args: Additional teardown arguments.
**kwargs: Additional teardown keyword arguments.
Raises:
NotImplementedError: Subclasses must implement the shutdown routine.
NotImplementedError: Must be implemented by subclasses.
"""
raise NotImplementedError()
def teardown_worker(self, worker_id: int, *args: Any, **kwargs: Any) -> None:
"""Release per-worker resources allocated by [`init_worker()`][agentlightning.Runner.init_worker].
"""Clean up worker-specific resources.
This method is called once per worker during shutdown to clean up
any resources specific to that worker.
Args:
worker_id: Identifier of the worker being torn down.
worker_id: The unique identifier of the worker being torn down.
*args: Additional teardown arguments.
**kwargs: Additional teardown keyword arguments.
Raises:
NotImplementedError: Subclasses must implement the shutdown routine.
NotImplementedError: Must be implemented by subclasses.
"""
raise NotImplementedError()
@@ -102,21 +122,18 @@ class Runner(ParallelWorkerBase, Generic[T_task]):
store: LightningStore,
hooks: Optional[Sequence[Hook]] = None,
worker_id: Optional[int] = None,
) -> Iterator[Runner[T_task]]:
"""Initialize and tear down a runner within a simple context manager.
The helper is primarily intended for debugging runner implementations
outside of a full [`Trainer`][agentlightning.Trainer] stack.
) -> Iterator[BaseRunner[T_task]]:
"""Context manager for quickly init and teardown the runner,
so that you can debug the runner without a trainer environment.
Args:
agent: Agent executed by this runner.
store: Backing [`LightningStore`][agentlightning.LightningStore].
If you don't have one, you can easily create one with
[`InMemoryLightningStore`][agentlightning.InMemoryLightningStore].
hooks: Optional sequence of hooks recognised by the runner.
Not all runners support hooks.
worker_id: Override the worker identifier used during setup. Defaults
to `0`.
agent: The LitAgent instance to be managed by this runner.
It should be the same agent that is to be run within the context.
store: The LightningStore instance for task coordination and data persistence.
If you don't have one, you can easily create one with `InMemoryLightningStore()`.
hooks: Optional sequence of Hook instances to be used by the runner.
Only some runners support hooks.
worker_id: Optional worker ID to be used by the runner.
"""
_initialized: bool = False
_worker_initialized: bool = False
@@ -146,11 +163,12 @@ class Runner(ParallelWorkerBase, Generic[T_task]):
them until interrupted by the event or when no more tasks are available.
Args:
event: Cooperative stop signal. When set, the runner should complete
the current unit of work and exit the loop.
event: Optional ExecutionEvent object that can be used to signal the runner
to stop gracefully. When set, the runner should finish its current
task and exit the iteration loop.
Raises:
NotImplementedError: Subclasses provide the iteration behavior.
NotImplementedError: Must be implemented by subclasses.
"""
raise NotImplementedError()
@@ -168,15 +186,18 @@ class Runner(ParallelWorkerBase, Generic[T_task]):
directly, bypassing the store's task queue.
Args:
input: Task payload consumed by the agent.
resources: Optional named resources scoped to this invocation.
mode: Optional rollout mode such as `"train"` or `"eval"`.
event: Cooperative stop signal for long-running tasks.
input: The task input to be processed by the agent.
resources: Optional named resources to be used for this specific task.
If not provided, the latest resources from the store will be used.
mode: Optional rollout mode (e.g., "train", "test"). If not provided,
the default mode will be used.
event: Optional ExecutionEvent object to signal interruption. When set, the
runner may abort the current execution.
Returns:
Completed rollout produced by the agent.
The completed rollout.
Raises:
NotImplementedError: Subclasses provide the execution behavior.
NotImplementedError: Must be implemented by subclasses.
"""
raise NotImplementedError()
+7 -7
View File
@@ -11,10 +11,10 @@ from agentlightning.adapter import TracerTraceToTriplet
from agentlightning.client import AgentLightningClient
from agentlightning.litagent import LitAgent
from agentlightning.litagent.litagent import is_v0_1_rollout_api
from agentlightning.tracer.base import Tracer
from agentlightning.tracer.base import BaseTracer
from agentlightning.types import RolloutLegacy, RolloutRawResultLegacy, Triplet
from .base import Runner
from .base import BaseRunner
logger = logging.getLogger(__name__)
@@ -23,7 +23,7 @@ __all__ = [
]
class LegacyAgentRunner(Runner[Any]):
class LegacyAgentRunner(BaseRunner[Any]):
"""Manages the agent's execution loop and integrates with AgentOps.
This class orchestrates the interaction between the agent (`LitAgent`) and
@@ -43,7 +43,7 @@ class LegacyAgentRunner(Runner[Any]):
self,
agent: LitAgent[Any],
client: AgentLightningClient,
tracer: Tracer,
tracer: BaseTracer,
triplet_exporter: TracerTraceToTriplet,
worker_id: Optional[int] = None,
max_tasks: Optional[int] = None,
@@ -58,7 +58,7 @@ class LegacyAgentRunner(Runner[Any]):
self.worker_id = worker_id
self.max_tasks = max_tasks
# These methods are overridden by Runner, getting them back to old behavior.
# These methods are overridden by BaseRunner, getting them back to old behavior.
def init(self, *args: Any, **kwargs: Any) -> None:
pass
@@ -180,7 +180,7 @@ class LegacyAgentRunner(Runner[Any]):
except Exception:
logger.exception(f"{self._log_prefix(rollout_id)} Exception during on_rollout_start hook.")
with self.tracer._trace_context_sync(name=f"rollout_{rollout_id}"): # pyright: ignore[reportPrivateUsage]
with self.tracer.trace_context(name=f"rollout_{rollout_id}"):
start_time = time.time()
rollout_method = self.agent.training_rollout if task.mode == "train" else self.agent.validation_rollout
# Pass the task input, not the whole task object
@@ -257,7 +257,7 @@ class LegacyAgentRunner(Runner[Any]):
except Exception:
logger.exception(f"{self._log_prefix(rollout_id)} Exception during on_rollout_start hook.")
async with self.tracer.trace_context(name=f"rollout_{rollout_id}"):
with self.tracer.trace_context(name=f"rollout_{rollout_id}"):
start_time = time.time()
rollout_method = (
self.agent.training_rollout_async if task.mode == "train" else self.agent.validation_rollout_async
+67 -106
View File
@@ -1,11 +1,6 @@
# Copyright (c) Microsoft. All rights reserved.
"""Legacy HTTP server compatible with the original Agent Lightning protocol.
The implementation in this module predates the modern store-powered runtime and
is kept for backwards compatibility with older deployments. New applications
should migrate to the store architecture where possible.
"""
"""Legacy server for the Agent Lightning framework. Deprecated in favor of agentlightning.store."""
from __future__ import annotations
@@ -34,15 +29,9 @@ logger = logging.getLogger(__name__)
class ServerDataStore:
"""Async-safe container for in-memory server state.
The store tracks queued tasks, claimed tasks, uploaded rollouts, and the
currently published resources. All interactions are guarded by asyncio locks
so that the FastAPI handlers can safely run in parallel.
!!! warning "Deprecated"
[`ServerDataStore`][agentlightning.server.ServerDataStore] is part of
the legacy client/server stack. Use [`LightningStore`][agentlightning.LightningStore] instead.
"""
A centralized, thread-safe, async, in-memory data store for the server's state.
This holds the task queue, versioned resources, and completed rollouts.
"""
def __init__(self):
@@ -65,18 +54,8 @@ class ServerDataStore:
resources_id: str | None = None,
metadata: Dict[str, Any] | None = None,
) -> str:
"""Enqueue a new task and return the generated rollout identifier.
Args:
sample: Payload that describes the task input.
mode: Phase in which the sample should be executed (`"train"`, `"val"`, or
`"test"`).
resources_id: Identifier of a resource bundle that the executor should
load before running the task.
metadata: Optional metadata forwarded to the executor.
Returns:
Unique rollout identifier assigned to the task.
"""
Adds a new task to the queue with specific metadata and returns its unique ID.
"""
rollout_id = f"rollout-{uuid.uuid4()}"
task = Task(
@@ -93,11 +72,9 @@ class ServerDataStore:
return rollout_id
async def get_next_task(self) -> Optional[Task]:
"""Retrieve the next task from the queue without blocking.
Returns:
Next [`Task`][agentlightning.Task] ready to execute, or ``None``
when the queue is empty.
"""
Retrieves the next task from the queue without blocking.
Returns None if the queue is empty.
"""
try:
async with self._results_lock:
@@ -118,10 +95,8 @@ class ServerDataStore:
return None
async def update_resources(self, update: ResourcesUpdate):
"""Persist a new resource bundle and mark it as the latest version.
Args:
update: Resource payload received from a client.
"""
Safely stores a new version of named resources and sets it as the latest.
"""
# TODO: evict old resources if necessary.
async with self._resources_lock:
@@ -130,38 +105,26 @@ class ServerDataStore:
logger.info(f"Resources updated. New version '{update.resources_id}' is now latest.")
async def get_resources_by_id(self, resources_id: str) -> Optional[ResourcesUpdate]:
"""Retrieve a specific resource bundle by identifier.
Args:
resources_id: Identifier that was previously published to the store.
Returns:
Matching [`ResourcesUpdate`][agentlightning.ResourcesUpdate]
instance, or ``None`` when the identifier is unknown.
"""
Safely retrieves a specific version of named resources by its ID.
"""
async with self._resources_lock:
resources = self._resource_versions.get(resources_id)
if resources:
return ResourcesUpdate(
resources_id=resources_id,
resources=resources,
create_time=time.time(),
update_time=time.time(),
version=1,
)
return ResourcesUpdate(resources_id=resources_id, resources=resources)
return None
async def get_latest_resources(self) -> Optional[ResourcesUpdate]:
"""Return the most recent resource bundle, if one exists."""
"""
Safely retrieves the latest version of named resources.
"""
if self._latest_resources_id:
return await self.get_resources_by_id(self._latest_resources_id)
return None
async def store_rollout(self, rollout: RolloutLegacy):
"""Persist a completed rollout for later inspection.
Args:
rollout: Rollout returned by a client.
"""
Safely stores a completed rollout from a client.
"""
async with self._results_lock:
self._processing_tasks.pop(rollout.rollout_id, None)
@@ -169,31 +132,27 @@ class ServerDataStore:
logger.info(f"Rollout received and stored: {rollout.rollout_id}")
async def retrieve_rollout(self, rollout_id: str) -> Optional[RolloutLegacy]:
"""Retrieve and remove a stored rollout by identifier.
Args:
rollout_id: Identifier of the rollout to fetch.
Returns:
Stored [`RolloutLegacy`][agentlightning.RolloutLegacy], or ``None``
when the identifier is unknown.
"""
Safely retrieves a single rollout by its ID, removing it from the store.
"""
async with self._results_lock:
return self._completed_rollouts.pop(rollout_id, None)
async def retrieve_completed_rollouts(self) -> List[RolloutLegacy]:
"""Return all completed rollouts and clear the internal buffer."""
"""
Retrieves all completed rollouts and clears the store.
"""
async with self._results_lock:
rollouts = list(self._completed_rollouts.values())
self._completed_rollouts.clear()
return rollouts
def get_processing_tasks(self) -> Dict[str, Task]:
"""Return a copy of currently processing tasks for timeout checking."""
"""Returns a copy of currently processing tasks for timeout checking."""
return self._processing_tasks.copy()
async def requeue_task(self, task: Task):
"""Requeue a task that timed out while being processed."""
"""Requeues a task that has timed out and removes it from processing."""
logger.warning(f"Requeuing task {task.rollout_id} after timeout (attempt {task.num_claims})")
async with self._results_lock:
# Remove from processing tasks
@@ -202,26 +161,21 @@ class ServerDataStore:
class AgentLightningServer:
"""High-level controller for the legacy Agent Lightning FastAPI server.
"""
The main SDK class for developers to control the Agent Lightning Server.
The controller orchestrates server start-up, task queueing, resource updates,
and retrieval of client rollouts. It is primarily used by existing systems that
still rely on the HTTP-based workflow.
!!! warning "Deprecated"
[`AgentLightningServer`][agentlightning.server.AgentLightningServer] is part of
the legacy client/server stack. Prefer the store-based runtime for new
integrations.
This class manages the server lifecycle, task queueing, resources updates,
and retrieval of results, providing a simple interface for the optimization logic.
"""
def __init__(self, host: str = "127.0.0.1", port: int = 8000, task_timeout_seconds: float = 300.0):
"""Initialize the controller.
"""
Initializes the server controller.
Args:
host: Hostname or IP address to bind the HTTP server to.
port: TCP port exposed by the server.
task_timeout_seconds: Seconds before a claimed task is considered stale and
re-queued.
host: The host to bind the server to.
port: The port to bind the server to.
task_timeout_seconds: Time in seconds after which a claimed task is considered stale and requeued.
"""
warnings.warn(
"AgentLightningServer is deprecated. Please use LightningStoreServer instead.", DeprecationWarning
@@ -246,7 +200,9 @@ class AgentLightningServer:
# --- ADDED: Lifespan context manager ---
@asynccontextmanager
async def _lifespan(self, app: FastAPI):
"""Manage server start-up and shutdown within the event loop."""
"""
Manages server startup and shutdown. This runs inside the server's event loop.
"""
logger.info("Server is starting up...")
self.loop = asyncio.get_running_loop()
self._store = ServerDataStore() # Initialize data store here
@@ -260,7 +216,9 @@ class AgentLightningServer:
self.loop = None
async def _check_and_requeue_stale_tasks(self):
"""Check for stale tasks and requeue them when they exceed the timeout."""
"""
Check for stale tasks and requeue them. Called reactively during get_next_task.
"""
current_time = time.time()
# Ensure store is initialized before checking
if not self._store:
@@ -275,11 +233,11 @@ class AgentLightningServer:
)
def _setup_routes(self):
"""Configure the FastAPI routes that make up the legacy HTTP API."""
"""Setup FastAPI routes."""
@self._app.get("/task", response_model=TaskIfAny)
async def next_task() -> TaskIfAny: # type: ignore
"""Provide the next available task to a client."""
"""Endpoint for clients to poll for the next available task."""
await self._check_and_requeue_stale_tasks()
if not self._store:
@@ -295,7 +253,7 @@ class AgentLightningServer:
@self._app.get("/resources/latest", response_model=ResourcesUpdate)
async def fetch_latest_resources() -> ResourcesUpdate: # type: ignore
"""Return the most recent resource bundle published to the server."""
"""Endpoint for clients to poll for the latest available resources."""
if not self._store:
raise HTTPException(status_code=503, detail="Server not fully initialized.")
resources_update = await self._store.get_latest_resources()
@@ -308,7 +266,7 @@ class AgentLightningServer:
async def fetch_resources_by_id( # type: ignore
resource_id: str = Path(..., description="The unique identifier for the resource version.")
) -> ResourcesUpdate:
"""Return a specific version of resources by identifier."""
"""Endpoint for clients to fetch a specific version of resources."""
if not self._store:
raise HTTPException(status_code=503, detail="Server not fully initialized.")
resources_update = await self._store.get_resources_by_id(resource_id)
@@ -319,7 +277,7 @@ class AgentLightningServer:
@self._app.post("/rollout", response_model=GenericResponse)
async def post_rollout(payload: RolloutLegacy) -> GenericResponse: # type: ignore
"""Persist the rollout reported by a client."""
"""Endpoint for clients to report a completed rollout."""
if not self._store:
raise HTTPException(status_code=503, detail="Server not fully initialized.")
await self._store.store_rollout(payload)
@@ -329,13 +287,13 @@ class AgentLightningServer:
)
async def start(self):
"""Start the FastAPI server in the background."""
"""Starts the FastAPI server in the background."""
logger.info(f"Starting server at {self.endpoint}")
asyncio.create_task(self._uvicorn_server.serve())
await asyncio.sleep(1) # Allow time for server to start up.
async def stop(self):
"""Stop the FastAPI server and wait for a graceful shutdown."""
"""Gracefully stops the running FastAPI server."""
if self._uvicorn_server.started:
logger.info("Stopping server...")
self._uvicorn_server.should_exit = True
@@ -343,7 +301,10 @@ class AgentLightningServer:
logger.info("Server stopped.")
async def run_forever(self):
"""Run the server indefinitely until `stop()` is invoked."""
"""
Runs the server indefinitely until stopped.
This is useful when async start and stop methods do not work.
"""
await self._uvicorn_server.serve()
async def queue_task(
@@ -353,37 +314,35 @@ class AgentLightningServer:
resources_id: str | None = None,
metadata: Dict[str, Any] | None = None,
) -> str:
"""Add a task to the queue for a client to process."""
"""
Adds a task to the queue for a client to process.
"""
if not self._store:
raise RuntimeError("Store not initialized. The server may not be running.")
return await self._store.add_task(sample, mode=mode, resources_id=resources_id, metadata=metadata)
async def update_resources(self, resources: NamedResources) -> str:
"""Publish a new resource bundle and return its generated identifier."""
"""
Updates the resources, creating a new version and setting it as the latest.
"""
if not self._store:
raise RuntimeError("Store not initialized. The server may not be running.")
resources_id = f"res-{uuid.uuid4()}"
update = ResourcesUpdate(
resources_id=resources_id, resources=resources, create_time=time.time(), update_time=time.time(), version=1
)
update = ResourcesUpdate(resources_id=resources_id, resources=resources)
await self._store.update_resources(update)
return resources_id
async def get_completed_rollout(self, rollout_id: str) -> Optional[RolloutLegacy]:
"""Retrieve a specific completed rollout by identifier."""
"""
Retrieves a specific completed rollout by its ID.
"""
if not self._store:
raise RuntimeError("Store not initialized. The server may not be running.")
return await self._store.retrieve_rollout(rollout_id)
async def poll_completed_rollout(self, rollout_id: str, timeout: Optional[float] = None) -> Optional[RolloutLegacy]:
"""Poll for a completed rollout until it becomes available or a timeout expires.
Args:
rollout_id: Identifier of the rollout to wait for.
timeout: Maximum number of seconds to wait. ``None`` waits indefinitely.
Returns:
Retrieved rollout, or ``None`` when the timeout is reached without success.
"""
Polls for a completed rollout by its ID, waiting up to `timeout` seconds.
"""
start_time = time.time()
while True:
@@ -395,7 +354,9 @@ class AgentLightningServer:
await asyncio.sleep(1)
async def retrieve_completed_rollouts(self) -> List[RolloutLegacy]:
"""Return every completed rollout and clear the internal buffer."""
"""
Retrieves all available completed trajectories and clears the internal store.
"""
if not self._store:
raise RuntimeError("Store not initialized. The server may not be running.")
return await self._store.retrieve_completed_rollouts()
+1 -2
View File
@@ -1,13 +1,12 @@
# Copyright (c) Microsoft. All rights reserved.
from .base import LightningStore, LightningStoreCapabilities
from .base import LightningStore
from .client_server import LightningStoreClient, LightningStoreServer
from .memory import InMemoryLightningStore
from .threading import LightningStoreThreaded
__all__ = [
"LightningStore",
"LightningStoreCapabilities",
"LightningStoreClient",
"LightningStoreServer",
"InMemoryLightningStore",
+80 -413
View File
@@ -2,7 +2,7 @@
from __future__ import annotations
from typing import Any, Dict, List, Literal, Optional, Sequence, TypedDict
from typing import Any, Dict, List, Literal, Optional, Sequence
from opentelemetry.sdk.trace import ReadableSpan
@@ -17,7 +17,6 @@ from agentlightning.types import (
RolloutStatus,
Span,
TaskInput,
Worker,
)
@@ -53,85 +52,33 @@ UNSET = _UnsetType()
Unset = _UnsetType # Alias for convenience
class LightningStoreCapabilities(TypedDict):
"""Capability of a LightningStore implementation."""
thread_safe: bool
"""Whether the store is thread-safe."""
async_safe: bool
"""Whether the store is async-safe."""
zero_copy: bool
"""Whether the store has only one copy across all threads/processes."""
class LightningStore:
"""Contract for the persistent control-plane that coordinates training rollouts.
A `LightningStore` mediates every interaction between algorithms and runners:
- **Rollout lifecycle:** accept new rollouts, queue them for execution, create attempts,
and drive the rollout status machine (`"queuing"` → `"preparing"` → `"running"` →
`{"succeeded","failed","cancelled"}` or `"requeuing"` when a retry is justified).
- **Attempt tracking:** record each execution attempt, including progress heartbeats,
retry sequencing, and terminal states such as `"timeout"` or `"unresponsive"`.
- **Span ingest:** capture structured telemetry emitted by runners (either as native
[`Span`][agentlightning.Span] objects or as `opentelemetry.sdk.trace.ReadableSpan`
instances) so that algorithms can reconstruct trajectories and rewards.
- **Resource versioning:** manage immutable snapshots of named resources
(prompt templates, model checkpoints, proxy endpoints, …) and expose a single
"latest" snapshot that runners can fetch just after claiming work.
Implementations must provide thread-safe/async-safe semantics: each coroutine should
appear atomic to callers even when multiple algorithms or runners call the API concurrently.
Unless stated otherwise, missing identifiers should result in a `ValueError`.
"""
A centralized, thread-safe, async, data store for the lightning's state.
This holds the task queue, versioned resources, and completed rollouts.
@property
def capabilities(self) -> LightningStoreCapabilities:
"""Return the capabilities of the store."""
return LightningStoreCapabilities(
thread_safe=False,
async_safe=False,
zero_copy=False,
)
The store has a built-in clock and it should be responsible for tracking the times.
All the time-based operations like retry, timeout, etc. should be handled by the store.
"""
async def start_rollout(
self,
input: TaskInput,
mode: Literal["train", "val", "test"] | None = None,
resources_id: str | None = None,
config: RolloutConfig | None = None,
metadata: Dict[str, Any] | None = None,
) -> AttemptedRollout:
"""Register a rollout and immediately create its first attempt.
"""
Add one incomplete rollout to the store, and get an attempt created for it.
This will immediately sets the rollout to a preparing state, and should be
used by whoever is going to execute the rollout.
!!! note
Use [`enqueue_rollout()`][agentlightning.LightningStore.enqueue_rollout] when the
caller only wants to submit work for later scheduling.
Return a special rollout with attempt object. Do not update it directly.
The rollout must be persisted with `status="preparing"` and an initial attempt
with `sequence_id == 1` so the caller can begin execution without visiting the
public queue. Implementations are expected to:
But if the rollout fails or timeouts, it's still possible that the watchdog
sends it back to the queue for retry.
1. Generate a unique `rollout_id` and `attempt_id`.
2. Record `start_time` for both rollout and attempt based on the current clock.
3. Copy `config` and `metadata` so later mutations do not leak shared references.
4. Resolve `resources_id` to the latest resource snapshot when `None` is supplied.
Args:
input: Arbitrary task payload supplied by an algorithm.
mode: Optional semantic mode for downstream analytics (`"train"`, `"val"`, `"test"`).
resources_id: Concrete resource snapshot to execute against; defaults to the latest stored snapshot.
config: Rollout retry/timeout policy. Should default to a fresh [`RolloutConfig`][agentlightning.RolloutConfig].
metadata: Free-form metadata persisted verbatim with the rollout.
Returns:
The fully-populated [`AttemptedRollout`][agentlightning.AttemptedRollout] including
the just-created attempt.
Raises:
NotImplementedError: Subclasses must provide durable storage for the rollout.
ValueError: Implementations should raise when `resources_id` does not exist.
To enqueue a rollout to the task queue, use `enqueue_rollout` instead.
"""
raise NotImplementedError()
@@ -140,101 +87,34 @@ class LightningStore:
input: TaskInput,
mode: Literal["train", "val", "test"] | None = None,
resources_id: str | None = None,
config: RolloutConfig | None = None,
metadata: Dict[str, Any] | None = None,
) -> Rollout:
"""Persist a rollout in `queuing` state so runners can claim it later.
!!! note
Different from [`start_rollout()`][agentlightning.LightningStore.start_rollout],
this method is called when the caller only wants to submit work for later scheduling.
Implementations must generate a unique `rollout_id`, stamp `start_time` with
the current time, default `config` to a fresh [`RolloutConfig`][agentlightning.RolloutConfig],
and insert the rollout at the tail of the scheduling queue. No attempt is created yet.
Args:
input: Arbitrary task payload supplied by an algorithm.
mode: Optional semantic mode indicator (`"train"`, `"val"`, `"test"`).
resources_id: Resource snapshot used when a runner eventually executes the rollout.
config: Fine-grained retry/timeout parameters to persist with the rollout.
metadata: Free-form metadata stored verbatim with the rollout record.
Returns:
The stored [`Rollout`][agentlightning.Rollout] in `queuing` status.
Raises:
NotImplementedError: Subclasses must persist the rollout.
ValueError: Implementations should raise when `resources_id` does not exist.
"""
Adds a new task to the queue with specific metadata and
returns the rollout object with its unique ID.
"""
raise NotImplementedError()
async def dequeue_rollout(self, worker_id: Optional[str] = None) -> Optional[AttemptedRollout]:
"""Claim the oldest queued rollout and transition it to `preparing`.
async def dequeue_rollout(self) -> Optional[AttemptedRollout]:
"""
Retrieves the next task from the queue without blocking.
Returns None if the queue is empty.
This function do not block.
Retrieval must be FIFO across rollouts that remain in `queuing` or `requeuing`
state. When a rollout is claimed, implementations must:
* Transition its status to `"preparing"`.
* Create a new attempt with `status="preparing"` and `sequence_id` equal to
the number of attempts already registered for the rollout plus one.
* Return an [`AttemptedRollout`][agentlightning.AttemptedRollout] snapshot so the
runner knows both rollout metadata and the attempt identifier.
* Optionally refresh the caller's [`Worker`][agentlightning.Worker] telemetry
(e.g., `last_dequeue_time`) when `worker_id` is provided.
Returns:
The next attempt to execute, or `None` when no eligible rollouts are queued.
Raises:
NotImplementedError: Subclasses must implement queue retrieval.
Will set the rollout status to preparing.
"""
raise NotImplementedError()
async def start_attempt(self, rollout_id: str) -> AttemptedRollout:
"""Create a manual retry attempt for an existing rollout.
This is typically invoked by runners that wish to retry outside of the
normal queue flow (for example in an online RL setup).
Implementations must validate that the rollout exists, allocate a fresh `attempt_id`,
increment the `sequence_id` monotonically, stamp the new attempt with `status="preparing"`,
and return an up-to-date [`AttemptedRollout`][agentlightning.AttemptedRollout].
Args:
rollout_id: Unique identifier of the rollout receiving a new attempt.
Returns:
The rollout paired with its newly-created attempt.
Raises:
NotImplementedError: Subclasses must implement attempt creation.
ValueError: Implementations must raise when `rollout_id` is unknown.
"""
Create a new attempt for a given rollout ID and return the attempt details.
"""
raise NotImplementedError()
async def add_span(self, span: Span) -> Span:
"""Persist a pre-constructed span emitted during rollout execution.
"""
Add a span to the store.
The provided [`Span`][agentlightning.Span] must already contain the `rollout_id`,
`attempt_id`, and `sequence_id`. Implementations must:
* Verify that both rollout and attempt exist.
* Ensure span ordering remains strictly increasing per attempt (rejecting or keeping duplicates).
* Treat the span arrival as a heartbeat: update the attempt's `last_heartbeat_time`
and transition both attempt and rollout to `"running"` if they were still
`"preparing"` or `"requeuing"`.
Args:
span: Fully populated span to persist.
Returns:
The stored span record (implementations may return a copy).
Raises:
NotImplementedError: Subclasses must implement span persistence.
ValueError: Implementations must raise when the referenced rollout or attempt is missing.
This method is responsible for updating the rollout/attempt status to "running" if needed.
"""
raise NotImplementedError()
@@ -245,232 +125,88 @@ class LightningStore:
readable_span: ReadableSpan,
sequence_id: int | None = None,
) -> Span:
"""Convert and persist an OpenTelemetry span for a particular attempt.
"""
Add an opentelemetry span to the store.
Implementations must transform the `readable_span` into a [`Span`][agentlightning.Span]
(typically via [`Span.from_opentelemetry()`][agentlightning.Span.from_opentelemetry]),
assign a strictly increasing `sequence_id` when one is not provided, and persist it
using the same semantics as [`add_span()`][agentlightning.LightningStore.add_span].
Args:
rollout_id: Identifier of the rollout that produced the span.
attempt_id: Attempt identifier the span belongs to.
readable_span: OpenTelemetry span in SDK form.
sequence_id: Optional explicit ordering hint. When omitted, call
[`get_next_span_sequence_id()`][agentlightning.LightningStore.get_next_span_sequence_id]
automatically.
Returns:
The stored span record.
Raises:
NotImplementedError: Subclasses must implement span persistence.
ValueError: Implementations must raise when the rollout or attempt is unknown.
If sequence_id is not provided, it will be fetched from `get_next_span_sequence_id` and assigned automatically.
"""
raise NotImplementedError()
async def query_rollouts(
self, *, status: Optional[Sequence[RolloutStatus]] = None, rollout_ids: Optional[Sequence[str]] = None
) -> List[Rollout]:
"""Retrieve rollouts filtered by status and/or explicit identifiers.
Args:
status: Optional whitelist of [`RolloutStatus`][agentlightning.RolloutStatus] values.
rollout_ids: Optional whitelist of rollout identifiers to include.
Returns:
A list of matching rollouts. Ordering is backend-defined but must be deterministic.
Raises:
NotImplementedError: Subclasses must implement the query.
"""
Query and retrieve rollouts filtered by their status.
If no status is provided, returns all rollouts.
"""
raise NotImplementedError()
async def query_attempts(self, rollout_id: str) -> List[Attempt]:
"""Return every attempt ever created for `rollout_id` in ascending sequence order.
Args:
rollout_id: Identifier of the rollout being inspected.
Returns:
Attempts sorted by `sequence_id` (oldest first). Returns an empty list when none exist.
Raises:
NotImplementedError: Subclasses must implement the query.
ValueError: Implementations must raise when the rollout does not exist.
"""
Query and retrieve all attempts associated with a specific rollout ID.
Returns an empty list if no attempts are found.
"""
raise NotImplementedError()
async def get_rollout_by_id(self, rollout_id: str) -> Optional[Rollout]:
"""Fetch a rollout by identifier without mutating its state.
Args:
rollout_id: Identifier to retrieve.
Returns:
The rollout when found, otherwise `None`.
Raises:
NotImplementedError: Subclasses must implement retrieval.
"""
Safely retrieves a specific rollout by its ID.
"""
raise NotImplementedError()
async def get_latest_attempt(self, rollout_id: str) -> Optional[Attempt]:
"""Fetch the attempt with the highest `sequence_id` for `rollout_id`.
Args:
rollout_id: Identifier to inspect.
Returns:
The most recent attempt or `None` when no attempts exist yet.
Raises:
NotImplementedError: Subclasses must implement retrieval.
ValueError: Implementations must raise when the rollout does not exist.
"""
raise NotImplementedError()
async def query_resources(self) -> List[ResourcesUpdate]:
"""List every stored resource snapshot in insertion order.
Returns:
A chronological list of [`ResourcesUpdate`][agentlightning.ResourcesUpdate] objects.
Raises:
NotImplementedError: Subclasses must implement retrieval.
Safely retrieves the latest attempt for a given rollout ID.
"""
raise NotImplementedError()
async def get_resources_by_id(self, resources_id: str) -> Optional[ResourcesUpdate]:
"""Return a specific named resource snapshot by identifier.
Args:
resources_id: Identifier of the snapshot.
Returns:
The stored [`ResourcesUpdate`][agentlightning.ResourcesUpdate], or `None` when missing.
Raises:
NotImplementedError: Subclasses must implement retrieval.
"""
Safely retrieves a specific version of named resources by its ID.
"""
raise NotImplementedError()
async def get_latest_resources(self) -> Optional[ResourcesUpdate]:
"""Fetch the latest resource snapshot marked as the global default.
Returns:
The current latest [`ResourcesUpdate`][agentlightning.ResourcesUpdate], or `None` when
no resources have been registered yet.
Raises:
NotImplementedError: Subclasses must implement retrieval.
"""
Safely retrieves the latest version of named resources.
"""
raise NotImplementedError()
async def get_next_span_sequence_id(self, rollout_id: str, attempt_id: str) -> int:
"""Allocate the next strictly increasing sequence number used to order spans.
"""
Get the next span sequence ID for a given rollout and attempt.
This should be used to assign a unique sequence ID to each span within an attempt.
Implementations must retain counters so repeated calls return `1, 2, ...` without
gaps unless spans were explicitly inserted with a custom `sequence_id`. The
counter may be scoped per rollout or per attempt, but the sequence must be
strictly increasing for spans emitted by the specified attempt so traces remain
totally ordered.
See [Distributed Tracing][distributed-tracing] for detailed motivations.
Args:
rollout_id: Identifier of the rollout emitting spans.
attempt_id: Attempt identifier for the upcoming span.
Returns:
The next integer sequence identifier, unique within the attempt.
Raises:
NotImplementedError: Subclasses must provide the allocator.
ValueError: Implementations must raise when the rollout or attempt does not exist.
Recommend getting the ID before the operation even begins to avoid racing conditions.
"""
raise NotImplementedError()
async def wait_for_rollouts(self, *, rollout_ids: List[str], timeout: Optional[float] = None) -> List[Rollout]:
"""Block until the targeted rollouts reach a terminal status or the timeout expires.
"""
Wait for specified rollouts to complete with a timeout.
Returns the completed rollouts, potentially incomplete if timeout is reached.
Terminal statuses are `"succeeded"`, `"failed"`, and `"cancelled"`. When the timeout
elapses, implementations should return the subset of rollouts that are already terminal
and omit the rest.
!!! warning
It's dangerous and might be event-loop blocking to call this function
with a long timeout. It's a good idea to poll for the method to check
if new completed rollouts can coming. Be careful in implementing the sleep logic
to avoid busy-waiting.
Args:
rollout_ids: Identifiers of rollouts to watch.
timeout: Maximum time in seconds to wait. `None` waits indefinitely.
Returns:
Rollouts that finished before the deadline, in arbitrary order.
Raises:
NotImplementedError: Subclasses must implement waiting semantics.
ValueError: Implementations must raise when a rollout identifier is unknown.
TODO: Add support for waiting for 20 new rollouts, or wait until 80% of the pending ids are completed.
"""
raise NotImplementedError()
async def query_spans(self, rollout_id: str, attempt_id: str | Literal["latest"] | None = None) -> List[Span]:
"""Return the stored spans for a rollout, optionally scoped to one attempt.
Spans must be returned in ascending `sequence_id` order. Implementations may raise
a `RuntimeError` when spans were evicted or expired.
Args:
rollout_id: Identifier of the rollout being inspected.
attempt_id: Attempt identifier to filter by. Pass `"latest"` to retrieve only the
most recent attempt, or `None` to return all spans across attempts.
Returns:
An ordered list of spans (possibly empty).
Raises:
NotImplementedError: Subclasses must implement the query.
ValueError: Implementations must raise when the rollout or attempt is unknown.
"""
Query and retrieve all spans associated with a specific rollout ID.
Returns an empty list if no spans are found.
"""
raise NotImplementedError()
async def add_resources(self, resources: NamedResources) -> ResourcesUpdate:
"""Persist a new immutable snapshot of named resources and mark it as latest.
Implementations must assign a fresh `resources_id` and ensure subsequent calls to
[`get_latest_resources()`][agentlightning.LightningStore.get_latest_resources] return the
snapshot produced here.
Args:
resources: Mapping of resource names to their serialized payloads.
Returns:
The stored [`ResourcesUpdate`][agentlightning.ResourcesUpdate] including its generated id.
Raises:
NotImplementedError: Subclasses must implement resource persistence.
"""
Safely stores a new version of named resources and sets it as the latest.
Not implemented by many stores yet.
"""
raise NotImplementedError()
async def update_resources(self, resources_id: str, resources: NamedResources) -> ResourcesUpdate:
"""Overwrite or extend an existing resource snapshot and mark it as latest.
This API is typically used by algorithms that maintain mutable resources (e.g., model
checkpoints) under a stable identifier.
Args:
resources_id: Identifier of the snapshot to replace.
resources: Updated mapping of resource names to payloads.
Returns:
The persisted [`ResourcesUpdate`][agentlightning.ResourcesUpdate].
Raises:
NotImplementedError: Subclasses must implement resource persistence.
ValueError: Implementations must raise when `resources_id` does not exist.
"""
Safely stores a new version or updates an existing version of named resources and sets it as the latest.
"""
raise NotImplementedError()
@@ -484,31 +220,22 @@ class LightningStore:
config: RolloutConfig | Unset = UNSET,
metadata: Optional[Dict[str, Any]] | Unset = UNSET,
) -> Rollout:
"""Update rollout metadata and, when provided, drive status transitions.
"""
Update the rollout status and related metadata.
Parameters default to the sentinel [`UNSET`][agentlightning.store.base.UNSET] to
distinguish omitted fields from explicit `None` assignments. Implementations must:
Not-listed fields here either cannot be updated, or should be auto-updated (e.g., end_time).
* Validate the rollout exists before mutating it.
* Replace each property when a concrete value (including `None`) is supplied.
* When the status switches into a terminal state, set `end_time` and signal any waiters.
* When the status re-enters a queueing state, ensure the rollout is enqueued exactly once.
When status is updated to a finished / problematic state, other states like task
queues will be updated accordingly.
Args:
rollout_id: Identifier of the rollout to update.
input: Replacement task payload; pass `None` to explicitly clear the input.
mode: Replacement rollout mode.
resources_id: Replacement resources snapshot reference.
status: Target rollout status.
config: Replacement retry/timeout configuration.
metadata: Replacement metadata dictionary.
Returns:
The updated rollout record.
Raises:
NotImplementedError: Subclasses must implement mutation logic.
ValueError: Implementations must raise when the rollout is unknown or the update is invalid.
rollout_id: Unique identifier for the rollout to update
input: New input data for the rollout. If set, will be updated. Can be updated to None
mode: New mode for the rollout. If set, will be updated. Can be updated to None
resources_id: New resources ID for the rollout. If set, will be updated. Can be updated to None
status: New status for the rollout. If set, will be updated
config: New config for the rollout. If set, will be updated
metadata: Dictionary of additional metadata to update. If set, will replace the existing metadata
"""
raise NotImplementedError()
@@ -521,78 +248,18 @@ class LightningStore:
last_heartbeat_time: float | Unset = UNSET,
metadata: Optional[Dict[str, Any]] | Unset = UNSET,
) -> Attempt:
"""Update attempt bookkeeping such as status, worker ownership, and heartbeats.
"""
Update a specific or latest attempt for a given rollout.
When `attempt_id` is `"latest"` the update must target the attempt with the highest
`sequence_id`; otherwise it must target the specific attempt. Implementations should
propagate status changes to the rollout (for example via [`propagate_status()`][agentlightning.store.utils.propagate_status])
once the latest attempt transitions to a terminal state.
Update the latest attempt will NOT affect the corresponding rollout status.
Similar to [`update_rollout()`][agentlightning.LightningStore.update_rollout],
parameters also default to the sentinel [`UNSET`][agentlightning.store.base.UNSET].
If `worker_id` is present, the worker status will be updated following the rules:
1. If attempt status is "succeeded" or "failed", the corresponding worker status will be set to "idle".
2. If attempt status is "unresponsive" or "timeout", the corresponding worker status will be set to "unknown".
3. Otherwise, the worker status will be set to "busy".
Args:
rollout_id: Identifier of the rollout whose attempt will be updated.
attempt_id: Attempt identifier or `"latest"` as a convenience.
status: Replacement attempt status. Terminal statuses must set `end_time`.
worker_id: Identifier for the worker currently processing the attempt.
last_heartbeat_time: Wall-clock timestamp (seconds) of the latest heartbeat/span.
metadata: Replacement metadata dictionary.
Returns:
The updated attempt record.
Raises:
NotImplementedError: Subclasses must implement mutation logic.
ValueError: Implementations must raise when the rollout or attempt is unknown.
"""
raise NotImplementedError()
async def query_workers(
self,
) -> List[Worker]:
"""Query all workers in the system.
Returns:
A list of all workers.
"""
raise NotImplementedError()
async def get_worker_by_id(self, worker_id: str) -> Optional[Worker]:
"""Retrieve a single worker by identifier.
Args:
worker_id: Identifier of the worker.
Returns:
The worker record if it exists, otherwise `None`.
Raises:
NotImplementedError: Subclasses must implement lookup semantics.
"""
raise NotImplementedError()
async def update_worker(
self,
worker_id: str,
heartbeat_stats: Dict[str, Any] | Unset = UNSET,
) -> Worker:
"""Record a heartbeat for `worker_id` and refresh telemetry.
Implementations must treat this API as heartbeat-only: it should snapshot
the latest stats when provided, stamp `last_heartbeat_time` with the
current wall clock, and rely on other store mutations (`dequeue_rollout`,
`update_attempt`, etc.) to drive the worker's busy/idle status,
assignment, and activity timestamps.
Args:
worker_id: Identifier of the worker to update.
heartbeat_stats: Replacement worker heartbeat statistics (non-null when provided).
rollout_id: Unique identifier for the rollout
attempt_id: Unique identifier for the attempt
status: Status to set for the attempt, update if provided
worker_id: Worker identifier, update if provided
last_heartbeat_time: Timestamp of the last heartbeat from the worker
metadata: Dictionary of additional metadata to update, will replace the existing metadata
"""
raise NotImplementedError()
File diff suppressed because it is too large Load Diff
+53 -452
View File
@@ -6,32 +6,13 @@ import asyncio
import functools
import hashlib
import logging
import sys
import threading
import time
import uuid
import weakref
from collections import deque
from collections.abc import Iterable
from collections.abc import Mapping as MappingABC
from typing import (
Any,
Callable,
Counter,
Dict,
List,
Literal,
Mapping,
Optional,
Sequence,
Set,
TypeVar,
Union,
cast,
)
from typing import Any, Callable, Counter, Dict, List, Literal, Optional, Sequence, TypeVar, cast
from opentelemetry.sdk.trace import ReadableSpan
from pydantic import BaseModel
from agentlightning.types import (
Attempt,
@@ -44,10 +25,9 @@ from agentlightning.types import (
RolloutStatus,
Span,
TaskInput,
Worker,
)
from .base import UNSET, LightningStore, LightningStoreCapabilities, Unset, is_finished, is_queuing
from .base import UNSET, LightningStore, Unset, is_finished, is_queuing
from .utils import healthcheck, propagate_status
T_callable = TypeVar("T_callable", bound=Callable[..., Any])
@@ -55,61 +35,6 @@ T_callable = TypeVar("T_callable", bound=Callable[..., Any])
logger = logging.getLogger(__name__)
class _LoopAwareAsyncLock:
"""Async lock that transparently rebinds to the current event loop.
The lock intentionally remains *thread-unsafe*: callers must only use it from
one thread at a time. If multiple threads interact with the store, each
thread gets its own event loop specific lock.
"""
def __init__(self) -> None:
self._locks: weakref.WeakKeyDictionary[asyncio.AbstractEventLoop, asyncio.Lock] = weakref.WeakKeyDictionary()
# When serializing and deserializing, we don't need to serialize the locks.
# Because another process will have its own set of event loops and its own lock.
def __getstate__(self) -> dict[str, Any]:
return {}
def __setstate__(self, state: dict[str, Any]) -> None:
self._locks = weakref.WeakKeyDictionary()
def _get_lock_for_current_loop(self) -> asyncio.Lock:
loop = asyncio.get_running_loop()
lock = self._locks.get(loop)
if lock is None:
lock = asyncio.Lock()
self._locks[loop] = lock
return lock
async def __aenter__(self) -> asyncio.Lock:
lock = self._get_lock_for_current_loop()
await lock.acquire()
return lock
async def __aexit__(self, exc_type: type[BaseException] | None, exc: BaseException | None, tb: Any) -> None:
loop = asyncio.get_running_loop()
lock = self._locks.get(loop)
if lock is None or not lock.locked():
raise RuntimeError("Lock released without being acquired")
lock.release()
def estimate_model_size(obj: Any) -> int:
"""Rough recursive size estimate for Pydantic BaseModel instances."""
if isinstance(obj, BaseModel):
values = cast(Iterable[Any], obj.__dict__.values())
return sum(estimate_model_size(value) for value in values) + sys.getsizeof(cast(object, obj))
if isinstance(obj, MappingABC):
mapping = cast(Mapping[Any, Any], obj)
return sum(estimate_model_size(value) for value in mapping.values()) + sys.getsizeof(cast(object, obj))
if isinstance(obj, (list, tuple, set)):
iterable = cast(Iterable[Any], obj)
return sum(estimate_model_size(value) for value in iterable) + sys.getsizeof(cast(object, obj))
return sys.getsizeof(cast(object, obj))
def _healthcheck_wrapper(func: T_callable) -> T_callable:
"""
Decorator to run the watchdog healthcheck **before** executing the decorated method.
@@ -157,19 +82,6 @@ def _generate_attempt_id() -> str:
return "at-" + short_id
def _detect_total_memory_bytes() -> int:
"""Best-effort detection of the total available system memory in bytes."""
try:
import psutil
return int(psutil.virtual_memory().total)
except ImportError:
# Fallback to 8GB if memory cannot be detected.
logger.error("psutil is not installed. Falling back to 8GB of memory in total.")
return 8 * 1024**3
class InMemoryLightningStore(LightningStore):
"""
In-memory implementation of LightningStore using Python data structures.
@@ -177,24 +89,10 @@ class InMemoryLightningStore(LightningStore):
The methods in this class should generally not call each other,
especially those that are locked.
Args:
eviction_memory_threshold: The threshold for evicting spans in bytes.
By default, it's 70% of the total VRAM available.
safe_memory_threshold: The threshold for safe memory usage in bytes.
By default, it's 80% of the eviction threshold.
span_size_estimator: A function to estimate the size of a span in bytes.
By default, it's a simple size estimator that uses sys.getsizeof.
"""
def __init__(
self,
*,
eviction_memory_threshold: float | int | None = None,
safe_memory_threshold: float | int | None = None,
span_size_estimator: Callable[[Span], int] | None = None,
):
self._lock = _LoopAwareAsyncLock()
def __init__(self):
self._lock = asyncio.Lock()
# Task queue and rollouts storage
self._task_queue: deque[Rollout] = deque()
@@ -207,93 +105,12 @@ class InMemoryLightningStore(LightningStore):
# Spans storage
self._spans: Dict[str, List[Span]] = {} # rollout_id -> list of spans
self._span_sequence_ids: Dict[str, int] = Counter() # rollout_id -> sequence_id
self._span_bytes_by_rollout: Dict[str, int] = Counter()
self._total_span_bytes: int = 0
self._evicted_rollout_span_sets: Set[str] = set()
self._memory_capacity_bytes = _detect_total_memory_bytes()
if self._memory_capacity_bytes <= 0:
raise ValueError("Detected memory capacity must be positive")
self._eviction_threshold_bytes = self._resolve_memory_threshold(
eviction_memory_threshold,
default_ratio=0.7,
capacity_bytes=self._memory_capacity_bytes,
name="eviction_memory_threshold",
minimum=1,
)
if safe_memory_threshold is None:
safe_memory_threshold = max(int(self._eviction_threshold_bytes * 0.8), 0)
self._safe_threshold_bytes = self._resolve_memory_threshold(
safe_memory_threshold,
default_ratio=self._eviction_threshold_bytes / self._memory_capacity_bytes,
capacity_bytes=self._memory_capacity_bytes,
name="safe_memory_threshold",
minimum=0,
)
if not (0 <= self._safe_threshold_bytes < self._eviction_threshold_bytes):
raise ValueError("safe_memory_threshold must be smaller than eviction_memory_threshold")
self._custom_span_size_estimator = span_size_estimator
# Attempt tracking
self._attempts: Dict[str, List[Attempt]] = {} # rollout_id -> list of attempts
# Completion tracking for wait_for_rollouts (cross-loop safe)
self._completion_events: Dict[str, threading.Event] = {}
# Worker tracking
self._workers: Dict[str, Worker] = {}
# Running rollouts cache, including preparing and running rollouts
self._running_rollout_ids: Set[str] = set()
def _get_or_create_worker(self, worker_id: str) -> Worker:
worker = self._workers.get(worker_id)
if worker is None:
worker = Worker(worker_id=worker_id)
self._workers[worker_id] = worker
return worker
def _sync_worker_with_attempt(self, attempt: Attempt) -> None:
worker_id = attempt.worker_id
if not worker_id:
return
worker = self._get_or_create_worker(worker_id)
now = time.time()
if attempt.status in ("succeeded", "failed"):
if worker.status != "idle":
worker.last_idle_time = now
worker.status = "idle"
worker.current_rollout_id = None
worker.current_attempt_id = None
elif attempt.status in ("timeout", "unresponsive"):
if worker.status != "unknown":
worker.last_idle_time = now
worker.status = "unknown"
worker.current_rollout_id = None
worker.current_attempt_id = None
else:
transitioned = worker.status != "busy" or worker.current_attempt_id != attempt.attempt_id
if transitioned:
worker.last_busy_time = now
worker.status = "busy"
worker.current_rollout_id = attempt.rollout_id
worker.current_attempt_id = attempt.attempt_id
Worker.model_validate(worker.model_dump())
@property
def capabilities(self) -> LightningStoreCapabilities:
"""Return the capabilities of the store."""
return LightningStoreCapabilities(
thread_safe=False,
async_safe=True,
zero_copy=False,
)
@_healthcheck_wrapper
async def start_rollout(
@@ -301,20 +118,15 @@ class InMemoryLightningStore(LightningStore):
input: TaskInput,
mode: Literal["train", "val", "test"] | None = None,
resources_id: str | None = None,
config: RolloutConfig | None = None,
metadata: Dict[str, Any] | None = None,
) -> AttemptedRollout:
"""Notify the store that I'm about to run a rollout.
See [`LightningStore.start_rollout()`][agentlightning.LightningStore.start_rollout] for semantics.
"""
Notify the store that I'm about to run a rollout.
"""
async with self._lock:
rollout_id = _generate_rollout_id()
current_time = time.time()
rollout_config = config.model_copy(deep=True) if config is not None else RolloutConfig()
rollout_metadata = dict(metadata) if metadata is not None else {}
rollout = Rollout(
rollout_id=rollout_id,
input=input,
@@ -322,10 +134,8 @@ class InMemoryLightningStore(LightningStore):
resources_id=resources_id or self._latest_resources_id,
start_time=current_time,
status="preparing",
config=rollout_config,
metadata=rollout_metadata,
metadata=metadata or {},
)
self._running_rollout_ids.add(rollout.rollout_id)
# Create the initial attempt
attempt_id = _generate_attempt_id()
@@ -351,20 +161,15 @@ class InMemoryLightningStore(LightningStore):
input: TaskInput,
mode: Literal["train", "val", "test"] | None = None,
resources_id: str | None = None,
config: RolloutConfig | None = None,
metadata: Dict[str, Any] | None = None,
) -> Rollout:
"""Adds a new task to the queue with specific metadata and returns the rollout.
See [`LightningStore.enqueue_rollout()`][agentlightning.LightningStore.enqueue_rollout] for semantics.
"""
Adds a new task to the queue with specific metadata and returns its unique ID.
"""
async with self._lock:
rollout_id = _generate_rollout_id()
current_time = time.time()
rollout_config = config.model_copy(deep=True) if config is not None else RolloutConfig()
rollout_metadata = dict(metadata) if metadata is not None else {}
rollout = Rollout(
rollout_id=rollout_id,
input=input,
@@ -372,8 +177,7 @@ class InMemoryLightningStore(LightningStore):
resources_id=resources_id or self._latest_resources_id,
start_time=current_time,
status="queuing", # should be queuing
config=rollout_config,
metadata=rollout_metadata,
metadata=metadata or {},
)
self._rollouts[rollout.rollout_id] = rollout
@@ -383,20 +187,14 @@ class InMemoryLightningStore(LightningStore):
return rollout
@_healthcheck_wrapper
async def dequeue_rollout(self, worker_id: Optional[str] = None) -> Optional[AttemptedRollout]:
"""Retrieves the next task from the queue without blocking.
Returns `None` if the queue is empty.
async def dequeue_rollout(self) -> Optional[AttemptedRollout]:
"""
Retrieves the next task from the queue without blocking.
Returns None if the queue is empty.
Will set the rollout status to preparing and create a new attempt.
See [`LightningStore.dequeue_rollout()`][agentlightning.LightningStore.dequeue_rollout] for semantics.
"""
async with self._lock:
if worker_id is not None:
worker = self._get_or_create_worker(worker_id)
worker.last_dequeue_time = time.time()
worker.status = "idle"
# Keep looking until we find a rollout that's still in queuing status
# or the queue is empty
while self._task_queue:
@@ -407,7 +205,6 @@ class InMemoryLightningStore(LightningStore):
if is_queuing(rollout):
# Update status to preparing
rollout.status = "preparing"
self._running_rollout_ids.add(rollout.rollout_id)
# Create a new attempt (could be first attempt or retry)
attempt_id = _generate_attempt_id()
@@ -429,9 +226,6 @@ class InMemoryLightningStore(LightningStore):
self._attempts[rollout.rollout_id] = []
self._attempts[rollout.rollout_id].append(attempt)
# Sync attempt status to rollout
await self._update_rollout_unlocked(rollout.rollout_id, status="preparing")
return AttemptedRollout(**rollout.model_dump(), attempt=attempt)
# If not in queuing state, skip this rollout and continue
@@ -442,9 +236,8 @@ class InMemoryLightningStore(LightningStore):
@_healthcheck_wrapper
async def start_attempt(self, rollout_id: str) -> AttemptedRollout:
"""Creates a new attempt for a given rollout ID and return the attempt details.
See [`LightningStore.start_attempt()`][agentlightning.LightningStore.start_attempt] for semantics.
"""
Create a new attempt for a given rollout ID and return the attempt details.
"""
async with self._lock:
# Get the rollout
@@ -476,9 +269,6 @@ class InMemoryLightningStore(LightningStore):
self._attempts[rollout_id] = []
self._attempts[rollout_id].append(attempt)
# Sync attempt status to rollout
await self._update_rollout_unlocked(rollout_id, status="preparing")
self._completion_events.setdefault(rollout.rollout_id, threading.Event())
return AttemptedRollout(**rollout.model_dump(), attempt=attempt)
@@ -487,10 +277,9 @@ class InMemoryLightningStore(LightningStore):
async def query_rollouts(
self, *, status: Optional[Sequence[RolloutStatus]] = None, rollout_ids: Optional[Sequence[str]] = None
) -> List[Rollout]:
"""Retrieves rollouts filtered by their status and rollout ids.
"""
Query and retrieve rollouts filtered by their status and rollout ids.
If no status is provided, returns all rollouts.
See [`LightningStore.query_rollouts()`][agentlightning.LightningStore.query_rollouts] for semantics.
"""
async with self._lock:
rollouts = list(self._rollouts.values())
@@ -505,82 +294,44 @@ class InMemoryLightningStore(LightningStore):
status_set = set(status)
rollouts = [rollout for rollout in rollouts if rollout.status in status_set]
# Attach the latest attempt to the rollout objects
rollouts = [self._rollout_to_attempted_rollout_unlocked(rollout) for rollout in rollouts]
return rollouts
@_healthcheck_wrapper
async def get_rollout_by_id(self, rollout_id: str) -> Optional[Union[Rollout, AttemptedRollout]]:
"""Retrieves a specific rollout by its ID.
See [`LightningStore.get_rollout_by_id()`][agentlightning.LightningStore.get_rollout_by_id] for semantics.
If the rollout has been attempted, the latest attempt will also be returned.
async def get_rollout_by_id(self, rollout_id: str) -> Optional[Rollout]:
"""
Safely retrieves a specific rollout by its ID.
"""
async with self._lock:
rollout = self._rollouts.get(rollout_id)
if rollout is None:
return None
return self._rollout_to_attempted_rollout_unlocked(rollout)
def _rollout_to_attempted_rollout_unlocked(self, rollout: Rollout) -> Union[Rollout, AttemptedRollout]:
"""Query the latest attempt for the rollout, and attach it to the rollout object.
If the rollout has no attempts, return the rollout object itself.
"""
latest_attempt = self._get_latest_attempt_unlocked(rollout.rollout_id)
if latest_attempt is None:
return rollout
else:
return AttemptedRollout(**rollout.model_dump(), attempt=latest_attempt)
def _get_latest_attempt_unlocked(self, rollout_id: str) -> Optional[Attempt]:
"""The unlocked version of `get_latest_attempt`."""
attempts = self._attempts.get(rollout_id, [])
return max(attempts, key=lambda a: a.sequence_id) if attempts else None
return self._rollouts.get(rollout_id)
@_healthcheck_wrapper
async def query_attempts(self, rollout_id: str) -> List[Attempt]:
"""Retrieves all attempts associated with a specific rollout ID.
"""
Query and retrieve all attempts associated with a specific rollout ID.
Returns an empty list if no attempts are found.
See [`LightningStore.query_attempts()`][agentlightning.LightningStore.query_attempts] for semantics.
"""
async with self._lock:
return self._attempts.get(rollout_id, [])
@_healthcheck_wrapper
async def get_latest_attempt(self, rollout_id: str) -> Optional[Attempt]:
"""Retrieves the latest attempt for a given rollout ID.
See [`LightningStore.get_latest_attempt()`][agentlightning.LightningStore.get_latest_attempt] for semantics.
"""
Safely retrieves the latest attempt for a given rollout ID.
"""
async with self._lock:
return self._get_latest_attempt_unlocked(rollout_id)
@_healthcheck_wrapper
async def query_resources(self) -> List[ResourcesUpdate]:
"""Return every stored resource snapshot in insertion order."""
async with self._lock:
return list(self._resources.values())
attempts = self._attempts.get(rollout_id, [])
if not attempts:
return None
return max(attempts, key=lambda a: a.sequence_id)
@_healthcheck_wrapper
async def add_resources(self, resources: NamedResources) -> ResourcesUpdate:
"""Stores a new version of named resources and sets it as the latest.
See [`LightningStore.add_resources()`][agentlightning.LightningStore.add_resources] for semantics.
"""
Safely stores a new version of named resources and sets it as the latest.
"""
resources_id = _generate_resources_id()
async with self._lock:
current_time = time.time()
update = ResourcesUpdate(
resources_id=resources_id,
resources=resources,
create_time=current_time,
update_time=current_time,
version=1,
)
update = ResourcesUpdate(resources_id=resources_id, resources=resources)
self._resources[resources_id] = update
self._latest_resources_id = resources_id
return update
@@ -589,45 +340,25 @@ class InMemoryLightningStore(LightningStore):
async def update_resources(self, resources_id: str, resources: NamedResources) -> ResourcesUpdate:
"""
Safely stores a new version of named resources and sets it as the latest.
See [`LightningStore.update_resources()`][agentlightning.LightningStore.update_resources] for semantics.
"""
async with self._lock:
current_time = time.time()
if resources_id not in self._resources:
update = ResourcesUpdate(
resources_id=resources_id,
resources=resources,
create_time=current_time,
update_time=current_time,
version=1,
)
else:
update = self._resources[resources_id].model_copy(
update={
"resources": resources,
"update_time": current_time,
"version": self._resources[resources_id].version + 1,
}
)
update = ResourcesUpdate(resources_id=resources_id, resources=resources)
self._resources[resources_id] = update
self._latest_resources_id = resources_id
return update
@_healthcheck_wrapper
async def get_resources_by_id(self, resources_id: str) -> Optional[ResourcesUpdate]:
"""Retrieves a specific version of named resources by its ID.
See [`LightningStore.get_resources_by_id()`][agentlightning.LightningStore.get_resources_by_id] for semantics.
"""
Safely retrieves a specific version of named resources by its ID.
"""
async with self._lock:
return self._resources.get(resources_id)
@_healthcheck_wrapper
async def get_latest_resources(self) -> Optional[ResourcesUpdate]:
"""Retrieves the latest version of named resources.
See [`LightningStore.get_latest_resources()`][agentlightning.LightningStore.get_latest_resources] for semantics.
"""
Safely retrieves the latest version of named resources.
"""
async with self._lock:
if self._latest_resources_id:
@@ -635,21 +366,17 @@ class InMemoryLightningStore(LightningStore):
return None
async def get_next_span_sequence_id(self, rollout_id: str, attempt_id: str) -> int:
"""Get the next span sequence ID for a given rollout and attempt.
"""
Get the next span sequence ID for a given rollout and attempt.
The number is strictly increasing for each rollout.
The store will not issue the same sequence ID twice.
See [`LightningStore.get_next_span_sequence_id()`][agentlightning.LightningStore.get_next_span_sequence_id] for semantics.
"""
async with self._lock:
self._span_sequence_ids[rollout_id] += 1
return self._span_sequence_ids[rollout_id]
async def add_span(self, span: Span) -> Span:
"""Persist a pre-converted span.
See [`LightningStore.add_span()`][agentlightning.LightningStore.add_span] for semantics.
"""
"""Persist a pre-converted span."""
async with self._lock:
self._span_sequence_ids[span.rollout_id] = max(self._span_sequence_ids[span.rollout_id], span.sequence_id)
return await self._add_span_unlocked(span)
@@ -657,10 +384,7 @@ class InMemoryLightningStore(LightningStore):
async def add_otel_span(
self, rollout_id: str, attempt_id: str, readable_span: ReadableSpan, sequence_id: int | None = None
) -> Span:
"""Add an opentelemetry span to the store.
See [`LightningStore.add_otel_span()`][agentlightning.LightningStore.add_otel_span] for semantics.
"""
"""Add an opentelemetry span to the store."""
async with self._lock:
if sequence_id is None:
# Issue a new sequence ID for the rollout
@@ -692,12 +416,10 @@ class InMemoryLightningStore(LightningStore):
if span.rollout_id not in self._spans:
self._spans[span.rollout_id] = []
self._spans[span.rollout_id].append(span)
self._account_span_size(span)
self._maybe_evict_spans()
# Update attempt heartbeat
current_attempt.last_heartbeat_time = time.time()
if current_attempt.status in ["preparing", "unresponsive"]:
if current_attempt.status in ["preparing", "unresponsive", "timeout"]:
current_attempt.status = "running"
# If the status has already timed out or failed, do not change it
@@ -706,7 +428,6 @@ class InMemoryLightningStore(LightningStore):
if current_attempt == latest_attempt:
if rollout.status == "preparing":
rollout.status = "running"
self._running_rollout_ids.add(rollout.rollout_id)
elif rollout.status in ["queuing", "requeuing"]:
try:
self._task_queue.remove(rollout)
@@ -715,89 +436,16 @@ class InMemoryLightningStore(LightningStore):
f"Trying to remove rollout {rollout.rollout_id} from the queue but it's not in the queue."
)
rollout.status = "running"
self._running_rollout_ids.add(rollout.rollout_id)
return span
@staticmethod
def _resolve_memory_threshold(
value: float | int | None,
*,
default_ratio: float,
capacity_bytes: int,
name: str,
minimum: int,
) -> int:
if value is None:
resolved = int(capacity_bytes * default_ratio)
elif isinstance(value, float):
if minimum == 0:
if not (0 <= value <= 1):
raise ValueError(f"{name} ratio must be between 0 and 1 inclusive")
else:
if not (0 < value <= 1):
raise ValueError(f"{name} ratio must be greater than 0 and at most 1")
resolved = int(capacity_bytes * value)
else:
value_int = value
if value_int < 0:
raise ValueError(f"{name} must be non-negative")
resolved = value_int
if resolved < minimum:
raise ValueError(f"{name} must be at least {minimum} bytes")
return resolved
def _account_span_size(self, span: Span) -> int:
if self._custom_span_size_estimator is not None:
size = max(int(self._custom_span_size_estimator(span)), 0)
else:
size = estimate_model_size(span)
self._span_bytes_by_rollout[span.rollout_id] += size
self._total_span_bytes += size
return size
def _maybe_evict_spans(self) -> None:
if self._total_span_bytes <= self._eviction_threshold_bytes:
return
candidates: List[tuple[float, str]] = []
for rollout_id, spans in self._spans.items():
if not spans:
continue
rollout = self._rollouts.get(rollout_id)
start_time = rollout.start_time if rollout is not None else (spans[0].start_time or 0.0)
candidates.append((start_time, rollout_id))
candidates.sort(key=lambda item: item[0])
logger.info(f"Evicting spans for {len(candidates)} rollouts to free up memory...")
memory_consumed_before = self._total_span_bytes
for _, rollout_id in candidates:
if self._total_span_bytes <= self._safe_threshold_bytes:
break
logger.debug(f"Evicting spans for rollout {rollout_id} to free up memory...")
self._evict_spans_for_rollout(rollout_id)
logger.info(f"Freed up {memory_consumed_before - self._total_span_bytes} bytes of memory")
def _evict_spans_for_rollout(self, rollout_id: str) -> None:
spans = self._spans.pop(rollout_id, [])
if not spans:
return
removed_bytes = self._span_bytes_by_rollout.pop(rollout_id, 0)
self._total_span_bytes = max(self._total_span_bytes - removed_bytes, 0)
self._evicted_rollout_span_sets.add(rollout_id)
@_healthcheck_wrapper
async def wait_for_rollouts(self, *, rollout_ids: List[str], timeout: Optional[float] = None) -> List[Rollout]:
"""Wait for specified rollouts to complete with a timeout.
"""
Wait for specified rollouts to complete with a timeout.
Returns the completed rollouts, potentially incomplete if timeout is reached.
This method does not change the state of the store.
See [`LightningStore.wait_for_rollouts()`][agentlightning.LightningStore.wait_for_rollouts] for semantics.
"""
completed_rollouts: List[Rollout] = []
@@ -850,12 +498,8 @@ class InMemoryLightningStore(LightningStore):
"""
Query and retrieve all spans associated with a specific rollout ID.
Returns an empty list if no spans are found.
See [`LightningStore.query_spans()`][agentlightning.LightningStore.query_spans] for semantics.
"""
async with self._lock:
if rollout_id in self._evicted_rollout_span_sets:
raise RuntimeError(f"Spans for rollout {rollout_id} have been evicted")
spans = self._spans.get(rollout_id, [])
if attempt_id is None:
return spans
@@ -879,9 +523,8 @@ class InMemoryLightningStore(LightningStore):
config: RolloutConfig | Unset = UNSET,
metadata: Optional[Dict[str, Any]] | Unset = UNSET,
) -> Rollout:
"""Update the rollout status and related metadata.
See [`LightningStore.update_rollout()`][agentlightning.LightningStore.update_rollout] for semantics.
"""
Update the rollout status and related metadata.
"""
async with self._lock:
return await self._update_rollout_unlocked(
@@ -904,9 +547,8 @@ class InMemoryLightningStore(LightningStore):
last_heartbeat_time: float | Unset = UNSET,
metadata: Optional[Dict[str, Any]] | Unset = UNSET,
) -> Attempt:
"""Update a specific or latest attempt for a given rollout.
See [`LightningStore.update_attempt()`][agentlightning.LightningStore.update_attempt] for semantics.
"""
Update a specific or latest attempt for a given rollout.
"""
async with self._lock:
attempt = await self._update_attempt_unlocked(
@@ -961,12 +603,6 @@ class InMemoryLightningStore(LightningStore):
elif is_queuing(rollout) and rollout not in self._task_queue:
self._task_queue.append(rollout)
# Updating running rollouts cache
if rollout.status in ["preparing", "running"]:
self._running_rollout_ids.add(rollout.rollout_id)
else:
self._running_rollout_ids.discard(rollout.rollout_id)
# If the rollout is no longer in a queueing state, remove it from the queue.
if not isinstance(status, Unset) and not is_queuing(rollout) and rollout in self._task_queue:
try:
@@ -1010,26 +646,19 @@ class InMemoryLightningStore(LightningStore):
if not attempt:
raise ValueError(f"Attempt {attempt_id} not found for rollout {rollout_id}")
worker_sync_required = False
# Update fields if they are not UNSET
if not isinstance(worker_id, Unset):
attempt.worker_id = worker_id
worker_sync_required = worker_sync_required or bool(worker_id)
if not isinstance(status, Unset):
attempt.status = status
# Also update end_time if the status indicates completion
if status in ["failed", "succeeded"]:
attempt.end_time = time.time()
worker_sync_required = worker_sync_required or bool(attempt.worker_id)
if not isinstance(worker_id, Unset):
attempt.worker_id = worker_id
if not isinstance(last_heartbeat_time, Unset):
attempt.last_heartbeat_time = last_heartbeat_time
if not isinstance(metadata, Unset):
attempt.metadata = metadata
if worker_sync_required and attempt.worker_id:
self._sync_worker_with_attempt(attempt)
# Re-validate the attempt to ensure legality
Attempt.model_validate(attempt.model_dump())
@@ -1047,40 +676,12 @@ class InMemoryLightningStore(LightningStore):
return attempt
@_healthcheck_wrapper
async def query_workers(self) -> List[Worker]:
"""Return the current snapshot of all workers."""
async with self._lock:
return list(self._workers.values())
@_healthcheck_wrapper
async def get_worker_by_id(self, worker_id: str) -> Optional[Worker]:
async with self._lock:
return self._workers.get(worker_id)
@_healthcheck_wrapper
async def update_worker(
self,
worker_id: str,
heartbeat_stats: Dict[str, Any] | Unset = UNSET,
) -> Worker:
"""Create or update a worker entry."""
async with self._lock:
worker = self._get_or_create_worker(worker_id)
if not isinstance(heartbeat_stats, Unset):
worker.heartbeat_stats = dict(heartbeat_stats)
worker.last_heartbeat_time = time.time()
Worker.model_validate(worker.model_dump())
return worker
async def _healthcheck(self) -> None:
"""Perform healthcheck against all running rollouts in the store."""
async with self._lock:
running_rollouts: List[AttemptedRollout] = []
for rollout_id in self._running_rollout_ids:
rollout = self._rollouts.get(rollout_id)
if rollout is not None and rollout.status in ["preparing", "running"]:
for rollout in self._rollouts.values():
if rollout.status in ["preparing", "running"]:
all_attempts = self._attempts.get(rollout.rollout_id, [])
if not all_attempts:
# The rollout is running but has no attempts, this should not happen
+5 -37
View File
@@ -18,10 +18,9 @@ from agentlightning.types import (
RolloutStatus,
Span,
TaskInput,
Worker,
)
from .base import UNSET, LightningStore, LightningStoreCapabilities, Unset
from .base import UNSET, LightningStore, Unset
class LightningStoreThreaded(LightningStore):
@@ -36,41 +35,29 @@ class LightningStoreThreaded(LightningStore):
self.store = store
self._lock = threading.Lock()
@property
def capabilities(self) -> LightningStoreCapabilities:
"""Return the capabilities of the store."""
capabilities = self.store.capabilities
return {
**capabilities,
"async_safe": True,
"thread_safe": True,
}
async def start_rollout(
self,
input: TaskInput,
mode: Literal["train", "val", "test"] | None = None,
resources_id: str | None = None,
config: RolloutConfig | None = None,
metadata: Dict[str, Any] | None = None,
) -> AttemptedRollout:
with self._lock:
return await self.store.start_rollout(input, mode, resources_id, config, metadata)
return await self.store.start_rollout(input, mode, resources_id, metadata)
async def enqueue_rollout(
self,
input: TaskInput,
mode: Literal["train", "val", "test"] | None = None,
resources_id: str | None = None,
config: RolloutConfig | None = None,
metadata: Dict[str, Any] | None = None,
) -> Rollout:
with self._lock:
return await self.store.enqueue_rollout(input, mode, resources_id, config, metadata)
return await self.store.enqueue_rollout(input, mode, resources_id, metadata)
async def dequeue_rollout(self, worker_id: Optional[str] = None) -> Optional[AttemptedRollout]:
async def dequeue_rollout(self) -> Optional[AttemptedRollout]:
with self._lock:
return await self.store.dequeue_rollout(worker_id=worker_id)
return await self.store.dequeue_rollout()
async def start_attempt(self, rollout_id: str) -> AttemptedRollout:
with self._lock:
@@ -182,22 +169,3 @@ class LightningStoreThreaded(LightningStore):
last_heartbeat_time=last_heartbeat_time,
metadata=metadata,
)
async def query_workers(self) -> List[Worker]:
with self._lock:
return await self.store.query_workers()
async def get_worker_by_id(self, worker_id: str) -> Optional[Worker]:
with self._lock:
return await self.store.get_worker_by_id(worker_id)
async def update_worker(
self,
worker_id: str,
heartbeat_stats: Dict[str, Any] | Unset = UNSET,
) -> Worker:
with self._lock:
return await self.store.update_worker(
worker_id=worker_id,
heartbeat_stats=heartbeat_stats,
)
-1
View File
@@ -57,7 +57,6 @@ async def healthcheck(
Perform health check on all running rollouts in the store.
This method should be called periodically to:
1. Update rollout status to failed to succeeded when the attempt is done
2. Check for unresponsive attempts (no heartbeat or spans for a while)
3. Check for timed-out rollouts (running too long since start_time)
+2 -2
View File
@@ -1,7 +1,7 @@
# Copyright (c) Microsoft. All rights reserved.
from .agentops import AgentOpsTracer
from .base import Tracer
from .base import BaseTracer
from .otel import OtelTracer
__all__ = ["AgentOpsTracer", "Tracer", "OtelTracer"]
__all__ = ["AgentOpsTracer", "BaseTracer", "OtelTracer"]
+131 -56
View File
@@ -6,21 +6,20 @@ import asyncio
import logging
import os
import threading
from contextlib import asynccontextmanager, contextmanager
from typing import TYPE_CHECKING, Any, AsyncGenerator, Awaitable, Iterator, List, Optional
from contextlib import contextmanager
from typing import TYPE_CHECKING, Any, Awaitable, Iterator, List, Optional
import agentops
import agentops.sdk.core
from agentops.sdk.core import TracingCore
from agentops.sdk.processors import SpanProcessor
from opentelemetry.instrumentation.utils import suppress_instrumentation
from opentelemetry.sdk.trace import ReadableSpan
from opentelemetry.trace.status import StatusCode
from agentlightning.instrumentation import instrument_all, uninstrument_all
from agentlightning.instrumentation.agentops import AgentOpsServerManager
from agentlightning.store.base import LightningStore
from .base import Tracer
from .base import BaseTracer
if TYPE_CHECKING:
from agentops.integration.callbacks.langchain import LangchainCallbackHandler
@@ -29,7 +28,7 @@ if TYPE_CHECKING:
logger = logging.getLogger(__name__)
class AgentOpsTracer(Tracer):
class AgentOpsTracer(BaseTracer):
"""Traces agent execution using AgentOps.
This tracer provides functionality to capture execution details using the
@@ -56,11 +55,46 @@ class AgentOpsTracer(Tracer):
self.instrument_managed = instrument_managed
self.daemon = daemon
self._agentops_server_manager = AgentOpsServerManager(self.daemon)
self._agentops_server_port_val: Optional[int] = None
if not self.agentops_managed:
logger.warning("agentops_managed=False. You are responsible for AgentOps setup.")
if not self.instrument_managed:
logger.warning("instrument_managed=False. You are responsible for all instrumentation.")
def __getstate__(self):
state = self.__dict__.copy()
state["_agentops_server_manager"] = None # Exclude the unpicklable server manager
# _agentops_server_port_val (int) is inherently picklable and will be included.
logger.debug(f"Getting state for pickling Trainer (PID {os.getpid()}). _agentops_server_manager excluded.")
return state
def __setstate__(self, state: Any):
self.__dict__.update(state)
# In child process, self._agentops_server_manager will be None.
logger.debug(f"Setting state for unpickled Trainer (PID {os.getpid()}). _agentops_server_manager is None.")
def init(self, *args: Any, **kwargs: Any):
if self.agentops_managed and self._agentops_server_manager:
self._agentops_server_manager.start()
self._agentops_server_port_val = self._agentops_server_manager.get_port()
if self._agentops_server_port_val is None:
if (
self._agentops_server_manager.server_process is not None
and self._agentops_server_manager.server_process.is_alive()
):
raise RuntimeError("AgentOps server started but port is None. Check server manager logic.")
elif (
self._agentops_server_port_val is None and self._agentops_server_manager.server_process is None
): # Server failed to start
raise RuntimeError("AgentOps server manager indicates server is not running and port is None.")
def teardown(self):
if self.agentops_managed:
self._agentops_server_manager.stop()
logger.info("AgentOps server stopped.")
def instrument(self, worker_id: int):
instrument_all()
@@ -76,9 +110,24 @@ class AgentOpsTracer(Tracer):
logger.info(f"[Worker {worker_id}] Instrumentation applied.")
if self.agentops_managed:
os.environ.setdefault("AGENTOPS_API_KEY", "dummy")
if self._agentops_server_port_val: # Use the stored, picklable port value
base_url = f"http://localhost:{self._agentops_server_port_val}"
env_vars_to_set = {
"AGENTOPS_API_KEY": "dummy",
"AGENTOPS_API_ENDPOINT": base_url,
"AGENTOPS_APP_URL": f"{base_url}/notavailable",
"AGENTOPS_EXPORTER_ENDPOINT": f"{base_url}/traces",
}
for key, value in env_vars_to_set.items():
os.environ[key] = value
logger.info(f"[Worker {worker_id}] Env var set: {key}={value}")
else:
logger.warning(
f"[Worker {worker_id}] AgentOps managed, but local server port is not available. Client may not connect as expected."
)
if not agentops.get_client().initialized:
agentops.init(auto_start_session=False) # type: ignore
agentops.init() # type: ignore
logger.info(f"[Worker {worker_id}] AgentOps client initialized.")
else:
logger.warning(f"[Worker {worker_id}] AgentOps client was already initialized.")
@@ -103,15 +152,15 @@ class AgentOpsTracer(Tracer):
self.uninstrument(worker_id)
logger.info(f"[Worker {worker_id}] Instrumentation removed.")
@asynccontextmanager
async def trace_context(
@contextmanager
def trace_context(
self,
name: Optional[str] = None,
*,
store: Optional[LightningStore] = None,
rollout_id: Optional[str] = None,
attempt_id: Optional[str] = None,
) -> AsyncGenerator[LightningSpanProcessor, None]:
) -> Iterator[LightningSpanProcessor]:
"""
Starts a new tracing context. This should be used as a context manager.
@@ -122,50 +171,20 @@ class AgentOpsTracer(Tracer):
attempt_id: Optional attempt ID to add the spans to.
Yields:
The [`LightningSpanProcessor`][agentlightning.tracer.agentops.LightningSpanProcessor] instance to collect spans.
The LightningSpanProcessor instance to collect spans.
"""
with self._trace_context_sync(
name=name, store=store, rollout_id=rollout_id, attempt_id=attempt_id
) as processor:
yield processor
@contextmanager
def _trace_context_sync(
self,
name: Optional[str] = None,
*,
store: Optional[LightningStore] = None,
rollout_id: Optional[str] = None,
attempt_id: Optional[str] = None,
) -> Iterator[LightningSpanProcessor]:
"""Implementation of `trace_context` for synchronous execution."""
if not self._lightning_span_processor:
raise RuntimeError("LightningSpanProcessor is not initialized. Call init_worker() first.")
kwargs: dict[str, Any] = {}
if name is not None:
kwargs["trace_name"] = name
elif rollout_id is not None:
kwargs["trace_name"] = rollout_id
trace = agentops.start_trace(**kwargs)
status = StatusCode.OK # type: ignore
try:
if store is not None and rollout_id is not None and attempt_id is not None:
ctx = self._lightning_span_processor.with_context(
store=store, rollout_id=rollout_id, attempt_id=attempt_id
)
with ctx as processor:
yield processor
elif store is None and rollout_id is None and attempt_id is None:
with self._lightning_span_processor:
yield self._lightning_span_processor
else:
raise ValueError("store, rollout_id, and attempt_id must be either all provided or all None")
except Exception as e:
status = StatusCode.ERROR # type: ignore
logger.error(f"Trace failed for rollout_id={rollout_id}, attempt_id={attempt_id}, error={e}")
finally:
agentops.end_trace(trace, end_state=status) # type: ignore
if store is not None and rollout_id is not None and attempt_id is not None:
ctx = self._lightning_span_processor.with_context(store=store, rollout_id=rollout_id, attempt_id=attempt_id)
with ctx as processor:
yield processor
elif store is None and rollout_id is None and attempt_id is None:
with self._lightning_span_processor:
yield self._lightning_span_processor
else:
raise ValueError("store, rollout_id, and attempt_id must be either all provided or all None")
def get_last_trace(self) -> List[ReadableSpan]:
"""
@@ -205,11 +224,42 @@ class AgentOpsTracer(Tracer):
get_langchain_callback_handler = get_langchain_handler # alias
class LightningSpanProcessor(SpanProcessor):
"""Span processor that subclasses OpenTelemetry's `SpanProcessor` and adds support to dump traces
to a [`LightningStore`][agentlightning.LightningStore].
"""
async def heartbeat(name="exporter-loop", period=0.5):
import asyncio
import time
last = time.perf_counter()
while True:
await asyncio.sleep(period)
now = time.perf_counter()
dt = now - last
last = now
if dt > period * 4: # e.g., >2s if period=0.5s
print("!!!!!!! [%s] loop stall detected: slept %.3fs (expected %.3fs)" % (name, dt, period))
import asyncio
import logging
# logging.basicConfig(level=logging.DEBUG)
# asyncio.get_event_loop().set_debug(True)
import time
def debug_dump(loop):
while True:
try:
print("=== Pending tasks ===")
for t in asyncio.all_tasks(loop):
if not t.done():
print(t, "awaiting", t.get_coro())
t.print_stack()
except Exception:
pass
time.sleep(5)
class LightningSpanProcessor(SpanProcessor):
def __init__(self):
self._spans: List[ReadableSpan] = []
@@ -229,8 +279,13 @@ class LightningSpanProcessor(SpanProcessor):
def _loop_runner(self):
loop = asyncio.new_event_loop()
self._loop = loop
self._loop.set_debug(True)
asyncio.set_event_loop(loop)
self._loop_ready.set()
thread = threading.Thread(target=debug_dump, args=(loop,), daemon=True)
thread.start()
# asyncio.create_task(heartbeat())
loop.run_forever()
loop.close()
@@ -319,6 +374,10 @@ class LightningSpanProcessor(SpanProcessor):
Args:
span: The span that has ended.
"""
import traceback
# print("ON_END")
# print(traceback.format_stack())
# Skip if span is not sampled
if not span.context or not span.context.trace_flags.sampled:
return
@@ -326,11 +385,27 @@ class LightningSpanProcessor(SpanProcessor):
if self._store and self._rollout_id and self._attempt_id:
try:
# Submit add_otel_span to the event loop and wait for it to complete
print("!!! before,")
print("Ready callbacks:", self._loop._ready)
print("Scheduled callbacks:", len(self._loop._scheduled))
if self._loop._scheduled:
print("First in the queue:", self._loop._scheduled[0])
print("..... Current thread: ", threading.current_thread())
print("..... Loop thread: ", self._loop_thread)
if self._loop_thread.ident == threading.current_thread().ident:
traceback.print_stack()
print("Span content: ", span.attributes)
from opentelemetry.instrumentation.utils import suppress_instrumentation
with suppress_instrumentation():
self._await_in_loop(
self._store.add_otel_span(self._rollout_id, self._attempt_id, span),
timeout=60.0,
timeout=30.0,
)
print("!!! after,")
print("All tasks")
print("Ready callbacks:", self._loop._ready)
print("Scheduled callbacks:", self._loop._scheduled)
except Exception:
# log; on_end MUST NOT raise
logger.exception(f"Error adding span to store: {span.name}")
+13 -26
View File
@@ -3,7 +3,8 @@
from __future__ import annotations
import logging
from typing import TYPE_CHECKING, Any, AsyncContextManager, Awaitable, Callable, ContextManager, List, Optional
from contextlib import contextmanager
from typing import TYPE_CHECKING, Any, Awaitable, Callable, Iterator, List, Optional
from opentelemetry.sdk.trace import ReadableSpan
@@ -11,12 +12,12 @@ from agentlightning.store.base import LightningStore
from agentlightning.types import ParallelWorkerBase
if TYPE_CHECKING:
from langchain_core.callbacks.base import BaseCallbackHandler # type: ignore
from langchain.callbacks.base import BaseCallbackHandler
logger = logging.getLogger(__name__)
class Tracer(ParallelWorkerBase):
class BaseTracer(ParallelWorkerBase):
"""
An abstract base class for tracers.
@@ -25,7 +26,7 @@ class Tracer(ParallelWorkerBase):
designed to be backend-agnostic, allowing for different implementations
(e.g., for AgentOps, OpenTelemetry, Docker, etc.).
The primary interaction pattern is through the [`trace_context`][agentlightning.Tracer.trace_context]
The primary interaction pattern is through the `trace_context`
context manager, which ensures that traces are properly started and captured,
even in the case of exceptions.
@@ -35,9 +36,9 @@ class Tracer(ParallelWorkerBase):
tracer = YourTracerImplementation()
try:
async with tracer.trace_context(name="my_traced_task"):
with tracer.trace_context(name="my_traced_task"):
# ... code to be traced ...
await run_my_agent_logic()
run_my_agent_logic()
except Exception as e:
print(f"An error occurred: {e}")
@@ -51,6 +52,7 @@ class Tracer(ParallelWorkerBase):
```
"""
@contextmanager
def trace_context(
self,
name: Optional[str] = None,
@@ -58,14 +60,14 @@ class Tracer(ParallelWorkerBase):
store: Optional[LightningStore] = None,
rollout_id: Optional[str] = None,
attempt_id: Optional[str] = None,
) -> AsyncContextManager[Any]:
) -> Iterator[Any]:
"""
Starts a new tracing context. This should be used as a context manager.
The implementation should handle the setup and teardown of the tracing
for the enclosed code block. It must ensure that any spans generated
within the `with` block are collected and made available via
[`get_last_trace`][agentlightning.Tracer.get_last_trace].
`get_last_trace`.
If a store is provided, the spans will be added to the store when tracing.
@@ -77,17 +79,6 @@ class Tracer(ParallelWorkerBase):
"""
raise NotImplementedError()
def _trace_context_sync(
self,
name: Optional[str] = None,
*,
store: Optional[LightningStore] = None,
rollout_id: Optional[str] = None,
attempt_id: Optional[str] = None,
) -> ContextManager[Any]:
"""Internal API for CI backward compatibility."""
raise NotImplementedError()
def get_last_trace(self) -> List[ReadableSpan]:
"""
Retrieves the raw list of captured spans from the most recent trace.
@@ -101,8 +92,6 @@ class Tracer(ParallelWorkerBase):
"""
A convenience wrapper to trace the execution of a single synchronous function.
Deprecated in favor of customizing Runners.
Args:
func: The synchronous function to execute and trace.
*args: Positional arguments to pass to the function.
@@ -111,15 +100,13 @@ class Tracer(ParallelWorkerBase):
Returns:
The return value of the function.
"""
with self._trace_context_sync(name=func.__name__):
with self.trace_context(name=func.__name__):
return func(*args, **kwargs)
async def trace_run_async(self, func: Callable[..., Awaitable[Any]], *args: Any, **kwargs: Any) -> Any:
"""
A convenience wrapper to trace the execution of a single asynchronous function.
Deprecated in favor of customizing Runners.
Args:
func: The asynchronous function to execute and trace.
*args: Positional arguments to pass to the function.
@@ -128,10 +115,10 @@ class Tracer(ParallelWorkerBase):
Returns:
The return value of the function.
"""
async with self.trace_context(name=func.__name__):
with self.trace_context(name=func.__name__):
return await func(*args, **kwargs)
def get_langchain_handler(self) -> Optional[BaseCallbackHandler]: # type: ignore
def get_langchain_handler(self) -> Optional[BaseCallbackHandler]:
"""Get a handler to install in langchain agent callback.
Agents are expected to use this handler in their agents to enable tracing.
+5 -16
View File
@@ -5,8 +5,8 @@ import logging
import multiprocessing
import queue
import uuid
from contextlib import asynccontextmanager, contextmanager
from typing import Any, AsyncGenerator, Awaitable, Callable, Dict, Iterator, List, Optional, Tuple
from contextlib import contextmanager
from typing import Any, Awaitable, Callable, Dict, Iterator, List, Optional, Tuple
from urllib.parse import urlparse
from httpdbg.hooks.all import httprecord
@@ -19,12 +19,12 @@ from opentelemetry.trace.span import (
TraceState,
)
from .base import Tracer
from .base import BaseTracer
logger = logging.getLogger(__name__)
class HttpTracer(Tracer):
class HttpTracer(BaseTracer):
"""
A tracer implementation that captures HTTP requests using httpdbg.
@@ -78,19 +78,8 @@ class HttpTracer(Tracer):
super().init_worker(worker_id)
logger.info(f"[Worker {worker_id}] HttpTracer initialized.")
@asynccontextmanager
async def trace_context(self, name: Optional[str] = None, **kwargs: Any) -> AsyncGenerator[HTTPRecords, None]:
"""
Starts a new HTTP tracing context. This should be used as a context manager.
Args:
name: Optional name for the tracing context.
"""
with self._trace_context_sync(name=name, **kwargs) as records:
yield records
@contextmanager
def _trace_context_sync(self, name: Optional[str] = None, **kwargs: Any) -> Iterator[HTTPRecords]:
def trace_context(self, name: Optional[str] = None, **kwargs: Any) -> Iterator[HTTPRecords]:
"""
Starts a new HTTP tracing context. This should be used as a context manager.
+7 -7
View File
@@ -3,8 +3,8 @@
from __future__ import annotations
import logging
from contextlib import asynccontextmanager
from typing import AsyncGenerator, List, Optional
from contextlib import contextmanager
from typing import Iterator, List, Optional
import opentelemetry.trace as trace_api
from opentelemetry.sdk.trace import ReadableSpan, TracerProvider
@@ -12,12 +12,12 @@ from opentelemetry.sdk.trace import ReadableSpan, TracerProvider
from agentlightning.store.base import LightningStore
from .agentops import LightningSpanProcessor # FIXME: This import should be from otel to agentops
from .base import Tracer
from .base import BaseTracer
logger = logging.getLogger(__name__)
class OtelTracer(Tracer):
class OtelTracer(BaseTracer):
"""Tracer that provides a basic OpenTelemetry tracer provider.
You should be able to collect agent-lightning signals like rewards with this tracer,
@@ -49,15 +49,15 @@ class OtelTracer(Tracer):
logger.info(f"[Worker {worker_id}] Tearing down OpenTelemetry tracer...")
self._tracer_provider = None
@asynccontextmanager
async def trace_context(
@contextmanager
def trace_context(
self,
name: Optional[str] = None,
*,
store: Optional[LightningStore] = None,
rollout_id: Optional[str] = None,
attempt_id: Optional[str] = None,
) -> AsyncGenerator[LightningSpanProcessor, None]:
) -> Iterator[LightningSpanProcessor]:
"""
Starts a new tracing context. This should be used as a context manager.
+4 -4
View File
@@ -9,11 +9,11 @@ import warnings
from typing import Any, List, Optional, TypeVar, Union
from agentlightning.adapter import TraceAdapter, TracerTraceToTriplet
from agentlightning.algorithm import Algorithm
from agentlightning.algorithm import BaseAlgorithm
from agentlightning.client import AgentLightningClient
from agentlightning.litagent import LitAgent
from agentlightning.runner import LegacyAgentRunner
from agentlightning.tracer.base import Tracer
from agentlightning.tracer.base import BaseTracer
from agentlightning.types import Dataset, ParallelWorkerBase
logger = logging.getLogger(__name__)
@@ -31,8 +31,8 @@ class TrainerLegacy(ParallelWorkerBase):
It won't be used in practice.
"""
self._dev = kwargs.pop("dev", False)
self.algorithm: Optional[Algorithm] = kwargs.pop("algorithm", None)
self.tracer: Tracer = kwargs.pop("tracer", None)
self.algorithm: Optional[BaseAlgorithm] = kwargs.pop("algorithm", None)
self.tracer: BaseTracer = kwargs.pop("tracer", None)
self.n_workers: int = kwargs.pop("n_workers", None)
self.max_tasks: Optional[int] = kwargs.pop("max_tasks", None)
self.daemon: bool = kwargs.pop("daemon", True)
+68 -186
View File
@@ -7,18 +7,18 @@ import warnings
from typing import Any, Callable, Dict, Optional, Sequence, TypeVar, Union
from agentlightning.adapter import TraceAdapter, TracerTraceToTriplet
from agentlightning.algorithm import Algorithm, Baseline, FastAlgorithm
from agentlightning.algorithm import BaseAlgorithm, Baseline, FastAlgorithm
from agentlightning.client import AgentLightningClient
from agentlightning.execution.base import ExecutionStrategy
from agentlightning.execution.client_server import ClientServerExecutionStrategy
from agentlightning.execution.events import ExecutionEvent
from agentlightning.litagent import LitAgent
from agentlightning.llm_proxy import LLMProxy
from agentlightning.runner import LitAgentRunner, Runner
from agentlightning.runner import BaseRunner, LitAgentRunner
from agentlightning.store.base import LightningStore
from agentlightning.store.memory import InMemoryLightningStore
from agentlightning.tracer.agentops import AgentOpsTracer
from agentlightning.tracer.base import Tracer
from agentlightning.tracer.base import BaseTracer
from agentlightning.types import Dataset, Hook, NamedResources
from .init_utils import build_component, instantiate_component
@@ -34,89 +34,44 @@ ComponentSpec = Union[T, type[T], Callable[[], T], str, Dict[str, Any], None]
class Trainer(TrainerLegacy):
"""High-level orchestration layer that wires Algorithm <-> Runner <-> Store.
"""Orchestrates the distributed execution of agent rollouts.
A [`Trainer`][agentlightning.Trainer] packages the moving parts of Agent-Lightning's
training loop into a single entry point:
The Trainer is responsible for launching one or more worker processes
that run the agent's execution loop. It manages multiprocessing,
handles graceful shutdown, and serves as the main entry point for
running a client-side agent fleet.
* **Algorithm lifecycle:** Instantiates or accepts an [`Algorithm`][agentlightning.Algorithm],
attaches the current [`LightningStore`][agentlightning.LightningStore], adapter, and
initial resources, then executes the algorithm role inside the configured execution strategy.
* **Runner fleet:** Spawns one or more [`Runner`][agentlightning.Runner] instances (defaulting
to [`LitAgentRunner`][agentlightning.LitAgentRunner]) that hydrate a [`LitAgent`][agentlightning.LitAgent],
claim rollouts, stream spans, and respect graceful termination signals from the execution strategy.
* **Execution strategy:** Delegates process management to an
[`ExecutionStrategy`][agentlightning.ExecutionStrategy] (shared memory, client/server, etc.),
so advanced users can swap orchestration backends without changing trainer code.
* **Telemetry plumbing:** Ensures tracers, adapters, and optional [`LLMProxy`][agentlightning.LLMProxy]
are wired into both algorithm and runners so telemetry flows back into the store.
The trainer exposes two convenience entry points:
[`fit()`][agentlightning.Trainer.fit] for full training and
[`dev()`][agentlightning.Trainer.dev] for fast, reproducible dry-runs. See the
[Train the First Agent](../how-to/train-first-agent.md) and
[Write the First Algorithm](../how-to/write-first-algorithm.md) tutorials for the broader context.
Attributes:
algorithm: An instance of `BaseAlgorithm` to use for training.
store: An instance of `LightningStore` to use for storing tasks and traces.
runner: An instance of `BaseRunner` to use for running the agent.
initial_resources: An instance of `Resources` to use for bootstrapping the fit/dev process.
The resources will be handed over to the algorithm.
Note that not all algorithms support seeding resources.
n_runners: Number of agent runners to run in parallel.
max_rollouts: Maximum number of rollouts to process per runner. If None,
workers run until no more rollouts are available.
strategy: An instance of `ExecutionStrategy` to use for spawning the algorithm and runners.
tracer: A tracer instance, or a string pointing to the class full name or a dictionary with a 'type' key
that specifies the class full name and other initialization parameters.
If None, a default `AgentOpsTracer` will be created with the current settings.
hooks: A sequence of `Hook` instances to be called at various lifecycle stages (e.g., on_trace_start,
on_trace_end, on_rollout_start, on_rollout_end).
adapter: An instance of `TracerTraceToTriplet` to export data consumble by algorithms from traces.
llm_proxy: An instance of `LLMProxy` to use for intercepting the LLM calls.
If not provided, algorithm will create one on its own.
n_workers: Number of agent workers to run in parallel. Deprecated in favor of `n_runners`.
max_tasks: Maximum number of tasks to process per runner. Deprecated in favor of `max_rollouts`.
daemon: Whether worker processes should be daemons. Daemon processes
are terminated automatically when the main process exits. Deprecated.
Only have effect with `fit_v0`.
triplet_exporter: An instance of `TracerTraceToTriplet` to export triplets from traces,
or a dictionary with the initialization parameters for the exporter.
Deprecated. Use `adapter` instead.
dev: If True, rollouts are run against the dev endpoint provided in `fit`.
Deprecated in favor of `dev()` method.
"""
algorithm: Optional[Algorithm]
"""An instance of [`Algorithm`][agentlightning.Algorithm] to use for training."""
store: LightningStore
"""An instance of [`LightningStore`][agentlightning.LightningStore] to use for storing tasks and traces."""
runner: Runner[Any]
"""An instance of [`Runner`][agentlightning.Runner] to use for running the agent."""
initial_resources: Optional[NamedResources]
"""An instance of [`NamedResources`][agentlightning.NamedResources] to use for bootstrapping the fit/dev process.
The resources will be handed over to the algorithm. Note that not all algorithms support seeding resources.
"""
n_runners: int
"""Number of agent runners to run in parallel."""
max_rollouts: Optional[int]
"""Maximum number of rollouts to process per runner. If None, workers run until no more rollouts are available."""
strategy: ExecutionStrategy
"""An instance of [`ExecutionStrategy`][agentlightning.ExecutionStrategy] to use for spawning the algorithm and runners."""
tracer: Tracer
"""A tracer instance, or a string pointing to the class full name or a dictionary with a 'type' key
that specifies the class full name and other initialization parameters.
If None, a default [`AgentOpsTracer`][agentlightning.AgentOpsTracer] will be created with the current settings."""
hooks: Sequence[Hook]
"""A sequence of [`Hook`][agentlightning.Hook] instances to be called at various lifecycle stages (e.g., `on_trace_start`,
`on_trace_end`, `on_rollout_start`, `on_rollout_end`)."""
adapter: TraceAdapter[Any]
"""An instance of [`TraceAdapter`][agentlightning.TraceAdapter] to export data consumble by algorithms from traces."""
llm_proxy: Optional[LLMProxy]
"""An instance of [`LLMProxy`][agentlightning.LLMProxy] to use for intercepting the LLM calls.
If not provided, algorithm may create one on its own."""
n_workers: int
"""Number of agent workers to run in parallel. Deprecated in favor of `n_runners`."""
max_tasks: Optional[int]
"""Maximum number of tasks to process per runner. Deprecated in favor of `max_rollouts`."""
daemon: bool
"""Whether worker processes should be daemons. Daemon processes
are terminated automatically when the main process exits. Deprecated.
Only have effect with `fit_v0`."""
triplet_exporter: TraceAdapter[Any]
"""An instance of [`TracerTraceToTriplet`][agentlightning.TracerTraceToTriplet] to export triplets from traces,
or a dictionary with the initialization parameters for the exporter.
Deprecated. Use [`adapter`][agentlightning.Trainer.adapter] instead."""
port: Optional[int]
"""Port forwarded to [`ClientServerExecutionStrategy`][agentlightning.ClientServerExecutionStrategy]."""
def __init__(
self,
*,
@@ -124,13 +79,12 @@ class Trainer(TrainerLegacy):
n_runners: Optional[int] = None,
max_rollouts: Optional[int] = None,
initial_resources: Optional[NamedResources] = None,
tracer: ComponentSpec[Tracer] = None,
tracer: ComponentSpec[BaseTracer] = None,
adapter: ComponentSpec[TraceAdapter[Any]] = None,
store: ComponentSpec[LightningStore] = None,
runner: ComponentSpec[Runner[Any]] = None,
runner: ComponentSpec[BaseRunner[Any]] = None,
strategy: ComponentSpec[ExecutionStrategy] = None,
port: Optional[int] = None,
algorithm: ComponentSpec[Algorithm] = None,
algorithm: ComponentSpec[BaseAlgorithm] = None,
llm_proxy: ComponentSpec[LLMProxy] = None,
n_workers: Optional[int] = None,
max_tasks: Optional[int] = None,
@@ -138,16 +92,6 @@ class Trainer(TrainerLegacy):
triplet_exporter: ComponentSpec[TracerTraceToTriplet] = None,
hooks: Optional[Union[Hook, Sequence[Hook]]] = None,
):
"""Configure the trainer and resolve user-provided component specifications.
Each keyword accepts either a concrete instance, a class, a callable factory, a
registry string, or a lightweight configuration dictionary (see
[`build_component()`][agentlightning.trainer.init_utils.build_component]).
When ``port`` is provided it is forwarded to
[`ClientServerExecutionStrategy`][agentlightning.ClientServerExecutionStrategy]
instances constructed (or supplied) for the trainer.
"""
# Do not call super().__init__() here.
# super().__init__() will call TrainerLegacy's initialization, which is not intended.
self.worker_id: Optional[int] = None
@@ -217,13 +161,7 @@ class Trainer(TrainerLegacy):
self.store = self._make_store(store)
self.runner = self._make_runner(runner)
self.port = port
self.strategy = self._make_strategy(
strategy,
n_runners=self.n_runners,
port=port,
)
self.strategy = self._make_strategy(strategy, n_runners=self.n_runners)
if hasattr(self.strategy, "n_runners"):
strategy_runners = getattr(self.strategy, "n_runners")
if isinstance(strategy_runners, int) and strategy_runners > 0:
@@ -241,8 +179,8 @@ class Trainer(TrainerLegacy):
"The cleanup must be handled manually."
)
def _make_tracer(self, tracer: ComponentSpec[Tracer]) -> Tracer:
"""Resolve the tracer component from user input, falling back to AgentOpsTracer."""
def _make_tracer(self, tracer: ComponentSpec[BaseTracer]) -> BaseTracer:
"""Creates a tracer instance based on the provided configuration."""
default_factory = lambda: AgentOpsTracer(
agentops_managed=True,
instrument_managed=True,
@@ -250,27 +188,26 @@ class Trainer(TrainerLegacy):
)
return build_component(
tracer,
expected_type=Tracer,
expected_type=BaseTracer,
spec_name="tracer",
default_factory=default_factory,
dict_requires_type=True,
invalid_spec_error_fmt="Invalid tracer type: {actual_type}. Expected Tracer, str, dict, or None.",
type_error_fmt="Tracer factory returned {type_name}, which is not a Tracer subclass.",
invalid_spec_error_fmt="Invalid tracer type: {actual_type}. Expected BaseTracer, str, dict, or None.",
type_error_fmt="Tracer factory returned {type_name}, which is not a BaseTracer subclass.",
)
def _make_algorithm(self, algorithm: ComponentSpec[Algorithm]) -> Optional[Algorithm]:
"""Resolve the algorithm component, allowing `None` for dev-mode dry runs."""
def _make_algorithm(self, algorithm: ComponentSpec[BaseAlgorithm]) -> Optional[BaseAlgorithm]:
"""Creates an algorithm instance based on the provided configuration."""
return build_component(
algorithm,
expected_type=Algorithm,
expected_type=BaseAlgorithm,
spec_name="algorithm",
allow_none=True,
invalid_spec_error_fmt="Invalid algorithm type: {actual_type}. Expected Algorithm, str, dict, or None.",
type_error_fmt="Algorithm factory returned {type_name}, which is not a Algorithm subclass.",
invalid_spec_error_fmt="Invalid algorithm type: {actual_type}. Expected BaseAlgorithm, str, dict, or None.",
type_error_fmt="Algorithm factory returned {type_name}, which is not a BaseAlgorithm subclass.",
)
def _make_adapter(self, adapter: ComponentSpec[TraceAdapter[Any]]) -> TraceAdapter[Any]:
"""Resolve the adapter used to transform spans into algorithm-ready payloads."""
return build_component(
adapter,
expected_type=TraceAdapter,
@@ -283,7 +220,6 @@ class Trainer(TrainerLegacy):
)
def _make_store(self, store: ComponentSpec[LightningStore]) -> LightningStore:
"""Resolve the store implementation backing rollouts, attempts, spans, and resources."""
return build_component(
store,
expected_type=LightningStore,
@@ -298,21 +234,13 @@ class Trainer(TrainerLegacy):
strategy: ComponentSpec[ExecutionStrategy],
*,
n_runners: int,
port: Optional[int] = None,
) -> ExecutionStrategy:
"""Resolve the execution strategy and seed defaults such as `n_runners`."""
if isinstance(strategy, ExecutionStrategy):
if port is not None and isinstance(strategy, ClientServerExecutionStrategy):
strategy.server_port = port
return strategy
optional_defaults: Dict[str, Callable[[], Any]] = {"n_runners": lambda: n_runners}
if port is not None:
optional_defaults["server_port"] = lambda: port
def default_factory() -> ExecutionStrategy:
if port is not None:
return ClientServerExecutionStrategy(n_runners=n_runners, server_port=port)
return ClientServerExecutionStrategy(n_runners=n_runners)
return ClientServerExecutionStrategy(n_runners=n_runners, role="both")
return build_component(
strategy,
@@ -331,7 +259,6 @@ class Trainer(TrainerLegacy):
*,
store: LightningStore,
) -> Optional[LLMProxy]:
"""Resolve an optional LLM proxy and ensure it shares the trainer's store instance."""
if isinstance(llm_proxy, LLMProxy):
return llm_proxy
@@ -350,27 +277,25 @@ class Trainer(TrainerLegacy):
type_error_fmt="llm_proxy factory returned {type_name}, which is not an LLMProxy subclass.",
)
def _make_runner(self, runner: ComponentSpec[Runner[Any]]) -> Runner[Any]:
"""Resolve the runner responsible for executing the agent inside each worker."""
def _make_runner(self, runner: ComponentSpec[BaseRunner[Any]]) -> BaseRunner[Any]:
optional_defaults: Dict[str, Callable[[], Any]] = {"tracer": lambda: self.tracer}
if self.max_rollouts is not None:
optional_defaults["max_rollouts"] = lambda: self.max_rollouts
def default_runner_factory() -> Runner[Any]:
def default_runner_factory() -> BaseRunner[Any]:
return instantiate_component(LitAgentRunner, optional_defaults=optional_defaults)
return build_component(
runner,
expected_type=Runner,
expected_type=BaseRunner,
spec_name="runner",
default_factory=default_runner_factory,
optional_defaults=optional_defaults,
invalid_spec_error_fmt="Invalid runner type: {actual_type}. Expected Runner, callable, str, dict, or None.",
type_error_fmt="Runner factory returned {type_name}, which is not a Runner subclass.",
invalid_spec_error_fmt="Invalid runner type: {actual_type}. Expected BaseRunner, callable, str, dict, or None.",
type_error_fmt="Runner factory returned {type_name}, which is not a BaseRunner subclass.",
)
def _normalize_hooks(self, hooks: Optional[Union[Hook, Sequence[Hook]]]) -> Sequence[Hook]:
"""Coerce hook inputs into an immutable sequence for runner initialization."""
if hooks is None:
return ()
if isinstance(hooks, Hook):
@@ -384,33 +309,13 @@ class Trainer(TrainerLegacy):
*,
val_dataset: Optional[Dataset[T_co]] = None,
) -> None:
"""Execute the full algorithm/runner training loop.
[`Trainer.fit`][agentlightning.Trainer.fit] packages the algorithm and runner bundles,
then hands them to the active [`ExecutionStrategy`][agentlightning.ExecutionStrategy].
The strategy rarely returns until:
* The algorithm exhausts the dataset(s) and stops enqueuing rollouts.
* `max_rollouts` causes individual runners to exit.
* An exception or interrupt cancels the shared [`ExecutionEvent`][agentlightning.ExecutionEvent].
"""Run the training loop using the configured strategy, store, and runner.
Args:
agent: [`LitAgent`][agentlightning.LitAgent] implementation executed by runners.
train_dataset: Optional iterable of rollout inputs consumed by the algorithm.
val_dataset: Optional iterable consumed by validation passes.
agent: The LitAgent instance to be trained on.
train_dataset: The dataset to train on.
val_dataset: The dataset to validate on.
"""
if isinstance(train_dataset, str):
logger.warning(
"Trainer.fit will no longer accepts a string URL in future version. "
"To continue using a string URL, please use Trainer.fit_v0 instead. "
"See documentation for how to migrate to latest version: https://microsoft.github.io/agent-lightning/stable/"
)
return self.fit_v0( # type: ignore
agent,
train_dataset,
val_dataset, # type: ignore
)
agent.set_trainer(self)
algorithm_bundle = functools.partial(
@@ -430,22 +335,15 @@ class Trainer(TrainerLegacy):
*,
val_dataset: Optional[Dataset[T_co]] = None,
) -> None:
"""Exercise the infrastructure using a fast, synchronous algorithm.
[`Trainer.dev`][agentlightning.Trainer.dev] mirrors [`fit()`][agentlightning.Trainer.fit] but
insists on an [`Algorithm`][agentlightning.Algorithm] subtype that also derives from
[`FastAlgorithm`][agentlightning.FastAlgorithm]. This keeps the loop responsive for
debugging while still touching the same store, runners, hooks, and tracer plumbing.
If no algorithm is provided, a default [`Baseline`][agentlightning.Baseline] algorithm will be used.
"""Dry run the training loop with a FastAlgorithm and the real runner.
Args:
agent: [`LitAgent`][agentlightning.LitAgent] implementation to execute.
train_dataset: Optional iterable passed to the algorithm.
val_dataset: Optional iterable passed to the algorithm.
agent: The LitAgent instance to be trained on.
train_dataset: The dataset to train on.
val_dataset: The dataset to validate on.
Raises:
TypeError: If the configured algorithm does not inherit from `FastAlgorithm`.
TypeError: If the configured algorithm is not a :class:`FastAlgorithm`.
"""
agent.set_trainer(self)
@@ -476,17 +374,8 @@ class Trainer(TrainerLegacy):
event: ExecutionEvent,
train_dataset: Optional[Dataset[T_co]],
val_dataset: Optional[Dataset[T_co]],
algorithm: Optional[Algorithm],
algorithm: Optional[BaseAlgorithm],
) -> None:
"""Internal entry point executed by the strategy for the algorithm role.
This coroutine is scheduled inside the strategy's process/thread and is responsible
for binding algorithm dependencies (store, adapter, initial resources, proxy) before
invoking [`Algorithm.run`][agentlightning.Algorithm.run].
When `algorithm` is `None` the bundle simply waits for the
shared `event` to signal shutdown so runners can still execute (useful for manual queue
seeding or external algorithms).
"""
if algorithm is not None:
algorithm.set_trainer(self)
algorithm.set_store(store)
@@ -521,14 +410,7 @@ class Trainer(TrainerLegacy):
async def _runner_bundle(
self, store: LightningStore, worker_id: int, event: ExecutionEvent, agent: LitAgent[T_co]
) -> None:
"""Internal entry point executed by the strategy for each runner role.
The bundle materializes the configured runner, binds the agent and hooks, associates
the worker with the shared store, and then drives the runner's [`iter`][agentlightning.Runner.iter]
loop until the execution event is set or an exception occurs. Cleanup mirrors the initialization
sequence to keep tracer state, hooks, and agent resources consistent across restarts.
"""
runner_instance: Runner[Any] | None = None
runner_instance: BaseRunner[Any] | None = None
runner_initialized = False
worker_initialized = False
try:
+58 -143
View File
@@ -1,7 +1,5 @@
# Copyright (c) Microsoft. All rights reserved.
"""Core data models shared across Agent Lightning components."""
from __future__ import annotations
from typing import (
@@ -27,8 +25,8 @@ from .tracer import Span
if TYPE_CHECKING:
from agentlightning.litagent import LitAgent
from agentlightning.runner.base import Runner
from agentlightning.tracer.base import Tracer
from agentlightning.runner.base import BaseRunner
from agentlightning.tracer.base import BaseTracer
__all__ = [
"Triplet",
@@ -49,15 +47,13 @@ __all__ = [
"Attempt",
"AttemptedRollout",
"Hook",
"Worker",
"WorkerStatus",
]
T_co = TypeVar("T_co", covariant=True)
class Triplet(BaseModel):
"""Single interaction turn captured during reinforcement learning."""
"""A standard structure for a single turn in a trajectory."""
prompt: Any
response: Any
@@ -66,11 +62,7 @@ class Triplet(BaseModel):
class RolloutLegacy(BaseModel):
"""Legacy reporting payload exchanged with the deprecated HTTP server.
!!! warning "Deprecated"
Use [`Rollout`][agentlightning.Rollout] instead.
"""
"""The standard reporting object from client to server."""
rollout_id: str
@@ -105,7 +97,6 @@ RolloutStatus = Literal[
"cancelled", # cancelled by user (or watchdog)
"requeuing", # retrying
]
"""The status of a rollout."""
AttemptStatus = Literal[
# A status is essentially a process.
@@ -117,83 +108,66 @@ AttemptStatus = Literal[
"unresponsive", # the worker has not reported results for a while
"timeout", # the worker has been emitting new logs, but have been working on the task for too long
]
"""The status of an attempt."""
RolloutMode = Literal["train", "val", "test"]
"""Possible rollout modes."""
class Attempt(BaseModel):
"""Execution attempt for a rollout, including metadata for retries."""
"""An attempt to execute a rollout. A rollout can have multiple attempts if retries are needed."""
rollout_id: str # the rollout this attempt belongs to
attempt_id: str # the universal id for current attempt
sequence_id: int # the sequence number of the attempt, starting from 1
start_time: float # time when the attempt has started
end_time: Optional[float] = None # time when the attempt has ended
rollout_id: str
"""The rollout which this attempt belongs to."""
attempt_id: str
"""The universal id for current attempt."""
sequence_id: int
"""The sequence number of the attempt, starting from 1."""
start_time: float
"""The time when the attempt has started."""
end_time: Optional[float] = None
"""The time when the attempt has ended."""
status: AttemptStatus = "preparing"
"""The status of the attempt."""
# The rollout worker which is executing this attempt
worker_id: Optional[str] = None
"""The rollout worker which is executing this attempt."""
last_heartbeat_time: Optional[float] = None
"""The last time when the worker has reported progress (i.e., a span)."""
last_heartbeat_time: Optional[float] = None # last time when the worker has reported progress
# A bucket for any other relevant information
metadata: Optional[Dict[str, Any]] = None
"""A bucket for any other relevant information."""
class RolloutConfig(BaseModel):
"""Configuration controlling rollout retries and timeouts."""
"""Configurations for rollout execution."""
timeout_seconds: Optional[float] = None
"""The timeout for the rollout, in seconds. None indicates no timeout."""
unresponsive_seconds: Optional[float] = None
"""The unresponsive timeout for the rollout, in seconds. None indicates no unresponsive timeout."""
max_attempts: int = Field(default=1, ge=1)
"""The maximum number of attempts for the rollout, including the first attempt."""
retry_condition: List[AttemptStatus] = Field(default_factory=cast(Callable[[], List[AttemptStatus]], list))
"""The list of statuses that should trigger a retry."""
timeout_seconds: Optional[float] = None # none indicates no timeout
unresponsive_seconds: Optional[float] = None # none indicates no unresponsive timeout
max_attempts: int = Field(default=1, ge=1) # including the first attempt
retry_condition: List[AttemptStatus] = Field(
default_factory=cast(Callable[[], List[AttemptStatus]], list)
) # list of statuses that should trigger a retry
class Rollout(BaseModel):
rollout_id: str
"""Unique identifier for the rollout."""
# Inputs
input: TaskInput
"""Task input used to generate the rollout."""
# Time to track the lifecycle of the rollout
start_time: float
"""Timestamp when the rollout started."""
end_time: Optional[float] = None
"""Timestamp when the rollout ended."""
mode: Optional[RolloutMode] = None
"""Execution mode such as `"train"`, `"val"` or `"test"`. See [`RolloutMode`][agentlightning.RolloutMode]."""
resources_id: Optional[str] = None
"""Identifier of the resources required to execute the rollout."""
# Overall scheduling/running information
status: RolloutStatus = "queuing"
"""Latest status emitted by the controller."""
config: RolloutConfig = Field(default_factory=RolloutConfig)
"""Retry and timeout configuration associated with the rollout."""
# A bucket for any other relevant information
metadata: Optional[Dict[str, Any]] = None
"""Additional metadata attached to the rollout."""
class AttemptedRollout(Rollout):
"""Rollout paired with the currently active attempt."""
"""A rollout along with its active attempt."""
attempt: Attempt
"""The attempt that is currently processing the rollout."""
@model_validator(mode="after")
def check_consistency(self) -> AttemptedRollout:
@@ -202,43 +176,12 @@ class AttemptedRollout(Rollout):
return self
WorkerStatus = Literal["idle", "busy", "unknown"]
class Worker(BaseModel):
"""Worker information. This is actually the same as Runner info."""
worker_id: str
"""The ID of the worker."""
status: WorkerStatus = "unknown"
"""The status of the worker."""
heartbeat_stats: Optional[Dict[str, Any]] = None
"""Statistics about the worker's heartbeat."""
last_heartbeat_time: Optional[float] = None
"""The last time when the worker has reported the stats."""
last_dequeue_time: Optional[float] = None
"""The last time when the worker has tried to dequeue a rollout."""
last_busy_time: Optional[float] = None
"""The last time when the worker has started an attempt and became busy."""
last_idle_time: Optional[float] = None
"""The last time when the worker has triggered the end of an attempt and became idle."""
current_rollout_id: Optional[str] = None
"""The ID of the current rollout that the worker is processing."""
current_attempt_id: Optional[str] = None
"""The ID of the current attempt that the worker is processing."""
TaskInput = Any
"""Task input type. Accepts arbitrary payloads."""
"""Task input type. Can be any type."""
class Task(BaseModel):
"""Rollout request served to client agents.
!!! warning "Deprecated"
The legacy HTTP client/server stack still uses this model. Prefer
[`LightningStore`][agentlightning.LightningStore] APIs for new workflows.
"""
"""A task (rollout request) to be processed by the client agent. Deprecated."""
rollout_id: str
input: TaskInput
@@ -256,23 +199,11 @@ class Task(BaseModel):
class TaskIfAny(BaseModel):
"""A task or indication that no task is available.
!!! warning "Deprecated"
Use [`LightningStore`][agentlightning.LightningStore] APIs for new workflows.
"""
is_available: bool
"""Indication that a task is available."""
task: Optional[Task] = None
RolloutRawResultLegacy = Union[None, float, List[Triplet], List[Dict[str, Any]], List[ReadableSpan], RolloutLegacy]
"""Legacy rollout result type.
!!! warning "Deprecated"
Use [`RolloutRawResult`][agentlightning.RolloutRawResult] instead.
"""
RolloutRawResult = Union[
None, # nothing (relies on tracer)
@@ -280,23 +211,11 @@ RolloutRawResult = Union[
List[ReadableSpan], # constructed OTEL spans by user
List[Span], # constructed Span objects by user
]
"""Rollout result type.
Possible return values of [`rollout`][agentlightning.LitAgent.rollout].
"""
class GenericResponse(BaseModel):
"""Generic server response used by compatibility endpoints.
!!! warning "Deprecated"
This response is no longer used by the new
[`LightningStore`][agentlightning.LightningStore] APIs.
Attributes:
status: Status string describing the result of the request.
message: Optional human readable explanation.
data: Arbitrary payload serialized as JSON.
"""
A generic response message that can be used for various purposes.
"""
status: str = "success"
@@ -305,18 +224,19 @@ class GenericResponse(BaseModel):
class ParallelWorkerBase:
"""Base class for workloads executed across multiple worker processes.
"""Base class for objects that can be parallelized across multiple worker processes.
The lifecycle is orchestrated by the main process:
This class defines the standard lifecycle for parallel processing:
* [`init()`][agentlightning.ParallelWorkerBase.init] prepares shared state.
* Each worker calls [`init_worker()`][agentlightning.ParallelWorkerBase.init_worker] during start-up.
* [`run()`][agentlightning.ParallelWorkerBase.run] performs the parallel workload.
* Workers call [`teardown_worker()`][agentlightning.ParallelWorkerBase.teardown_worker] before exiting.
* The main process finalizes through [`teardown()`][agentlightning.ParallelWorkerBase.teardown].
Main Process:
1. init() - Initialize the object in the main process
2. spawn workers and call init_worker() in each worker
3. run() - Execute the main workload in parallel across workers
4. teardown_worker() - Clean up resources in each worker
5. teardown() - Final cleanup in the main process
Subclasses must implement [`run()`][agentlightning.ParallelWorkerBase.run]
and can override other lifecycle hooks.
Subclasses should implement the run() method and optionally override
the lifecycle methods for custom initialization and cleanup behavior.
"""
def __init__(self) -> None:
@@ -324,30 +244,25 @@ class ParallelWorkerBase:
self.worker_id: Optional[int] = None
def init(self, *args: Any, **kwargs: Any) -> None:
"""Initialize before spawning the workers. This method can be overridden by subclasses."""
pass
def init_worker(self, worker_id: int, *args: Any, **kwargs: Any) -> None:
"""Initialize the worker. This method can be overridden by subclasses."""
self.worker_id = worker_id
def run(self, *args: Any, **kwargs: Any) -> Any:
"""Run the workload. This method can be overridden by subclasses."""
pass
def teardown_worker(self, worker_id: int, *args: Any, **kwargs: Any) -> None:
"""Teardown the worker. This method can be overridden by subclasses."""
pass
def teardown(self, *args: Any, **kwargs: Any) -> None:
"""Teardown after the workers have exited. This method can be overridden by subclasses."""
pass
class Dataset(Protocol, Generic[T_co]):
"""The general interface for a dataset.
It's currently implemented as a protocol, having a similar interface to `torch.utils.data.Dataset`.
It's currently implemented as a protocol, having a similar interface to torch.utils.data.Dataset.
You don't have to inherit from this class; you can use a simple list if you want to.
"""
@@ -360,42 +275,42 @@ class Hook(ParallelWorkerBase):
"""Base class for defining hooks in the agent runner's lifecycle."""
async def on_trace_start(
self, *, agent: LitAgent[Any], runner: Runner[Any], tracer: Tracer, rollout: Rollout
self, *, agent: LitAgent[Any], runner: BaseRunner[Any], tracer: BaseTracer, rollout: Rollout
) -> None:
"""Hook called immediately after the tracer enters the trace context but before the rollout begins.
Args:
agent: The [`LitAgent`][agentlightning.LitAgent] instance associated with the runner.
runner: The [`Runner`][agentlightning.Runner] managing the rollout.
tracer: The [`Tracer`][agentlightning.Tracer] instance associated with the runner.
rollout: The [`Rollout`][agentlightning.Rollout] object that will be processed.
agent: The :class:`LitAgent` instance associated with the runner.
runner: The :class:`BaseRunner` managing the rollout.
tracer: The :class:`BaseTracer` instance associated with the runner.
rollout: The :class:`Rollout` object that will be processed.
Subclasses can override this method to implement custom logic such as logging,
metric collection, or resource setup. By default, this is a no-op.
"""
async def on_trace_end(
self, *, agent: LitAgent[Any], runner: Runner[Any], tracer: Tracer, rollout: Rollout
self, *, agent: LitAgent[Any], runner: BaseRunner[Any], tracer: BaseTracer, rollout: Rollout
) -> None:
"""Hook called immediately after the rollout completes but before the tracer exits the trace context.
Args:
agent: The [`LitAgent`][agentlightning.LitAgent] instance associated with the runner.
runner: The [`Runner`][agentlightning.Runner] managing the rollout.
tracer: The [`Tracer`][agentlightning.Tracer] instance associated with the runner.
rollout: The [`Rollout`][agentlightning.Rollout] object that has been processed.
agent: The :class:`LitAgent` instance associated with the runner.
runner: The :class:`BaseRunner` managing the rollout.
tracer: The :class:`BaseTracer` instance associated with the runner.
rollout: The :class:`Rollout` object that has been processed.
Subclasses can override this method to implement custom logic such as logging,
metric collection, or resource cleanup. By default, this is a no-op.
"""
async def on_rollout_start(self, *, agent: LitAgent[Any], runner: Runner[Any], rollout: Rollout) -> None:
async def on_rollout_start(self, *, agent: LitAgent[Any], runner: BaseRunner[Any], rollout: Rollout) -> None:
"""Hook called immediately before a rollout *attempt* begins.
Args:
agent: The [`LitAgent`][agentlightning.LitAgent] instance associated with the runner.
runner: The [`Runner`][agentlightning.Runner] managing the rollout.
rollout: The [`Rollout`][agentlightning.Rollout] object that will be processed.
agent: The :class:`LitAgent` instance associated with the runner.
runner: The :class:`BaseRunner` managing the rollout.
rollout: The :class:`Rollout` object that will be processed.
Subclasses can override this method to implement custom logic such as
logging, metric collection, or resource setup. By default, this is a
@@ -406,16 +321,16 @@ class Hook(ParallelWorkerBase):
self,
*,
agent: LitAgent[Any],
runner: Runner[Any],
runner: BaseRunner[Any],
rollout: Rollout,
spans: Union[List[ReadableSpan], List[Span]],
) -> None:
"""Hook called after a rollout *attempt* completes.
Args:
agent: The [`LitAgent`][agentlightning.LitAgent] instance associated with the runner.
runner: The [`Runner`][agentlightning.Runner] managing the rollout.
rollout: The [`Rollout`][agentlightning.Rollout] object that has been processed.
agent: The :class:`LitAgent` instance associated with the runner.
runner: The :class:`BaseRunner` managing the rollout.
rollout: The :class:`Rollout` object that has been processed.
spans: The spans that have been added to the store.
Subclasses can override this method for cleanup or additional
+42 -57
View File
@@ -2,8 +2,6 @@
from __future__ import annotations
"""Typed representations of tunable resources shared between Agent Lightning components."""
import inspect
import logging
from typing import (
@@ -34,40 +32,40 @@ __all__ = [
class Resource(BaseModel):
"""Base class for tunable resources distributed to executors."""
"""
Base class for all tunable resources.
"""
resource_type: Any
"""Alias of the resource type."""
class LLM(Resource):
"""Resource that identifies an LLM endpoint and its configuration."""
"""
Provide an LLM endpoint and model name as a resource.
Attributes:
endpoint (str): The URL of the LLM API endpoint.
model (str): The identifier for the model to be used (e.g., 'gpt-4o').
sampling_parameters (SamplingParameters): A dictionary of hyperparameters
for model inference, such as temperature, top_p, etc.
"""
resource_type: Literal["llm"] = "llm"
endpoint: str
"""The URL of the LLM API endpoint."""
model: str
"""The identifier for the model to be used (e.g., 'gpt-4o')."""
api_key: Optional[str] = None
"""Optional secret used to authenticate requests."""
sampling_parameters: Dict[str, Any] = Field(default_factory=dict)
"""A dictionary of hyperparameters for model inference, such as temperature, top_p, etc."""
def get_base_url(self, *args: Any, **kwargs: Any) -> str:
"""Return the base URL consumed by OpenAI-compatible clients.
"""The base_url to put into openai.OpenAI.
Users are encouraged to use `get_base_url(rollout_id, attempt_id)` to get
the LLM endpoint instead of accessing `.endpoint` directly.
Users are encouraged to use `base_url` to get the LLM endpoint instead of accessing `endpoint` directly.
"""
return self.endpoint
class ProxyLLM(LLM):
"""LLM resource that rewrites endpoints through [`LLMProxy`][agentlightning.LLMProxy].
The proxy injects rollout- and attempt-specific routing information into the
endpoint so that downstream services can attribute requests correctly.
"""
"""Proxy LLM resource that is tailored by `llm_proxy.LLMProxy`."""
resource_type: Literal["proxy_llm"] = "proxy_llm" # type: ignore
_initialized: bool = False
@@ -78,7 +76,7 @@ class ProxyLLM(LLM):
object.__setattr__(self, "_initialized", True)
def __getattribute__(self, name: str) -> Any:
"""Emit a warning when `endpoint` is accessed directly after initialization."""
"""Override to emit a warning when endpoint is accessed directly."""
# Check if we're accessing endpoint after initialization and not from base_url
if name == "endpoint":
try:
@@ -99,7 +97,7 @@ class ProxyLLM(LLM):
return super().__getattribute__(name)
def with_attempted_rollout(self, rollout: AttemptedRollout) -> LLM:
"""Bake rollout metadata into a concrete [`LLM`][agentlightning.LLM] instance."""
"""Bake the rollout and attempt id into the endpoint."""
return LLM(
endpoint=self.get_base_url(rollout.rollout_id, rollout.attempt.attempt_id),
model=self.model,
@@ -108,18 +106,6 @@ class ProxyLLM(LLM):
)
def get_base_url(self, rollout_id: Optional[str], attempt_id: Optional[str]) -> str:
"""Return the routed endpoint for a specific rollout/attempt pair.
Args:
rollout_id: Identifier of the rollout making the request.
attempt_id: Identifier of the attempt within that rollout.
Returns:
Fully qualified endpoint including rollout metadata.
Raises:
ValueError: If exactly one of ``rollout_id`` or ``attempt_id`` is provided.
"""
if rollout_id is None and attempt_id is None:
return self.endpoint
@@ -144,20 +130,22 @@ class ProxyLLM(LLM):
class PromptTemplate(Resource):
"""Resource describing a reusable prompt template."""
"""
A prompt template as a resource.
Attributes:
template (str): The template string. The format depends on the engine.
engine (Literal['jinja', 'f-string', 'poml']): The templating engine
to use for rendering the prompt. I imagine users can use their own
customized engines, but algos can only well operate on a subset of them.
"""
resource_type: Literal["prompt_template"] = "prompt_template"
template: str
"""The template string. The format depends on the engine."""
engine: Literal["jinja", "f-string", "poml"]
"""The templating engine to use for rendering the prompt."""
def format(self, **kwargs: Any) -> str:
"""Format the prompt using keyword arguments.
!!! warning
Only the `f-string` engine is supported for now.
"""
"""Format the prompt template with the given kwargs."""
if self.engine == "f-string":
return self.template.format(**kwargs)
else:
@@ -170,35 +158,32 @@ class PromptTemplate(Resource):
# TODO: migrate to use a registry
ResourceUnion = Annotated[Union[LLM, ProxyLLM, PromptTemplate], Field(discriminator="resource_type")]
NamedResources = Dict[str, ResourceUnion]
"""Mapping from resource names to their configured instances.
"""
A dictionary-like class to hold named resources.
Examples:
```python
Example:
resources: NamedResources = {
"main_llm": LLM(
'main_llm': LLM(
endpoint="http://localhost:8080",
model="llama3",
sampling_parameters={"temperature": 0.7, "max_tokens": 100},
sampling_parameters={'temperature': 0.7, 'max_tokens': 100}
),
"system_prompt": PromptTemplate(
'system_prompt': PromptTemplate(
template="You are a helpful assistant.",
engine="f-string",
),
engine='f-string'
)
}
```
"""
class ResourcesUpdate(BaseModel):
"""Update payload broadcast to clients when resources change."""
"""
A resource update message to be sent from the server to clients.
This message contains a dictionary of resources that clients should use
for subsequent tasks. It is used to update the resources available to
clients dynamically.
"""
resources_id: str
"""Identifier used to version the resources."""
create_time: float
"""Timestamp of the creation time of the resources."""
update_time: float
"""Timestamp of the last update time of the resources."""
version: int
"""Version of the resources."""
resources: NamedResources
"""Mapping of resource names to their definitions."""
+34 -134
View File
@@ -2,19 +2,17 @@
from __future__ import annotations
"""Data models that mirror OpenTelemetry spans for Agent Lightning."""
import json
from enum import Enum
from typing import Any, Dict, List, Optional, Sequence, Union
from opentelemetry import trace as trace_api
from opentelemetry.sdk.resources import Resource
from opentelemetry.sdk.resources import Resource as OtelResource
from opentelemetry.sdk.trace import Event as OtelEvent
from opentelemetry.sdk.trace import ReadableSpan
from opentelemetry.sdk.trace.id_generator import RandomIdGenerator
from opentelemetry.trace.status import Status as OtelStatus
from pydantic import BaseModel, ConfigDict
from pydantic import BaseModel
__all__ = [
"AttributeValue",
@@ -24,7 +22,7 @@ __all__ = [
"TraceStatus",
"Event",
"Link",
"OtelResource",
"Resource",
"Span",
"SpanNames",
"SpanAttributeNames",
@@ -33,13 +31,9 @@ __all__ = [
def convert_timestamp(timestamp: Optional[int]) -> Optional[float]:
"""Normalize OpenTelemetry timestamps to seconds.
"""Convert timestamp from nanoseconds to seconds if needed.
Args:
timestamp: Timestamp expressed either in seconds or nanoseconds.
Returns:
Timestamp in seconds when `timestamp` is provided; otherwise `None`.
Auto-detects format: if > 1e12, assumes nanoseconds; otherwise seconds.
"""
if not timestamp:
return None
@@ -47,15 +41,7 @@ def convert_timestamp(timestamp: Optional[int]) -> Optional[float]:
def extract_extra_fields(src: Any, excluded_fields: List[str]) -> Dict[str, Any]:
"""Capture custom attributes from an OpenTelemetry object.
Args:
src: Object that exposes a `__dict__` of potential attributes.
excluded_fields: Attribute names that should be removed from the output.
Returns:
Dictionary containing JSON-serializable representations of the remaining fields.
"""
"""Extract extra fields from source object, excluding specified fields and private fields."""
excluded_fields_set = set(excluded_fields) | set(["_" + k for k in excluded_fields])
# Exclude the function fields
excluded_fields_set |= set(src.__class__.__dict__.keys())
@@ -76,31 +62,23 @@ AttributeValue = Union[
Sequence[int],
Sequence[float],
]
"""Possible values for OpenTelemetry attributes."""
Attributes = Dict[str, AttributeValue]
"""Mapping from attribute names to their values. Same as OpenTelemetry `Attributes` type."""
TraceState = Dict[str, str]
"""Mapping from trace state key to its value. Same as OpenTelemetry `TraceState` type."""
class SpanContext(BaseModel):
"""Pydantic representation of `opentelemetry.trace.SpanContext` values."""
"""Corresponding to opentelemetry.trace.SpanContext"""
trace_id: str
"""The trace ID of the span."""
span_id: str
"""The span ID of the span."""
is_remote: bool
"""Whether the span is remote."""
trace_state: TraceState
"""Mapping from trace state key to its value."""
model_config = ConfigDict(extra="allow")
class Config:
allow_extra = True
@classmethod
def from_opentelemetry(cls, src: trace_api.SpanContext) -> "SpanContext":
"""Construct a [`SpanContext`][agentlightning.SpanContext] from OpenTelemetry data."""
return cls(
trace_id=trace_api.format_trace_id(src.trace_id),
span_id=trace_api.format_span_id(src.span_id),
@@ -111,19 +89,16 @@ class SpanContext(BaseModel):
class TraceStatus(BaseModel):
"""Serializable variant of `opentelemetry.trace.Status`."""
"""Corresponding to opentelemetry.trace.Status"""
status_code: str
"""The status code of the span. Same as OpenTelemetry `Status.status_code` type."""
description: Optional[str] = None
"""The description of the span. Same as OpenTelemetry `Status.description` type."""
model_config = ConfigDict(extra="allow")
class Config:
allow_extra = True
@classmethod
def from_opentelemetry(cls, src: OtelStatus) -> "TraceStatus":
"""Create a [`TraceStatus`][agentlightning.TraceStatus] from OpenTelemetry metadata."""
return cls(
status_code=src.status_code.name,
description=src.description,
@@ -132,21 +107,17 @@ class TraceStatus(BaseModel):
class Event(BaseModel):
"""Serializable representation of OpenTelemetry `Event` values."""
"""Corresponding to opentelemetry.trace.Event"""
name: str
"""The name of the event."""
attributes: Attributes
"""Mapping from attribute names to their values. Same as OpenTelemetry `Attributes` type."""
timestamp: Optional[float] = None
"""The timestamp of the event. Same as OpenTelemetry `Event.timestamp` type."""
model_config = ConfigDict(extra="allow")
class Config:
allow_extra = True
@classmethod
def from_opentelemetry(cls, src: OtelEvent) -> "Event":
"""Create an [`Event`][agentlightning.Event] from an OpenTelemetry event."""
return cls(
name=src.name,
attributes=dict(src.attributes) if src.attributes else {},
@@ -156,19 +127,16 @@ class Event(BaseModel):
class Link(BaseModel):
"""Serializable representation of OpenTelemetry `Link` values."""
"""Corresponding to opentelemetry.trace.Link"""
context: SpanContext
"""The context of the link."""
attributes: Optional[Attributes] = None
"""Optional attributes."""
model_config = ConfigDict(extra="allow")
class Config:
allow_extra = True
@classmethod
def from_opentelemetry(cls, src: trace_api.Link) -> "Link":
"""Create a [`Link`][agentlightning.Link] from an OpenTelemetry link."""
return cls(
context=SpanContext.from_opentelemetry(src.context),
attributes=dict(src.attributes) if src.attributes else None,
@@ -176,24 +144,14 @@ class Link(BaseModel):
)
class OtelResource(BaseModel):
"""Serializable representation of OpenTelemetry `Resource` values.
Named as `OtelResource` to avoid confusion with the [`Resource`][agentlightning.Resource] class.
Users will very rarely need to construct this class directly. Most of the times,
they deal with the [`Resource`][agentlightning.Resource] class instead, which describes
a very different concept.
"""
class Resource(BaseModel):
"""Corresponding to opentelemetry.sdk.resources.Resource"""
attributes: Attributes
"""Mapping from attribute names to their values. Same as OpenTelemetry `Attributes` type."""
schema_url: str
"""The schema URL of the resource."""
@classmethod
def from_opentelemetry(cls, src: Resource) -> "OtelResource":
"""Create a [`Resource`][agentlightning.Resource] from an OpenTelemetry resource."""
def from_opentelemetry(cls, src: OtelResource) -> "Resource":
return cls(
attributes=dict(src.attributes) if src.attributes else {},
schema_url=src.schema_url if src.schema_url else "",
@@ -202,58 +160,35 @@ class OtelResource(BaseModel):
class Span(BaseModel):
"""Agent Lightning's canonical span model used for persistence and analytics.
The model captures the most relevant fields from
`opentelemetry.sdk.trace.ReadableSpan` instances while preserving unmodeled
attributes in Pydantic `BaseModel`'s extra storage. This keeps the serialized format
stable even as upstream OpenTelemetry types evolve.
"""
model_config = ConfigDict(extra="allow")
class Config:
allow_extra = True # allow extra fields if needed
rollout_id: str
"""The rollout which this span belongs to."""
attempt_id: str
"""The attempt which this span belongs to."""
# The ID to make spans ordered within a single attempt
sequence_id: int
"""The ID to make spans ordered within a single attempt."""
# Current ID (in hex, formatted via trace_api.format_*)
trace_id: str # one rollout can have traces coming from multiple places
"""The trace ID of the span. One rollout/attempt can have multiple traces.
This ID comes from the OpenTelemetry trace ID generator.
"""
span_id: str
"""The span ID of the span. This ID comes from the OpenTelemetry span ID generator."""
parent_id: Optional[str]
"""The parent span ID of the span."""
# Core ReadableSpan fields
name: str
"""The name of the span. See [OpenTelemetry docs](https://opentelemetry.io/docs/concepts/signals/traces/)."""
status: TraceStatus
"""The status of the span. See [OpenTelemetry docs](https://opentelemetry.io/docs/concepts/signals/traces/)."""
attributes: Attributes
"""The attributes of the span. See [OpenTelemetry docs](https://opentelemetry.io/docs/concepts/signals/traces/)."""
events: List[Event]
"""The events of the span. See [OpenTelemetry docs](https://opentelemetry.io/docs/concepts/signals/traces/)."""
links: List[Link]
"""The links of the span. See [OpenTelemetry docs](https://opentelemetry.io/docs/concepts/signals/traces/)."""
# Timestamps
start_time: Optional[float]
"""The start time of the span. See [OpenTelemetry docs](https://opentelemetry.io/docs/concepts/signals/traces/)."""
end_time: Optional[float]
"""The end time of the span. See [OpenTelemetry docs](https://opentelemetry.io/docs/concepts/signals/traces/)."""
# Other parsable fields
context: Optional[SpanContext]
"""The context of the span. See [OpenTelemetry docs](https://opentelemetry.io/docs/concepts/signals/traces/)."""
parent: Optional[SpanContext]
"""The parent context of the span. See [OpenTelemetry docs](https://opentelemetry.io/docs/concepts/signals/traces/)."""
resource: OtelResource
"""The resource of the span. See [OpenTelemetry docs](https://opentelemetry.io/docs/concepts/signals/traces/)."""
resource: Resource
# Preserve other fields in the readable span as extra fields
# Make sure that are json serializable (so no bytes, complex objects, ...)
@@ -266,17 +201,6 @@ class Span(BaseModel):
attempt_id: str,
sequence_id: int,
) -> "Span":
"""Convert an OpenTelemetry span into the Agent Lightning data model.
Args:
src: Span captured by OpenTelemetry.
rollout_id: Identifier for the rollout that produced the span.
attempt_id: Identifier of the attempt within the rollout.
sequence_id: Monotonically increasing identifier assigned to the span.
Returns:
Parsed [`Span`][agentlightning.Span] instance suitable for persistence.
"""
context = src.get_span_context()
if context is None:
trace_id = span_id = 0
@@ -299,7 +223,7 @@ class Span(BaseModel):
end_time=convert_timestamp(src.end_time),
context=SpanContext.from_opentelemetry(context) if context else None,
parent=(SpanContext.from_opentelemetry(src.parent) if src.parent else None),
resource=OtelResource.from_opentelemetry(src.resource),
resource=Resource.from_opentelemetry(src.resource),
**extract_extra_fields(
src,
[
@@ -337,28 +261,8 @@ class Span(BaseModel):
parent_id: Optional[str] = None,
start_time: Optional[float] = None,
end_time: Optional[float] = None,
resource: Optional[OtelResource] = None,
resource: Optional[Resource] = None,
) -> "Span":
"""Build a synthetic span from raw attributes.
Different from the [`from_opentelemetry`][agentlightning.Span.from_opentelemetry] method,
all parameters other than `attributes` are optional and will be generated if not provided.
Args:
attributes: Span attributes to persist.
rollout_id: Optional rollout identifier associated with the span.
attempt_id: Optional attempt identifier associated with the span.
sequence_id: Optional sequence number to preserve ordering.
name: Optional human-readable span name.
trace_id: Custom trace identifier. When omitted, a random identifier is generated.
span_id: Custom span identifier. When omitted, a random identifier is generated.
parent_id: Optional parent span identifier.
start_time: Span start timestamp in seconds.
end_time: Span end timestamp in seconds.
resource: Explicit resource information to attach to the span.
Returns:
[`Span`][agentlightning.Span] populated with the provided attributes.
"""
id_generator = RandomIdGenerator()
trace_id = trace_id or trace_api.format_trace_id(id_generator.generate_trace_id())
@@ -380,7 +284,7 @@ class Span(BaseModel):
trace_state={},
),
name=name or SpanNames.VIRTUAL.value,
resource=resource or OtelResource(attributes={}, schema_url=""),
resource=resource or Resource(attributes={}, schema_url=""),
attributes=attributes,
status=TraceStatus(status_code="OK"),
events=[],
@@ -399,28 +303,24 @@ class Span(BaseModel):
class SpanNames(str, Enum):
"""Enumerated span names recognised by Agent-lightning."""
"""Standard span name values for AgentLightning.
Currently reward, message, object and exception spans are supported.
We will add more spans related to error handling in the future.
"""
REWARD = "agentlightning.reward"
"""The name of the reward span."""
MESSAGE = "agentlightning.message"
"""The name of the message span."""
OBJECT = "agentlightning.object"
"""The name of the object span."""
EXCEPTION = "agentlightning.exception"
"""The name of the exception span."""
VIRTUAL = "agentlightning.virtual"
"""The name of the virtual span. It represents derived spans without concrete operations."""
class SpanAttributeNames(str, Enum):
"""Canonical attribute names written by Agent Lightning emitters."""
"""Standard attribute names for AgentLightning spans."""
MESSAGE = "message"
"""The name of the message attribute."""
OBJECT = "object"
"""The name of the object attribute."""
SpanLike = Union[ReadableSpan, Span]
"""Union type of OpenTelemetry `ReadableSpan` and Agent-lightning [`Span`][agentlightning.Span]."""
-1
View File
@@ -1 +0,0 @@
# Copyright (c) Microsoft. All rights reserved.
File diff suppressed because it is too large Load Diff
-72
View File
@@ -1,72 +0,0 @@
# Copyright (c) Microsoft. All rights reserved.
from __future__ import annotations
import platform
import socket
from contextlib import suppress
from datetime import datetime
from typing import Any, Dict, List, cast
import psutil
from gpustat import GPUStat, GPUStatCollection
def system_snapshot(include_gpu: bool = False) -> Dict[str, Any]:
# CPU
cpu = {
"cpu_name": platform.processor(),
"cpu_cores": psutil.cpu_count(logical=False),
"cpu_threads": psutil.cpu_count(logical=True),
"cpu_usage_pct": psutil.cpu_percent(0.05),
}
# Memory
vm = psutil.virtual_memory()
mem = {
"mem_used_gb": round(vm.used / (2**30), 2),
"mem_total_gb": round(vm.total / (2**30), 2),
"mem_pct": vm.percent,
}
# Disk
du = psutil.disk_usage("/")
disk = {
"disk_used_gb": round(du.used / (2**30), 2),
"disk_total_gb": round(du.total / (2**30), 2),
"disk_pct": du.percent,
}
# GPU
gpus: List[Dict[str, Any]] = []
with suppress(Exception):
for g in GPUStatCollection.new_query().gpus: # type: ignore
g = cast(GPUStat, g)
gpus.append(
{
"gpu": g.name, # type: ignore
"util_pct": g.utilization,
"mem_used_mb": g.memory_used,
"mem_total_mb": g.memory_total,
"temp_c": g.temperature,
}
)
# Network
net = psutil.net_io_counters()
netinfo = {
"bytes_sent_mb": round(net.bytes_sent / (2**20), 2),
"bytes_recv_mb": round(net.bytes_recv / (2**20), 2),
}
# OS / meta
return {
"timestamp": datetime.now().isoformat(timespec="seconds"),
"host": socket.gethostname(),
"os": platform.platform(),
**cpu,
**mem,
**disk,
**netinfo,
**({"gpus": gpus} if include_gpu else {}),
}
+1 -1
View File
@@ -32,7 +32,7 @@ class PatchedvLLMServer(_unwrap_ray_remote(AsyncvLLMServer)):
async def chat_completion(self, raw_request: Request):
"""OpenAI-compatible HTTP endpoint.
API reference: [OpenAI-compatible server documentation](https://docs.vllm.ai/en/latest/serving/openai_compatible_server.html)
API reference: https://docs.vllm.ai/en/latest/serving/openai_compatible_server.html
"""
request_json = await raw_request.json()
request = ChatCompletionRequest(**request_json)
+20 -58
View File
@@ -7,7 +7,6 @@ import socket
import threading
import time
import uuid
from collections import defaultdict
from collections.abc import Mapping
from typing import Any, Dict, List, Literal, Optional, Tuple
@@ -18,13 +17,13 @@ from flask import Flask, Response, abort, request
from tensordict import TensorDict
from verl import DataProto
from agentlightning import LLM, AgentLightningServer, NamedResources, RolloutLegacy, setup_logging
from agentlightning import LLM, AgentLightningServer, NamedResources, RolloutLegacy, configure_logger
from agentlightning.adapter.triplet import TracerTraceToTriplet, TraceToTripletBase
from agentlightning.llm_proxy import LLMProxy, ModelConfig
from agentlightning.store.base import LightningStore
from agentlightning.types import Rollout, RolloutConfig, Task
setup_logging()
configure_logger()
__all__ = [
"AgentModeDaemon",
@@ -294,7 +293,7 @@ class AgentModeDaemon:
self._proxy_thread.start()
print(f"Proxy server running on port {self.proxy_port}")
async def _update_proxy_server_v1(self):
def _update_proxy_server_v1(self):
model_name = self.train_information.get("model")
if not model_name:
raise ValueError("Model name is not set.")
@@ -313,7 +312,12 @@ class AgentModeDaemon:
],
)
await self.llm_proxy.restart()
if self.llm_proxy.is_running():
# FIXME: Need to switch to a different port right now
# because the forked processes carried the old fd
self.llm_proxy.restart(_port=_find_available_port())
else:
self.llm_proxy.start()
def start(self):
"""Starts the main AgentLightningServer and the proxy server."""
@@ -347,7 +351,7 @@ class AgentModeDaemon:
if server_addresses != self.backend_llm_server_addresses:
self.backend_llm_server_addresses = server_addresses
if self.mode == "v1" and not self.llm_proxy.is_running():
await self._update_proxy_server_v1()
self._update_proxy_server_v1()
self.is_train = is_train
# 1. Update resources on the server for clients to use
@@ -554,28 +558,13 @@ class AgentModeDaemon:
assert len(self._completed_rollouts_v0) == self._total_tasks_queued
sample_stat_list: List[Dict[str, Any]] = []
sample_stat_list_by_source: Dict[str, List[Dict[str, Any]]] = defaultdict(
list
) # FIXME: Evaluate whether grouping stats by source is actually needed.
for rollout_id, rollout in self._completed_rollouts_v0.items():
for _, rollout in self._completed_rollouts_v0.items():
final_reward = self._fillna_reward(rollout)
if not rollout.triplets:
print(f"Warning: No triplets found for test rollout {rollout.rollout_id}.")
sample_stat_list.append({"reward": final_reward})
continue
response_length_list = [len(triplet.response.get("token_ids", [])) for triplet in rollout.triplets]
if "data_source" in self._task_id_to_original_sample[rollout_id]:
# When a test sample includes a 'data_source' field, record per-source statistics for test results.
data_source = self._task_id_to_original_sample[rollout_id]["data_source"]
sample_stat_list_by_source[data_source].append(
{
"sum_response_length": np.sum(response_length_list),
"mean_response_length": np.mean(response_length_list) if response_length_list else 0,
"turn_count": len(rollout.triplets),
"reward": final_reward,
}
)
sample_stat_list.append(
{
"sum_response_length": np.sum(response_length_list),
@@ -584,45 +573,18 @@ class AgentModeDaemon:
"reward": final_reward,
}
)
metric_dict: Dict[str, Any] = {}
stats_w_trace = [stat for stat in sample_stat_list if "sum_response_length" in stat]
stats_w_trace_by_source = {
data_source: [stat for stat in sample_stats if "sum_response_length" in stat]
for data_source, sample_stats in sample_stat_list_by_source.items()
return {
"val/n_rollouts": len(sample_stat_list),
"val/n_rollouts_w_trace": len(stats_w_trace),
"val/reward": np.mean(
[stat["reward"] for stat in sample_stat_list]
), # each rollout must have a reward (fillna if missing)
"val/mean_response_length": np.mean([stat["mean_response_length"] for stat in stats_w_trace]),
"val/sum_response_length": np.mean([stat["sum_response_length"] for stat in stats_w_trace]),
"val/turn_count": np.mean([stat["turn_count"] for stat in stats_w_trace]),
}
for data_source, sample_stats in sample_stat_list_by_source.items():
metric_dict.update(
{
f"val/{data_source}/n_rollouts": len(sample_stats),
f"val/{data_source}/n_rollouts_w_trace": len(stats_w_trace_by_source[data_source]),
f"val/{data_source}/reward": np.mean(
[stat["reward"] for stat in sample_stats]
), # each rollout must have a reward (fillna if missing)
f"val/{data_source}/mean_response_length": np.mean(
[stat["mean_response_length"] for stat in stats_w_trace_by_source[data_source]]
),
f"val/{data_source}/sum_response_length": np.mean(
[stat["sum_response_length"] for stat in stats_w_trace_by_source[data_source]]
),
f"val/{data_source}/turn_count": np.mean(
[stat["turn_count"] for stat in stats_w_trace_by_source[data_source]]
),
}
)
metric_dict.update(
{
"val/n_rollouts": len(sample_stat_list),
"val/n_rollouts_w_trace": len(stats_w_trace),
"val/reward": np.mean(
[stat["reward"] for stat in sample_stat_list]
), # each rollout must have a reward (fillna if missing)
"val/mean_response_length": np.mean([stat["mean_response_length"] for stat in stats_w_trace]),
"val/sum_response_length": np.mean([stat["sum_response_length"] for stat in stats_w_trace]),
"val/turn_count": np.mean([stat["turn_count"] for stat in stats_w_trace]),
}
)
return metric_dict
def get_train_data_batch(self, max_prompt_length: int, max_response_length: int, device: torch.device):
"""
+1 -9
View File
@@ -2,12 +2,10 @@
# type: ignore
from importlib.metadata import version
from typing import Any
import hydra
import ray
from packaging import version as packaging_version
from verl.trainer.main_ppo import create_rl_sampler
from verl.trainer.ppo.reward import load_reward_manager
@@ -41,17 +39,11 @@ def run_ppo(
) -> None:
if not ray.is_initialized():
# this is for local ray cluster
try:
# verl >= 0.6.0
num_cpus = config.ray_kwargs.ray_init.num_cpus
except AttributeError:
# verl < 0.6.0
num_cpus = config.ray_init.num_cpus
ray.init(
runtime_env={
"env_vars": {"TOKENIZERS_PARALLELISM": "true", "NCCL_DEBUG": "WARN", "VLLM_LOGGING_LEVEL": "WARN"}
},
num_cpus=num_cpus,
num_cpus=config.ray_init.num_cpus,
)
runner = TaskRunner.remote()
+6 -119
View File
@@ -12,7 +12,6 @@ from typing import Dict, Tuple
import numpy as np
import torch
import verl
from codetiming import Timer
from omegaconf import OmegaConf
from tqdm import tqdm
@@ -20,7 +19,7 @@ from verl import DataProto
from verl.protocol import pad_dataproto_to_divisor, unpad_dataproto
from verl.trainer.ppo.core_algos import agg_loss
from verl.trainer.ppo.metric_utils import (
_compute_response_info,
compute_data_metrics,
compute_throughout_metrics,
compute_timing_metrics,
)
@@ -54,108 +53,6 @@ def _timer(name: str, timing_raw: Dict[str, float]):
timing_raw[name] += timer.last
# This function is adapted from verl.
# We introduce a new parameter `suffix` to distinguish between metrics computed
# before and after AgentLightnings post-processing.
# - "Before" refers to raw reward and advantage values.
# - "After" refers to values computed following post-processing, which involves:
# (1) Dropping prompts that exceed the maximum allowed length.
# (2) Adjusting the batch size to be a multiple of the mini PPO size.
# Different suffixes are used to label these two stages accordingly.
def compute_data_metrics(batch: DataProto, use_critic: bool = True, suffix: str = "") -> Dict[str, Any]:
"""
Computes various metrics from a batch of data for PPO training.
This function calculates metrics related to scores, rewards, advantages, returns, values,
and sequence lengths from a batch of data. It provides statistical information (mean, max, min)
for each metric category.
Args:
batch: A DataProto object containing batch data with token-level scores, rewards, advantages, etc.
use_critic: Whether to include critic-specific metrics. Defaults to True.
Returns:
A dictionary of metrics including:
- critic/score/mean, max, min: Statistics about sequence scores
- critic/rewards/mean, max, min: Statistics about sequence rewards
- critic/advantages/mean, max, min: Statistics about advantages
- critic/returns/mean, max, min: Statistics about returns
- critic/values/mean, max, min: Statistics about critic values (if use_critic=True)
- critic/vf_explained_var: Explained variance of the value function (if use_critic=True)
- response_length/mean, max, min, clip_ratio: Statistics about response lengths
- prompt_length/mean, max, min, clip_ratio: Statistics about prompt lengths
"""
sequence_score = batch.batch["token_level_scores"].sum(-1)
sequence_reward = batch.batch["token_level_rewards"].sum(-1)
advantages = batch.batch["advantages"]
returns = batch.batch["returns"]
max_response_length = batch.batch["responses"].shape[-1]
prompt_mask = batch.batch["attention_mask"][:, :-max_response_length].bool()
response_mask = batch.batch["attention_mask"][:, -max_response_length:].bool()
max_prompt_length = prompt_mask.size(-1)
response_info = _compute_response_info(batch)
prompt_length = response_info["prompt_length"]
response_length = response_info["response_length"]
valid_adv = torch.masked_select(advantages, response_mask)
valid_returns = torch.masked_select(returns, response_mask)
if use_critic:
values = batch.batch["values"]
valid_values = torch.masked_select(values, response_mask)
return_diff_var = torch.var(valid_returns - valid_values)
return_var = torch.var(valid_returns)
metrics = {
# score
"critic/score/mean" + suffix: torch.mean(sequence_score).detach().item(),
"critic/score/max" + suffix: torch.max(sequence_score).detach().item(),
"critic/score/min" + suffix: torch.min(sequence_score).detach().item(),
# reward
"critic/rewards/mean" + suffix: torch.mean(sequence_reward).detach().item(),
"critic/rewards/max" + suffix: torch.max(sequence_reward).detach().item(),
"critic/rewards/min" + suffix: torch.min(sequence_reward).detach().item(),
# adv
"critic/advantages/mean" + suffix: torch.mean(valid_adv).detach().item(),
"critic/advantages/max" + suffix: torch.max(valid_adv).detach().item(),
"critic/advantages/min" + suffix: torch.min(valid_adv).detach().item(),
# returns
"critic/returns/mean" + suffix: torch.mean(valid_returns).detach().item(),
"critic/returns/max" + suffix: torch.max(valid_returns).detach().item(),
"critic/returns/min" + suffix: torch.min(valid_returns).detach().item(),
**(
{
# values
"critic/values/mean" + suffix: torch.mean(valid_values).detach().item(),
"critic/values/max" + suffix: torch.max(valid_values).detach().item(),
"critic/values/min" + suffix: torch.min(valid_values).detach().item(),
# vf explained var
"critic/vf_explained_var" + suffix: (1.0 - return_diff_var / (return_var + 1e-5)).detach().item(),
}
if use_critic
else {}
),
# response length
"response_length/mean" + suffix: torch.mean(response_length).detach().item(),
"response_length/max" + suffix: torch.max(response_length).detach().item(),
"response_length/min" + suffix: torch.min(response_length).detach().item(),
"response_length/clip_ratio"
+ suffix: torch.mean(torch.eq(response_length, max_response_length).float()).detach().item(),
# prompt length
"prompt_length/mean" + suffix: torch.mean(prompt_length).detach().item(),
"prompt_length/max" + suffix: torch.max(prompt_length).detach().item(),
"prompt_length/min" + suffix: torch.min(prompt_length).detach().item(),
"prompt_length/clip_ratio"
+ suffix: torch.mean(torch.eq(prompt_length, max_prompt_length).float()).detach().item(),
}
return metrics
class AgentLightningTrainer(RayPPOTrainer):
"""
Specialized PPO trainer for agent-based reinforcement learning.
@@ -166,7 +63,6 @@ class AgentLightningTrainer(RayPPOTrainer):
RayPPOTrainer and focusing on the agent mode workflow.
Key differences from RayPPOTrainer:
1. Uses AgentModeDaemon for server communication
2. Simplified data flow without pop/union operations
3. Direct batch processing through agent daemon
@@ -318,9 +214,6 @@ class AgentLightningTrainer(RayPPOTrainer):
config=self.config.algorithm,
)
# Calculate the metrics before processing. Refer to the comments of function `compute_data_metrics` for details.
metrics.update(compute_data_metrics(batch=batch, use_critic=self.use_critic, suffix="_before_processing"))
# after advantages are assinged, we begin to drop (1) long prompt (2) floor to ppo minisize
keep_indices = (~batch.batch["is_drop_mask"]).nonzero(as_tuple=True)[0]
metrics["training/n_triplets_prompt_too_long"] = (
@@ -380,7 +273,7 @@ class AgentLightningTrainer(RayPPOTrainer):
)
# compute training metrics
metrics.update(compute_data_metrics(batch=batch, use_critic=self.use_critic, suffix="_after_processing"))
metrics.update(compute_data_metrics(batch=batch, use_critic=self.use_critic))
metrics.update(compute_timing_metrics(batch=batch, timing_raw=timing_raw))
# TODO: implement actual tflpo and theoretical tflpo
n_gpus = self.resource_pool_manager.get_n_gpus()
@@ -404,20 +297,14 @@ class AgentLightningTrainer(RayPPOTrainer):
assert self.async_rollout_mode, "If agent mode is enabled, async server must be enabled"
if self.adapter is not None and not isinstance(self.adapter, TraceToTripletBase):
raise ValueError("Adapter must be a TraceToTripletBase for currently VERL implementation.")
verl_version = verl.__version__
if verl_version == "0.5.0":
# Note (Zhiyuan): To avoid further patch into vllm async server, using the same sentence to get the naming here.
# However, it is possible that verl updates the naming and causes incompatibility.
# Reference: https://github.com/volcengine/verl/blob/5b5e09d9cc20625e436d01f69d9cc739ff681c54/verl/workers/rollout/vllm_rollout/vllm_async_server.py#L217
model = "/".join(self.config.actor_rollout_ref.model.path.split("/")[-2:])
else:
# For other versions (e.g., 0.6.0), we use the full path to the model.
model = self.config.actor_rollout_ref.model.path
self.agent_mode_daemon = AgentModeDaemon(
self.config.agentlightning.port,
self.config.actor_rollout_ref.rollout.n,
train_information={
"model": model,
# Note (Zhiyuan): To avoid further patch into vllm async server, using the same sentence to get the naming here.
# However, it is possible that verl updates the naming and causes incompatibility.
# Reference: https://github.com/volcengine/verl/blob/5b5e09d9cc20625e436d01f69d9cc739ff681c54/verl/workers/rollout/vllm_rollout/vllm_async_server.py#L217
"model": "/".join(self.config.actor_rollout_ref.model.path.split("/")[-2:]),
"temperature": self.config.actor_rollout_ref.rollout.temperature,
},
tokenizer=self.tokenizer,
-132
View File
@@ -1,132 +0,0 @@
# Logs
logs
*.log
npm-debug.log*
yarn-debug.log*
yarn-error.log*
lerna-debug.log*
.pnpm-debug.log*
# Diagnostic reports (https://nodejs.org/api/report.html)
report.[0-9]*.[0-9]*.[0-9]*.[0-9]*.json
# Runtime data
pids
*.pid
*.seed
*.pid.lock
# Directory for instrumented libs generated by jscoverage/JSCover
lib-cov
# Coverage directory used by tools like istanbul
coverage
*.lcov
# nyc test coverage
.nyc_output
# Grunt intermediate storage (https://gruntjs.com/creating-plugins#storing-task-files)
.grunt
# Bower dependency directory (https://bower.io/)
bower_components
# node-waf configuration
.lock-wscript
# Compiled binary addons (https://nodejs.org/api/addons.html)
build/Release
# Dependency directories
node_modules/
jspm_packages/
# Snowpack dependency directory (https://snowpack.dev/)
web_modules/
# TypeScript cache
*.tsbuildinfo
# Optional npm cache directory
.npm
# Optional eslint cache
.eslintcache
# Optional stylelint cache
.stylelintcache
# Microbundle cache
.rpt2_cache/
.rts2_cache_cjs/
.rts2_cache_es/
.rts2_cache_umd/
# Optional REPL history
.node_repl_history
# Output of 'npm pack'
*.tgz
# Yarn Integrity file
.yarn-integrity
# dotenv environment variable files
.env
.env.development.local
.env.test.local
.env.production.local
.env.local
# parcel-bundler cache (https://parceljs.org/)
.cache
.parcel-cache
# Next.js build output
.next
out
# Nuxt.js build / generate output
.nuxt
dist
# Gatsby files
.cache/
# Comment in the public line in if your project uses Gatsby and not Next.js
# https://nextjs.org/blog/next-9-1#public-directory-support
# public
# vuepress build output
.vuepress/dist
# vuepress v2.x temp and cache directory
.temp
.cache
# Docusaurus cache and generated files
.docusaurus
# Serverless directories
.serverless/
# FuseBox cache
.fusebox/
# DynamoDB Local files
.dynamodb/
# TernJS port file
.tern-port
# Stores VSCode versions used for testing VSCode extensions
.vscode-test
# yarn v2
.yarn/cache
.yarn/unplugged
.yarn/build-state.yml
.yarn/install-state.gz
.pnp.*
.DS_Store
-47
View File
@@ -1,47 +0,0 @@
// Copyright (c) Microsoft. All rights reserved.
/** @type {import("@ianvs/prettier-plugin-sort-imports").PrettierConfig} */
const config = {
printWidth: 120,
singleQuote: true,
tabWidth: 2,
useTabs: false,
semi: true,
quoteProps: 'consistent',
jsxSingleQuote: true,
trailingComma: 'all',
bracketSpacing: true,
objectWrap: 'preserve',
arrowParens: 'always',
proseWrap: 'preserve',
endOfLine: 'lf',
plugins: ['@ianvs/prettier-plugin-sort-imports'],
importOrder: [
'.*styles.css$',
'',
'dayjs',
'^react$',
'^next$',
'^next/.*$',
'<BUILTIN_MODULES>',
'<THIRD_PARTY_MODULES>',
'^@mantine/(.*)$',
'^@mantinex/(.*)$',
'^@mantine-tests/(.*)$',
'^@docs/(.*)$',
'^@/.*$',
'^../(?!.*.css$).*$',
'^./(?!.*.css$).*$',
'\\.css$',
],
overrides: [
{
files: '*.mdx',
options: {
printWidth: 120,
},
},
],
};
export default config;
-12
View File
@@ -1,12 +0,0 @@
// Copyright (c) Microsoft. All rights reserved.
// Centralized constants that keep Storybook fixtures deterministic so Chromatic
// snapshots do not drift when the build environment changes.
export const STORY_DATE_NOW_MS = 1762775145209;
export const STORY_DATE_NOW_SECONDS = Math.floor(STORY_DATE_NOW_MS / 1000);
// Use a fixed origin so any code that would normally read window.location.*
// in the app can rely on the same value from Storybook fixtures. Prefer HTTPS
// so Chromatic (which is served over HTTPS) avoids mixed-content fetch errors.
export const STORY_BASE_URL = 'https://storybook.agentlightning.invalid';
export const STORY_LOCATION_HREF = `${STORY_BASE_URL}/storybook`;
-20
View File
@@ -1,20 +0,0 @@
// Copyright (c) Microsoft. All rights reserved.
import type { StorybookConfig } from '@storybook/react-vite';
const config: StorybookConfig = {
core: {
disableWhatsNewNotifications: true,
disableTelemetry: true,
enableCrashReports: false,
},
stories: ['../src/**/*.mdx', '../src/**/*.story.@(js|jsx|ts|tsx)'],
staticDirs: ['../static'],
addons: ['@storybook/addon-themes', '@storybook/addon-vitest'],
framework: {
name: '@storybook/react-vite',
options: {},
},
};
export default config;
-13
View File
@@ -1,13 +0,0 @@
// Copyright (c) Microsoft. All rights reserved.
export const allModes = {
MD: {
viewport: 'md',
},
LG: {
viewport: 'lg',
},
XL: {
viewport: 'xl',
},
} as const;
-79
View File
@@ -1,79 +0,0 @@
// Copyright (c) Microsoft. All rights reserved.
import '@mantine/core/styles.css';
import 'mantine-datatable/styles.css';
import '../src/styles/theme.css';
import '../src/styles/app.css';
import { initialize, mswLoader } from 'msw-storybook-addon';
import { ColorSchemeScript, MantineProvider } from '@mantine/core';
import { shadcnCssVariableResolver } from '../src/cssVariableResolver';
import { theme as mantineTheme } from '../src/theme';
import { STORY_DATE_NOW_MS } from './constants';
type ColorSchemeValue = 'light' | 'dark';
initialize({
onUnhandledRequest: 'bypass',
serviceWorker: {
url: '/mockServiceWorker.js',
},
});
const fixedDateNow = (() => {
const patched = Date.now as typeof Date.now & { __storybookPatched?: boolean };
if (patched.__storybookPatched) {
return patched;
}
const replacement = (() => STORY_DATE_NOW_MS) as typeof Date.now & { __storybookPatched?: boolean };
replacement.__storybookPatched = true;
return replacement;
})();
Date.now = fixedDateNow;
export const parameters = {
layout: 'fullscreen',
options: {
showPanel: false,
// @ts-expect-error storybook throws build error for (a: any, b: any)
storySort: (a, b) => a.title.localeCompare(b.title, undefined, { numeric: true }),
},
backgrounds: { disable: true },
viewport: {
options: {
md: { name: 'md', styles: { width: '1280px', height: '800px' } },
lg: { name: 'lg', styles: { width: '1920px', height: '1080px' } },
xl: { name: 'xl', styles: { width: '2560px', height: '1440px' } },
},
},
};
export const globalTypes = {
theme: {
name: 'Theme',
description: 'Mantine color scheme',
defaultValue: 'light',
toolbar: {
icon: 'mirror',
items: [
{ value: 'light', title: 'Light' },
{ value: 'dark', title: 'Dark' },
],
},
},
};
export const decorators = [
(Story: any, context: any) => {
const scheme = (context.parameters.theme ?? context.globals.theme ?? 'light') as ColorSchemeValue;
return (
<MantineProvider theme={mantineTheme} cssVariablesResolver={shadcnCssVariableResolver} forceColorScheme={scheme}>
<ColorSchemeScript />
<Story />
</MantineProvider>
);
},
];
export const loaders = [mswLoader];
-8
View File
@@ -1,8 +0,0 @@
// Copyright (c) Microsoft. All rights reserved.
import { setProjectAnnotations } from '@storybook/react-vite';
import * as projectAnnotations from './preview';
// This is an important step to apply the right configuration when testing your stories.
// More info at: https://storybook.js.org/docs/api/portable-stories/portable-stories-vitest#setprojectannotations
setProjectAnnotations([projectAnnotations]);
-5
View File
@@ -1,5 +0,0 @@
# Generated files
dist
# Theme files
theme.css
-28
View File
@@ -1,28 +0,0 @@
{
"extends": ["stylelint-config-standard-scss"],
"rules": {
"custom-property-pattern": null,
"selector-class-pattern": null,
"scss/no-duplicate-mixins": null,
"declaration-empty-line-before": null,
"declaration-block-no-redundant-longhand-properties": null,
"alpha-value-notation": null,
"custom-property-empty-line-before": null,
"property-no-vendor-prefix": null,
"color-function-notation": null,
"length-zero-no-unit": null,
"selector-not-notation": null,
"no-descending-specificity": null,
"comment-empty-line-before": null,
"scss/at-mixin-pattern": null,
"scss/at-rule-no-unknown": null,
"value-keyword-case": null,
"media-feature-range-notation": null,
"selector-pseudo-class-no-unknown": [
true,
{
"ignorePseudoClasses": ["global"]
}
]
}
}
-27
View File
@@ -1,27 +0,0 @@
# Agent-lightning Dashboard
This is the dashboard for Agent-lightning. It is a web application that allows you to inspect your Agent-lightning store and debug running experiments.
The dashboard is built with React, Mantine UI, and Storybook.
## npm scripts
## Build and dev scripts
- `dev` start development server
- `build` build production version of the app
- `preview` locally preview production build
### Testing scripts
- `eslint` - runs ESLint
- `stylelint` - runs Stylelint
- `prettier` - runs Prettier
- `typecheck` - runs TypeScript typecheck
- `vitest` runs vitest tests
- `chromatic` runs chromatic tests
### Other scripts
- `storybook` starts storybook dev server
- `build-storybook` build production storybook bundle to `storybook-static`
-50
View File
@@ -1,50 +0,0 @@
// Copyright (c) Microsoft. All rights reserved.
// @ts-check
import stylistic from '@stylistic/eslint-plugin';
import mantine from 'eslint-config-mantine';
import { defineConfig } from 'eslint/config';
import tseslint from 'typescript-eslint';
export default defineConfig([
// These are arrays → safe to spread
...tseslint.configs.recommended,
stylistic.configs.customize({ semi: true }),
// mantine is often a single object → include as-is (or spread only if it's actually an array)
...(Array.isArray(mantine) ? mantine : [mantine]),
// ignores go as their own entry
{ ignores: ['**/*.{mjs,cjs,js,d.ts,d.mts}'] },
// file-specific rules
{
files: ['**/*.story.tsx'],
rules: { 'no-console': 'off' },
},
// project/TS settings + your custom rules
{
languageOptions: {
parserOptions: {
tsconfigRootDir: process.cwd(),
project: ['./tsconfig.json'],
},
},
rules: {
// Disabling conflict rules with prettier
'@stylistic/brace-style': ['error', '1tbs', { allowSingleLine: false }],
'@stylistic/no-trailing-spaces': 'error',
'@stylistic/no-multiple-empty-lines': ['error', { max: 2, maxEOF: 1 }],
'@stylistic/jsx-quotes': ['error', 'prefer-single'],
'@stylistic/multiline-ternary': 'off',
'@stylistic/arrow-parens': ['error', 'always'],
'@stylistic/jsx-closing-bracket-location': 'off',
'@stylistic/operator-linebreak': 'off',
'@stylistic/jsx-newline': 'off',
'@stylistic/jsx-one-expression-per-line': 'off',
'@stylistic/indent': 'off',
'@stylistic/indent-binary-ops': 'off',
},
},
]);
-10890
View File
File diff suppressed because it is too large Load Diff
-75
View File
@@ -1,75 +0,0 @@
{
"name": "agent-lightning-dashboard",
"type": "module",
"version": "0.2.2",
"scripts": {
"dev": "vite",
"build": "tsc && vite build",
"preview": "vite preview",
"typecheck": "tsc --noEmit",
"eslint": "eslint .",
"stylelint": "stylelint '**/*.css'",
"prettier": "prettier --check \"**/*.{ts,tsx,mjs,cjs}\"",
"vitest": "vitest run --project unit",
"vitest-storybook": "vitest run --project storybook",
"storybook": "storybook dev -p 6006",
"build-storybook": "storybook build",
"chromatic": "chromatic"
},
"dependencies": {
"@mantine/core": "8.3.5",
"@mantine/hooks": "8.3.5",
"@monaco-editor/react": "^4.7.0",
"@reduxjs/toolkit": "^2.9.2",
"@tabler/icons-react": "^3.35.0",
"clsx": "^2.1.1",
"dayjs": "^1.11.18",
"mantine-datatable": "^8.2.0",
"react": "^19.2.0",
"react-dom": "^19.2.0",
"react-redux": "^9.2.0",
"react-router-dom": "^7.9.4"
},
"devDependencies": {
"@eslint/js": "^9.37.0",
"@ianvs/prettier-plugin-sort-imports": "^4.7.0",
"@storybook/addon-themes": "^9.1.10",
"@storybook/addon-vitest": "^9.1.16",
"@storybook/react": "^9.1.10",
"@storybook/react-vite": "^9.1.10",
"@stylistic/eslint-plugin": "^5.5.0",
"@testing-library/dom": "^10.4.1",
"@testing-library/jest-dom": "^6.9.1",
"@testing-library/react": "^16.3.0",
"@testing-library/user-event": "^14.6.1",
"@types/node": "^24.7.1",
"@types/react": "^19.2.2",
"@types/react-dom": "^19.2.1",
"@vitejs/plugin-react": "^5.0.4",
"chromatic": "^13.3.3",
"eslint": "^9.37.0",
"eslint-config-mantine": "^4.0.3",
"eslint-plugin-jsx-a11y": "^6.10.2",
"eslint-plugin-react": "^7.37.5",
"identity-obj-proxy": "^3.0.0",
"jsdom": "^27.0.0",
"msw": "^2.11.6",
"msw-storybook-addon": "^2.0.6",
"postcss": "^8.5.6",
"postcss-preset-mantine": "1.18.0",
"postcss-simple-vars": "^7.0.1",
"prettier": "^3.6.2",
"prop-types": "^15.8.1",
"storybook": "^9.1.10",
"stylelint": "^16.25.0",
"stylelint-config-standard-scss": "^16.0.0",
"typescript": "^5.9.3",
"typescript-eslint": "^8.46.0",
"vite": "^7.1.9",
"vite-tsconfig-paths": "^5.1.4",
"vitest": "^4.0.0",
"playwright": "^1.56.1",
"@vitest/browser-playwright": "4.0.4",
"@vitest/coverage-v8": "4.0.4"
}
}

Some files were not shown because too many files have changed in this diff Show More