Compare commits
2 commits
ebf778fc3f
...
541b720e26
| Author | SHA1 | Date | |
|---|---|---|---|
| 541b720e26 | |||
| 4a10aebf2a |
33 changed files with 7056 additions and 21 deletions
8
.ci.yml
Normal file
8
.ci.yml
Normal file
|
|
@ -0,0 +1,8 @@
|
|||
pipelines:
|
||||
smoke:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
hello:
|
||||
run: echo "CI is alive"
|
||||
118
Cargo.lock
generated
118
Cargo.lock
generated
|
|
@ -397,6 +397,21 @@ version = "1.5.0"
|
|||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6e4de3bc4ea267985becf712dc6d9eed8b04c953b3fcfb339ebc87acd9804901"
|
||||
|
||||
[[package]]
|
||||
name = "ci-relay"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"clap",
|
||||
"hex",
|
||||
"hmac",
|
||||
"iroh",
|
||||
"serde_json",
|
||||
"sha2 0.10.9",
|
||||
"swactor-ci",
|
||||
"tiny_http",
|
||||
"tokio",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ciborium"
|
||||
version = "0.2.2"
|
||||
|
|
@ -1063,6 +1078,7 @@ checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
|
|||
dependencies = [
|
||||
"block-buffer 0.10.4",
|
||||
"crypto-common 0.1.7",
|
||||
"subtle",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
|
@ -1383,9 +1399,9 @@ dependencies = [
|
|||
|
||||
[[package]]
|
||||
name = "futures"
|
||||
version = "0.3.31"
|
||||
version = "0.3.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "65bc07b1a8bc7c85c5f2e110c476c7389b4554ba72af57d8445ea63a576b0876"
|
||||
checksum = "8b147ee9d1f6d097cef9ce628cd2ee62288d963e16fb287bd9286455b241382d"
|
||||
dependencies = [
|
||||
"futures-channel",
|
||||
"futures-core",
|
||||
|
|
@ -1411,9 +1427,9 @@ dependencies = [
|
|||
|
||||
[[package]]
|
||||
name = "futures-channel"
|
||||
version = "0.3.31"
|
||||
version = "0.3.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2dff15bf788c671c1934e366d07e30c1814a8ef514e1af724a602e8a2fbe1b10"
|
||||
checksum = "07bbe89c50d7a535e539b8c17bc0b49bdb77747034daa8087407d655f3f7cc1d"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"futures-sink",
|
||||
|
|
@ -1421,15 +1437,15 @@ dependencies = [
|
|||
|
||||
[[package]]
|
||||
name = "futures-core"
|
||||
version = "0.3.31"
|
||||
version = "0.3.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "05f29059c0c2090612e8d742178b0580d2dc940c837851ad723096f87af6663e"
|
||||
checksum = "7e3450815272ef58cec6d564423f6e755e25379b217b0bc688e295ba24df6b1d"
|
||||
|
||||
[[package]]
|
||||
name = "futures-executor"
|
||||
version = "0.3.31"
|
||||
version = "0.3.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1e28d1d997f585e54aebc3f97d39e72338912123a67330d723fdbb564d646c9f"
|
||||
checksum = "baf29c38818342a3b26b5b923639e7b1f4a61fc5e76102d4b1981c6dc7a7579d"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"futures-task",
|
||||
|
|
@ -1438,9 +1454,9 @@ dependencies = [
|
|||
|
||||
[[package]]
|
||||
name = "futures-io"
|
||||
version = "0.3.31"
|
||||
version = "0.3.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9e5c1b78ca4aae1ac06c48a526a655760685149f0d465d21f37abfe57ce075c6"
|
||||
checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718"
|
||||
|
||||
[[package]]
|
||||
name = "futures-lite"
|
||||
|
|
@ -1457,9 +1473,9 @@ dependencies = [
|
|||
|
||||
[[package]]
|
||||
name = "futures-macro"
|
||||
version = "0.3.31"
|
||||
version = "0.3.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "162ee34ebcb7c64a8abebc059ce0fee27c2262618d7b60ed8faf72fef13c3650"
|
||||
checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
|
|
@ -1468,21 +1484,21 @@ dependencies = [
|
|||
|
||||
[[package]]
|
||||
name = "futures-sink"
|
||||
version = "0.3.31"
|
||||
version = "0.3.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e575fab7d1e0dcb8d0c7bcf9a63ee213816ab51902e6d244a95819acacf1d4f7"
|
||||
checksum = "c39754e157331b013978ec91992bde1ac089843443c49cbc7f46150b0fad0893"
|
||||
|
||||
[[package]]
|
||||
name = "futures-task"
|
||||
version = "0.3.31"
|
||||
version = "0.3.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f90f7dce0722e95104fcb095585910c0977252f286e354b5e3bd38902cd99988"
|
||||
checksum = "037711b3d59c33004d3856fbdc83b99d4ff37a24768fa1be9ce3538a1cde4393"
|
||||
|
||||
[[package]]
|
||||
name = "futures-util"
|
||||
version = "0.3.31"
|
||||
version = "0.3.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9fa08315bb612088cc391249efdc3bc77536f16c91f6cf495e6fbe85b20a4a81"
|
||||
checksum = "389ca41296e6190b48053de0321d02a77f32f8a5d2461dd38762c0593805c6d6"
|
||||
dependencies = [
|
||||
"futures-channel",
|
||||
"futures-core",
|
||||
|
|
@ -1492,7 +1508,6 @@ dependencies = [
|
|||
"futures-task",
|
||||
"memchr",
|
||||
"pin-project-lite",
|
||||
"pin-utils",
|
||||
"slab",
|
||||
]
|
||||
|
||||
|
|
@ -1703,6 +1718,12 @@ version = "0.5.2"
|
|||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c"
|
||||
|
||||
[[package]]
|
||||
name = "hex"
|
||||
version = "0.4.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70"
|
||||
|
||||
[[package]]
|
||||
name = "hickory-proto"
|
||||
version = "0.25.2"
|
||||
|
|
@ -1756,6 +1777,15 @@ dependencies = [
|
|||
"tracing",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "hmac"
|
||||
version = "0.12.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6c49c37c09c17a53d937dfbb742eb3a961d65a994e6bcdcf37e7399d0cc8ab5e"
|
||||
dependencies = [
|
||||
"digest 0.10.7",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "http"
|
||||
version = "1.4.0"
|
||||
|
|
@ -2479,6 +2509,20 @@ version = "1.0.0"
|
|||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "11d3d7f243d5c5a8b9bb5d6dd2b1602c0cb0b9db1621bafc7ed66e35ff9fe092"
|
||||
|
||||
[[package]]
|
||||
name = "local-runner"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"clap",
|
||||
"ctrlc",
|
||||
"iroh",
|
||||
"runtime-dashboard",
|
||||
"serde_json",
|
||||
"swactor",
|
||||
"swactor-ci",
|
||||
"tokio",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "lock_api"
|
||||
version = "0.4.14"
|
||||
|
|
@ -3786,6 +3830,7 @@ dependencies = [
|
|||
"serde",
|
||||
"serde_json",
|
||||
"swactor",
|
||||
"swactor-ci",
|
||||
"tiny_http",
|
||||
"tracing",
|
||||
"tracing-subscriber",
|
||||
|
|
@ -4057,6 +4102,19 @@ dependencies = [
|
|||
"serde",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_yaml"
|
||||
version = "0.9.34+deprecated"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6a8b1a1a2ebf674015cc02edccce75287f1a0130d394307b36743c2f5d504b47"
|
||||
dependencies = [
|
||||
"indexmap",
|
||||
"itoa",
|
||||
"ryu",
|
||||
"serde",
|
||||
"unsafe-libyaml",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "sha1_smol"
|
||||
version = "1.0.1"
|
||||
|
|
@ -4178,6 +4236,7 @@ dependencies = [
|
|||
"serde_json",
|
||||
"simulation",
|
||||
"swactor",
|
||||
"swactor-ci",
|
||||
"tiny_http",
|
||||
"toml",
|
||||
]
|
||||
|
|
@ -4375,6 +4434,21 @@ dependencies = [
|
|||
"wat",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "swactor-ci"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"hex",
|
||||
"hmac",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"serde_yaml",
|
||||
"sha2 0.10.9",
|
||||
"swactor",
|
||||
"tiny_http",
|
||||
"ureq",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "swactor-datastore"
|
||||
version = "0.1.0"
|
||||
|
|
@ -4973,6 +5047,12 @@ version = "0.2.4"
|
|||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7264e107f553ccae879d21fbea1d6724ac785e8c3bfc762137959b5802826ef3"
|
||||
|
||||
[[package]]
|
||||
name = "unsafe-libyaml"
|
||||
version = "0.2.11"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "673aac59facbab8a9007c7f6108d11f63b603f7cabff99fabf650fea5c32b861"
|
||||
|
||||
[[package]]
|
||||
name = "untrusted"
|
||||
version = "0.9.0"
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
[workspace]
|
||||
members = [".", "crates/python", "crates/wasm", "crates/bin-runner", "crates/simulation", "crates/runtime-dashboard", "crates/distribution", "crates/std", "crates/datastore", "tests/docker"]
|
||||
members = [".", "crates/python", "crates/wasm", "crates/bin-runner", "crates/simulation", "crates/runtime-dashboard", "crates/distribution", "crates/std", "crates/datastore", "tests/docker", "crates/ci", "crates/local-runner", "crates/ci-relay"]
|
||||
exclude = ["tools/depgraph"]
|
||||
|
||||
[package]
|
||||
|
|
|
|||
357
HOW_TO_TEST_DRIVE.md
Normal file
357
HOW_TO_TEST_DRIVE.md
Normal file
|
|
@ -0,0 +1,357 @@
|
|||
# How to Test Drive the Local CI Runner over LAN
|
||||
|
||||
This guide walks through running the local CI runner on the Thinkpad and triggering it from Forgejo on your main laptop. By the end, pushes to your Forgejo repos will automatically run CI pipelines on the Thinkpad.
|
||||
|
||||
## Network Layout
|
||||
|
||||
```
|
||||
┌──────────────────────┐ LAN ┌──────────────────────┐
|
||||
│ Main Laptop │◄───────────────────►│ Thinkpad │
|
||||
│ │ │ │
|
||||
│ Forgejo instance │ webhook POST ───► │ local-runner │
|
||||
│ (e.g. :3000) │ ◄── status API ──── │ (e.g. :8787) │
|
||||
│ │ │ dashboard (:9090) │
|
||||
└──────────────────────┘ └──────────────────────┘
|
||||
```
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- **Main laptop**: Forgejo running and accessible on LAN (e.g. `http://192.168.1.100:3000`)
|
||||
- **Thinkpad**: Rust toolchain installed, this repo cloned, network-reachable from the main laptop
|
||||
- Both machines on the same LAN (or routable to each other)
|
||||
|
||||
Find each machine's LAN IP:
|
||||
|
||||
```bash
|
||||
# On each machine
|
||||
ip addr show | grep 'inet ' | grep -v 127.0.0.1
|
||||
# or
|
||||
hostname -I
|
||||
```
|
||||
|
||||
We'll use these example IPs throughout the guide:
|
||||
- Main laptop (Forgejo): `192.168.1.100`
|
||||
- Thinkpad (CI runner): `192.168.1.200`
|
||||
|
||||
Replace them with your actual IPs.
|
||||
|
||||
---
|
||||
|
||||
## Step 1: Create a Forgejo API Token
|
||||
|
||||
On your main laptop, open Forgejo in a browser:
|
||||
|
||||
1. Go to **Settings > Applications** (top-right user menu > Settings > Applications)
|
||||
2. Create a new token with at least these permissions:
|
||||
- `repo`: read/write (needed to post commit statuses)
|
||||
3. Copy the token (e.g. `abc123def456`)
|
||||
|
||||
---
|
||||
|
||||
## Step 2: Write a `.ci.yml` for Your Repository
|
||||
|
||||
Create a `.ci.yml` file on the Thinkpad. This defines what pipelines and jobs to run.
|
||||
|
||||
Example for a Rust project:
|
||||
|
||||
```yaml
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
exclude: ["master"]
|
||||
jobs:
|
||||
fmt:
|
||||
run: cargo fmt -- --check
|
||||
clippy:
|
||||
run: cargo clippy -- -D warnings
|
||||
test:
|
||||
needs: [fmt, clippy]
|
||||
run: cargo test
|
||||
timeout: 300
|
||||
|
||||
release:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["master"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test --all-features
|
||||
timeout: 600
|
||||
bench:
|
||||
needs: [test]
|
||||
run: cargo bench
|
||||
```
|
||||
|
||||
Key points:
|
||||
- `triggers[].event`: one of `push`, `tag`, `merge`
|
||||
- `triggers[].branches`: glob patterns (`"*"` matches all, `"feature-*"` matches prefixed)
|
||||
- `triggers[].exclude`: branches to skip
|
||||
- `jobs[].run`: a single command string, or a list of commands (run sequentially)
|
||||
- `jobs[].needs`: list of jobs that must pass first (DAG dependencies)
|
||||
- `jobs[].timeout`: max seconds before the job is killed (default: 300)
|
||||
- `jobs[].env`: extra environment variables as key-value pairs
|
||||
|
||||
Save this file somewhere accessible on the Thinkpad, e.g. `/home/user/ci/my-project.ci.yml`.
|
||||
|
||||
---
|
||||
|
||||
## Step 3: Build the Local Runner on the Thinkpad
|
||||
|
||||
SSH into the Thinkpad (or work directly on it):
|
||||
|
||||
```bash
|
||||
# Clone the repo if not already present
|
||||
git clone <repo-url> ~/swactor
|
||||
cd ~/swactor
|
||||
|
||||
# Build the local-runner binary
|
||||
cargo build --release -p local-runner
|
||||
```
|
||||
|
||||
The binary will be at `target/release/local-runner`.
|
||||
|
||||
---
|
||||
|
||||
## Step 4: Create a Working Directory
|
||||
|
||||
The runner clones your repo into a working directory for each pipeline. Create it:
|
||||
|
||||
```bash
|
||||
mkdir -p ~/ci-work
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Step 5: Start the Local Runner
|
||||
|
||||
```bash
|
||||
./target/release/local-runner \
|
||||
--port 8787 \
|
||||
--forgejo-url http://192.168.1.100:3000 \
|
||||
--forgejo-token abc123def456 \
|
||||
--secret my-webhook-secret \
|
||||
--yaml /home/user/ci/my-project.ci.yml \
|
||||
--work-dir /home/user/ci-work \
|
||||
--repo-url http://192.168.1.100:3000/user/repo.git \
|
||||
--dashboard-port 9090
|
||||
```
|
||||
|
||||
| Flag | Description |
|
||||
|------|-------------|
|
||||
| `--port` | Port the webhook listener binds to (default: `8787`) |
|
||||
| `--forgejo-url` | URL of your Forgejo instance on the main laptop |
|
||||
| `--forgejo-token` | API token created in Step 1 |
|
||||
| `--secret` | Webhook secret (must match what you set in Forgejo, see Step 6) |
|
||||
| `--yaml` | Path to the `.ci.yml` file on disk |
|
||||
| `--work-dir` | Base directory for git clones (one subdirectory per pipeline) |
|
||||
| `--repo-url` | Git clone URL for the repository (HTTP or SSH) |
|
||||
| `--dashboard-port` | Optional: enables the web dashboard on this port |
|
||||
|
||||
You should see:
|
||||
|
||||
```
|
||||
Webhook listener on http://0.0.0.0:8787
|
||||
Local CI runner started
|
||||
Webhook: http://0.0.0.0:8787
|
||||
YAML: /home/user/ci/my-project.ci.yml
|
||||
Workdir: /home/user/ci-work
|
||||
Dashboard: http://0.0.0.0:9090
|
||||
```
|
||||
|
||||
The runner is now listening for webhooks.
|
||||
|
||||
---
|
||||
|
||||
## Step 6: Configure the Forgejo Webhook
|
||||
|
||||
On your main laptop, open Forgejo and go to the repo's settings:
|
||||
|
||||
1. Navigate to **Settings > Webhooks > Add Webhook > Forgejo**
|
||||
2. Fill in:
|
||||
- **Target URL**: `http://192.168.1.200:8787` (Thinkpad's LAN IP and runner port)
|
||||
- **Secret**: `my-webhook-secret` (must match `--secret` from Step 5)
|
||||
- **Trigger On**: Choose which events to send:
|
||||
- **Push Events** (for `push` triggers)
|
||||
- **Create Events** (for `tag` triggers)
|
||||
- **Pull Request Events** (for `merge` triggers)
|
||||
- **Branch filter**: leave blank to send all branches, or set a pattern
|
||||
- **Active**: checked
|
||||
3. Click **Add Webhook**
|
||||
|
||||
### Test the webhook connection
|
||||
|
||||
After adding the webhook, Forgejo shows a **Test Delivery** button. Click it to send a test ping. Check the Thinkpad terminal for output. Forgejo also shows the response status — you should see `200 OK`.
|
||||
|
||||
---
|
||||
|
||||
## Step 7: Push and Watch
|
||||
|
||||
From your main laptop (or anywhere with push access):
|
||||
|
||||
```bash
|
||||
cd ~/my-project
|
||||
git checkout -b test-ci
|
||||
echo "// test" >> src/main.rs
|
||||
git add src/main.rs
|
||||
git commit -m "test CI"
|
||||
git push origin test-ci
|
||||
```
|
||||
|
||||
On the Thinkpad terminal, you'll see the runner:
|
||||
|
||||
1. Receive the webhook
|
||||
2. Match triggers in `.ci.yml`
|
||||
3. Queue the pipeline
|
||||
4. Clone the repo and checkout the commit SHA
|
||||
5. Execute jobs one at a time, respecting the DAG order
|
||||
6. Stream stdout/stderr in real time
|
||||
7. Report commit statuses back to Forgejo
|
||||
|
||||
Back in Forgejo, the commit will show status checks (pending, then success/failure) next to the SHA.
|
||||
|
||||
---
|
||||
|
||||
## Step 8: View the Dashboard (Optional)
|
||||
|
||||
If you started with `--dashboard-port 9090`, open a browser on any LAN machine:
|
||||
|
||||
```
|
||||
http://192.168.1.200:9090
|
||||
```
|
||||
|
||||
This shows the swactor runtime dashboard with CI-specific panels: active pipelines, recent pipelines, job statuses, and actor system metrics.
|
||||
|
||||
---
|
||||
|
||||
## Behavior Reference
|
||||
|
||||
### One-at-a-Time Execution
|
||||
|
||||
Jobs run serially — only one job executes at any moment across all pipelines. This guarantees benchmark isolation with no resource contention.
|
||||
|
||||
### Queue Supersede
|
||||
|
||||
If you push twice to the same branch quickly:
|
||||
|
||||
- **First push** is already running: it finishes normally
|
||||
- **Second push** is queued: it runs after the first finishes
|
||||
- **Third push** arrives while second is still queued: the second is **superseded** (marked as error, skipped), and the third takes its place in the queue
|
||||
|
||||
Only queued pipelines get superseded — a running pipeline always runs to completion.
|
||||
|
||||
### DAG Dependencies
|
||||
|
||||
Within a pipeline, jobs respect their `needs` dependencies. If `test` needs `[fmt, clippy]`, then `fmt` runs first, then `clippy`, then `test`. If `fmt` fails, `test` is skipped.
|
||||
|
||||
### CI Environment Variables
|
||||
|
||||
Every job command has these injected:
|
||||
|
||||
| Variable | Value |
|
||||
|----------|-------|
|
||||
| `CI` | `true` |
|
||||
| `CI_COMMIT_SHA` | The commit being tested |
|
||||
| `CI_BRANCH` | The branch name |
|
||||
| `CI_PIPELINE_ID` | Numeric pipeline identifier |
|
||||
| `CI_JOB_NAME` | Name of the current job |
|
||||
|
||||
Plus any `env` keys from the job definition in `.ci.yml`.
|
||||
|
||||
### Commit Status Reporting
|
||||
|
||||
The runner posts status updates to the Forgejo API for each pipeline and each job:
|
||||
|
||||
- `pending` when a pipeline/job is queued
|
||||
- `success` when all jobs pass
|
||||
- `failure` when a job fails
|
||||
- `error` when a pipeline is superseded
|
||||
|
||||
These appear as commit status checks in Forgejo's UI.
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Webhook not reaching the Thinkpad
|
||||
|
||||
- Verify the Thinkpad's firewall allows inbound on the webhook port:
|
||||
```bash
|
||||
# On Thinkpad
|
||||
sudo ufw allow 8787/tcp # if using ufw
|
||||
# or
|
||||
sudo iptables -A INPUT -p tcp --dport 8787 -j ACCEPT
|
||||
```
|
||||
- Confirm connectivity from main laptop:
|
||||
```bash
|
||||
curl -v http://192.168.1.200:8787
|
||||
# Should get "method not allowed" (405) — that means the server is reachable
|
||||
```
|
||||
|
||||
### Signature mismatch (401)
|
||||
|
||||
- The `--secret` flag on the runner must exactly match the **Secret** field in Forgejo's webhook config
|
||||
- If you don't want signature verification, set both to empty strings (omit `--secret` and leave Secret blank in Forgejo)
|
||||
|
||||
### Git clone fails
|
||||
|
||||
- Make sure `--repo-url` is reachable from the Thinkpad:
|
||||
```bash
|
||||
# On Thinkpad
|
||||
git ls-remote http://192.168.1.100:3000/user/repo.git
|
||||
```
|
||||
- If the repo is private, use an authenticated URL:
|
||||
```
|
||||
http://user:password@192.168.1.100:3000/user/repo.git
|
||||
```
|
||||
Or use SSH:
|
||||
```
|
||||
git@192.168.1.100:user/repo.git
|
||||
```
|
||||
|
||||
### Status updates not appearing in Forgejo
|
||||
|
||||
- Verify the API token has `repo` write permissions
|
||||
- Check the Thinkpad terminal for `StatusReporter: failed to post status` errors
|
||||
- Test the token manually:
|
||||
```bash
|
||||
curl -H "Authorization: token abc123def456" \
|
||||
http://192.168.1.100:3000/api/v1/user
|
||||
```
|
||||
|
||||
### Jobs failing unexpectedly
|
||||
|
||||
- Check that the Thinkpad has the necessary toolchain (cargo, rustup, etc.)
|
||||
- The working directory for each pipeline is `{work-dir}/pipeline-{id}/` — you can inspect it
|
||||
- Job stdout/stderr is streamed to the runner's terminal output
|
||||
|
||||
---
|
||||
|
||||
## Quick-Start Cheat Sheet
|
||||
|
||||
```bash
|
||||
# === Thinkpad ===
|
||||
cd ~/swactor
|
||||
cargo build --release -p local-runner
|
||||
mkdir -p ~/ci-work
|
||||
|
||||
./target/release/local-runner \
|
||||
--port 8787 \
|
||||
--forgejo-url http://LAPTOP_IP:3000 \
|
||||
--forgejo-token YOUR_TOKEN \
|
||||
--secret YOUR_SECRET \
|
||||
--yaml /path/to/.ci.yml \
|
||||
--work-dir ~/ci-work \
|
||||
--repo-url http://LAPTOP_IP:3000/user/repo.git \
|
||||
--dashboard-port 9090
|
||||
|
||||
# === Main Laptop (Forgejo) ===
|
||||
# Repo > Settings > Webhooks > Add Webhook:
|
||||
# URL: http://THINKPAD_IP:8787
|
||||
# Secret: YOUR_SECRET
|
||||
# Events: Push, Create, Pull Request
|
||||
|
||||
# === Test it ===
|
||||
git push origin my-branch # triggers CI on the Thinkpad
|
||||
```
|
||||
19
crates/ci-relay/Cargo.toml
Normal file
19
crates/ci-relay/Cargo.toml
Normal file
|
|
@ -0,0 +1,19 @@
|
|||
[package]
|
||||
name = "ci-relay"
|
||||
version = "0.1.0"
|
||||
edition = "2024"
|
||||
|
||||
[[bin]]
|
||||
name = "ci-relay"
|
||||
path = "src/main.rs"
|
||||
|
||||
[dependencies]
|
||||
swactor-ci = { path = "../ci", features = ["local"] }
|
||||
iroh = "0.96"
|
||||
tokio = { version = "1", features = ["rt-multi-thread"] }
|
||||
tiny_http = "0.12"
|
||||
serde_json = "1"
|
||||
hmac = "0.12"
|
||||
sha2 = "0.10"
|
||||
hex = "0.4"
|
||||
clap = { version = "4", features = ["derive"] }
|
||||
243
crates/ci-relay/src/main.rs
Normal file
243
crates/ci-relay/src/main.rs
Normal file
|
|
@ -0,0 +1,243 @@
|
|||
//! ci-relay — webhook relay for VPS side.
|
||||
//!
|
||||
//! Receives Forgejo webhook POSTs over HTTP, then forwards the parsed
|
||||
//! `WebhookEvent` payloads to the Thinkpad local-runner over iroh.
|
||||
|
||||
use std::sync::Arc;
|
||||
use std::time::Duration;
|
||||
|
||||
use clap::Parser;
|
||||
use iroh::{Endpoint, RelayMode};
|
||||
use tokio::sync::Mutex as TokioMutex;
|
||||
|
||||
use swactor_ci::webhook_server::parse_webhook_json;
|
||||
use swactor_ci::{EventType, WebhookEvent};
|
||||
|
||||
/// ALPN protocol identifier for CI relay traffic over iroh.
|
||||
const ALPN: &[u8] = b"swactor/ci/1";
|
||||
|
||||
/// Wire tag for WebhookEvent messages.
|
||||
const WEBHOOK_TAG: &str = "ci::WebhookEvent";
|
||||
|
||||
#[derive(Parser)]
|
||||
#[command(name = "ci-relay", about = "Webhook relay: Forgejo → iroh → local-runner")]
|
||||
struct Args {
|
||||
/// HTTP port for receiving Forgejo webhooks.
|
||||
#[arg(long, default_value = "8787")]
|
||||
port: u16,
|
||||
|
||||
/// Webhook secret for HMAC-SHA256 verification (empty to skip).
|
||||
#[arg(long, default_value = "")]
|
||||
secret: String,
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let args = Args::parse();
|
||||
|
||||
let rt = tokio::runtime::Builder::new_multi_thread()
|
||||
.enable_all()
|
||||
.build()
|
||||
.expect("failed to build tokio runtime");
|
||||
|
||||
let endpoint = rt.block_on(async {
|
||||
Endpoint::builder()
|
||||
.alpns(vec![ALPN.to_vec()])
|
||||
.relay_mode(RelayMode::Default)
|
||||
.bind()
|
||||
.await
|
||||
.expect("failed to bind iroh endpoint")
|
||||
});
|
||||
|
||||
let node_id = endpoint.id();
|
||||
eprintln!("ci-relay started");
|
||||
eprintln!(" Iroh Node ID: {node_id}");
|
||||
eprintln!(" Webhook HTTP: http://0.0.0.0:{}", args.port);
|
||||
eprintln!();
|
||||
eprintln!("Waiting for runner to connect...");
|
||||
|
||||
// Shared state: the active connection from the Thinkpad runner.
|
||||
let connection: Arc<TokioMutex<Option<iroh::endpoint::Connection>>> =
|
||||
Arc::new(TokioMutex::new(None));
|
||||
|
||||
// Spawn a task that accepts inbound iroh connections from the runner.
|
||||
{
|
||||
let endpoint = endpoint.clone();
|
||||
let connection = Arc::clone(&connection);
|
||||
rt.spawn(async move {
|
||||
loop {
|
||||
match endpoint.accept().await {
|
||||
Some(incoming) => match incoming.await {
|
||||
Ok(conn) => {
|
||||
let remote = conn.remote_id();
|
||||
eprintln!("Runner connected: {remote}");
|
||||
*connection.lock().await = Some(conn);
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!("iroh accept error: {e}");
|
||||
}
|
||||
},
|
||||
None => {
|
||||
eprintln!("iroh endpoint closed");
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Run the HTTP webhook listener on a standard thread (blocking).
|
||||
let secret = args.secret.clone();
|
||||
let server = tiny_http::Server::http(format!("0.0.0.0:{}", args.port))
|
||||
.expect("failed to start HTTP server");
|
||||
|
||||
eprintln!("Listening for webhooks...");
|
||||
|
||||
for mut request in server.incoming_requests() {
|
||||
let response = handle_webhook(&mut request, &secret, &connection, &rt);
|
||||
let _ = request.respond(response);
|
||||
}
|
||||
}
|
||||
|
||||
/// Handle an incoming webhook HTTP request.
|
||||
///
|
||||
/// Parses and verifies the webhook, then forwards the event over iroh.
|
||||
fn handle_webhook(
|
||||
request: &mut tiny_http::Request,
|
||||
secret: &str,
|
||||
connection: &Arc<TokioMutex<Option<iroh::endpoint::Connection>>>,
|
||||
rt: &tokio::runtime::Runtime,
|
||||
) -> tiny_http::Response<std::io::Cursor<Vec<u8>>> {
|
||||
use hmac::{Hmac, Mac};
|
||||
use sha2::Sha256;
|
||||
|
||||
if request.method() != &tiny_http::Method::Post {
|
||||
return tiny_http::Response::from_string("method not allowed").with_status_code(405);
|
||||
}
|
||||
|
||||
// Read body.
|
||||
let mut body = String::new();
|
||||
if let Err(e) = std::io::Read::read_to_string(&mut request.as_reader(), &mut body) {
|
||||
eprintln!("webhook: failed to read body: {e}");
|
||||
return tiny_http::Response::from_string("bad request").with_status_code(400);
|
||||
}
|
||||
|
||||
// Verify HMAC-SHA256 signature if secret is non-empty.
|
||||
if !secret.is_empty() {
|
||||
let sig_header = request
|
||||
.headers()
|
||||
.iter()
|
||||
.find(|h| h.field.equiv("X-Forgejo-Signature"))
|
||||
.map(|h| h.value.as_str().to_string());
|
||||
|
||||
match sig_header {
|
||||
Some(sig_hex) => {
|
||||
type HmacSha256 = Hmac<Sha256>;
|
||||
let mut mac =
|
||||
HmacSha256::new_from_slice(secret.as_bytes()).expect("HMAC key creation");
|
||||
hmac::Mac::update(&mut mac, body.as_bytes());
|
||||
let expected = hex::encode(mac.finalize().into_bytes());
|
||||
if sig_hex != expected {
|
||||
eprintln!("webhook: signature mismatch");
|
||||
return tiny_http::Response::from_string("unauthorized").with_status_code(401);
|
||||
}
|
||||
}
|
||||
None => {
|
||||
eprintln!("webhook: missing signature header");
|
||||
return tiny_http::Response::from_string("unauthorized").with_status_code(401);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Determine event type from Forgejo header.
|
||||
let event_header = request
|
||||
.headers()
|
||||
.iter()
|
||||
.find(|h| h.field.equiv("X-Forgejo-Event"))
|
||||
.map(|h| h.value.as_str().to_string())
|
||||
.unwrap_or_default();
|
||||
|
||||
let event_type = match event_header.as_str() {
|
||||
"push" => EventType::Push,
|
||||
"create" => EventType::Tag,
|
||||
"pull_request" => EventType::Merge,
|
||||
other => {
|
||||
eprintln!("webhook: ignoring event type '{other}'");
|
||||
return tiny_http::Response::from_string("ignored").with_status_code(200);
|
||||
}
|
||||
};
|
||||
|
||||
// Parse JSON body.
|
||||
let json: serde_json::Value = match serde_json::from_str(&body) {
|
||||
Ok(v) => v,
|
||||
Err(e) => {
|
||||
eprintln!("webhook: failed to parse JSON: {e}");
|
||||
return tiny_http::Response::from_string("bad json").with_status_code(400);
|
||||
}
|
||||
};
|
||||
|
||||
let webhook_event = match parse_webhook_json(&json, event_type) {
|
||||
Some(e) => e,
|
||||
None => {
|
||||
eprintln!("webhook: could not extract fields from JSON");
|
||||
return tiny_http::Response::from_string("bad payload").with_status_code(400);
|
||||
}
|
||||
};
|
||||
|
||||
eprintln!(
|
||||
"webhook: {} {} on {}/{}",
|
||||
webhook_event.commit_sha.get(..8).unwrap_or(&webhook_event.commit_sha),
|
||||
webhook_event.branch,
|
||||
webhook_event.repo_owner,
|
||||
webhook_event.repo_name,
|
||||
);
|
||||
|
||||
// Forward over iroh.
|
||||
match forward_event(&webhook_event, connection, rt) {
|
||||
Ok(()) => {
|
||||
eprintln!(" → forwarded to runner");
|
||||
tiny_http::Response::from_string("ok").with_status_code(200)
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(" → forward failed: {e}");
|
||||
tiny_http::Response::from_string("relay error").with_status_code(502)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Serialize and send a WebhookEvent over the iroh connection.
|
||||
fn forward_event(
|
||||
event: &WebhookEvent,
|
||||
connection: &Arc<TokioMutex<Option<iroh::endpoint::Connection>>>,
|
||||
rt: &tokio::runtime::Runtime,
|
||||
) -> Result<(), Box<dyn std::error::Error>> {
|
||||
let payload = serde_json::to_vec(event)?;
|
||||
|
||||
rt.block_on(async {
|
||||
let guard = connection.lock().await;
|
||||
let conn = guard.as_ref().ok_or("no runner connected")?;
|
||||
|
||||
let mut send = conn.open_uni().await?;
|
||||
write_tagged_message(&mut send, WEBHOOK_TAG.as_bytes(), &payload).await?;
|
||||
send.finish()?;
|
||||
|
||||
// Wait briefly for the stream to flush.
|
||||
tokio::time::sleep(Duration::from_millis(50)).await;
|
||||
|
||||
Ok(())
|
||||
})
|
||||
}
|
||||
|
||||
/// Write a tagged message to a QUIC send stream.
|
||||
///
|
||||
/// Frame format: `[4B tag_len][tag_bytes][payload_bytes]`
|
||||
async fn write_tagged_message(
|
||||
send: &mut iroh::endpoint::SendStream,
|
||||
tag: &[u8],
|
||||
payload: &[u8],
|
||||
) -> Result<(), Box<dyn std::error::Error>> {
|
||||
let tag_len = (tag.len() as u32).to_be_bytes();
|
||||
send.write_all(&tag_len).await?;
|
||||
send.write_all(tag).await?;
|
||||
send.write_all(payload).await?;
|
||||
Ok(())
|
||||
}
|
||||
29
crates/ci/Cargo.toml
Normal file
29
crates/ci/Cargo.toml
Normal file
|
|
@ -0,0 +1,29 @@
|
|||
[package]
|
||||
name = "swactor-ci"
|
||||
version = "0.1.0"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
swactor = { path = "../..", features = ["serde"] }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde_yaml = "0.9"
|
||||
serde_json = "1"
|
||||
|
||||
# Local runner library dependencies
|
||||
tiny_http = { version = "0.12", optional = true }
|
||||
ureq = { version = "2", optional = true }
|
||||
hmac = { version = "0.12", optional = true }
|
||||
sha2 = { version = "0.10", optional = true }
|
||||
hex = { version = "0.4", optional = true }
|
||||
|
||||
[features]
|
||||
default = []
|
||||
local = [
|
||||
"dep:tiny_http",
|
||||
"dep:ureq",
|
||||
"dep:hmac",
|
||||
"dep:sha2",
|
||||
"dep:hex",
|
||||
]
|
||||
|
||||
[dev-dependencies]
|
||||
426
crates/ci/src/coordinator.rs
Normal file
426
crates/ci/src/coordinator.rs
Normal file
|
|
@ -0,0 +1,426 @@
|
|||
//! Coordinator actor: central brain of the CI system.
|
||||
//!
|
||||
//! Receives webhook events, manages pipeline lifecycles, dispatches jobs
|
||||
//! to the Provisioner and RunnerSupervisor actors.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use swactor::actor::{ActorAddress, ActorInterface, Ctx};
|
||||
|
||||
use crate::pipeline::PipelineExecution;
|
||||
use crate::yaml::{self, CiYaml};
|
||||
use crate::{
|
||||
CiConfig, JobComplete, JobId, JobProgress, JobStatus, PipelineId, ProvisionRequest,
|
||||
ProvisionResponse, StatusUpdate, TerminateRequest, WebhookEvent,
|
||||
};
|
||||
|
||||
/// Messages the Coordinator can receive.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum CoordinatorMsg {
|
||||
/// A webhook event from Forgejo.
|
||||
Webhook(WebhookEvent),
|
||||
/// The CI YAML config to use (loaded externally or from the repo).
|
||||
SetCiYaml(CiYaml),
|
||||
/// Response from the Provisioner.
|
||||
ProvisionResponse(ProvisionResponse),
|
||||
/// Streamed output from a RunnerSupervisor.
|
||||
JobProgress(JobProgress),
|
||||
/// Final result from a RunnerSupervisor.
|
||||
JobComplete(JobComplete),
|
||||
/// Notification that the provisioner is offline (detected via SWIM).
|
||||
ProvisionerOffline,
|
||||
/// Notification that the provisioner is back online.
|
||||
ProvisionerOnline,
|
||||
}
|
||||
|
||||
/// The Coordinator actor state.
|
||||
pub struct Coordinator {
|
||||
config: CiConfig,
|
||||
ci_yaml: Option<CiYaml>,
|
||||
pipelines: HashMap<PipelineId, PipelineExecution>,
|
||||
next_pipeline_id: u64,
|
||||
provisioner_addr: Option<ActorAddress>,
|
||||
provisioner_online: bool,
|
||||
/// Maps job_id → runner supervisor address.
|
||||
runner_addrs: HashMap<JobId, ActorAddress>,
|
||||
/// Captured status updates (for testing/simulation).
|
||||
status_updates: Vec<StatusUpdate>,
|
||||
/// Jobs waiting for the provisioner to come online.
|
||||
queued_provisions: Vec<ProvisionRequest>,
|
||||
}
|
||||
|
||||
impl Coordinator {
|
||||
pub fn new(config: CiConfig) -> Self {
|
||||
Self {
|
||||
config,
|
||||
ci_yaml: None,
|
||||
pipelines: HashMap::new(),
|
||||
next_pipeline_id: 1,
|
||||
provisioner_addr: None,
|
||||
provisioner_online: false,
|
||||
runner_addrs: HashMap::new(),
|
||||
status_updates: Vec::new(),
|
||||
queued_provisions: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn with_provisioner(mut self, addr: ActorAddress) -> Self {
|
||||
self.provisioner_addr = Some(addr);
|
||||
self.provisioner_online = true;
|
||||
self
|
||||
}
|
||||
|
||||
pub fn with_ci_yaml(mut self, yaml: CiYaml) -> Self {
|
||||
self.ci_yaml = Some(yaml);
|
||||
self
|
||||
}
|
||||
|
||||
pub fn pipelines(&self) -> &HashMap<PipelineId, PipelineExecution> {
|
||||
&self.pipelines
|
||||
}
|
||||
|
||||
pub fn status_updates(&self) -> &[StatusUpdate] {
|
||||
&self.status_updates
|
||||
}
|
||||
|
||||
fn handle_webhook(&mut self, ctx: &Ctx, event: WebhookEvent) {
|
||||
let ci = match &self.ci_yaml {
|
||||
Some(ci) => ci.clone(),
|
||||
None => return,
|
||||
};
|
||||
|
||||
let matched = yaml::matching_pipelines(&ci, &event);
|
||||
for pipeline_name in matched {
|
||||
let pipeline_def = &ci.pipelines[&pipeline_name];
|
||||
let pipeline_id = PipelineId(self.next_pipeline_id);
|
||||
self.next_pipeline_id += 1;
|
||||
|
||||
// Build job definitions.
|
||||
let job_defs: Vec<_> = pipeline_def
|
||||
.jobs
|
||||
.iter()
|
||||
.map(|(name, def)| yaml::to_job_definition(name, def))
|
||||
.collect();
|
||||
|
||||
let pipeline = PipelineExecution::new(
|
||||
pipeline_id,
|
||||
pipeline_name.clone(),
|
||||
event.repo_owner.clone(),
|
||||
event.repo_name.clone(),
|
||||
event.commit_sha.clone(),
|
||||
event.branch.clone(),
|
||||
job_defs,
|
||||
);
|
||||
|
||||
// Set pending status on Forgejo.
|
||||
self.emit_status_update(StatusUpdate {
|
||||
repo_owner: event.repo_owner.clone(),
|
||||
repo_name: event.repo_name.clone(),
|
||||
commit_sha: event.commit_sha.clone(),
|
||||
state: "pending".into(),
|
||||
context: format!("ci/{pipeline_name}"),
|
||||
description: format!("Pipeline '{pipeline_name}' is pending"),
|
||||
});
|
||||
|
||||
self.pipelines.insert(pipeline_id, pipeline);
|
||||
|
||||
// Start eligible jobs.
|
||||
self.advance_pipeline(ctx, pipeline_id);
|
||||
}
|
||||
}
|
||||
|
||||
fn advance_pipeline(&mut self, ctx: &Ctx, pipeline_id: PipelineId) {
|
||||
let pipeline = match self.pipelines.get(&pipeline_id) {
|
||||
Some(p) => p,
|
||||
None => return,
|
||||
};
|
||||
|
||||
// If pipeline is already terminal, emit final status.
|
||||
if pipeline.status.is_terminal() {
|
||||
let update = StatusUpdate {
|
||||
repo_owner: pipeline.repo_owner.clone(),
|
||||
repo_name: pipeline.repo_name.clone(),
|
||||
commit_sha: pipeline.commit_sha.clone(),
|
||||
state: pipeline.status.forgejo_state().into(),
|
||||
context: format!("ci/{}", pipeline.pipeline_name),
|
||||
description: format!(
|
||||
"Pipeline '{}' {}",
|
||||
pipeline.pipeline_name,
|
||||
pipeline.status.forgejo_state()
|
||||
),
|
||||
};
|
||||
self.emit_status_update(update);
|
||||
return;
|
||||
}
|
||||
|
||||
let eligible = pipeline.eligible_jobs();
|
||||
let repo_owner = pipeline.repo_owner.clone();
|
||||
let repo_name = pipeline.repo_name.clone();
|
||||
let commit_sha = pipeline.commit_sha.clone();
|
||||
|
||||
for job_name in eligible {
|
||||
let job_id = JobId {
|
||||
pipeline_id,
|
||||
job_name: job_name.clone(),
|
||||
};
|
||||
|
||||
let spec = {
|
||||
let pipeline = self.pipelines.get(&pipeline_id).unwrap();
|
||||
let job = &pipeline.jobs[&job_name];
|
||||
crate::InstanceSpec {
|
||||
docker_required: job.definition.docker,
|
||||
..Default::default()
|
||||
}
|
||||
};
|
||||
|
||||
if self.provisioner_online {
|
||||
// Request provisioning.
|
||||
if let Some(prov_addr) = self.provisioner_addr {
|
||||
let request = ProvisionRequest {
|
||||
job_id: job_id.clone(),
|
||||
instance_spec: spec,
|
||||
};
|
||||
let _ = ctx.send(prov_addr, crate::provisioner::ProvisionerMsg::Provision(request));
|
||||
}
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&pipeline_id) {
|
||||
if let Some(job) = pipeline.jobs.get_mut(&job_name) {
|
||||
job.status = JobStatus::Provisioning;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Queue for later.
|
||||
let request = ProvisionRequest {
|
||||
job_id: job_id.clone(),
|
||||
instance_spec: spec,
|
||||
};
|
||||
self.queued_provisions.push(request);
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&pipeline_id) {
|
||||
if let Some(job) = pipeline.jobs.get_mut(&job_name) {
|
||||
job.status = JobStatus::WaitingForProvisioner;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Emit per-job status.
|
||||
self.emit_status_update(StatusUpdate {
|
||||
repo_owner: repo_owner.clone(),
|
||||
repo_name: repo_name.clone(),
|
||||
commit_sha: commit_sha.clone(),
|
||||
state: "pending".into(),
|
||||
context: format!("ci/{job_name}"),
|
||||
description: format!("Job '{job_name}' is provisioning"),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
fn handle_provision_response(&mut self, ctx: &Ctx, response: ProvisionResponse) {
|
||||
let pipeline_id = response.job_id.pipeline_id;
|
||||
let job_name = response.job_id.job_name.clone();
|
||||
|
||||
match response.result {
|
||||
Ok(instance) => {
|
||||
// Store instance_id for cleanup.
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&pipeline_id) {
|
||||
if let Some(job) = pipeline.jobs.get_mut(&job_name) {
|
||||
job.instance_id = Some(instance.instance_id.clone());
|
||||
job.status = JobStatus::Running;
|
||||
}
|
||||
}
|
||||
|
||||
// Spawn a RunnerSupervisor for this job.
|
||||
let pipeline = &self.pipelines[&pipeline_id];
|
||||
let job = &pipeline.jobs[&job_name];
|
||||
|
||||
let start_job = crate::StartJob {
|
||||
job_id: response.job_id.clone(),
|
||||
instance: instance.clone(),
|
||||
repo_url: format!(
|
||||
"{}/{}/{}",
|
||||
self.config.forgejo_url, pipeline.repo_owner, pipeline.repo_name
|
||||
),
|
||||
commit_sha: pipeline.commit_sha.clone(),
|
||||
job_def: job.definition.clone(),
|
||||
};
|
||||
|
||||
let runner = crate::runner::RunnerSupervisor::new(
|
||||
ctx.self_addr(),
|
||||
start_job,
|
||||
);
|
||||
|
||||
match ctx.spawn(runner) {
|
||||
Ok(runner_addr) => {
|
||||
self.runner_addrs.insert(response.job_id, runner_addr);
|
||||
}
|
||||
Err(_) => {
|
||||
// Failed to spawn runner — mark job as failed.
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&pipeline_id) {
|
||||
pipeline.set_job_status(
|
||||
&job_name,
|
||||
JobStatus::Failed {
|
||||
reason: "failed to spawn runner".into(),
|
||||
},
|
||||
);
|
||||
}
|
||||
self.advance_pipeline(ctx, pipeline_id);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(err) => {
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&pipeline_id) {
|
||||
pipeline.set_job_status(
|
||||
&job_name,
|
||||
JobStatus::Failed {
|
||||
reason: err.to_string(),
|
||||
},
|
||||
);
|
||||
}
|
||||
self.advance_pipeline(ctx, pipeline_id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn handle_job_complete(&mut self, ctx: &Ctx, complete: JobComplete) {
|
||||
let pipeline_id = complete.job_id.pipeline_id;
|
||||
let job_name = complete.job_id.job_name.clone();
|
||||
|
||||
// Send terminate request for the instance.
|
||||
if let Some(pipeline) = self.pipelines.get(&pipeline_id) {
|
||||
if let Some(job) = pipeline.jobs.get(&job_name) {
|
||||
if let Some(ref instance_id) = job.instance_id {
|
||||
let terminate = TerminateRequest {
|
||||
job_id: complete.job_id.clone(),
|
||||
instance_id: instance_id.clone(),
|
||||
};
|
||||
if let Some(prov_addr) = self.provisioner_addr {
|
||||
let _ = ctx.send(
|
||||
prov_addr,
|
||||
crate::provisioner::ProvisionerMsg::Terminate(terminate),
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Update job status.
|
||||
let status = match complete.result {
|
||||
Ok(_) => JobStatus::Passed,
|
||||
Err(ref failure) => JobStatus::Failed {
|
||||
reason: failure.to_string(),
|
||||
},
|
||||
};
|
||||
|
||||
let (repo_owner, repo_name, commit_sha) = {
|
||||
let pipeline = match self.pipelines.get(&pipeline_id) {
|
||||
Some(p) => p,
|
||||
None => return,
|
||||
};
|
||||
(
|
||||
pipeline.repo_owner.clone(),
|
||||
pipeline.repo_name.clone(),
|
||||
pipeline.commit_sha.clone(),
|
||||
)
|
||||
};
|
||||
|
||||
// Emit per-job final status.
|
||||
self.emit_status_update(StatusUpdate {
|
||||
repo_owner,
|
||||
repo_name,
|
||||
commit_sha,
|
||||
state: match &status {
|
||||
JobStatus::Passed => "success".into(),
|
||||
_ => "failure".into(),
|
||||
},
|
||||
context: format!("ci/{job_name}"),
|
||||
description: format!("Job '{job_name}' completed"),
|
||||
});
|
||||
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&pipeline_id) {
|
||||
pipeline.set_job_status(&job_name, status);
|
||||
}
|
||||
|
||||
// Remove runner address.
|
||||
self.runner_addrs.remove(&complete.job_id);
|
||||
|
||||
// Advance pipeline to schedule downstream jobs.
|
||||
self.advance_pipeline(ctx, pipeline_id);
|
||||
}
|
||||
|
||||
fn handle_provisioner_offline(&mut self) {
|
||||
self.provisioner_online = false;
|
||||
|
||||
// Mark all provisioning jobs as waiting.
|
||||
for pipeline in self.pipelines.values_mut() {
|
||||
for job in pipeline.jobs.values_mut() {
|
||||
if job.status == JobStatus::Provisioning {
|
||||
job.status = JobStatus::WaitingForProvisioner;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn handle_provisioner_online(&mut self, ctx: &Ctx) {
|
||||
self.provisioner_online = true;
|
||||
|
||||
// Flush queued provision requests.
|
||||
let queued = std::mem::take(&mut self.queued_provisions);
|
||||
for request in queued {
|
||||
if let Some(prov_addr) = self.provisioner_addr {
|
||||
let _ = ctx.send(prov_addr, crate::provisioner::ProvisionerMsg::Provision(request));
|
||||
}
|
||||
}
|
||||
|
||||
// Re-advance pipelines that have waiting jobs.
|
||||
let pipeline_ids: Vec<PipelineId> = self.pipelines.keys().copied().collect();
|
||||
for pid in pipeline_ids {
|
||||
// Move waiting jobs back to provisioning.
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&pid) {
|
||||
let waiting_jobs: Vec<String> = pipeline
|
||||
.jobs
|
||||
.iter()
|
||||
.filter(|(_, j)| j.status == JobStatus::WaitingForProvisioner)
|
||||
.map(|(name, _)| name.clone())
|
||||
.collect();
|
||||
|
||||
for job_name in waiting_jobs {
|
||||
if let Some(job) = pipeline.jobs.get_mut(&job_name) {
|
||||
job.status = JobStatus::Pending;
|
||||
}
|
||||
}
|
||||
}
|
||||
self.advance_pipeline(ctx, pid);
|
||||
}
|
||||
}
|
||||
|
||||
fn emit_status_update(&mut self, update: StatusUpdate) {
|
||||
self.status_updates.push(update);
|
||||
}
|
||||
}
|
||||
|
||||
impl ActorInterface for Coordinator {
|
||||
type Incoming = CoordinatorMsg;
|
||||
type Response = ();
|
||||
|
||||
fn handle(&mut self, ctx: &Ctx, msg: CoordinatorMsg) {
|
||||
match msg {
|
||||
CoordinatorMsg::Webhook(event) => self.handle_webhook(ctx, event),
|
||||
CoordinatorMsg::SetCiYaml(yaml) => {
|
||||
self.ci_yaml = Some(yaml);
|
||||
}
|
||||
CoordinatorMsg::ProvisionResponse(resp) => {
|
||||
self.handle_provision_response(ctx, resp);
|
||||
}
|
||||
CoordinatorMsg::JobProgress(progress) => {
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&progress.job_id.pipeline_id) {
|
||||
if let Some(job) = pipeline.jobs.get_mut(&progress.job_id.job_name) {
|
||||
job.output_lines.push(progress.output_line);
|
||||
}
|
||||
}
|
||||
}
|
||||
CoordinatorMsg::JobComplete(complete) => {
|
||||
self.handle_job_complete(ctx, complete);
|
||||
}
|
||||
CoordinatorMsg::ProvisionerOffline => self.handle_provisioner_offline(),
|
||||
CoordinatorMsg::ProvisionerOnline => self.handle_provisioner_online(ctx),
|
||||
}
|
||||
}
|
||||
}
|
||||
314
crates/ci/src/lib.rs
Normal file
314
crates/ci/src/lib.rs
Normal file
|
|
@ -0,0 +1,314 @@
|
|||
pub mod coordinator;
|
||||
pub mod local_coordinator;
|
||||
pub mod local_runner;
|
||||
pub mod pipeline;
|
||||
pub mod provisioner;
|
||||
pub mod runner;
|
||||
pub mod status_reporter;
|
||||
pub mod webhook_server;
|
||||
pub mod yaml;
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::fmt;
|
||||
use std::net::IpAddr;
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
// ─── Core Identifiers ───────────────────────────────────────────────────────
|
||||
|
||||
/// Unique identifier for a pipeline execution.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
|
||||
pub struct PipelineId(pub u64);
|
||||
|
||||
impl fmt::Display for PipelineId {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
write!(f, "pipeline-{}", self.0)
|
||||
}
|
||||
}
|
||||
|
||||
/// Unique identifier for a job within a pipeline.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)]
|
||||
pub struct JobId {
|
||||
pub pipeline_id: PipelineId,
|
||||
pub job_name: String,
|
||||
}
|
||||
|
||||
impl fmt::Display for JobId {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
write!(f, "{}/{}", self.pipeline_id, self.job_name)
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Job Status ─────────────────────────────────────────────────────────────
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub enum JobStatus {
|
||||
Pending,
|
||||
WaitingForProvisioner,
|
||||
Provisioning,
|
||||
Running,
|
||||
Passed,
|
||||
Failed { reason: String },
|
||||
Skipped,
|
||||
Interrupted,
|
||||
}
|
||||
|
||||
impl JobStatus {
|
||||
pub fn is_terminal(&self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
JobStatus::Passed | JobStatus::Failed { .. } | JobStatus::Skipped | JobStatus::Interrupted
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Pipeline Status ────────────────────────────────────────────────────────
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub enum PipelineStatus {
|
||||
Pending,
|
||||
Running,
|
||||
Passed,
|
||||
Failed,
|
||||
Error { reason: String },
|
||||
}
|
||||
|
||||
impl PipelineStatus {
|
||||
pub fn is_terminal(&self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
PipelineStatus::Passed | PipelineStatus::Failed | PipelineStatus::Error { .. }
|
||||
)
|
||||
}
|
||||
|
||||
/// Convert to Forgejo commit status string.
|
||||
pub fn forgejo_state(&self) -> &'static str {
|
||||
match self {
|
||||
PipelineStatus::Pending => "pending",
|
||||
PipelineStatus::Running => "pending",
|
||||
PipelineStatus::Passed => "success",
|
||||
PipelineStatus::Failed => "failure",
|
||||
PipelineStatus::Error { .. } => "error",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Instance Types ─────────────────────────────────────────────────────────
|
||||
|
||||
/// Specification for a spot instance.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct InstanceSpec {
|
||||
pub min_cpus: u32,
|
||||
pub min_ram_mb: u32,
|
||||
pub min_disk_gb: u32,
|
||||
pub docker_required: bool,
|
||||
pub region_preferences: Vec<String>,
|
||||
}
|
||||
|
||||
impl Default for InstanceSpec {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
min_cpus: 2,
|
||||
min_ram_mb: 2048,
|
||||
min_disk_gb: 20,
|
||||
docker_required: false,
|
||||
region_preferences: Vec::new(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Connection details for a provisioned spot instance.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct InstanceReady {
|
||||
pub instance_id: String,
|
||||
pub ip: IpAddr,
|
||||
pub ssh_port: u16,
|
||||
pub ssh_host_key: String,
|
||||
}
|
||||
|
||||
// ─── Provisioner ↔ Coordinator Messages ─────────────────────────────────────
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct ProvisionRequest {
|
||||
pub job_id: JobId,
|
||||
pub instance_spec: InstanceSpec,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct ProvisionResponse {
|
||||
pub job_id: JobId,
|
||||
pub result: Result<InstanceReady, ProvisionError>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub enum ProvisionError {
|
||||
NoCapacity,
|
||||
ProviderError(String),
|
||||
Timeout,
|
||||
ProvisionerOffline,
|
||||
}
|
||||
|
||||
impl fmt::Display for ProvisionError {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
match self {
|
||||
ProvisionError::NoCapacity => write!(f, "no capacity available"),
|
||||
ProvisionError::ProviderError(msg) => write!(f, "provider error: {msg}"),
|
||||
ProvisionError::Timeout => write!(f, "provisioning timed out"),
|
||||
ProvisionError::ProvisionerOffline => write!(f, "provisioner is offline"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct TerminateRequest {
|
||||
pub job_id: JobId,
|
||||
pub instance_id: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct TerminateAck {
|
||||
pub job_id: JobId,
|
||||
}
|
||||
|
||||
// ─── Coordinator ↔ RunnerSupervisor Messages ────────────────────────────────
|
||||
|
||||
/// Sent from Coordinator to RunnerSupervisor to begin a job.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct StartJob {
|
||||
pub job_id: JobId,
|
||||
pub instance: InstanceReady,
|
||||
pub repo_url: String,
|
||||
pub commit_sha: String,
|
||||
pub job_def: JobDefinition,
|
||||
}
|
||||
|
||||
/// Streamed output from RunnerSupervisor back to Coordinator.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct JobProgress {
|
||||
pub job_id: JobId,
|
||||
pub output_line: String,
|
||||
}
|
||||
|
||||
/// Final result from RunnerSupervisor.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct JobComplete {
|
||||
pub job_id: JobId,
|
||||
pub result: Result<JobSuccess, JobFailure>,
|
||||
pub artifacts: Vec<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct JobSuccess;
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub enum JobFailure {
|
||||
CommandFailed { exit_code: i32, last_lines: Vec<String> },
|
||||
SshError(String),
|
||||
ExecError(String),
|
||||
Timeout,
|
||||
Interrupted,
|
||||
}
|
||||
|
||||
impl fmt::Display for JobFailure {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
match self {
|
||||
JobFailure::CommandFailed { exit_code, .. } => {
|
||||
write!(f, "command exited with code {exit_code}")
|
||||
}
|
||||
JobFailure::SshError(msg) => write!(f, "SSH error: {msg}"),
|
||||
JobFailure::ExecError(msg) => write!(f, "exec error: {msg}"),
|
||||
JobFailure::Timeout => write!(f, "job timed out"),
|
||||
JobFailure::Interrupted => write!(f, "spot instance interrupted"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Job Definition ─────────────────────────────────────────────────────────
|
||||
|
||||
/// A parsed job from the .ci.yml file.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct JobDefinition {
|
||||
pub name: String,
|
||||
pub run: Vec<String>,
|
||||
pub needs: Vec<String>,
|
||||
pub timeout_secs: u64,
|
||||
pub docker: bool,
|
||||
pub artifacts: Vec<String>,
|
||||
pub env: HashMap<String, String>,
|
||||
}
|
||||
|
||||
// ─── Webhook Types ──────────────────────────────────────────────────────────
|
||||
|
||||
/// Parsed webhook event from Forgejo.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct WebhookEvent {
|
||||
pub event_type: EventType,
|
||||
pub repo_owner: String,
|
||||
pub repo_name: String,
|
||||
pub branch: String,
|
||||
pub commit_sha: String,
|
||||
pub tag: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub enum EventType {
|
||||
Push,
|
||||
Tag,
|
||||
Merge,
|
||||
}
|
||||
|
||||
// ─── Forgejo Status Updates ─────────────────────────────────────────────────
|
||||
|
||||
/// A commit status update to send to Forgejo's API.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct StatusUpdate {
|
||||
pub repo_owner: String,
|
||||
pub repo_name: String,
|
||||
pub commit_sha: String,
|
||||
pub state: String,
|
||||
pub context: String,
|
||||
pub description: String,
|
||||
}
|
||||
|
||||
// ─── Coordinator Config ─────────────────────────────────────────────────────
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct CiConfig {
|
||||
pub webhook_port: u16,
|
||||
pub webhook_secret: String,
|
||||
pub forgejo_url: String,
|
||||
pub forgejo_token: String,
|
||||
pub data_dir: String,
|
||||
}
|
||||
|
||||
impl Default for CiConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
webhook_port: 8787,
|
||||
webhook_secret: String::new(),
|
||||
forgejo_url: String::new(),
|
||||
forgejo_token: String::new(),
|
||||
data_dir: "./ci-data".into(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Local CI Types ──────────────────────────────────────────────────────────
|
||||
|
||||
/// Job execution request for local runner (no InstanceReady needed).
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct LocalStartJob {
|
||||
pub job_id: JobId,
|
||||
pub work_dir: String,
|
||||
pub job_def: JobDefinition,
|
||||
pub env_overrides: HashMap<String, String>,
|
||||
}
|
||||
|
||||
/// Configuration for the local CI runner.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct LocalCiConfig {
|
||||
pub ci: CiConfig,
|
||||
pub repo_url: String,
|
||||
pub work_dir: String,
|
||||
pub ci_yaml_path: String,
|
||||
}
|
||||
485
crates/ci/src/local_coordinator.rs
Normal file
485
crates/ci/src/local_coordinator.rs
Normal file
|
|
@ -0,0 +1,485 @@
|
|||
//! LocalCoordinator actor: single-machine CI brain.
|
||||
//!
|
||||
//! Receives webhook events, queues pipelines, executes jobs one at a time
|
||||
//! directly on the host. Supports branch-level supersede for queued pipelines.
|
||||
|
||||
use std::collections::{HashMap, VecDeque};
|
||||
use std::process::Command;
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
use swactor::actor::{ActorAddress, ActorInterface, Ctx};
|
||||
|
||||
use crate::pipeline::PipelineExecution;
|
||||
use crate::status_reporter::StatusReporterMsg;
|
||||
use crate::yaml::{self, CiYaml};
|
||||
use crate::{
|
||||
JobComplete, JobId, JobProgress, JobStatus, LocalCiConfig, LocalStartJob, PipelineId,
|
||||
PipelineStatus, StatusUpdate, WebhookEvent,
|
||||
};
|
||||
|
||||
/// Messages the LocalCoordinator can receive.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum LocalCoordinatorMsg {
|
||||
Webhook(WebhookEvent),
|
||||
SetCiYaml(CiYaml),
|
||||
JobProgress(JobProgress),
|
||||
JobComplete(JobComplete),
|
||||
GitReady {
|
||||
pipeline_id: PipelineId,
|
||||
job_name: String,
|
||||
work_dir: String,
|
||||
},
|
||||
}
|
||||
|
||||
/// Lightweight snapshot of coordinator state (no dashboard dependency).
|
||||
/// The binary layer converts this to `CiSnapshot` for the dashboard.
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct LocalCiSnapshot {
|
||||
pub active_pipelines: Vec<PipelineExecution>,
|
||||
pub recent_pipelines: Vec<PipelineExecution>,
|
||||
pub has_running_job: bool,
|
||||
}
|
||||
|
||||
/// The LocalCoordinator actor state.
|
||||
pub struct LocalCoordinator {
|
||||
config: LocalCiConfig,
|
||||
ci_yaml: Option<CiYaml>,
|
||||
pipelines: HashMap<PipelineId, PipelineExecution>,
|
||||
next_pipeline_id: u64,
|
||||
/// FIFO queue of pipeline IDs awaiting execution.
|
||||
queue: VecDeque<PipelineId>,
|
||||
/// The pipeline currently being executed.
|
||||
active_pipeline: Option<PipelineId>,
|
||||
/// At most one running job (job_id, runner actor address).
|
||||
running_job: Option<(JobId, ActorAddress)>,
|
||||
/// Status reporter actor address.
|
||||
status_reporter_addr: Option<ActorAddress>,
|
||||
/// Bounded ring of finished pipelines.
|
||||
completed: VecDeque<PipelineExecution>,
|
||||
/// Shared snapshot for external consumers (e.g. dashboard binary).
|
||||
ci_snapshot: Arc<Mutex<LocalCiSnapshot>>,
|
||||
}
|
||||
|
||||
impl LocalCoordinator {
|
||||
pub fn new(config: LocalCiConfig) -> Self {
|
||||
Self {
|
||||
config,
|
||||
ci_yaml: None,
|
||||
pipelines: HashMap::new(),
|
||||
next_pipeline_id: 1,
|
||||
queue: VecDeque::new(),
|
||||
active_pipeline: None,
|
||||
running_job: None,
|
||||
status_reporter_addr: None,
|
||||
completed: VecDeque::new(),
|
||||
ci_snapshot: Arc::new(Mutex::new(LocalCiSnapshot::default())),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn with_status_reporter(mut self, addr: ActorAddress) -> Self {
|
||||
self.status_reporter_addr = Some(addr);
|
||||
self
|
||||
}
|
||||
|
||||
pub fn with_ci_yaml(mut self, yaml: CiYaml) -> Self {
|
||||
self.ci_yaml = Some(yaml);
|
||||
self
|
||||
}
|
||||
|
||||
pub fn ci_snapshot(&self) -> Arc<Mutex<LocalCiSnapshot>> {
|
||||
Arc::clone(&self.ci_snapshot)
|
||||
}
|
||||
|
||||
pub fn pipelines(&self) -> &HashMap<PipelineId, PipelineExecution> {
|
||||
&self.pipelines
|
||||
}
|
||||
|
||||
pub fn completed(&self) -> &VecDeque<PipelineExecution> {
|
||||
&self.completed
|
||||
}
|
||||
|
||||
pub fn queue(&self) -> &VecDeque<PipelineId> {
|
||||
&self.queue
|
||||
}
|
||||
|
||||
pub fn active_pipeline(&self) -> Option<PipelineId> {
|
||||
self.active_pipeline
|
||||
}
|
||||
|
||||
pub fn running_job(&self) -> Option<&(JobId, ActorAddress)> {
|
||||
self.running_job.as_ref()
|
||||
}
|
||||
|
||||
fn handle_webhook(&mut self, ctx: &Ctx, event: WebhookEvent) {
|
||||
let ci = match &self.ci_yaml {
|
||||
Some(ci) => ci.clone(),
|
||||
None => return,
|
||||
};
|
||||
|
||||
let matched = yaml::matching_pipelines(&ci, &event);
|
||||
for pipeline_name in matched {
|
||||
let pipeline_def = &ci.pipelines[&pipeline_name];
|
||||
let pipeline_id = PipelineId(self.next_pipeline_id);
|
||||
self.next_pipeline_id += 1;
|
||||
|
||||
let job_defs: Vec<_> = pipeline_def
|
||||
.jobs
|
||||
.iter()
|
||||
.map(|(name, def)| yaml::to_job_definition(name, def))
|
||||
.collect();
|
||||
|
||||
let pipeline = PipelineExecution::new(
|
||||
pipeline_id,
|
||||
pipeline_name.clone(),
|
||||
event.repo_owner.clone(),
|
||||
event.repo_name.clone(),
|
||||
event.commit_sha.clone(),
|
||||
event.branch.clone(),
|
||||
job_defs,
|
||||
);
|
||||
|
||||
self.emit_status(
|
||||
ctx,
|
||||
StatusUpdate {
|
||||
repo_owner: event.repo_owner.clone(),
|
||||
repo_name: event.repo_name.clone(),
|
||||
commit_sha: event.commit_sha.clone(),
|
||||
state: "pending".into(),
|
||||
context: format!("ci/{pipeline_name}"),
|
||||
description: format!("Pipeline '{pipeline_name}' is pending"),
|
||||
},
|
||||
);
|
||||
|
||||
self.pipelines.insert(pipeline_id, pipeline);
|
||||
self.enqueue_pipeline(pipeline_id, &event.branch);
|
||||
}
|
||||
|
||||
self.try_schedule_next(ctx);
|
||||
}
|
||||
|
||||
/// Enqueue a pipeline, superseding any queued pipeline for the same branch.
|
||||
fn enqueue_pipeline(&mut self, pipeline_id: PipelineId, branch: &str) {
|
||||
// Scan queue for entry with same branch (not the active pipeline).
|
||||
let supersede_idx = self.queue.iter().position(|&qid| {
|
||||
self.pipelines
|
||||
.get(&qid)
|
||||
.map(|p| p.branch == branch)
|
||||
.unwrap_or(false)
|
||||
});
|
||||
|
||||
if let Some(idx) = supersede_idx {
|
||||
let old_id = self.queue[idx];
|
||||
// Mark old pipeline as superseded.
|
||||
if let Some(old_pipeline) = self.pipelines.get_mut(&old_id) {
|
||||
old_pipeline.status = PipelineStatus::Error {
|
||||
reason: "superseded".into(),
|
||||
};
|
||||
// Mark all pending jobs as skipped.
|
||||
let job_names: Vec<String> = old_pipeline.jobs.keys().cloned().collect();
|
||||
for name in job_names {
|
||||
if old_pipeline.jobs[&name].status == JobStatus::Pending {
|
||||
old_pipeline.set_job_status(&name, JobStatus::Skipped);
|
||||
}
|
||||
}
|
||||
}
|
||||
// Archive the superseded pipeline.
|
||||
if let Some(old_pipeline) = self.pipelines.remove(&old_id) {
|
||||
self.archive_pipeline(old_pipeline);
|
||||
}
|
||||
// Replace queue entry.
|
||||
self.queue[idx] = pipeline_id;
|
||||
} else {
|
||||
self.queue.push_back(pipeline_id);
|
||||
}
|
||||
}
|
||||
|
||||
/// Core scheduling: one job at a time.
|
||||
fn try_schedule_next(&mut self, ctx: &Ctx) {
|
||||
// If a job is already running, nothing to do.
|
||||
if self.running_job.is_some() {
|
||||
return;
|
||||
}
|
||||
|
||||
// If we have an active pipeline, try to find eligible jobs.
|
||||
if let Some(active_id) = self.active_pipeline {
|
||||
if let Some(pipeline) = self.pipelines.get(&active_id) {
|
||||
let eligible = pipeline.eligible_jobs();
|
||||
if !eligible.is_empty() {
|
||||
let job_name = eligible[0].clone();
|
||||
self.start_job(ctx, active_id, &job_name);
|
||||
return;
|
||||
}
|
||||
|
||||
// No eligible jobs — check if pipeline is terminal.
|
||||
if pipeline.status.is_terminal() {
|
||||
let pipeline = self.pipelines.remove(&active_id).unwrap();
|
||||
self.emit_status(
|
||||
ctx,
|
||||
StatusUpdate {
|
||||
repo_owner: pipeline.repo_owner.clone(),
|
||||
repo_name: pipeline.repo_name.clone(),
|
||||
commit_sha: pipeline.commit_sha.clone(),
|
||||
state: pipeline.status.forgejo_state().into(),
|
||||
context: format!("ci/{}", pipeline.pipeline_name),
|
||||
description: format!(
|
||||
"Pipeline '{}' {}",
|
||||
pipeline.pipeline_name,
|
||||
pipeline.status.forgejo_state()
|
||||
),
|
||||
},
|
||||
);
|
||||
self.archive_pipeline(pipeline);
|
||||
self.active_pipeline = None;
|
||||
// Recurse to pick next from queue.
|
||||
self.try_schedule_next(ctx);
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Pipeline exists but no eligible jobs and not terminal — waiting for running job.
|
||||
return;
|
||||
}
|
||||
|
||||
// No active pipeline — pop from queue.
|
||||
if let Some(next_id) = self.queue.pop_front() {
|
||||
self.active_pipeline = Some(next_id);
|
||||
self.try_schedule_next(ctx);
|
||||
}
|
||||
}
|
||||
|
||||
fn start_job(&mut self, ctx: &Ctx, pipeline_id: PipelineId, job_name: &str) {
|
||||
let pipeline = match self.pipelines.get_mut(&pipeline_id) {
|
||||
Some(p) => p,
|
||||
None => return,
|
||||
};
|
||||
let job = match pipeline.jobs.get_mut(job_name) {
|
||||
Some(j) => j,
|
||||
None => return,
|
||||
};
|
||||
|
||||
job.status = JobStatus::Running;
|
||||
|
||||
let work_dir = format!(
|
||||
"{}/pipeline-{}",
|
||||
self.config.work_dir, pipeline_id.0
|
||||
);
|
||||
|
||||
// Build CI env overrides.
|
||||
let mut env_overrides = HashMap::new();
|
||||
env_overrides.insert("CI".into(), "true".into());
|
||||
env_overrides.insert("CI_COMMIT_SHA".into(), pipeline.commit_sha.clone());
|
||||
env_overrides.insert("CI_BRANCH".into(), pipeline.branch.clone());
|
||||
env_overrides.insert("CI_PIPELINE_ID".into(), pipeline_id.0.to_string());
|
||||
env_overrides.insert("CI_JOB_NAME".into(), job_name.to_string());
|
||||
|
||||
let start_job = LocalStartJob {
|
||||
job_id: job.job_id.clone(),
|
||||
work_dir: work_dir.clone(),
|
||||
job_def: job.definition.clone(),
|
||||
env_overrides,
|
||||
};
|
||||
|
||||
// Perform git checkout inline (blocks this worker, acceptable for local runner).
|
||||
let sha = pipeline.commit_sha.clone();
|
||||
let repo_url = self.config.repo_url.clone();
|
||||
let git_ok = self.git_checkout(&repo_url, &sha, &work_dir);
|
||||
|
||||
if !git_ok {
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&pipeline_id) {
|
||||
pipeline.set_job_status(
|
||||
job_name,
|
||||
JobStatus::Failed {
|
||||
reason: "git checkout failed".into(),
|
||||
},
|
||||
);
|
||||
}
|
||||
self.try_schedule_next(ctx);
|
||||
return;
|
||||
}
|
||||
|
||||
// Emit per-job running status.
|
||||
if let Some(pipeline) = self.pipelines.get(&pipeline_id) {
|
||||
self.emit_status(
|
||||
ctx,
|
||||
StatusUpdate {
|
||||
repo_owner: pipeline.repo_owner.clone(),
|
||||
repo_name: pipeline.repo_name.clone(),
|
||||
commit_sha: pipeline.commit_sha.clone(),
|
||||
state: "pending".into(),
|
||||
context: format!("ci/{job_name}"),
|
||||
description: format!("Job '{job_name}' is running"),
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
// Spawn LocalRunner actor.
|
||||
let runner =
|
||||
crate::local_runner::LocalRunner::new(ctx.self_addr(), start_job);
|
||||
|
||||
match ctx.spawn(runner) {
|
||||
Ok(runner_addr) => {
|
||||
let job_id = JobId {
|
||||
pipeline_id,
|
||||
job_name: job_name.to_string(),
|
||||
};
|
||||
self.running_job = Some((job_id, runner_addr));
|
||||
}
|
||||
Err(_) => {
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&pipeline_id) {
|
||||
pipeline.set_job_status(
|
||||
job_name,
|
||||
JobStatus::Failed {
|
||||
reason: "failed to spawn runner".into(),
|
||||
},
|
||||
);
|
||||
}
|
||||
self.try_schedule_next(ctx);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn git_checkout(&self, repo_url: &str, sha: &str, work_dir: &str) -> bool {
|
||||
let path = std::path::Path::new(work_dir);
|
||||
if path.join(".git").exists() {
|
||||
// Already cloned — fetch and checkout.
|
||||
let fetch = Command::new("git")
|
||||
.args(["fetch", "origin"])
|
||||
.current_dir(work_dir)
|
||||
.output();
|
||||
if fetch.is_err() || !fetch.unwrap().status.success() {
|
||||
return false;
|
||||
}
|
||||
let checkout = Command::new("git")
|
||||
.args(["checkout", sha])
|
||||
.current_dir(work_dir)
|
||||
.output();
|
||||
checkout.map(|o| o.status.success()).unwrap_or(false)
|
||||
} else {
|
||||
// Fresh clone.
|
||||
if let Some(parent) = path.parent() {
|
||||
let _ = std::fs::create_dir_all(parent);
|
||||
}
|
||||
let clone = Command::new("git")
|
||||
.args(["clone", repo_url, work_dir])
|
||||
.output();
|
||||
if clone.is_err() || !clone.as_ref().unwrap().status.success() {
|
||||
return false;
|
||||
}
|
||||
let checkout = Command::new("git")
|
||||
.args(["checkout", sha])
|
||||
.current_dir(work_dir)
|
||||
.output();
|
||||
checkout.map(|o| o.status.success()).unwrap_or(false)
|
||||
}
|
||||
}
|
||||
|
||||
fn handle_job_complete(&mut self, ctx: &Ctx, complete: JobComplete) {
|
||||
let pipeline_id = complete.job_id.pipeline_id;
|
||||
let job_name = complete.job_id.job_name.clone();
|
||||
|
||||
let status = match complete.result {
|
||||
Ok(_) => JobStatus::Passed,
|
||||
Err(ref failure) => JobStatus::Failed {
|
||||
reason: failure.to_string(),
|
||||
},
|
||||
};
|
||||
|
||||
// Emit per-job final status.
|
||||
if let Some(pipeline) = self.pipelines.get(&pipeline_id) {
|
||||
self.emit_status(
|
||||
ctx,
|
||||
StatusUpdate {
|
||||
repo_owner: pipeline.repo_owner.clone(),
|
||||
repo_name: pipeline.repo_name.clone(),
|
||||
commit_sha: pipeline.commit_sha.clone(),
|
||||
state: match &status {
|
||||
JobStatus::Passed => "success".into(),
|
||||
_ => "failure".into(),
|
||||
},
|
||||
context: format!("ci/{job_name}"),
|
||||
description: format!("Job '{job_name}' completed"),
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&pipeline_id) {
|
||||
pipeline.set_job_status(&job_name, status);
|
||||
}
|
||||
|
||||
// Clear running job.
|
||||
self.running_job = None;
|
||||
|
||||
// Schedule next.
|
||||
self.try_schedule_next(ctx);
|
||||
}
|
||||
|
||||
fn emit_status(&self, ctx: &Ctx, update: StatusUpdate) {
|
||||
if let Some(reporter_addr) = self.status_reporter_addr {
|
||||
let _ = ctx.send(
|
||||
reporter_addr,
|
||||
StatusReporterMsg::Report {
|
||||
update,
|
||||
forgejo_url: self.config.ci.forgejo_url.clone(),
|
||||
forgejo_token: self.config.ci.forgejo_token.clone(),
|
||||
},
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
fn archive_pipeline(&mut self, pipeline: PipelineExecution) {
|
||||
self.completed.push_back(pipeline);
|
||||
if self.completed.len() > 50 {
|
||||
self.completed.pop_front();
|
||||
}
|
||||
}
|
||||
|
||||
fn update_snapshot(&self) {
|
||||
let active: Vec<PipelineExecution> = self.pipelines.values().cloned().collect();
|
||||
let recent: Vec<PipelineExecution> = self
|
||||
.completed
|
||||
.iter()
|
||||
.rev()
|
||||
.take(20)
|
||||
.cloned()
|
||||
.collect();
|
||||
|
||||
if let Ok(mut snap) = self.ci_snapshot.lock() {
|
||||
snap.active_pipelines = active;
|
||||
snap.recent_pipelines = recent;
|
||||
snap.has_running_job = self.running_job.is_some();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl ActorInterface for LocalCoordinator {
|
||||
type Incoming = LocalCoordinatorMsg;
|
||||
type Response = ();
|
||||
|
||||
fn handle(&mut self, ctx: &Ctx, msg: LocalCoordinatorMsg) {
|
||||
match msg {
|
||||
LocalCoordinatorMsg::Webhook(event) => self.handle_webhook(ctx, event),
|
||||
LocalCoordinatorMsg::SetCiYaml(yaml) => {
|
||||
self.ci_yaml = Some(yaml);
|
||||
}
|
||||
LocalCoordinatorMsg::JobProgress(progress) => {
|
||||
if let Some(pipeline) = self.pipelines.get_mut(&progress.job_id.pipeline_id) {
|
||||
if let Some(job) = pipeline.jobs.get_mut(&progress.job_id.job_name) {
|
||||
job.output_lines.push(progress.output_line);
|
||||
}
|
||||
}
|
||||
}
|
||||
LocalCoordinatorMsg::JobComplete(complete) => {
|
||||
self.handle_job_complete(ctx, complete);
|
||||
}
|
||||
LocalCoordinatorMsg::GitReady {
|
||||
pipeline_id,
|
||||
job_name,
|
||||
work_dir,
|
||||
} => {
|
||||
// Git ready is used in the async variant; for now handled inline in start_job.
|
||||
let _ = (pipeline_id, job_name, work_dir);
|
||||
}
|
||||
}
|
||||
|
||||
self.update_snapshot();
|
||||
}
|
||||
}
|
||||
237
crates/ci/src/local_runner.rs
Normal file
237
crates/ci/src/local_runner.rs
Normal file
|
|
@ -0,0 +1,237 @@
|
|||
//! LocalRunner actor: executes job commands directly on the host via shell.
|
||||
//!
|
||||
//! Short-lived actor, one per job. Spawned by LocalCoordinator when a job
|
||||
//! is ready to execute.
|
||||
|
||||
use std::io::BufRead;
|
||||
use std::process::{Command, Stdio};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
use swactor::actor::{ActorAddress, ActorInterface, Ctx};
|
||||
|
||||
use crate::local_coordinator::LocalCoordinatorMsg;
|
||||
use crate::{JobComplete, JobFailure, JobProgress, JobSuccess, LocalStartJob};
|
||||
|
||||
/// Messages the LocalRunner can receive.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum LocalRunnerMsg {
|
||||
/// Begin executing the job (sent to self in on_start).
|
||||
Execute,
|
||||
/// Simulated: job completed (for testing without real shell).
|
||||
SimComplete(Result<(), String>),
|
||||
}
|
||||
|
||||
/// LocalRunner actor state.
|
||||
pub struct LocalRunner {
|
||||
coordinator_addr: ActorAddress,
|
||||
start_job: LocalStartJob,
|
||||
}
|
||||
|
||||
impl LocalRunner {
|
||||
pub fn new(coordinator_addr: ActorAddress, start_job: LocalStartJob) -> Self {
|
||||
Self {
|
||||
coordinator_addr,
|
||||
start_job,
|
||||
}
|
||||
}
|
||||
|
||||
/// Execute all commands in the job definition, streaming output back.
|
||||
fn execute(&self, ctx: &Ctx) {
|
||||
let job_id = &self.start_job.job_id;
|
||||
let work_dir = &self.start_job.work_dir;
|
||||
let timeout_secs = self.start_job.job_def.timeout_secs;
|
||||
|
||||
for cmd_str in &self.start_job.job_def.run {
|
||||
// Send progress: command being run.
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
LocalCoordinatorMsg::JobProgress(JobProgress {
|
||||
job_id: job_id.clone(),
|
||||
output_line: format!("$ {cmd_str}"),
|
||||
}),
|
||||
);
|
||||
|
||||
let child_result = Command::new("sh")
|
||||
.arg("-c")
|
||||
.arg(cmd_str)
|
||||
.current_dir(work_dir)
|
||||
.envs(&self.start_job.env_overrides)
|
||||
.envs(&self.start_job.job_def.env)
|
||||
.stdout(Stdio::piped())
|
||||
.stderr(Stdio::piped())
|
||||
.spawn();
|
||||
|
||||
let mut child = match child_result {
|
||||
Ok(c) => c,
|
||||
Err(e) => {
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
LocalCoordinatorMsg::JobComplete(JobComplete {
|
||||
job_id: job_id.clone(),
|
||||
result: Err(JobFailure::ExecError(e.to_string())),
|
||||
artifacts: Vec::new(),
|
||||
}),
|
||||
);
|
||||
ctx.stop_self();
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
// Timeout mechanism: share child handle, spawn thread that kills after timeout.
|
||||
let kill_flag = Arc::new(Mutex::new(false));
|
||||
let kill_flag_clone = Arc::clone(&kill_flag);
|
||||
// Use a pipe to signal the timeout thread when the command finishes.
|
||||
let (done_tx, done_rx) = std::sync::mpsc::channel::<()>();
|
||||
let timeout_handle = std::thread::spawn(move || {
|
||||
// Wait for either timeout or command completion.
|
||||
if done_rx.recv_timeout(std::time::Duration::from_secs(timeout_secs)).is_err() {
|
||||
*kill_flag_clone.lock().unwrap() = true;
|
||||
}
|
||||
});
|
||||
|
||||
// Read stdout line-by-line.
|
||||
let stdout = child.stdout.take();
|
||||
let stderr = child.stderr.take();
|
||||
|
||||
let mut last_lines: Vec<String> = Vec::new();
|
||||
|
||||
if let Some(stdout) = stdout {
|
||||
let reader = std::io::BufReader::new(stdout);
|
||||
for line in reader.lines() {
|
||||
if let Ok(line) = line {
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
LocalCoordinatorMsg::JobProgress(JobProgress {
|
||||
job_id: job_id.clone(),
|
||||
output_line: line.clone(),
|
||||
}),
|
||||
);
|
||||
last_lines.push(line);
|
||||
if last_lines.len() > 50 {
|
||||
last_lines.remove(0);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(stderr) = stderr {
|
||||
let reader = std::io::BufReader::new(stderr);
|
||||
for line in reader.lines() {
|
||||
if let Ok(line) = line {
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
LocalCoordinatorMsg::JobProgress(JobProgress {
|
||||
job_id: job_id.clone(),
|
||||
output_line: format!("[stderr] {line}"),
|
||||
}),
|
||||
);
|
||||
last_lines.push(line);
|
||||
if last_lines.len() > 50 {
|
||||
last_lines.remove(0);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let status = child.wait();
|
||||
|
||||
// Signal timeout thread that command finished.
|
||||
let _ = done_tx.send(());
|
||||
let _ = timeout_handle.join();
|
||||
|
||||
// Check if killed by timeout.
|
||||
if *kill_flag.lock().unwrap() {
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
LocalCoordinatorMsg::JobComplete(JobComplete {
|
||||
job_id: job_id.clone(),
|
||||
result: Err(JobFailure::Timeout),
|
||||
artifacts: Vec::new(),
|
||||
}),
|
||||
);
|
||||
ctx.stop_self();
|
||||
return;
|
||||
}
|
||||
|
||||
match status {
|
||||
Ok(exit) if exit.success() => {
|
||||
// Command passed, continue to next.
|
||||
}
|
||||
Ok(exit) => {
|
||||
let exit_code = exit.code().unwrap_or(-1);
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
LocalCoordinatorMsg::JobComplete(JobComplete {
|
||||
job_id: job_id.clone(),
|
||||
result: Err(JobFailure::CommandFailed {
|
||||
exit_code,
|
||||
last_lines,
|
||||
}),
|
||||
artifacts: Vec::new(),
|
||||
}),
|
||||
);
|
||||
ctx.stop_self();
|
||||
return;
|
||||
}
|
||||
Err(e) => {
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
LocalCoordinatorMsg::JobComplete(JobComplete {
|
||||
job_id: job_id.clone(),
|
||||
result: Err(JobFailure::ExecError(e.to_string())),
|
||||
artifacts: Vec::new(),
|
||||
}),
|
||||
);
|
||||
ctx.stop_self();
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// All commands passed.
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
LocalCoordinatorMsg::JobComplete(JobComplete {
|
||||
job_id: job_id.clone(),
|
||||
result: Ok(JobSuccess),
|
||||
artifacts: Vec::new(),
|
||||
}),
|
||||
);
|
||||
ctx.stop_self();
|
||||
}
|
||||
}
|
||||
|
||||
impl ActorInterface for LocalRunner {
|
||||
type Incoming = LocalRunnerMsg;
|
||||
type Response = ();
|
||||
|
||||
fn on_start(&mut self, ctx: &Ctx) {
|
||||
let _ = ctx.send(ctx.self_addr(), LocalRunnerMsg::Execute);
|
||||
}
|
||||
|
||||
fn handle(&mut self, ctx: &Ctx, msg: LocalRunnerMsg) {
|
||||
match msg {
|
||||
LocalRunnerMsg::Execute => {
|
||||
self.execute(ctx);
|
||||
}
|
||||
LocalRunnerMsg::SimComplete(result) => {
|
||||
let complete = JobComplete {
|
||||
job_id: self.start_job.job_id.clone(),
|
||||
result: match result {
|
||||
Ok(()) => Ok(JobSuccess),
|
||||
Err(msg) => Err(JobFailure::CommandFailed {
|
||||
exit_code: 1,
|
||||
last_lines: vec![msg],
|
||||
}),
|
||||
},
|
||||
artifacts: Vec::new(),
|
||||
};
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
LocalCoordinatorMsg::JobComplete(complete),
|
||||
);
|
||||
ctx.stop_self();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
391
crates/ci/src/pipeline.rs
Normal file
391
crates/ci/src/pipeline.rs
Normal file
|
|
@ -0,0 +1,391 @@
|
|||
//! Pipeline resolution, DAG execution logic, and job ordering.
|
||||
|
||||
use std::collections::{HashMap, HashSet, VecDeque};
|
||||
|
||||
use crate::{JobDefinition, JobId, JobStatus, PipelineId, PipelineStatus};
|
||||
|
||||
/// A pipeline execution: tracks the DAG of jobs and their statuses.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct PipelineExecution {
|
||||
pub pipeline_id: PipelineId,
|
||||
pub pipeline_name: String,
|
||||
pub repo_owner: String,
|
||||
pub repo_name: String,
|
||||
pub commit_sha: String,
|
||||
pub branch: String,
|
||||
pub status: PipelineStatus,
|
||||
pub jobs: HashMap<String, JobExecution>,
|
||||
}
|
||||
|
||||
/// State of a single job within a pipeline execution.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct JobExecution {
|
||||
pub job_id: JobId,
|
||||
pub definition: JobDefinition,
|
||||
pub status: JobStatus,
|
||||
pub output_lines: Vec<String>,
|
||||
pub instance_id: Option<String>,
|
||||
}
|
||||
|
||||
impl PipelineExecution {
|
||||
/// Create a new pipeline execution from a set of job definitions.
|
||||
pub fn new(
|
||||
pipeline_id: PipelineId,
|
||||
pipeline_name: String,
|
||||
repo_owner: String,
|
||||
repo_name: String,
|
||||
commit_sha: String,
|
||||
branch: String,
|
||||
job_defs: Vec<JobDefinition>,
|
||||
) -> Self {
|
||||
let mut jobs = HashMap::new();
|
||||
for def in job_defs {
|
||||
let job_id = JobId {
|
||||
pipeline_id,
|
||||
job_name: def.name.clone(),
|
||||
};
|
||||
jobs.insert(
|
||||
def.name.clone(),
|
||||
JobExecution {
|
||||
job_id,
|
||||
definition: def,
|
||||
status: JobStatus::Pending,
|
||||
output_lines: Vec::new(),
|
||||
instance_id: None,
|
||||
},
|
||||
);
|
||||
}
|
||||
Self {
|
||||
pipeline_id,
|
||||
pipeline_name,
|
||||
repo_owner,
|
||||
repo_name,
|
||||
commit_sha,
|
||||
branch,
|
||||
status: PipelineStatus::Pending,
|
||||
jobs,
|
||||
}
|
||||
}
|
||||
|
||||
/// Return job names that are eligible for execution: Pending with all needs satisfied.
|
||||
pub fn eligible_jobs(&self) -> Vec<String> {
|
||||
self.jobs
|
||||
.values()
|
||||
.filter(|job| {
|
||||
job.status == JobStatus::Pending
|
||||
&& job.definition.needs.iter().all(|dep| {
|
||||
self.jobs
|
||||
.get(dep)
|
||||
.map(|d| d.status == JobStatus::Passed)
|
||||
.unwrap_or(false)
|
||||
})
|
||||
})
|
||||
.map(|job| job.definition.name.clone())
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Mark a job as a given status. If a job fails, propagate skip to dependents.
|
||||
pub fn set_job_status(&mut self, job_name: &str, status: JobStatus) {
|
||||
if let Some(job) = self.jobs.get_mut(job_name) {
|
||||
job.status = status.clone();
|
||||
}
|
||||
|
||||
// If failure or interruption, skip all transitive dependents.
|
||||
if matches!(
|
||||
status,
|
||||
JobStatus::Failed { .. } | JobStatus::Interrupted
|
||||
) {
|
||||
let to_skip = self.transitive_dependents(job_name);
|
||||
for dep_name in to_skip {
|
||||
if let Some(dep_job) = self.jobs.get_mut(&dep_name) {
|
||||
if dep_job.status == JobStatus::Pending {
|
||||
dep_job.status = JobStatus::Skipped;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Update pipeline status.
|
||||
self.update_pipeline_status();
|
||||
}
|
||||
|
||||
/// Compute the overall pipeline status from individual job statuses.
|
||||
fn update_pipeline_status(&mut self) {
|
||||
let all_terminal = self.jobs.values().all(|j| j.status.is_terminal());
|
||||
let any_running = self.jobs.values().any(|j| {
|
||||
matches!(
|
||||
j.status,
|
||||
JobStatus::Running | JobStatus::Provisioning | JobStatus::WaitingForProvisioner
|
||||
)
|
||||
});
|
||||
let any_failed = self.jobs.values().any(|j| {
|
||||
matches!(
|
||||
j.status,
|
||||
JobStatus::Failed { .. } | JobStatus::Interrupted
|
||||
)
|
||||
});
|
||||
|
||||
if all_terminal {
|
||||
self.status = if any_failed {
|
||||
PipelineStatus::Failed
|
||||
} else {
|
||||
PipelineStatus::Passed
|
||||
};
|
||||
} else if any_running
|
||||
|| self
|
||||
.jobs
|
||||
.values()
|
||||
.any(|j| j.status == JobStatus::Passed)
|
||||
{
|
||||
self.status = PipelineStatus::Running;
|
||||
}
|
||||
}
|
||||
|
||||
/// Find all transitive dependents of a job (jobs that directly or indirectly need it).
|
||||
fn transitive_dependents(&self, job_name: &str) -> Vec<String> {
|
||||
let mut result = Vec::new();
|
||||
let mut queue: VecDeque<&str> = VecDeque::new();
|
||||
queue.push_back(job_name);
|
||||
let mut visited = HashSet::new();
|
||||
|
||||
while let Some(current) = queue.pop_front() {
|
||||
for (name, job) in &self.jobs {
|
||||
if job.definition.needs.iter().any(|n| n == current) && visited.insert(name.clone())
|
||||
{
|
||||
result.push(name.clone());
|
||||
queue.push_back(name.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
result
|
||||
}
|
||||
}
|
||||
|
||||
// ─── DAG Validation ─────────────────────────────────────────────────────────
|
||||
|
||||
/// Topological sort of job definitions. Returns ordered job names or an error if cyclic.
|
||||
pub fn topological_sort(jobs: &HashMap<String, JobDefinition>) -> Result<Vec<String>, DagError> {
|
||||
let mut in_degree: HashMap<&str, usize> = HashMap::new();
|
||||
let mut dependents: HashMap<&str, Vec<&str>> = HashMap::new();
|
||||
|
||||
for (name, def) in jobs {
|
||||
in_degree.entry(name.as_str()).or_insert(0);
|
||||
for dep in &def.needs {
|
||||
if !jobs.contains_key(dep) {
|
||||
return Err(DagError::MissingDependency {
|
||||
job: name.clone(),
|
||||
dependency: dep.clone(),
|
||||
});
|
||||
}
|
||||
dependents.entry(dep.as_str()).or_default().push(name.as_str());
|
||||
*in_degree.entry(name.as_str()).or_insert(0) += 1;
|
||||
}
|
||||
}
|
||||
|
||||
let mut queue: VecDeque<&str> = in_degree
|
||||
.iter()
|
||||
.filter(|(_, deg)| **deg == 0)
|
||||
.map(|(&name, _)| name)
|
||||
.collect();
|
||||
|
||||
// Sort the initial queue for deterministic ordering.
|
||||
let mut sorted_queue: Vec<&str> = queue.drain(..).collect();
|
||||
sorted_queue.sort();
|
||||
queue.extend(sorted_queue);
|
||||
|
||||
let mut result = Vec::new();
|
||||
while let Some(node) = queue.pop_front() {
|
||||
result.push(node.to_string());
|
||||
if let Some(deps) = dependents.get(node) {
|
||||
let mut next_nodes = Vec::new();
|
||||
for &dep in deps {
|
||||
if let Some(deg) = in_degree.get_mut(dep) {
|
||||
*deg -= 1;
|
||||
if *deg == 0 {
|
||||
next_nodes.push(dep);
|
||||
}
|
||||
}
|
||||
}
|
||||
next_nodes.sort();
|
||||
queue.extend(next_nodes);
|
||||
}
|
||||
}
|
||||
|
||||
if result.len() != jobs.len() {
|
||||
return Err(DagError::Cycle);
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum DagError {
|
||||
Cycle,
|
||||
MissingDependency { job: String, dependency: String },
|
||||
}
|
||||
|
||||
impl std::fmt::Display for DagError {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
DagError::Cycle => write!(f, "job dependency cycle detected"),
|
||||
DagError::MissingDependency { job, dependency } => {
|
||||
write!(f, "job '{job}' depends on unknown job '{dependency}'")
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for DagError {}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::*;
|
||||
use crate::JobDefinition;
|
||||
|
||||
fn make_job(name: &str, needs: &[&str]) -> JobDefinition {
|
||||
JobDefinition {
|
||||
name: name.into(),
|
||||
run: vec!["echo test".into()],
|
||||
needs: needs.iter().map(|s| s.to_string()).collect(),
|
||||
timeout_secs: 300,
|
||||
docker: false,
|
||||
artifacts: Vec::new(),
|
||||
env: HashMap::new(),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn topological_sort_linear_chain() {
|
||||
let mut jobs = HashMap::new();
|
||||
jobs.insert("a".into(), make_job("a", &[]));
|
||||
jobs.insert("b".into(), make_job("b", &["a"]));
|
||||
jobs.insert("c".into(), make_job("c", &["b"]));
|
||||
|
||||
let order = topological_sort(&jobs).unwrap();
|
||||
assert_eq!(order, vec!["a", "b", "c"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn topological_sort_diamond() {
|
||||
let mut jobs = HashMap::new();
|
||||
jobs.insert("a".into(), make_job("a", &[]));
|
||||
jobs.insert("b".into(), make_job("b", &["a"]));
|
||||
jobs.insert("c".into(), make_job("c", &["a"]));
|
||||
jobs.insert("d".into(), make_job("d", &["b", "c"]));
|
||||
|
||||
let order = topological_sort(&jobs).unwrap();
|
||||
let pos = |name: &str| order.iter().position(|n| n == name).unwrap();
|
||||
assert!(pos("a") < pos("b"));
|
||||
assert!(pos("a") < pos("c"));
|
||||
assert!(pos("b") < pos("d"));
|
||||
assert!(pos("c") < pos("d"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn topological_sort_detects_cycle() {
|
||||
let mut jobs = HashMap::new();
|
||||
jobs.insert("a".into(), make_job("a", &["b"]));
|
||||
jobs.insert("b".into(), make_job("b", &["a"]));
|
||||
|
||||
let result = topological_sort(&jobs);
|
||||
assert!(matches!(result, Err(DagError::Cycle)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn topological_sort_independent_jobs() {
|
||||
let mut jobs = HashMap::new();
|
||||
jobs.insert("a".into(), make_job("a", &[]));
|
||||
jobs.insert("b".into(), make_job("b", &[]));
|
||||
jobs.insert("c".into(), make_job("c", &[]));
|
||||
|
||||
let order = topological_sort(&jobs).unwrap();
|
||||
// All jobs present, order is alphabetical for independent nodes
|
||||
assert_eq!(order.len(), 3);
|
||||
assert_eq!(order, vec!["a", "b", "c"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pipeline_eligible_jobs_respects_dag() {
|
||||
let jobs = vec![
|
||||
make_job("fmt", &[]),
|
||||
make_job("clippy", &[]),
|
||||
make_job("test", &["fmt", "clippy"]),
|
||||
];
|
||||
|
||||
let mut pipeline = PipelineExecution::new(
|
||||
PipelineId(1),
|
||||
"check".into(),
|
||||
"user".into(),
|
||||
"repo".into(),
|
||||
"abc123".into(),
|
||||
"main".into(),
|
||||
jobs,
|
||||
);
|
||||
|
||||
// Initially, fmt and clippy are eligible
|
||||
let mut eligible = pipeline.eligible_jobs();
|
||||
eligible.sort();
|
||||
assert_eq!(eligible, vec!["clippy", "fmt"]);
|
||||
|
||||
// After fmt passes, test is still not eligible (clippy pending)
|
||||
pipeline.set_job_status("fmt", JobStatus::Passed);
|
||||
let eligible = pipeline.eligible_jobs();
|
||||
assert_eq!(eligible, vec!["clippy"]);
|
||||
|
||||
// After clippy passes, test becomes eligible
|
||||
pipeline.set_job_status("clippy", JobStatus::Passed);
|
||||
let eligible = pipeline.eligible_jobs();
|
||||
assert_eq!(eligible, vec!["test"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pipeline_failure_skips_dependents() {
|
||||
let jobs = vec![
|
||||
make_job("fmt", &[]),
|
||||
make_job("test", &["fmt"]),
|
||||
make_job("bench", &["test"]),
|
||||
];
|
||||
|
||||
let mut pipeline = PipelineExecution::new(
|
||||
PipelineId(1),
|
||||
"check".into(),
|
||||
"user".into(),
|
||||
"repo".into(),
|
||||
"abc123".into(),
|
||||
"main".into(),
|
||||
jobs,
|
||||
);
|
||||
|
||||
// fmt fails → test and bench should be skipped
|
||||
pipeline.set_job_status(
|
||||
"fmt",
|
||||
JobStatus::Failed {
|
||||
reason: "formatting error".into(),
|
||||
},
|
||||
);
|
||||
|
||||
assert_eq!(pipeline.jobs["test"].status, JobStatus::Skipped);
|
||||
assert_eq!(pipeline.jobs["bench"].status, JobStatus::Skipped);
|
||||
assert_eq!(pipeline.status, PipelineStatus::Failed);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pipeline_all_pass_yields_passed() {
|
||||
let jobs = vec![make_job("a", &[]), make_job("b", &["a"])];
|
||||
|
||||
let mut pipeline = PipelineExecution::new(
|
||||
PipelineId(1),
|
||||
"check".into(),
|
||||
"user".into(),
|
||||
"repo".into(),
|
||||
"abc123".into(),
|
||||
"main".into(),
|
||||
jobs,
|
||||
);
|
||||
|
||||
pipeline.set_job_status("a", JobStatus::Passed);
|
||||
pipeline.set_job_status("b", JobStatus::Passed);
|
||||
assert_eq!(pipeline.status, PipelineStatus::Passed);
|
||||
}
|
||||
}
|
||||
86
crates/ci/src/provisioner.rs
Normal file
86
crates/ci/src/provisioner.rs
Normal file
|
|
@ -0,0 +1,86 @@
|
|||
//! Provisioner actor: manages spot instance lifecycle via pluggable provider scripts.
|
||||
//!
|
||||
//! In real deployment, runs on the developer's laptop and calls cloud provider APIs.
|
||||
//! In simulation, provisions are driven by the simulation harness.
|
||||
|
||||
use swactor::actor::{ActorAddress, ActorInterface, Ctx};
|
||||
|
||||
use crate::coordinator::CoordinatorMsg;
|
||||
use crate::{
|
||||
InstanceReady, ProvisionError, ProvisionRequest, ProvisionResponse, TerminateRequest,
|
||||
};
|
||||
|
||||
/// Messages the Provisioner can receive.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum ProvisionerMsg {
|
||||
/// Request to provision a new spot instance.
|
||||
Provision(ProvisionRequest),
|
||||
/// Request to terminate a spot instance.
|
||||
Terminate(TerminateRequest),
|
||||
/// Simulated: provisioning result delivered asynchronously.
|
||||
SimProvisionResult {
|
||||
request: ProvisionRequest,
|
||||
result: Result<InstanceReady, ProvisionError>,
|
||||
},
|
||||
}
|
||||
|
||||
/// Provisioner actor state.
|
||||
///
|
||||
/// In real deployment, this would invoke provider scripts.
|
||||
/// In simulation, the sim harness controls provision outcomes.
|
||||
pub struct Provisioner {
|
||||
coordinator_addr: ActorAddress,
|
||||
/// Active instances tracked for cleanup.
|
||||
active_instances: Vec<String>,
|
||||
}
|
||||
|
||||
impl Provisioner {
|
||||
pub fn new(coordinator_addr: ActorAddress) -> Self {
|
||||
Self {
|
||||
coordinator_addr,
|
||||
active_instances: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn active_instances(&self) -> &[String] {
|
||||
&self.active_instances
|
||||
}
|
||||
}
|
||||
|
||||
impl ActorInterface for Provisioner {
|
||||
type Incoming = ProvisionerMsg;
|
||||
type Response = ();
|
||||
|
||||
fn handle(&mut self, ctx: &Ctx, msg: ProvisionerMsg) {
|
||||
match msg {
|
||||
ProvisionerMsg::Provision(_request) => {
|
||||
// In real deployment: invoke provider script, await result.
|
||||
// In simulation: the sim harness sends SimProvisionResult.
|
||||
}
|
||||
ProvisionerMsg::Terminate(request) => {
|
||||
self.active_instances.retain(|id| id != &request.instance_id);
|
||||
// In real deployment: invoke provider destroy script.
|
||||
// In simulation: just track the termination.
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
CoordinatorMsg::ProvisionResponse(ProvisionResponse {
|
||||
job_id: request.job_id.clone(),
|
||||
result: Err(ProvisionError::NoCapacity), // placeholder, terminate doesn't need response
|
||||
}),
|
||||
);
|
||||
}
|
||||
ProvisionerMsg::SimProvisionResult { request, result } => {
|
||||
if let Ok(ref instance) = result {
|
||||
self.active_instances.push(instance.instance_id.clone());
|
||||
}
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
CoordinatorMsg::ProvisionResponse(ProvisionResponse {
|
||||
job_id: request.job_id,
|
||||
result,
|
||||
}),
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
84
crates/ci/src/runner.rs
Normal file
84
crates/ci/src/runner.rs
Normal file
|
|
@ -0,0 +1,84 @@
|
|||
//! RunnerSupervisor actor: manages SSH session and job execution on a spot instance.
|
||||
//!
|
||||
//! Spawned per-job by the Coordinator. Owns the connection to the spot instance.
|
||||
|
||||
use swactor::actor::{ActorAddress, ActorInterface, Ctx};
|
||||
|
||||
use crate::coordinator::CoordinatorMsg;
|
||||
use crate::{JobComplete, JobFailure, JobProgress, JobSuccess, StartJob};
|
||||
|
||||
/// Messages the RunnerSupervisor can receive.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum RunnerMsg {
|
||||
/// Begin executing the job (sent immediately after spawn via on_start).
|
||||
Execute,
|
||||
/// Simulated: job command output line.
|
||||
OutputLine(String),
|
||||
/// Simulated: job completed successfully.
|
||||
SimComplete(Result<(), String>),
|
||||
}
|
||||
|
||||
/// RunnerSupervisor actor state.
|
||||
///
|
||||
/// In real deployment, this would manage an SSH connection.
|
||||
/// In simulation, job execution is driven by external messages.
|
||||
pub struct RunnerSupervisor {
|
||||
coordinator_addr: ActorAddress,
|
||||
start_job: StartJob,
|
||||
}
|
||||
|
||||
impl RunnerSupervisor {
|
||||
pub fn new(coordinator_addr: ActorAddress, start_job: StartJob) -> Self {
|
||||
Self {
|
||||
coordinator_addr,
|
||||
start_job,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl ActorInterface for RunnerSupervisor {
|
||||
type Incoming = RunnerMsg;
|
||||
type Response = ();
|
||||
|
||||
fn on_start(&mut self, ctx: &Ctx) {
|
||||
// In simulation, the sim harness will send SimComplete messages.
|
||||
// In real deployment, this would initiate SSH connection + command execution.
|
||||
let _ = ctx.send(ctx.self_addr(), RunnerMsg::Execute);
|
||||
}
|
||||
|
||||
fn handle(&mut self, ctx: &Ctx, msg: RunnerMsg) {
|
||||
match msg {
|
||||
RunnerMsg::Execute => {
|
||||
// In real mode, we'd SSH into the instance and run commands.
|
||||
// In simulation, this is a no-op; SimComplete drives completion.
|
||||
}
|
||||
RunnerMsg::OutputLine(line) => {
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
CoordinatorMsg::JobProgress(JobProgress {
|
||||
job_id: self.start_job.job_id.clone(),
|
||||
output_line: line,
|
||||
}),
|
||||
);
|
||||
}
|
||||
RunnerMsg::SimComplete(result) => {
|
||||
let complete = JobComplete {
|
||||
job_id: self.start_job.job_id.clone(),
|
||||
result: match result {
|
||||
Ok(()) => Ok(JobSuccess),
|
||||
Err(msg) => Err(JobFailure::CommandFailed {
|
||||
exit_code: 1,
|
||||
last_lines: vec![msg],
|
||||
}),
|
||||
},
|
||||
artifacts: Vec::new(),
|
||||
};
|
||||
let _ = ctx.send(
|
||||
self.coordinator_addr,
|
||||
CoordinatorMsg::JobComplete(complete),
|
||||
);
|
||||
ctx.stop_self();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
75
crates/ci/src/status_reporter.rs
Normal file
75
crates/ci/src/status_reporter.rs
Normal file
|
|
@ -0,0 +1,75 @@
|
|||
//! StatusReporter actor: fire-and-forget Forgejo commit status updates.
|
||||
//!
|
||||
//! Receives status update messages and POSTs them to the Forgejo API.
|
||||
|
||||
use swactor::actor::{ActorInterface, Ctx};
|
||||
|
||||
use crate::StatusUpdate;
|
||||
|
||||
/// Messages the StatusReporter can receive.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum StatusReporterMsg {
|
||||
Report {
|
||||
update: StatusUpdate,
|
||||
forgejo_url: String,
|
||||
forgejo_token: String,
|
||||
},
|
||||
}
|
||||
|
||||
/// StatusReporter actor state.
|
||||
pub struct StatusReporter;
|
||||
|
||||
impl StatusReporter {
|
||||
pub fn new() -> Self {
|
||||
Self
|
||||
}
|
||||
|
||||
#[cfg(feature = "local")]
|
||||
fn post_status(update: &StatusUpdate, forgejo_url: &str, forgejo_token: &str) {
|
||||
let url = format!(
|
||||
"{}/api/v1/repos/{}/{}/statuses/{}",
|
||||
forgejo_url.trim_end_matches('/'),
|
||||
update.repo_owner,
|
||||
update.repo_name,
|
||||
update.commit_sha,
|
||||
);
|
||||
|
||||
let body = serde_json::json!({
|
||||
"state": update.state,
|
||||
"context": update.context,
|
||||
"description": update.description,
|
||||
});
|
||||
|
||||
let result = ureq::post(&url)
|
||||
.set("Authorization", &format!("token {forgejo_token}"))
|
||||
.set("Content-Type", "application/json")
|
||||
.send_string(&body.to_string());
|
||||
|
||||
if let Err(e) = result {
|
||||
eprintln!("StatusReporter: failed to post status to {url}: {e}");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl ActorInterface for StatusReporter {
|
||||
type Incoming = StatusReporterMsg;
|
||||
type Response = ();
|
||||
|
||||
fn handle(&mut self, _ctx: &Ctx, msg: StatusReporterMsg) {
|
||||
match msg {
|
||||
StatusReporterMsg::Report {
|
||||
update,
|
||||
forgejo_url,
|
||||
forgejo_token,
|
||||
} => {
|
||||
#[cfg(feature = "local")]
|
||||
Self::post_status(&update, &forgejo_url, &forgejo_token);
|
||||
|
||||
#[cfg(not(feature = "local"))]
|
||||
{
|
||||
let _ = (update, forgejo_url, forgejo_token);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
188
crates/ci/src/webhook_server.rs
Normal file
188
crates/ci/src/webhook_server.rs
Normal file
|
|
@ -0,0 +1,188 @@
|
|||
//! Webhook HTTP listener: receives Forgejo webhook POSTs and forwards
|
||||
//! them to the LocalCoordinator actor.
|
||||
//!
|
||||
//! Runs as a standard thread (not an actor) using `tiny_http`.
|
||||
|
||||
use crate::{EventType, WebhookEvent};
|
||||
|
||||
/// Start the webhook listener in a new thread.
|
||||
///
|
||||
/// Returns a join handle for the listener thread.
|
||||
#[cfg(feature = "local")]
|
||||
pub fn start_webhook_listener(
|
||||
port: u16,
|
||||
secret: String,
|
||||
runtime: std::sync::Arc<swactor::runtime::Runtime>,
|
||||
coordinator_addr: swactor::actor::ActorAddress,
|
||||
) -> std::thread::JoinHandle<()> {
|
||||
std::thread::Builder::new()
|
||||
.name("webhook-listener".into())
|
||||
.spawn(move || {
|
||||
let server = tiny_http::Server::http(format!("0.0.0.0:{port}"))
|
||||
.expect("failed to start webhook server");
|
||||
|
||||
eprintln!("Webhook listener on http://0.0.0.0:{port}");
|
||||
|
||||
for mut request in server.incoming_requests() {
|
||||
let response = handle_request(&mut request, &secret, &runtime, coordinator_addr);
|
||||
let _ = request.respond(response);
|
||||
}
|
||||
})
|
||||
.expect("failed to spawn webhook listener thread")
|
||||
}
|
||||
|
||||
#[cfg(feature = "local")]
|
||||
fn handle_request(
|
||||
request: &mut tiny_http::Request,
|
||||
secret: &str,
|
||||
runtime: &std::sync::Arc<swactor::runtime::Runtime>,
|
||||
coordinator_addr: swactor::actor::ActorAddress,
|
||||
) -> tiny_http::Response<std::io::Cursor<Vec<u8>>> {
|
||||
use hmac::{Hmac, Mac};
|
||||
use sha2::Sha256;
|
||||
use crate::local_coordinator::LocalCoordinatorMsg;
|
||||
|
||||
// Only accept POST.
|
||||
if request.method() != &tiny_http::Method::Post {
|
||||
return tiny_http::Response::from_string("method not allowed")
|
||||
.with_status_code(405);
|
||||
}
|
||||
|
||||
// Read body.
|
||||
let mut body = String::new();
|
||||
if let Err(e) = std::io::Read::read_to_string(&mut request.as_reader(), &mut body) {
|
||||
eprintln!("webhook: failed to read body: {e}");
|
||||
return tiny_http::Response::from_string("bad request")
|
||||
.with_status_code(400);
|
||||
}
|
||||
|
||||
// Verify HMAC-SHA256 signature if secret is non-empty.
|
||||
if !secret.is_empty() {
|
||||
let sig_header = request
|
||||
.headers()
|
||||
.iter()
|
||||
.find(|h| h.field.equiv("X-Forgejo-Signature"))
|
||||
.map(|h| h.value.as_str().to_string());
|
||||
|
||||
match sig_header {
|
||||
Some(sig_hex) => {
|
||||
type HmacSha256 = Hmac<Sha256>;
|
||||
let mut mac = HmacSha256::new_from_slice(secret.as_bytes())
|
||||
.expect("HMAC key creation");
|
||||
hmac::Mac::update(&mut mac, body.as_bytes());
|
||||
let expected = hex::encode(mac.finalize().into_bytes());
|
||||
if sig_hex != expected {
|
||||
eprintln!("webhook: signature mismatch");
|
||||
return tiny_http::Response::from_string("unauthorized")
|
||||
.with_status_code(401);
|
||||
}
|
||||
}
|
||||
None => {
|
||||
eprintln!("webhook: missing signature header");
|
||||
return tiny_http::Response::from_string("unauthorized")
|
||||
.with_status_code(401);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Determine event type from Forgejo header.
|
||||
let event_header = request
|
||||
.headers()
|
||||
.iter()
|
||||
.find(|h| h.field.equiv("X-Forgejo-Event"))
|
||||
.map(|h| h.value.as_str().to_string())
|
||||
.unwrap_or_default();
|
||||
|
||||
let event_type = match event_header.as_str() {
|
||||
"push" => EventType::Push,
|
||||
"create" => EventType::Tag,
|
||||
"pull_request" => EventType::Merge,
|
||||
other => {
|
||||
eprintln!("webhook: ignoring event type '{other}'");
|
||||
return tiny_http::Response::from_string("ignored").with_status_code(200);
|
||||
}
|
||||
};
|
||||
|
||||
// Parse JSON body to extract fields.
|
||||
let json: serde_json::Value = match serde_json::from_str(&body) {
|
||||
Ok(v) => v,
|
||||
Err(e) => {
|
||||
eprintln!("webhook: failed to parse JSON: {e}");
|
||||
return tiny_http::Response::from_string("bad json").with_status_code(400);
|
||||
}
|
||||
};
|
||||
|
||||
let webhook_event = match parse_webhook_json(&json, event_type) {
|
||||
Some(e) => e,
|
||||
None => {
|
||||
eprintln!("webhook: could not extract webhook fields from JSON");
|
||||
return tiny_http::Response::from_string("bad payload").with_status_code(400);
|
||||
}
|
||||
};
|
||||
|
||||
// Send to coordinator.
|
||||
let _ = runtime.send_to(coordinator_addr, LocalCoordinatorMsg::Webhook(webhook_event));
|
||||
|
||||
tiny_http::Response::from_string("ok").with_status_code(200)
|
||||
}
|
||||
|
||||
/// Parse a Forgejo webhook JSON payload into a WebhookEvent.
|
||||
pub fn parse_webhook_json(json: &serde_json::Value, event_type: EventType) -> Option<WebhookEvent> {
|
||||
let repo = json.get("repository")?;
|
||||
let repo_owner = repo
|
||||
.get("owner")
|
||||
.and_then(|o| o.get("login"))
|
||||
.or_else(|| repo.get("owner").and_then(|o| o.get("username")))
|
||||
.and_then(|v| v.as_str())?
|
||||
.to_string();
|
||||
let repo_name = repo.get("name").and_then(|v| v.as_str())?.to_string();
|
||||
|
||||
let (branch, commit_sha, tag) = match event_type {
|
||||
EventType::Push => {
|
||||
let reference = json.get("ref").and_then(|v| v.as_str()).unwrap_or("");
|
||||
let branch = reference.strip_prefix("refs/heads/").unwrap_or(reference);
|
||||
let sha = json
|
||||
.get("after")
|
||||
.and_then(|v| v.as_str())
|
||||
.unwrap_or("")
|
||||
.to_string();
|
||||
(branch.to_string(), sha, None)
|
||||
}
|
||||
EventType::Tag => {
|
||||
let reference = json.get("ref").and_then(|v| v.as_str()).unwrap_or("");
|
||||
let tag_name = reference.strip_prefix("refs/tags/").unwrap_or(reference);
|
||||
let sha = json
|
||||
.get("sha")
|
||||
.or_else(|| json.get("after"))
|
||||
.and_then(|v| v.as_str())
|
||||
.unwrap_or("")
|
||||
.to_string();
|
||||
(String::new(), sha, Some(tag_name.to_string()))
|
||||
}
|
||||
EventType::Merge => {
|
||||
let pr = json.get("pull_request")?;
|
||||
let branch = pr
|
||||
.get("head")
|
||||
.and_then(|h| h.get("ref"))
|
||||
.and_then(|v| v.as_str())
|
||||
.unwrap_or("")
|
||||
.to_string();
|
||||
let sha = pr
|
||||
.get("head")
|
||||
.and_then(|h| h.get("sha"))
|
||||
.and_then(|v| v.as_str())
|
||||
.unwrap_or("")
|
||||
.to_string();
|
||||
(branch, sha, None)
|
||||
}
|
||||
};
|
||||
|
||||
Some(WebhookEvent {
|
||||
event_type,
|
||||
repo_owner,
|
||||
repo_name,
|
||||
branch,
|
||||
commit_sha,
|
||||
tag,
|
||||
})
|
||||
}
|
||||
432
crates/ci/src/yaml.rs
Normal file
432
crates/ci/src/yaml.rs
Normal file
|
|
@ -0,0 +1,432 @@
|
|||
//! Parser for `.ci.yml` pipeline configuration files.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::{EventType, JobDefinition, WebhookEvent};
|
||||
|
||||
/// Root of a `.ci.yml` file.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct CiYaml {
|
||||
pub pipelines: HashMap<String, PipelineDef>,
|
||||
}
|
||||
|
||||
/// A single pipeline definition.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct PipelineDef {
|
||||
pub triggers: Vec<TriggerDef>,
|
||||
pub jobs: HashMap<String, JobDef>,
|
||||
}
|
||||
|
||||
/// A trigger condition.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct TriggerDef {
|
||||
pub event: TriggerEvent,
|
||||
#[serde(default)]
|
||||
pub branches: Vec<String>,
|
||||
#[serde(default)]
|
||||
pub exclude: Vec<String>,
|
||||
#[serde(default)]
|
||||
pub pattern: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum TriggerEvent {
|
||||
Push,
|
||||
Tag,
|
||||
Merge,
|
||||
}
|
||||
|
||||
/// A job definition in YAML form.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct JobDef {
|
||||
pub run: RunCommand,
|
||||
#[serde(default)]
|
||||
pub needs: Vec<String>,
|
||||
#[serde(default)]
|
||||
pub timeout: Option<u64>,
|
||||
#[serde(default)]
|
||||
pub docker: bool,
|
||||
#[serde(default)]
|
||||
pub artifacts: Vec<String>,
|
||||
#[serde(default)]
|
||||
pub env: HashMap<String, String>,
|
||||
}
|
||||
|
||||
/// `run` can be a single string or an array of strings.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
#[serde(untagged)]
|
||||
pub enum RunCommand {
|
||||
Single(String),
|
||||
Multiple(Vec<String>),
|
||||
}
|
||||
|
||||
impl RunCommand {
|
||||
pub fn into_vec(self) -> Vec<String> {
|
||||
match self {
|
||||
RunCommand::Single(s) => vec![s],
|
||||
RunCommand::Multiple(v) => v,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Parsing ────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Parse a `.ci.yml` string into a `CiYaml`.
|
||||
pub fn parse_ci_yaml(input: &str) -> Result<CiYaml, ParseError> {
|
||||
let ci: CiYaml = serde_yaml::from_str(input).map_err(ParseError::Yaml)?;
|
||||
|
||||
// Validate: check for unknown job references in `needs`
|
||||
for (pipeline_name, pipeline) in &ci.pipelines {
|
||||
let job_names: Vec<&str> = pipeline.jobs.keys().map(|s| s.as_str()).collect();
|
||||
for (job_name, job) in &pipeline.jobs {
|
||||
for dep in &job.needs {
|
||||
if !job_names.contains(&dep.as_str()) {
|
||||
return Err(ParseError::UnknownDependency {
|
||||
pipeline: pipeline_name.clone(),
|
||||
job: job_name.clone(),
|
||||
dependency: dep.clone(),
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(ci)
|
||||
}
|
||||
|
||||
/// Convert a YAML `JobDef` to the runtime `JobDefinition`.
|
||||
pub fn to_job_definition(name: &str, def: &JobDef) -> JobDefinition {
|
||||
JobDefinition {
|
||||
name: name.to_string(),
|
||||
run: def.run.clone().into_vec(),
|
||||
needs: def.needs.clone(),
|
||||
timeout_secs: def.timeout.unwrap_or(300),
|
||||
docker: def.docker,
|
||||
artifacts: def.artifacts.clone(),
|
||||
env: def.env.clone(),
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Trigger Matching ───────────────────────────────────────────────────────
|
||||
|
||||
/// Returns the names of pipelines whose triggers match the given webhook event.
|
||||
pub fn matching_pipelines(ci: &CiYaml, event: &WebhookEvent) -> Vec<String> {
|
||||
ci.pipelines
|
||||
.iter()
|
||||
.filter(|(_, pipeline)| pipeline.triggers.iter().any(|t| trigger_matches(t, event)))
|
||||
.map(|(name, _)| name.clone())
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Check whether a single trigger matches a webhook event.
|
||||
fn trigger_matches(trigger: &TriggerDef, event: &WebhookEvent) -> bool {
|
||||
// Event type must match.
|
||||
let event_matches = match (&trigger.event, &event.event_type) {
|
||||
(TriggerEvent::Push, EventType::Push) => true,
|
||||
(TriggerEvent::Tag, EventType::Tag) => true,
|
||||
(TriggerEvent::Merge, EventType::Merge) => true,
|
||||
_ => false,
|
||||
};
|
||||
if !event_matches {
|
||||
return false;
|
||||
}
|
||||
|
||||
// For tag events, check pattern.
|
||||
if trigger.event == TriggerEvent::Tag {
|
||||
if let Some(ref pattern) = trigger.pattern {
|
||||
return glob_matches(pattern, event.tag.as_deref().unwrap_or(""));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// For push/merge, check branch filters.
|
||||
let branch = &event.branch;
|
||||
|
||||
// If exclude patterns are specified and branch matches any, reject.
|
||||
if trigger.exclude.iter().any(|pat| glob_matches(pat, branch)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// If branch patterns are specified, at least one must match.
|
||||
if trigger.branches.is_empty() {
|
||||
return true;
|
||||
}
|
||||
trigger.branches.iter().any(|pat| glob_matches(pat, branch))
|
||||
}
|
||||
|
||||
/// Simple glob matching supporting `*` (any chars) and `?` (one char).
|
||||
pub fn glob_matches(pattern: &str, text: &str) -> bool {
|
||||
glob_matches_inner(pattern.as_bytes(), text.as_bytes())
|
||||
}
|
||||
|
||||
fn glob_matches_inner(pat: &[u8], text: &[u8]) -> bool {
|
||||
match (pat.first(), text.first()) {
|
||||
(None, None) => true,
|
||||
(Some(b'*'), _) => {
|
||||
// '*' matches zero or more characters
|
||||
glob_matches_inner(&pat[1..], text)
|
||||
|| (!text.is_empty() && glob_matches_inner(pat, &text[1..]))
|
||||
}
|
||||
(Some(b'?'), Some(_)) => glob_matches_inner(&pat[1..], &text[1..]),
|
||||
(Some(a), Some(b)) if a == b => glob_matches_inner(&pat[1..], &text[1..]),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Errors ─────────────────────────────────────────────────────────────────
|
||||
|
||||
#[derive(Debug)]
|
||||
pub enum ParseError {
|
||||
Yaml(serde_yaml::Error),
|
||||
UnknownDependency {
|
||||
pipeline: String,
|
||||
job: String,
|
||||
dependency: String,
|
||||
},
|
||||
}
|
||||
|
||||
impl std::fmt::Display for ParseError {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
ParseError::Yaml(e) => write!(f, "YAML parse error: {e}"),
|
||||
ParseError::UnknownDependency {
|
||||
pipeline,
|
||||
job,
|
||||
dependency,
|
||||
} => write!(
|
||||
f,
|
||||
"pipeline '{pipeline}', job '{job}': unknown dependency '{dependency}'"
|
||||
),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for ParseError {}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn parse_minimal_ci_yaml() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test
|
||||
"#;
|
||||
let ci = parse_ci_yaml(yaml).unwrap();
|
||||
assert_eq!(ci.pipelines.len(), 1);
|
||||
assert!(ci.pipelines.contains_key("check"));
|
||||
let check = &ci.pipelines["check"];
|
||||
assert_eq!(check.jobs.len(), 1);
|
||||
assert!(check.jobs.contains_key("test"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_full_ci_yaml() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
exclude: ["master"]
|
||||
jobs:
|
||||
fmt:
|
||||
run: cargo fmt -- --check
|
||||
clippy:
|
||||
run: cargo clippy --all-features -- -D warnings
|
||||
test:
|
||||
needs: [fmt, clippy]
|
||||
run: cargo test
|
||||
full:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["master"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test --all-features
|
||||
timeout: 600
|
||||
bench:
|
||||
needs: [test]
|
||||
run: cargo bench -- --output-format json
|
||||
artifacts: ["target/criterion/**"]
|
||||
docker:
|
||||
needs: [test]
|
||||
run: cargo test -p docker-tests -- --ignored
|
||||
docker: true
|
||||
release:
|
||||
triggers:
|
||||
- event: tag
|
||||
pattern: "v*"
|
||||
jobs:
|
||||
build:
|
||||
run: cargo build --release
|
||||
artifacts: ["target/release/swactor-node"]
|
||||
"#;
|
||||
let ci = parse_ci_yaml(yaml).unwrap();
|
||||
assert_eq!(ci.pipelines.len(), 3);
|
||||
assert!(ci.pipelines.contains_key("check"));
|
||||
assert!(ci.pipelines.contains_key("full"));
|
||||
assert!(ci.pipelines.contains_key("release"));
|
||||
|
||||
let full = &ci.pipelines["full"];
|
||||
assert_eq!(full.jobs.len(), 3);
|
||||
assert_eq!(full.jobs["bench"].needs, vec!["test"]);
|
||||
assert!(full.jobs["docker"].docker);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unknown_dependency_rejected() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
test:
|
||||
needs: [nonexistent]
|
||||
run: cargo test
|
||||
"#;
|
||||
let result = parse_ci_yaml(yaml);
|
||||
assert!(result.is_err());
|
||||
let err = result.unwrap_err();
|
||||
assert!(
|
||||
matches!(err, ParseError::UnknownDependency { .. }),
|
||||
"expected UnknownDependency, got: {err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trigger_matching_push_branch() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
exclude: ["master"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test
|
||||
full:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["master"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test
|
||||
"#;
|
||||
let ci = parse_ci_yaml(yaml).unwrap();
|
||||
|
||||
// Push to feature branch → matches "check" only
|
||||
let event = WebhookEvent {
|
||||
event_type: EventType::Push,
|
||||
repo_owner: "user".into(),
|
||||
repo_name: "repo".into(),
|
||||
branch: "feature-x".into(),
|
||||
commit_sha: "abc123".into(),
|
||||
tag: None,
|
||||
};
|
||||
let mut matched = matching_pipelines(&ci, &event);
|
||||
matched.sort();
|
||||
assert_eq!(matched, vec!["check"]);
|
||||
|
||||
// Push to master → matches "full" only (check excludes master)
|
||||
let event = WebhookEvent {
|
||||
event_type: EventType::Push,
|
||||
repo_owner: "user".into(),
|
||||
repo_name: "repo".into(),
|
||||
branch: "master".into(),
|
||||
commit_sha: "abc123".into(),
|
||||
tag: None,
|
||||
};
|
||||
let mut matched = matching_pipelines(&ci, &event);
|
||||
matched.sort();
|
||||
assert_eq!(matched, vec!["full"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trigger_matching_tag() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
release:
|
||||
triggers:
|
||||
- event: tag
|
||||
pattern: "v*"
|
||||
jobs:
|
||||
build:
|
||||
run: cargo build --release
|
||||
"#;
|
||||
let ci = parse_ci_yaml(yaml).unwrap();
|
||||
|
||||
let event = WebhookEvent {
|
||||
event_type: EventType::Tag,
|
||||
repo_owner: "user".into(),
|
||||
repo_name: "repo".into(),
|
||||
branch: "master".into(),
|
||||
commit_sha: "abc123".into(),
|
||||
tag: Some("v1.0.0".into()),
|
||||
};
|
||||
let matched = matching_pipelines(&ci, &event);
|
||||
assert_eq!(matched, vec!["release"]);
|
||||
|
||||
// Non-matching tag
|
||||
let event = WebhookEvent {
|
||||
event_type: EventType::Tag,
|
||||
repo_owner: "user".into(),
|
||||
repo_name: "repo".into(),
|
||||
branch: "master".into(),
|
||||
commit_sha: "abc123".into(),
|
||||
tag: Some("nightly-1".into()),
|
||||
};
|
||||
let matched = matching_pipelines(&ci, &event);
|
||||
assert!(matched.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn glob_matching() {
|
||||
assert!(glob_matches("*", "anything"));
|
||||
assert!(glob_matches("v*", "v1.0.0"));
|
||||
assert!(!glob_matches("v*", "nightly"));
|
||||
assert!(glob_matches("feature-?", "feature-x"));
|
||||
assert!(!glob_matches("feature-?", "feature-xy"));
|
||||
assert!(glob_matches("master", "master"));
|
||||
assert!(!glob_matches("master", "main"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_command_single_and_multiple() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
single:
|
||||
run: cargo test
|
||||
multi:
|
||||
run:
|
||||
- cargo fmt -- --check
|
||||
- cargo test
|
||||
"#;
|
||||
let ci = parse_ci_yaml(yaml).unwrap();
|
||||
let check = &ci.pipelines["check"];
|
||||
|
||||
let single = to_job_definition("single", &check.jobs["single"]);
|
||||
assert_eq!(single.run, vec!["cargo test"]);
|
||||
|
||||
let multi = to_job_definition("multi", &check.jobs["multi"]);
|
||||
assert_eq!(multi.run, vec!["cargo fmt -- --check", "cargo test"]);
|
||||
}
|
||||
}
|
||||
18
crates/local-runner/Cargo.toml
Normal file
18
crates/local-runner/Cargo.toml
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
[package]
|
||||
name = "local-runner"
|
||||
version = "0.1.0"
|
||||
edition = "2024"
|
||||
|
||||
[[bin]]
|
||||
name = "local-runner"
|
||||
path = "src/main.rs"
|
||||
|
||||
[dependencies]
|
||||
swactor = { path = "../..", features = ["serde"] }
|
||||
swactor-ci = { path = "../ci", features = ["local"] }
|
||||
runtime-dashboard = { path = "../runtime-dashboard", features = ["ci"] }
|
||||
clap = { version = "4", features = ["derive"] }
|
||||
ctrlc = "3"
|
||||
iroh = "0.96"
|
||||
tokio = { version = "1", features = ["rt-multi-thread"] }
|
||||
serde_json = "1"
|
||||
361
crates/local-runner/src/main.rs
Normal file
361
crates/local-runner/src/main.rs
Normal file
|
|
@ -0,0 +1,361 @@
|
|||
//! local-runner — single-machine CI runner for the Thinkpad.
|
||||
//!
|
||||
//! Receives Forgejo webhooks, queues pipelines, and executes jobs
|
||||
//! one at a time for benchmark isolation.
|
||||
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::thread;
|
||||
use std::time::Duration;
|
||||
|
||||
use clap::Parser;
|
||||
|
||||
use swactor::actor::ActorAddress;
|
||||
use swactor::config::RuntimeConfig;
|
||||
use swactor::runtime::Runtime;
|
||||
|
||||
use runtime_dashboard::ci_collector::{
|
||||
CiSnapshot, CiStatsProvider, JobSnapshot, PipelineSnapshot, ProvisionerStatus,
|
||||
};
|
||||
|
||||
use swactor_ci::local_coordinator::{LocalCiSnapshot, LocalCoordinator, LocalCoordinatorMsg};
|
||||
use swactor_ci::pipeline::PipelineExecution;
|
||||
use swactor_ci::status_reporter::StatusReporter;
|
||||
use swactor_ci::webhook_server;
|
||||
use swactor_ci::yaml;
|
||||
use swactor_ci::{CiConfig, LocalCiConfig};
|
||||
|
||||
#[derive(Parser)]
|
||||
#[command(name = "local-runner", about = "Swactor local CI runner")]
|
||||
struct Args {
|
||||
/// Webhook listen port.
|
||||
#[arg(long, default_value = "8787")]
|
||||
port: u16,
|
||||
|
||||
/// Forgejo instance URL.
|
||||
#[arg(long, default_value = "")]
|
||||
forgejo_url: String,
|
||||
|
||||
/// Forgejo API token.
|
||||
#[arg(long, default_value = "")]
|
||||
forgejo_token: String,
|
||||
|
||||
/// Webhook secret for HMAC verification (empty to skip).
|
||||
#[arg(long, default_value = "")]
|
||||
secret: String,
|
||||
|
||||
/// Path to .ci.yml file.
|
||||
#[arg(long, default_value = ".ci.yml")]
|
||||
yaml: String,
|
||||
|
||||
/// Base directory for git checkouts.
|
||||
#[arg(long, default_value = "./ci-work")]
|
||||
work_dir: String,
|
||||
|
||||
/// Git clone URL for the repository.
|
||||
#[arg(long, default_value = "")]
|
||||
repo_url: String,
|
||||
|
||||
/// Dashboard HTTP port (omit to disable).
|
||||
#[arg(long)]
|
||||
dashboard_port: Option<u16>,
|
||||
|
||||
/// Iroh Node ID of the ci-relay on the VPS (hex).
|
||||
/// When set, webhooks arrive via iroh instead of HTTP.
|
||||
#[arg(long)]
|
||||
relay_node_id: Option<String>,
|
||||
}
|
||||
|
||||
/// Bridge from LocalCiSnapshot to CiSnapshot for the dashboard.
|
||||
struct LocalCiSnapshotProvider {
|
||||
snapshot: Arc<Mutex<LocalCiSnapshot>>,
|
||||
}
|
||||
|
||||
impl CiStatsProvider for LocalCiSnapshotProvider {
|
||||
fn snapshot(&self) -> CiSnapshot {
|
||||
let local = self.snapshot.lock().unwrap().clone();
|
||||
CiSnapshot {
|
||||
active_pipelines: local
|
||||
.active_pipelines
|
||||
.iter()
|
||||
.map(pipeline_to_dashboard)
|
||||
.collect(),
|
||||
recent_pipelines: local
|
||||
.recent_pipelines
|
||||
.iter()
|
||||
.map(pipeline_to_dashboard)
|
||||
.collect(),
|
||||
provisioner_status: ProvisionerStatus::Online,
|
||||
active_instances: if local.has_running_job { 1 } else { 0 },
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn pipeline_to_dashboard(p: &PipelineExecution) -> PipelineSnapshot {
|
||||
PipelineSnapshot {
|
||||
pipeline_id: p.pipeline_id,
|
||||
pipeline_name: p.pipeline_name.clone(),
|
||||
repo_owner: p.repo_owner.clone(),
|
||||
repo_name: p.repo_name.clone(),
|
||||
commit_sha: p.commit_sha.clone(),
|
||||
branch: p.branch.clone(),
|
||||
status: p.status.clone(),
|
||||
jobs: p
|
||||
.jobs
|
||||
.values()
|
||||
.map(|j| JobSnapshot {
|
||||
job_id: j.job_id.clone(),
|
||||
job_name: j.definition.name.clone(),
|
||||
status: j.status.clone(),
|
||||
output_line_count: j.output_lines.len(),
|
||||
})
|
||||
.collect(),
|
||||
}
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let args = Args::parse();
|
||||
let stop = Arc::new(AtomicBool::new(false));
|
||||
|
||||
// Signal handler.
|
||||
{
|
||||
let stop = Arc::clone(&stop);
|
||||
ctrlc::set_handler(move || {
|
||||
stop.store(true, Ordering::Relaxed);
|
||||
})
|
||||
.expect("failed to set signal handler");
|
||||
}
|
||||
|
||||
// Optionally start dashboard.
|
||||
let dash = args.dashboard_port.map(|port| {
|
||||
let d = runtime_dashboard::start_dashboard(runtime_dashboard::DashboardConfig {
|
||||
port,
|
||||
..Default::default()
|
||||
});
|
||||
d.install_tracing();
|
||||
d
|
||||
});
|
||||
|
||||
// Create 2-thread runtime.
|
||||
let num_threads = 2;
|
||||
let collector = runtime_dashboard::collector::StatsCollector::new(num_threads);
|
||||
let mut rt = Runtime::new(RuntimeConfig {
|
||||
num_threads,
|
||||
max_actors: 256,
|
||||
channel_buffer_size: 2000,
|
||||
..Default::default()
|
||||
});
|
||||
rt.set_stats_hook(collector.clone());
|
||||
|
||||
// Build config.
|
||||
let ci_config = CiConfig {
|
||||
webhook_port: args.port,
|
||||
webhook_secret: args.secret.clone(),
|
||||
forgejo_url: args.forgejo_url.clone(),
|
||||
forgejo_token: args.forgejo_token.clone(),
|
||||
data_dir: args.work_dir.clone(),
|
||||
};
|
||||
let local_config = LocalCiConfig {
|
||||
ci: ci_config,
|
||||
repo_url: args.repo_url.clone(),
|
||||
work_dir: args.work_dir.clone(),
|
||||
ci_yaml_path: args.yaml.clone(),
|
||||
};
|
||||
|
||||
// Spawn StatusReporter.
|
||||
let reporter_addr = rt
|
||||
.spawn(StatusReporter::new())
|
||||
.expect("failed to spawn StatusReporter");
|
||||
|
||||
// Spawn LocalCoordinator.
|
||||
let coordinator = LocalCoordinator::new(local_config).with_status_reporter(reporter_addr);
|
||||
let ci_snapshot = coordinator.ci_snapshot();
|
||||
let coordinator_addr = rt
|
||||
.spawn(coordinator)
|
||||
.expect("failed to spawn LocalCoordinator");
|
||||
|
||||
// Load CI YAML from disk.
|
||||
let yaml_content = std::fs::read_to_string(&args.yaml)
|
||||
.unwrap_or_else(|e| panic!("failed to read {}: {e}", args.yaml));
|
||||
let ci_yaml = yaml::parse_ci_yaml(&yaml_content)
|
||||
.unwrap_or_else(|e| panic!("failed to parse CI YAML: {e}"));
|
||||
|
||||
// Start runtime.
|
||||
let handle = rt.run().expect("failed to start runtime");
|
||||
|
||||
// Send CiYaml to coordinator.
|
||||
let _ = handle
|
||||
.runtime
|
||||
.send_to(coordinator_addr, LocalCoordinatorMsg::SetCiYaml(ci_yaml));
|
||||
|
||||
// Wire dashboard.
|
||||
if let Some(ref d) = dash {
|
||||
d.set_runtime(Arc::clone(&handle.runtime), collector);
|
||||
let provider = Arc::new(LocalCiSnapshotProvider {
|
||||
snapshot: ci_snapshot,
|
||||
});
|
||||
d.set_ci(provider);
|
||||
}
|
||||
|
||||
// Start webhook source: iroh relay or HTTP listener.
|
||||
if let Some(ref relay_id_hex) = args.relay_node_id {
|
||||
start_iroh_receiver(
|
||||
relay_id_hex,
|
||||
Arc::clone(&handle.runtime),
|
||||
coordinator_addr,
|
||||
Arc::clone(&stop),
|
||||
);
|
||||
} else {
|
||||
let _webhook_handle = webhook_server::start_webhook_listener(
|
||||
args.port,
|
||||
args.secret,
|
||||
Arc::clone(&handle.runtime),
|
||||
coordinator_addr,
|
||||
);
|
||||
}
|
||||
|
||||
eprintln!("Local CI runner started");
|
||||
if args.relay_node_id.is_some() {
|
||||
eprintln!(" Webhook: via iroh relay");
|
||||
} else {
|
||||
eprintln!(" Webhook: http://0.0.0.0:{}", args.port);
|
||||
}
|
||||
eprintln!(" YAML: {}", args.yaml);
|
||||
eprintln!(" Workdir: {}", args.work_dir);
|
||||
if let Some(port) = args.dashboard_port {
|
||||
eprintln!(" Dashboard: http://0.0.0.0:{port}");
|
||||
}
|
||||
|
||||
// Main loop — just wait for ctrlc.
|
||||
while !stop.load(Ordering::Relaxed) {
|
||||
thread::sleep(Duration::from_millis(100));
|
||||
}
|
||||
|
||||
eprintln!("\nShutting down...");
|
||||
handle.shutdown();
|
||||
if let Some(d) = dash {
|
||||
d.shutdown();
|
||||
}
|
||||
handle.join();
|
||||
}
|
||||
|
||||
// ─── Iroh Webhook Receiver ──────────────────────────────────────────────────
|
||||
|
||||
/// ALPN protocol identifier — must match ci-relay.
|
||||
const CI_ALPN: &[u8] = b"swactor/ci/1";
|
||||
|
||||
/// Connect to the VPS ci-relay via iroh and receive WebhookEvents.
|
||||
///
|
||||
/// Runs in a background thread with its own tokio runtime.
|
||||
fn start_iroh_receiver(
|
||||
relay_id_hex: &str,
|
||||
swactor_rt: Arc<Runtime>,
|
||||
coordinator_addr: ActorAddress,
|
||||
stop: Arc<AtomicBool>,
|
||||
) {
|
||||
let relay_key: iroh::PublicKey = relay_id_hex
|
||||
.parse()
|
||||
.unwrap_or_else(|e| panic!("invalid relay node ID '{relay_id_hex}': {e}"));
|
||||
|
||||
thread::Builder::new()
|
||||
.name("iroh-receiver".into())
|
||||
.spawn(move || {
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.enable_all()
|
||||
.build()
|
||||
.expect("failed to build tokio runtime for iroh receiver");
|
||||
|
||||
rt.block_on(async move {
|
||||
let endpoint = iroh::Endpoint::builder()
|
||||
.alpns(vec![CI_ALPN.to_vec()])
|
||||
.relay_mode(iroh::RelayMode::Default)
|
||||
.bind()
|
||||
.await
|
||||
.expect("failed to bind iroh endpoint");
|
||||
|
||||
eprintln!(" Iroh local ID: {}", endpoint.id());
|
||||
eprintln!(" Connecting to relay {relay_key}...");
|
||||
|
||||
let conn = endpoint
|
||||
.connect(relay_key, CI_ALPN)
|
||||
.await
|
||||
.expect("failed to connect to ci-relay");
|
||||
|
||||
eprintln!(" Connected to relay!");
|
||||
|
||||
// Receive loop: the relay opens uni streams to send us events.
|
||||
while !stop.load(Ordering::Relaxed) {
|
||||
match tokio::time::timeout(Duration::from_secs(1), conn.accept_uni()).await {
|
||||
Ok(Ok(mut recv)) => {
|
||||
match read_tagged_message(&mut recv).await {
|
||||
Ok((tag, payload)) => {
|
||||
if tag == "ci::WebhookEvent" {
|
||||
match serde_json::from_slice::<swactor_ci::WebhookEvent>(
|
||||
&payload,
|
||||
) {
|
||||
Ok(event) => {
|
||||
eprintln!(
|
||||
"iroh: received webhook {} on {}",
|
||||
event
|
||||
.commit_sha
|
||||
.get(..8)
|
||||
.unwrap_or(&event.commit_sha),
|
||||
event.branch,
|
||||
);
|
||||
let _ = swactor_rt.send_to(
|
||||
coordinator_addr,
|
||||
LocalCoordinatorMsg::Webhook(event),
|
||||
);
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!("iroh: failed to deserialize event: {e}")
|
||||
}
|
||||
}
|
||||
} else {
|
||||
eprintln!("iroh: unknown tag '{tag}', ignoring");
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!("iroh: read error: {e}");
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(Err(e)) => {
|
||||
eprintln!("iroh: connection error: {e}");
|
||||
break;
|
||||
}
|
||||
Err(_) => {
|
||||
// Timeout — just loop and check stop flag.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
endpoint.close().await;
|
||||
});
|
||||
})
|
||||
.expect("failed to spawn iroh-receiver thread");
|
||||
}
|
||||
|
||||
/// Read a tagged message from a QUIC recv stream.
|
||||
///
|
||||
/// Frame format: `[4B tag_len][tag_bytes][payload_bytes]`
|
||||
async fn read_tagged_message(
|
||||
recv: &mut iroh::endpoint::RecvStream,
|
||||
) -> Result<(String, Vec<u8>), Box<dyn std::error::Error>> {
|
||||
let mut tag_len_buf = [0u8; 4];
|
||||
recv.read_exact(&mut tag_len_buf).await?;
|
||||
let tag_len = u32::from_be_bytes(tag_len_buf) as usize;
|
||||
|
||||
if tag_len > 1024 {
|
||||
return Err("tag too large".into());
|
||||
}
|
||||
|
||||
let mut tag_buf = vec![0u8; tag_len];
|
||||
recv.read_exact(&mut tag_buf).await?;
|
||||
let tag = String::from_utf8(tag_buf)?;
|
||||
|
||||
let payload = recv.read_to_end(64 * 1024).await?;
|
||||
|
||||
Ok((tag, payload))
|
||||
}
|
||||
|
|
@ -17,6 +17,7 @@ distribution = { path = "../distribution", optional = true }
|
|||
clap = { version = "4", features = ["derive"], optional = true }
|
||||
ctrlc = "3"
|
||||
iroh = { version = "0.96", optional = true }
|
||||
swactor-ci = { path = "../ci", optional = true }
|
||||
|
||||
[features]
|
||||
default = ["distribution"]
|
||||
|
|
@ -25,6 +26,7 @@ distribution = ["dep:distribution"]
|
|||
node = ["distribution", "dep:clap", "swactor/transport", "tcp"]
|
||||
tcp = ["distribution/tcp"]
|
||||
iroh = ["distribution/iroh", "dep:iroh"]
|
||||
ci = ["dep:swactor-ci"]
|
||||
|
||||
[[bin]]
|
||||
name = "swactor-tui"
|
||||
|
|
|
|||
210
crates/runtime-dashboard/src/ci_collector.rs
Normal file
210
crates/runtime-dashboard/src/ci_collector.rs
Normal file
|
|
@ -0,0 +1,210 @@
|
|||
//! CI Dashboard extension: provides HTTP API endpoints and stats for CI pipelines.
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use swactor_ci::{JobId, JobStatus, PipelineId, PipelineStatus};
|
||||
|
||||
// ─── Stats Provider ─────────────────────────────────────────────────────────
|
||||
|
||||
/// Trait for providing CI snapshot data to the dashboard.
|
||||
pub trait CiStatsProvider: Send + Sync {
|
||||
fn snapshot(&self) -> CiSnapshot;
|
||||
}
|
||||
|
||||
/// Point-in-time snapshot of CI system state.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct CiSnapshot {
|
||||
pub active_pipelines: Vec<PipelineSnapshot>,
|
||||
pub recent_pipelines: Vec<PipelineSnapshot>,
|
||||
pub provisioner_status: ProvisionerStatus,
|
||||
pub active_instances: usize,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub enum ProvisionerStatus {
|
||||
Online,
|
||||
Offline,
|
||||
Unknown,
|
||||
}
|
||||
|
||||
/// Snapshot of a single pipeline execution.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct PipelineSnapshot {
|
||||
pub pipeline_id: PipelineId,
|
||||
pub pipeline_name: String,
|
||||
pub repo_owner: String,
|
||||
pub repo_name: String,
|
||||
pub commit_sha: String,
|
||||
pub branch: String,
|
||||
pub status: PipelineStatus,
|
||||
pub jobs: Vec<JobSnapshot>,
|
||||
}
|
||||
|
||||
/// Snapshot of a single job within a pipeline.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct JobSnapshot {
|
||||
pub job_id: JobId,
|
||||
pub job_name: String,
|
||||
pub status: JobStatus,
|
||||
pub output_line_count: usize,
|
||||
}
|
||||
|
||||
// ─── HTTP API Responses ─────────────────────────────────────────────────────
|
||||
|
||||
/// Response for GET /api/ci/pipelines
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct PipelineListResponse {
|
||||
pub pipelines: Vec<PipelineSnapshot>,
|
||||
}
|
||||
|
||||
/// Response for GET /api/ci/pipelines/{id}
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct PipelineDetailResponse {
|
||||
pub pipeline: PipelineSnapshot,
|
||||
}
|
||||
|
||||
/// Response for GET /api/ci/status
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct SystemStatusResponse {
|
||||
pub provisioner_status: ProvisionerStatus,
|
||||
pub active_pipelines: usize,
|
||||
pub active_instances: usize,
|
||||
}
|
||||
|
||||
/// Response for GET /api/ci/pipelines/{id}/jobs/{job_id}/log
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct JobLogResponse {
|
||||
pub job_id: JobId,
|
||||
pub lines: Vec<String>,
|
||||
}
|
||||
|
||||
// ─── Route Matching ─────────────────────────────────────────────────────────
|
||||
|
||||
/// Parsed API route for the CI dashboard.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum CiRoute {
|
||||
ListPipelines,
|
||||
PipelineDetail { id: u64 },
|
||||
JobLog { pipeline_id: u64, job_name: String },
|
||||
Artifact { job_id: String, path: String },
|
||||
SystemStatus,
|
||||
NotFound,
|
||||
}
|
||||
|
||||
/// Parse a request path into a CiRoute.
|
||||
pub fn parse_route(path: &str) -> CiRoute {
|
||||
let parts: Vec<&str> = path.trim_start_matches('/').split('/').collect();
|
||||
match parts.as_slice() {
|
||||
["api", "ci", "pipelines"] => CiRoute::ListPipelines,
|
||||
["api", "ci", "pipelines", id] => {
|
||||
if let Ok(id) = id.parse() {
|
||||
CiRoute::PipelineDetail { id }
|
||||
} else {
|
||||
CiRoute::NotFound
|
||||
}
|
||||
}
|
||||
["api", "ci", "pipelines", id, "jobs", job_name, "log"] => {
|
||||
if let Ok(pipeline_id) = id.parse() {
|
||||
CiRoute::JobLog {
|
||||
pipeline_id,
|
||||
job_name: job_name.to_string(),
|
||||
}
|
||||
} else {
|
||||
CiRoute::NotFound
|
||||
}
|
||||
}
|
||||
["api", "ci", "artifacts", job_id, rest @ ..] if !rest.is_empty() => CiRoute::Artifact {
|
||||
job_id: job_id.to_string(),
|
||||
path: rest.join("/"),
|
||||
},
|
||||
["api", "ci", "status"] => CiRoute::SystemStatus,
|
||||
_ => CiRoute::NotFound,
|
||||
}
|
||||
}
|
||||
|
||||
/// Render a CiSnapshot into a JSON response for the given route.
|
||||
pub fn handle_route(route: &CiRoute, snapshot: &CiSnapshot) -> Option<String> {
|
||||
match route {
|
||||
CiRoute::ListPipelines => {
|
||||
let resp = PipelineListResponse {
|
||||
pipelines: snapshot
|
||||
.active_pipelines
|
||||
.iter()
|
||||
.chain(snapshot.recent_pipelines.iter())
|
||||
.cloned()
|
||||
.collect(),
|
||||
};
|
||||
serde_json::to_string(&resp).ok()
|
||||
}
|
||||
CiRoute::PipelineDetail { id } => {
|
||||
let pipeline = snapshot
|
||||
.active_pipelines
|
||||
.iter()
|
||||
.chain(snapshot.recent_pipelines.iter())
|
||||
.find(|p| p.pipeline_id.0 == *id)?;
|
||||
let resp = PipelineDetailResponse {
|
||||
pipeline: pipeline.clone(),
|
||||
};
|
||||
serde_json::to_string(&resp).ok()
|
||||
}
|
||||
CiRoute::SystemStatus => {
|
||||
let resp = SystemStatusResponse {
|
||||
provisioner_status: snapshot.provisioner_status.clone(),
|
||||
active_pipelines: snapshot.active_pipelines.len(),
|
||||
active_instances: snapshot.active_instances,
|
||||
};
|
||||
serde_json::to_string(&resp).ok()
|
||||
}
|
||||
CiRoute::JobLog { .. } | CiRoute::Artifact { .. } => {
|
||||
// These require access to stored data beyond the snapshot.
|
||||
None
|
||||
}
|
||||
CiRoute::NotFound => None,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn route_parsing() {
|
||||
assert_eq!(parse_route("/api/ci/pipelines"), CiRoute::ListPipelines);
|
||||
assert_eq!(
|
||||
parse_route("/api/ci/pipelines/42"),
|
||||
CiRoute::PipelineDetail { id: 42 }
|
||||
);
|
||||
assert_eq!(
|
||||
parse_route("/api/ci/pipelines/1/jobs/test/log"),
|
||||
CiRoute::JobLog {
|
||||
pipeline_id: 1,
|
||||
job_name: "test".into()
|
||||
}
|
||||
);
|
||||
assert_eq!(
|
||||
parse_route("/api/ci/artifacts/job-1/target/release/bin"),
|
||||
CiRoute::Artifact {
|
||||
job_id: "job-1".into(),
|
||||
path: "target/release/bin".into()
|
||||
}
|
||||
);
|
||||
assert_eq!(parse_route("/api/ci/status"), CiRoute::SystemStatus);
|
||||
assert_eq!(parse_route("/api/ci/unknown"), CiRoute::NotFound);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn handle_system_status() {
|
||||
let snapshot = CiSnapshot {
|
||||
active_pipelines: vec![],
|
||||
recent_pipelines: vec![],
|
||||
provisioner_status: ProvisionerStatus::Online,
|
||||
active_instances: 2,
|
||||
};
|
||||
|
||||
let route = CiRoute::SystemStatus;
|
||||
let json = handle_route(&route, &snapshot).unwrap();
|
||||
let resp: SystemStatusResponse = serde_json::from_str(&json).unwrap();
|
||||
assert_eq!(resp.provisioner_status, ProvisionerStatus::Online);
|
||||
assert_eq!(resp.active_instances, 2);
|
||||
}
|
||||
}
|
||||
|
|
@ -20,6 +20,9 @@ mod distribution_html;
|
|||
#[cfg(feature = "distribution")]
|
||||
pub mod distribution_collector;
|
||||
|
||||
#[cfg(feature = "ci")]
|
||||
pub mod ci_collector;
|
||||
|
||||
use std::io;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
|
@ -91,6 +94,8 @@ pub struct DashboardHandle {
|
|||
recording: bool,
|
||||
#[cfg(feature = "distribution")]
|
||||
distribution: Arc<Mutex<Option<Arc<dyn distribution_collector::DistributionStatsProvider>>>>,
|
||||
#[cfg(feature = "ci")]
|
||||
ci: Arc<Mutex<Option<Arc<dyn ci_collector::CiStatsProvider>>>>,
|
||||
}
|
||||
|
||||
impl DashboardHandle {
|
||||
|
|
@ -123,6 +128,12 @@ impl DashboardHandle {
|
|||
*self.distribution.lock().unwrap() = Some(provider);
|
||||
}
|
||||
|
||||
/// Attach a CI stats provider, enabling the `/api/ci/*` endpoints.
|
||||
#[cfg(feature = "ci")]
|
||||
pub fn set_ci(&self, provider: Arc<dyn ci_collector::CiStatsProvider>) {
|
||||
*self.ci.lock().unwrap() = Some(provider);
|
||||
}
|
||||
|
||||
/// Access the time-series history store (for TUI sparklines, etc.).
|
||||
pub fn history(&self) -> &Arc<DashboardHistory> {
|
||||
&self.history
|
||||
|
|
@ -178,6 +189,10 @@ pub fn start_dashboard(config: DashboardConfig) -> DashboardHandle {
|
|||
let distribution: Arc<Mutex<Option<Arc<dyn distribution_collector::DistributionStatsProvider>>>> =
|
||||
Arc::new(Mutex::new(None));
|
||||
|
||||
#[cfg(feature = "ci")]
|
||||
let ci: Arc<Mutex<Option<Arc<dyn ci_collector::CiStatsProvider>>>> =
|
||||
Arc::new(Mutex::new(None));
|
||||
|
||||
server::spawn_http_server(
|
||||
Arc::clone(&store),
|
||||
Arc::clone(&runtime),
|
||||
|
|
@ -187,6 +202,8 @@ pub fn start_dashboard(config: DashboardConfig) -> DashboardHandle {
|
|||
config.port,
|
||||
#[cfg(feature = "distribution")]
|
||||
Arc::clone(&distribution),
|
||||
#[cfg(feature = "ci")]
|
||||
Arc::clone(&ci),
|
||||
);
|
||||
|
||||
// Start stats recorder thread when recording is enabled
|
||||
|
|
@ -229,6 +246,8 @@ pub fn start_dashboard(config: DashboardConfig) -> DashboardHandle {
|
|||
recording: config.record,
|
||||
#[cfg(feature = "distribution")]
|
||||
distribution,
|
||||
#[cfg(feature = "ci")]
|
||||
ci,
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -24,6 +24,9 @@ use crate::distribution_collector::DistributionStatsProvider;
|
|||
#[cfg(feature = "distribution")]
|
||||
use crate::distribution_html::DISTRIBUTION_HTML;
|
||||
|
||||
#[cfg(feature = "ci")]
|
||||
use crate::ci_collector::CiStatsProvider;
|
||||
|
||||
/// Format a server-sent event.
|
||||
fn format_sse(event: &str, data: &str) -> Vec<u8> {
|
||||
format!("event: {event}\ndata: {data}\n\n").into_bytes()
|
||||
|
|
@ -128,6 +131,8 @@ pub(crate) fn spawn_http_server(
|
|||
port: u16,
|
||||
#[cfg(feature = "distribution")]
|
||||
distribution: Arc<Mutex<Option<Arc<dyn DistributionStatsProvider>>>>,
|
||||
#[cfg(feature = "ci")]
|
||||
ci: Arc<Mutex<Option<Arc<dyn CiStatsProvider>>>>,
|
||||
) {
|
||||
let addr = format!("0.0.0.0:{port}");
|
||||
let server = tiny_http::Server::http(&addr).expect("failed to bind HTTP server");
|
||||
|
|
@ -144,6 +149,8 @@ pub(crate) fn spawn_http_server(
|
|||
let cmd_router = Arc::clone(&cmd_router);
|
||||
#[cfg(feature = "distribution")]
|
||||
let distribution = Arc::clone(&distribution);
|
||||
#[cfg(feature = "ci")]
|
||||
let ci = Arc::clone(&ci);
|
||||
thread::spawn(move || {
|
||||
loop {
|
||||
let request = match server.recv() {
|
||||
|
|
@ -169,6 +176,8 @@ pub(crate) fn spawn_http_server(
|
|||
Arc::clone(&history),
|
||||
#[cfg(feature = "distribution")]
|
||||
Arc::clone(&distribution),
|
||||
#[cfg(feature = "ci")]
|
||||
Arc::clone(&ci),
|
||||
);
|
||||
}
|
||||
"/api/stats" => {
|
||||
|
|
@ -207,6 +216,10 @@ pub(crate) fn spawn_http_server(
|
|||
"/api/logs" => {
|
||||
handle_logs_api(request, &url, Arc::clone(&store));
|
||||
}
|
||||
#[cfg(feature = "ci")]
|
||||
_ if path.starts_with("/api/ci/") => {
|
||||
handle_ci_api(request, path, Arc::clone(&ci));
|
||||
}
|
||||
_ if path.starts_with("/actor/") => {
|
||||
let hex = &path[7..]; // strip "/actor/"
|
||||
respond_actor_detail(request, hex);
|
||||
|
|
@ -239,6 +252,8 @@ fn handle_live_sse(
|
|||
history: Arc<DashboardHistory>,
|
||||
#[cfg(feature = "distribution")]
|
||||
distribution: Arc<Mutex<Option<Arc<dyn DistributionStatsProvider>>>>,
|
||||
#[cfg(feature = "ci")]
|
||||
ci: Arc<Mutex<Option<Arc<dyn CiStatsProvider>>>>,
|
||||
) {
|
||||
let (tx, rx) = mpsc::channel::<Vec<u8>>();
|
||||
let response = make_sse_response(rx);
|
||||
|
|
@ -309,6 +324,20 @@ fn handle_live_sse(
|
|||
}
|
||||
}
|
||||
|
||||
// Send CI snapshot if provider is attached
|
||||
#[cfg(feature = "ci")]
|
||||
{
|
||||
let maybe_ci = ci.lock().unwrap().clone();
|
||||
if let Some(provider) = maybe_ci {
|
||||
let snapshot = provider.snapshot();
|
||||
if let Ok(json) = serde_json::to_string(&snapshot) {
|
||||
if tx.send(format_sse("ci", &json)).is_err() {
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Send new activity events
|
||||
let (batch, new_cursor) = store.read_from(cursor);
|
||||
if !batch.is_empty() {
|
||||
|
|
@ -423,6 +452,35 @@ fn handle_distribution_api(
|
|||
let _ = request.respond(response);
|
||||
}
|
||||
|
||||
#[cfg(feature = "ci")]
|
||||
fn handle_ci_api(
|
||||
request: tiny_http::Request,
|
||||
path: &str,
|
||||
ci: Arc<Mutex<Option<Arc<dyn CiStatsProvider>>>>,
|
||||
) {
|
||||
use crate::ci_collector;
|
||||
|
||||
let route = ci_collector::parse_route(path);
|
||||
let json = match ci.lock().unwrap().as_ref() {
|
||||
Some(provider) => {
|
||||
let snapshot = provider.snapshot();
|
||||
ci_collector::handle_route(&route, &snapshot)
|
||||
.unwrap_or_else(|| r#"{"error":"not found"}"#.to_string())
|
||||
}
|
||||
None => serde_json::json!({
|
||||
"error": "CI provider not attached"
|
||||
})
|
||||
.to_string(),
|
||||
};
|
||||
|
||||
let response = tiny_http::Response::from_string(json).with_header(
|
||||
"Content-Type: application/json"
|
||||
.parse::<tiny_http::Header>()
|
||||
.unwrap(),
|
||||
);
|
||||
let _ = request.respond(response);
|
||||
}
|
||||
|
||||
fn handle_topology_api(
|
||||
request: tiny_http::Request,
|
||||
runtime: Arc<Mutex<Option<Arc<Runtime>>>>,
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ edition = "2024"
|
|||
default = []
|
||||
gossip = ["dep:log"]
|
||||
dashboard = ["gossip", "dep:tiny_http", "dep:toml"]
|
||||
ci = ["dep:swactor-ci"]
|
||||
|
||||
[dependencies]
|
||||
distribution = { path = "../distribution" }
|
||||
|
|
@ -17,9 +18,11 @@ getrandom = "0.2"
|
|||
log = { version = "0.4", optional = true }
|
||||
tiny_http = { version = "0.12", optional = true }
|
||||
toml = { version = "0.8", optional = true }
|
||||
swactor-ci = { path = "../ci", optional = true }
|
||||
|
||||
[dev-dependencies]
|
||||
simulation = { path = ".", features = ["gossip"] }
|
||||
simulation = { path = ".", features = ["gossip", "ci"] }
|
||||
swactor-ci = { path = "../ci" }
|
||||
|
||||
[[example]]
|
||||
name = "gossip_sim"
|
||||
|
|
|
|||
552
crates/simulation/src/ci/local_sim.rs
Normal file
552
crates/simulation/src/ci/local_sim.rs
Normal file
|
|
@ -0,0 +1,552 @@
|
|||
//! Local CI simulation: deterministic, round-based execution of the local
|
||||
//! coordinator's queue + scheduling logic.
|
||||
//!
|
||||
//! No actors, no IO. Models the one-at-a-time scheduling with supersede.
|
||||
//! Follows the same pattern as `sim.rs`.
|
||||
|
||||
use std::collections::{HashMap, VecDeque};
|
||||
|
||||
use swactor_ci::pipeline::PipelineExecution;
|
||||
use swactor_ci::yaml::{self, CiYaml};
|
||||
use swactor_ci::{JobId, JobStatus, PipelineId, PipelineStatus, StatusUpdate, WebhookEvent};
|
||||
|
||||
// ─── Simulation Config ──────────────────────────────────────────────────────
|
||||
|
||||
/// Configuration for a local CI simulation run.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct LocalSimConfig {
|
||||
pub name: String,
|
||||
pub num_rounds: usize,
|
||||
pub ci_yaml: String,
|
||||
pub webhook_schedule: Vec<(usize, WebhookEvent)>,
|
||||
/// Rounds a job takes to execute.
|
||||
pub job_duration: usize,
|
||||
/// Force specific jobs to fail: (round, job_name_substring).
|
||||
pub job_failure_schedule: Vec<(usize, String)>,
|
||||
}
|
||||
|
||||
impl Default for LocalSimConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
name: "local-sim".into(),
|
||||
num_rounds: 50,
|
||||
ci_yaml: String::new(),
|
||||
webhook_schedule: Vec::new(),
|
||||
job_duration: 3,
|
||||
job_failure_schedule: Vec::new(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Simulation Trace ───────────────────────────────────────────────────────
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum LocalSimEvent {
|
||||
WebhookReceived { commit_sha: String },
|
||||
PipelineCreated { pipeline_id: PipelineId, name: String },
|
||||
PipelineSuperseded { pipeline_id: PipelineId },
|
||||
PipelineCompleted { pipeline_id: PipelineId, status: PipelineStatus },
|
||||
JobStarted { job_id: JobId },
|
||||
JobCompleted { job_id: JobId, passed: bool },
|
||||
JobSkipped { job_id: JobId },
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct LocalSimSnapshot {
|
||||
pub queued_pipelines: usize,
|
||||
pub active_pipeline: Option<PipelineId>,
|
||||
pub running_job: Option<JobId>,
|
||||
pub completed_pipelines: usize,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct LocalSimTrace {
|
||||
pub name: String,
|
||||
pub events: Vec<(usize, LocalSimEvent)>,
|
||||
pub snapshots: Vec<LocalSimSnapshot>,
|
||||
pub status_updates: Vec<StatusUpdate>,
|
||||
pub num_rounds: usize,
|
||||
pub final_pipelines: Vec<PipelineExecution>,
|
||||
}
|
||||
|
||||
// ─── Simulation State ───────────────────────────────────────────────────────
|
||||
|
||||
struct RunningJob {
|
||||
job_id: JobId,
|
||||
started_round: usize,
|
||||
}
|
||||
|
||||
/// Run a local CI simulation and return the trace.
|
||||
pub fn run_simulation(config: LocalSimConfig) -> LocalSimTrace {
|
||||
let ci_yaml: CiYaml =
|
||||
yaml::parse_ci_yaml(&config.ci_yaml).expect("LocalSimConfig.ci_yaml must be valid YAML");
|
||||
|
||||
let mut events: Vec<(usize, LocalSimEvent)> = Vec::new();
|
||||
let mut snapshots: Vec<LocalSimSnapshot> = Vec::new();
|
||||
let mut all_status_updates: Vec<StatusUpdate> = Vec::new();
|
||||
|
||||
// Coordinator state.
|
||||
let mut pipelines: HashMap<PipelineId, PipelineExecution> = HashMap::new();
|
||||
let mut next_pipeline_id: u64 = 1;
|
||||
let mut queue: VecDeque<PipelineId> = VecDeque::new();
|
||||
let mut active_pipeline: Option<PipelineId> = None;
|
||||
let mut running_job: Option<RunningJob> = None;
|
||||
let mut completed_pipelines: Vec<PipelineExecution> = Vec::new();
|
||||
|
||||
for round in 1..=config.num_rounds {
|
||||
// 1. Inject webhook events for this round.
|
||||
for (sched_round, event) in &config.webhook_schedule {
|
||||
if *sched_round == round {
|
||||
events.push((
|
||||
round,
|
||||
LocalSimEvent::WebhookReceived {
|
||||
commit_sha: event.commit_sha.clone(),
|
||||
},
|
||||
));
|
||||
|
||||
let matched = yaml::matching_pipelines(&ci_yaml, event);
|
||||
for pipeline_name in matched {
|
||||
let pipeline_id = PipelineId(next_pipeline_id);
|
||||
next_pipeline_id += 1;
|
||||
|
||||
let pipeline_def = &ci_yaml.pipelines[&pipeline_name];
|
||||
let job_defs: Vec<_> = pipeline_def
|
||||
.jobs
|
||||
.iter()
|
||||
.map(|(name, def)| yaml::to_job_definition(name, def))
|
||||
.collect();
|
||||
|
||||
let pipeline = PipelineExecution::new(
|
||||
pipeline_id,
|
||||
pipeline_name.clone(),
|
||||
event.repo_owner.clone(),
|
||||
event.repo_name.clone(),
|
||||
event.commit_sha.clone(),
|
||||
event.branch.clone(),
|
||||
job_defs,
|
||||
);
|
||||
|
||||
events.push((
|
||||
round,
|
||||
LocalSimEvent::PipelineCreated {
|
||||
pipeline_id,
|
||||
name: pipeline_name.clone(),
|
||||
},
|
||||
));
|
||||
|
||||
all_status_updates.push(StatusUpdate {
|
||||
repo_owner: event.repo_owner.clone(),
|
||||
repo_name: event.repo_name.clone(),
|
||||
commit_sha: event.commit_sha.clone(),
|
||||
state: "pending".into(),
|
||||
context: format!("ci/{pipeline_name}"),
|
||||
description: format!("Pipeline '{pipeline_name}' is pending"),
|
||||
});
|
||||
|
||||
pipelines.insert(pipeline_id, pipeline);
|
||||
|
||||
// Enqueue with supersede logic.
|
||||
let supersede_idx = queue.iter().position(|&qid| {
|
||||
pipelines
|
||||
.get(&qid)
|
||||
.map(|p| p.branch == event.branch)
|
||||
.unwrap_or(false)
|
||||
});
|
||||
|
||||
if let Some(idx) = supersede_idx {
|
||||
let old_id = queue[idx];
|
||||
if let Some(old_pipeline) = pipelines.get_mut(&old_id) {
|
||||
old_pipeline.status = PipelineStatus::Error {
|
||||
reason: "superseded".into(),
|
||||
};
|
||||
let job_names: Vec<String> =
|
||||
old_pipeline.jobs.keys().cloned().collect();
|
||||
for name in job_names {
|
||||
if old_pipeline.jobs[&name].status == JobStatus::Pending {
|
||||
old_pipeline.set_job_status(&name, JobStatus::Skipped);
|
||||
}
|
||||
}
|
||||
}
|
||||
events.push((
|
||||
round,
|
||||
LocalSimEvent::PipelineSuperseded {
|
||||
pipeline_id: old_id,
|
||||
},
|
||||
));
|
||||
if let Some(old_pipeline) = pipelines.remove(&old_id) {
|
||||
// Emit terminal status for superseded pipeline.
|
||||
all_status_updates.push(StatusUpdate {
|
||||
repo_owner: old_pipeline.repo_owner.clone(),
|
||||
repo_name: old_pipeline.repo_name.clone(),
|
||||
commit_sha: old_pipeline.commit_sha.clone(),
|
||||
state: "error".into(),
|
||||
context: format!("ci/{}", old_pipeline.pipeline_name),
|
||||
description: "superseded".into(),
|
||||
});
|
||||
completed_pipelines.push(old_pipeline);
|
||||
}
|
||||
queue[idx] = pipeline_id;
|
||||
} else {
|
||||
queue.push_back(pipeline_id);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Complete running job if it has reached duration.
|
||||
if let Some(ref rj) = running_job {
|
||||
if round - rj.started_round >= config.job_duration {
|
||||
let job_id = rj.job_id.clone();
|
||||
|
||||
let should_fail = config
|
||||
.job_failure_schedule
|
||||
.iter()
|
||||
.any(|(r, name_sub)| *r <= round && job_id.job_name.contains(name_sub.as_str()));
|
||||
|
||||
let passed = !should_fail;
|
||||
|
||||
if passed {
|
||||
if let Some(pipeline) = pipelines.get_mut(&job_id.pipeline_id) {
|
||||
pipeline.set_job_status(&job_id.job_name, JobStatus::Passed);
|
||||
}
|
||||
} else {
|
||||
if let Some(pipeline) = pipelines.get_mut(&job_id.pipeline_id) {
|
||||
pipeline.set_job_status(
|
||||
&job_id.job_name,
|
||||
JobStatus::Failed {
|
||||
reason: "command failed".into(),
|
||||
},
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
events.push((
|
||||
round,
|
||||
LocalSimEvent::JobCompleted {
|
||||
job_id: job_id.clone(),
|
||||
passed,
|
||||
},
|
||||
));
|
||||
|
||||
// Emit skipped events for any jobs that were skipped due to failure.
|
||||
if !passed {
|
||||
if let Some(pipeline) = pipelines.get(&job_id.pipeline_id) {
|
||||
for (_name, job) in &pipeline.jobs {
|
||||
if job.status == JobStatus::Skipped {
|
||||
events.push((
|
||||
round,
|
||||
LocalSimEvent::JobSkipped {
|
||||
job_id: job.job_id.clone(),
|
||||
},
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
running_job = None;
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Schedule next (one-at-a-time).
|
||||
schedule_next(
|
||||
&mut pipelines,
|
||||
&mut queue,
|
||||
&mut active_pipeline,
|
||||
&mut running_job,
|
||||
&mut completed_pipelines,
|
||||
&mut events,
|
||||
&mut all_status_updates,
|
||||
round,
|
||||
);
|
||||
|
||||
// 4. Snapshot.
|
||||
snapshots.push(LocalSimSnapshot {
|
||||
queued_pipelines: queue.len(),
|
||||
active_pipeline,
|
||||
running_job: running_job.as_ref().map(|rj| rj.job_id.clone()),
|
||||
completed_pipelines: completed_pipelines.len(),
|
||||
});
|
||||
}
|
||||
|
||||
// Collect remaining active pipelines into final output.
|
||||
let mut final_pipelines: Vec<PipelineExecution> = pipelines.into_values().collect();
|
||||
final_pipelines.extend(completed_pipelines);
|
||||
|
||||
LocalSimTrace {
|
||||
name: config.name,
|
||||
events,
|
||||
snapshots,
|
||||
status_updates: all_status_updates,
|
||||
num_rounds: config.num_rounds,
|
||||
final_pipelines,
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn schedule_next(
|
||||
pipelines: &mut HashMap<PipelineId, PipelineExecution>,
|
||||
queue: &mut VecDeque<PipelineId>,
|
||||
active_pipeline: &mut Option<PipelineId>,
|
||||
running_job: &mut Option<RunningJob>,
|
||||
completed_pipelines: &mut Vec<PipelineExecution>,
|
||||
events: &mut Vec<(usize, LocalSimEvent)>,
|
||||
status_updates: &mut Vec<StatusUpdate>,
|
||||
round: usize,
|
||||
) {
|
||||
// If a job is running, nothing to do.
|
||||
if running_job.is_some() {
|
||||
return;
|
||||
}
|
||||
|
||||
// If we have an active pipeline, try eligible jobs.
|
||||
if let Some(active_id) = *active_pipeline {
|
||||
if let Some(pipeline) = pipelines.get(&active_id) {
|
||||
let eligible = pipeline.eligible_jobs();
|
||||
if !eligible.is_empty() {
|
||||
let job_name = eligible[0].clone();
|
||||
let job_id = JobId {
|
||||
pipeline_id: active_id,
|
||||
job_name: job_name.clone(),
|
||||
};
|
||||
|
||||
// Mark as running.
|
||||
if let Some(pipeline) = pipelines.get_mut(&active_id) {
|
||||
if let Some(job) = pipeline.jobs.get_mut(&job_name) {
|
||||
job.status = JobStatus::Running;
|
||||
}
|
||||
}
|
||||
|
||||
events.push((round, LocalSimEvent::JobStarted { job_id: job_id.clone() }));
|
||||
|
||||
*running_job = Some(RunningJob {
|
||||
job_id,
|
||||
started_round: round,
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
// No eligible jobs — check terminal.
|
||||
if pipeline.status.is_terminal() {
|
||||
let pipeline = pipelines.remove(&active_id).unwrap();
|
||||
let status = pipeline.status.clone();
|
||||
|
||||
status_updates.push(StatusUpdate {
|
||||
repo_owner: pipeline.repo_owner.clone(),
|
||||
repo_name: pipeline.repo_name.clone(),
|
||||
commit_sha: pipeline.commit_sha.clone(),
|
||||
state: pipeline.status.forgejo_state().into(),
|
||||
context: format!("ci/{}", pipeline.pipeline_name),
|
||||
description: format!(
|
||||
"Pipeline '{}' {}",
|
||||
pipeline.pipeline_name,
|
||||
pipeline.status.forgejo_state()
|
||||
),
|
||||
});
|
||||
|
||||
events.push((
|
||||
round,
|
||||
LocalSimEvent::PipelineCompleted {
|
||||
pipeline_id: active_id,
|
||||
status,
|
||||
},
|
||||
));
|
||||
|
||||
completed_pipelines.push(pipeline);
|
||||
*active_pipeline = None;
|
||||
|
||||
// Recurse.
|
||||
schedule_next(
|
||||
pipelines,
|
||||
queue,
|
||||
active_pipeline,
|
||||
running_job,
|
||||
completed_pipelines,
|
||||
events,
|
||||
status_updates,
|
||||
round,
|
||||
);
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Pipeline exists but no eligible jobs and not terminal — waiting.
|
||||
return;
|
||||
}
|
||||
|
||||
// No active pipeline — pop from queue.
|
||||
if let Some(next_id) = queue.pop_front() {
|
||||
*active_pipeline = Some(next_id);
|
||||
schedule_next(
|
||||
pipelines,
|
||||
queue,
|
||||
active_pipeline,
|
||||
running_job,
|
||||
completed_pipelines,
|
||||
events,
|
||||
status_updates,
|
||||
round,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Properties ─────────────────────────────────────────────────────────────
|
||||
|
||||
/// At most one job running in any snapshot.
|
||||
pub fn check_one_at_a_time(trace: &LocalSimTrace) -> bool {
|
||||
trace
|
||||
.snapshots
|
||||
.iter()
|
||||
.all(|s| s.running_job.is_some() as usize <= 1)
|
||||
}
|
||||
|
||||
/// Superseded pipelines never have a Running job.
|
||||
pub fn check_superseded_no_running(trace: &LocalSimTrace) -> bool {
|
||||
let superseded: Vec<PipelineId> = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter_map(|(_, e)| match e {
|
||||
LocalSimEvent::PipelineSuperseded { pipeline_id } => Some(*pipeline_id),
|
||||
_ => None,
|
||||
})
|
||||
.collect();
|
||||
|
||||
let started_jobs: Vec<&JobId> = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter_map(|(_, e)| match e {
|
||||
LocalSimEvent::JobStarted { job_id } => Some(job_id),
|
||||
_ => None,
|
||||
})
|
||||
.collect();
|
||||
|
||||
for pid in &superseded {
|
||||
if started_jobs
|
||||
.iter()
|
||||
.any(|jid| jid.pipeline_id == *pid)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// All non-superseded pipelines reach terminal status.
|
||||
pub fn check_termination(trace: &LocalSimTrace) -> bool {
|
||||
let superseded: Vec<PipelineId> = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter_map(|(_, e)| match e {
|
||||
LocalSimEvent::PipelineSuperseded { pipeline_id } => Some(*pipeline_id),
|
||||
_ => None,
|
||||
})
|
||||
.collect();
|
||||
|
||||
for pipeline in &trace.final_pipelines {
|
||||
if superseded.contains(&pipeline.pipeline_id) {
|
||||
continue;
|
||||
}
|
||||
if !pipeline.status.is_terminal() {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Within a pipeline, jobs respect dependency order.
|
||||
pub fn check_dag_ordering(trace: &LocalSimTrace) -> bool {
|
||||
let mut started: HashMap<(u64, &str), usize> = HashMap::new();
|
||||
let mut completed: HashMap<(u64, &str), usize> = HashMap::new();
|
||||
|
||||
for (round, event) in &trace.events {
|
||||
match event {
|
||||
LocalSimEvent::JobStarted { job_id } => {
|
||||
started.insert(
|
||||
(job_id.pipeline_id.0, job_id.job_name.as_str()),
|
||||
*round,
|
||||
);
|
||||
}
|
||||
LocalSimEvent::JobCompleted { job_id, .. } => {
|
||||
completed.insert(
|
||||
(job_id.pipeline_id.0, job_id.job_name.as_str()),
|
||||
*round,
|
||||
);
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
for pipeline in &trace.final_pipelines {
|
||||
for (name, job) in &pipeline.jobs {
|
||||
if let Some(&start_round) = started.get(&(pipeline.pipeline_id.0, name.as_str())) {
|
||||
for dep in &job.definition.needs {
|
||||
if let Some(&dep_complete_round) =
|
||||
completed.get(&(pipeline.pipeline_id.0, dep.as_str()))
|
||||
{
|
||||
if dep_complete_round > start_round {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Different branches execute in queue order (FIFO).
|
||||
pub fn check_fifo_order(trace: &LocalSimTrace) -> bool {
|
||||
// Collect pipeline creation order and first job start per pipeline.
|
||||
let mut creation_order: Vec<PipelineId> = Vec::new();
|
||||
let mut first_start: HashMap<PipelineId, usize> = HashMap::new();
|
||||
|
||||
for (round, event) in &trace.events {
|
||||
if let LocalSimEvent::PipelineCreated { pipeline_id, .. } = event {
|
||||
creation_order.push(*pipeline_id);
|
||||
}
|
||||
if let LocalSimEvent::JobStarted { job_id } = event {
|
||||
first_start
|
||||
.entry(job_id.pipeline_id)
|
||||
.or_insert(*round);
|
||||
}
|
||||
}
|
||||
|
||||
// For each pair of pipelines created in order, if both started, the earlier-created
|
||||
// one should have started no later.
|
||||
for i in 0..creation_order.len() {
|
||||
for j in (i + 1)..creation_order.len() {
|
||||
let pid_a = creation_order[i];
|
||||
let pid_b = creation_order[j];
|
||||
if let (Some(&start_a), Some(&start_b)) =
|
||||
(first_start.get(&pid_a), first_start.get(&pid_b))
|
||||
{
|
||||
if start_a > start_b {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Every webhook produces a terminal status (success/failure/error).
|
||||
pub fn check_all_webhooks_terminate(trace: &LocalSimTrace) -> bool {
|
||||
let webhook_commits: Vec<&str> = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter_map(|(_, e)| match e {
|
||||
LocalSimEvent::WebhookReceived { commit_sha } => Some(commit_sha.as_str()),
|
||||
_ => None,
|
||||
})
|
||||
.collect();
|
||||
|
||||
for sha in webhook_commits {
|
||||
let has_terminal = trace.status_updates.iter().any(|u| {
|
||||
u.commit_sha == sha
|
||||
&& (u.state == "success" || u.state == "failure" || u.state == "error")
|
||||
});
|
||||
if !has_terminal {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
2
crates/simulation/src/ci/mod.rs
Normal file
2
crates/simulation/src/ci/mod.rs
Normal file
|
|
@ -0,0 +1,2 @@
|
|||
pub mod local_sim;
|
||||
pub mod sim;
|
||||
610
crates/simulation/src/ci/sim.rs
Normal file
610
crates/simulation/src/ci/sim.rs
Normal file
|
|
@ -0,0 +1,610 @@
|
|||
//! CI protocol simulation: deterministic, round-based execution of the CI pipeline
|
||||
//! lifecycle without real IO (no SSH, no cloud API, no HTTP).
|
||||
//!
|
||||
//! Follows the same pattern as `crates/simulation/src/distribution/sim.rs`:
|
||||
//! configure → run rounds → collect trace → analyze properties.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use swactor_ci::pipeline::PipelineExecution;
|
||||
use swactor_ci::yaml::{self, CiYaml};
|
||||
use swactor_ci::{
|
||||
JobId, JobStatus, PipelineId, ProvisionRequest, StatusUpdate, WebhookEvent,
|
||||
};
|
||||
|
||||
// ─── Simulation Config ──────────────────────────────────────────────────────
|
||||
|
||||
/// Configuration for a CI simulation run.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct CiSimConfig {
|
||||
pub name: String,
|
||||
pub num_rounds: usize,
|
||||
|
||||
/// CI YAML to use for all simulated repos.
|
||||
pub ci_yaml: String,
|
||||
|
||||
/// Webhook events to inject at specific rounds.
|
||||
pub webhook_schedule: Vec<(usize, WebhookEvent)>,
|
||||
|
||||
/// Rounds of latency for provisioning to complete.
|
||||
pub provision_latency: usize,
|
||||
|
||||
/// Probability that provisioning fails (0.0-1.0).
|
||||
pub provision_failure_rate: f64,
|
||||
|
||||
/// Rounds of latency for a job to complete.
|
||||
pub job_duration: usize,
|
||||
|
||||
/// Force specific jobs to fail: (round, job_name_substring).
|
||||
pub job_failure_schedule: Vec<(usize, String)>,
|
||||
|
||||
/// Rounds during which the provisioner is offline: (start_round, end_round).
|
||||
pub provisioner_offline_schedule: Vec<(usize, usize)>,
|
||||
|
||||
/// Round at which a specific job's instance is interrupted.
|
||||
pub instance_interrupt_schedule: Vec<(usize, String)>,
|
||||
}
|
||||
|
||||
impl Default for CiSimConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
name: "ci-sim".into(),
|
||||
num_rounds: 50,
|
||||
ci_yaml: String::new(),
|
||||
webhook_schedule: Vec::new(),
|
||||
provision_latency: 2,
|
||||
provision_failure_rate: 0.0,
|
||||
job_duration: 3,
|
||||
job_failure_schedule: Vec::new(),
|
||||
provisioner_offline_schedule: Vec::new(),
|
||||
instance_interrupt_schedule: Vec::new(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Simulation Trace ───────────────────────────────────────────────────────
|
||||
|
||||
/// Event recorded during simulation.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum CiSimEvent {
|
||||
WebhookReceived { commit_sha: String },
|
||||
PipelineCreated { pipeline_id: PipelineId, name: String },
|
||||
ProvisionRequested { job_id: JobId },
|
||||
ProvisionCompleted { job_id: JobId, success: bool },
|
||||
JobStarted { job_id: JobId },
|
||||
JobCompleted { job_id: JobId, passed: bool },
|
||||
JobSkipped { job_id: JobId },
|
||||
InstanceTerminated { instance_id: String },
|
||||
ProvisionerWentOffline,
|
||||
ProvisionerCameOnline,
|
||||
StatusUpdateEmitted(StatusUpdate),
|
||||
}
|
||||
|
||||
/// Per-round snapshot of simulation state.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct CiSimSnapshot {
|
||||
pub active_pipelines: usize,
|
||||
pub completed_pipelines: usize,
|
||||
pub active_provisions: usize,
|
||||
pub active_jobs: usize,
|
||||
pub provisioner_online: bool,
|
||||
pub active_instances: usize,
|
||||
}
|
||||
|
||||
/// Complete trace output from a CI simulation.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct CiSimTrace {
|
||||
pub name: String,
|
||||
pub events: Vec<(usize, CiSimEvent)>,
|
||||
pub snapshots: Vec<CiSimSnapshot>,
|
||||
pub status_updates: Vec<StatusUpdate>,
|
||||
pub num_rounds: usize,
|
||||
/// Final state of all pipelines.
|
||||
pub final_pipelines: Vec<PipelineExecution>,
|
||||
/// Instances that were provisioned.
|
||||
pub provisioned_instances: Vec<String>,
|
||||
/// Instances that were terminated.
|
||||
pub terminated_instances: Vec<String>,
|
||||
}
|
||||
|
||||
// ─── Simulation State ───────────────────────────────────────────────────────
|
||||
|
||||
/// Tracks an in-flight provision request.
|
||||
struct PendingProvision {
|
||||
request: ProvisionRequest,
|
||||
started_round: usize,
|
||||
}
|
||||
|
||||
/// Tracks an in-flight job execution.
|
||||
struct RunningJob {
|
||||
job_id: JobId,
|
||||
started_round: usize,
|
||||
instance_id: String,
|
||||
}
|
||||
|
||||
/// Run a CI simulation and return the trace.
|
||||
pub fn run_simulation(config: CiSimConfig) -> CiSimTrace {
|
||||
let ci_yaml: CiYaml = yaml::parse_ci_yaml(&config.ci_yaml)
|
||||
.expect("CiSimConfig.ci_yaml must be valid YAML");
|
||||
|
||||
let mut events: Vec<(usize, CiSimEvent)> = Vec::new();
|
||||
let mut snapshots: Vec<CiSimSnapshot> = Vec::new();
|
||||
|
||||
// Coordinator state (simulated directly, not as actor).
|
||||
let mut pipelines: HashMap<PipelineId, PipelineExecution> = HashMap::new();
|
||||
let mut next_pipeline_id: u64 = 1;
|
||||
let mut all_status_updates: Vec<StatusUpdate> = Vec::new();
|
||||
|
||||
// Provisioner state.
|
||||
let mut provisioner_online = true;
|
||||
let mut pending_provisions: Vec<PendingProvision> = Vec::new();
|
||||
let mut queued_provisions: Vec<ProvisionRequest> = Vec::new();
|
||||
let mut provisioned_instances: Vec<String> = Vec::new();
|
||||
let mut terminated_instances: Vec<String> = Vec::new();
|
||||
let mut active_instances: Vec<String> = Vec::new();
|
||||
let mut next_instance_id: u64 = 1;
|
||||
|
||||
// Runner state.
|
||||
let mut running_jobs: Vec<RunningJob> = Vec::new();
|
||||
|
||||
// Simple deterministic "RNG" for provision failure decisions.
|
||||
let mut rng_counter: u64 = 0x853c49e6748fea9b;
|
||||
let mut det_random = || -> f64 {
|
||||
rng_counter = rng_counter.wrapping_mul(6364136223846793005).wrapping_add(1);
|
||||
(rng_counter >> 33) as f64 / (u32::MAX as f64)
|
||||
};
|
||||
|
||||
for round in 1..=config.num_rounds {
|
||||
// 1. Apply provisioner online/offline schedule.
|
||||
let should_be_offline = config
|
||||
.provisioner_offline_schedule
|
||||
.iter()
|
||||
.any(|(start, end)| round >= *start && round <= *end);
|
||||
|
||||
if should_be_offline && provisioner_online {
|
||||
provisioner_online = false;
|
||||
events.push((round, CiSimEvent::ProvisionerWentOffline));
|
||||
|
||||
// Move pending provisions to queue.
|
||||
for pending in pending_provisions.drain(..) {
|
||||
queued_provisions.push(pending.request);
|
||||
}
|
||||
} else if !should_be_offline && !provisioner_online {
|
||||
provisioner_online = true;
|
||||
events.push((round, CiSimEvent::ProvisionerCameOnline));
|
||||
|
||||
// Flush queued provisions.
|
||||
for request in queued_provisions.drain(..) {
|
||||
pending_provisions.push(PendingProvision {
|
||||
request,
|
||||
started_round: round,
|
||||
});
|
||||
}
|
||||
|
||||
// Also resubmit any WaitingForProvisioner jobs.
|
||||
let mut resubmits = Vec::new();
|
||||
for pipeline in pipelines.values_mut() {
|
||||
for job in pipeline.jobs.values_mut() {
|
||||
if job.status == JobStatus::WaitingForProvisioner {
|
||||
job.status = JobStatus::Provisioning;
|
||||
resubmits.push(ProvisionRequest {
|
||||
job_id: job.job_id.clone(),
|
||||
instance_spec: swactor_ci::InstanceSpec {
|
||||
docker_required: job.definition.docker,
|
||||
..Default::default()
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
for request in resubmits {
|
||||
events.push((round, CiSimEvent::ProvisionRequested { job_id: request.job_id.clone() }));
|
||||
pending_provisions.push(PendingProvision {
|
||||
request,
|
||||
started_round: round,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Inject webhook events for this round.
|
||||
for (sched_round, event) in &config.webhook_schedule {
|
||||
if *sched_round == round {
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::WebhookReceived {
|
||||
commit_sha: event.commit_sha.clone(),
|
||||
},
|
||||
));
|
||||
|
||||
let matched = yaml::matching_pipelines(&ci_yaml, event);
|
||||
for pipeline_name in matched {
|
||||
let pipeline_id = PipelineId(next_pipeline_id);
|
||||
next_pipeline_id += 1;
|
||||
|
||||
let pipeline_def = &ci_yaml.pipelines[&pipeline_name];
|
||||
let job_defs: Vec<_> = pipeline_def
|
||||
.jobs
|
||||
.iter()
|
||||
.map(|(name, def)| yaml::to_job_definition(name, def))
|
||||
.collect();
|
||||
|
||||
let pipeline = PipelineExecution::new(
|
||||
pipeline_id,
|
||||
pipeline_name.clone(),
|
||||
event.repo_owner.clone(),
|
||||
event.repo_name.clone(),
|
||||
event.commit_sha.clone(),
|
||||
event.branch.clone(),
|
||||
job_defs,
|
||||
);
|
||||
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::PipelineCreated {
|
||||
pipeline_id,
|
||||
name: pipeline_name.clone(),
|
||||
},
|
||||
));
|
||||
|
||||
// Emit pending status.
|
||||
all_status_updates.push(StatusUpdate {
|
||||
repo_owner: event.repo_owner.clone(),
|
||||
repo_name: event.repo_name.clone(),
|
||||
commit_sha: event.commit_sha.clone(),
|
||||
state: "pending".into(),
|
||||
context: format!("ci/{pipeline_name}"),
|
||||
description: format!("Pipeline '{pipeline_name}' is pending"),
|
||||
});
|
||||
|
||||
pipelines.insert(pipeline_id, pipeline);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Complete provisions that have reached latency.
|
||||
let mut completed_provisions = Vec::new();
|
||||
pending_provisions.retain(|pending| {
|
||||
if round - pending.started_round >= config.provision_latency {
|
||||
completed_provisions.push(pending.request.clone());
|
||||
false
|
||||
} else {
|
||||
true
|
||||
}
|
||||
});
|
||||
|
||||
for request in completed_provisions {
|
||||
let should_fail = det_random() < config.provision_failure_rate;
|
||||
|
||||
if should_fail {
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::ProvisionCompleted {
|
||||
job_id: request.job_id.clone(),
|
||||
success: false,
|
||||
},
|
||||
));
|
||||
|
||||
if let Some(pipeline) = pipelines.get_mut(&request.job_id.pipeline_id) {
|
||||
pipeline.set_job_status(
|
||||
&request.job_id.job_name,
|
||||
JobStatus::Failed {
|
||||
reason: "provision failed".into(),
|
||||
},
|
||||
);
|
||||
}
|
||||
} else {
|
||||
let instance_id = format!("instance-{next_instance_id}");
|
||||
next_instance_id += 1;
|
||||
provisioned_instances.push(instance_id.clone());
|
||||
active_instances.push(instance_id.clone());
|
||||
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::ProvisionCompleted {
|
||||
job_id: request.job_id.clone(),
|
||||
success: true,
|
||||
},
|
||||
));
|
||||
|
||||
// Mark job as running and record instance.
|
||||
if let Some(pipeline) = pipelines.get_mut(&request.job_id.pipeline_id) {
|
||||
if let Some(job) = pipeline.jobs.get_mut(&request.job_id.job_name) {
|
||||
job.status = JobStatus::Running;
|
||||
job.instance_id = Some(instance_id.clone());
|
||||
}
|
||||
}
|
||||
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::JobStarted {
|
||||
job_id: request.job_id.clone(),
|
||||
},
|
||||
));
|
||||
|
||||
running_jobs.push(RunningJob {
|
||||
job_id: request.job_id,
|
||||
started_round: round,
|
||||
instance_id,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Apply instance interruptions.
|
||||
for (interrupt_round, job_name_sub) in &config.instance_interrupt_schedule {
|
||||
if *interrupt_round == round {
|
||||
running_jobs.retain(|rj| {
|
||||
if rj.job_id.job_name.contains(job_name_sub.as_str()) {
|
||||
// Instance interrupted.
|
||||
if let Some(pipeline) = pipelines.get_mut(&rj.job_id.pipeline_id) {
|
||||
pipeline.set_job_status(&rj.job_id.job_name, JobStatus::Interrupted);
|
||||
}
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::JobCompleted {
|
||||
job_id: rj.job_id.clone(),
|
||||
passed: false,
|
||||
},
|
||||
));
|
||||
// Terminate the instance.
|
||||
active_instances.retain(|id| id != &rj.instance_id);
|
||||
terminated_instances.push(rj.instance_id.clone());
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::InstanceTerminated {
|
||||
instance_id: rj.instance_id.clone(),
|
||||
},
|
||||
));
|
||||
false
|
||||
} else {
|
||||
true
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// 5. Complete jobs that have reached duration.
|
||||
let mut newly_completed = Vec::new();
|
||||
running_jobs.retain(|rj| {
|
||||
if round - rj.started_round >= config.job_duration {
|
||||
newly_completed.push((rj.job_id.clone(), rj.instance_id.clone()));
|
||||
false
|
||||
} else {
|
||||
true
|
||||
}
|
||||
});
|
||||
|
||||
for (job_id, instance_id) in newly_completed {
|
||||
// Check if this job should fail per the schedule.
|
||||
let should_fail = config
|
||||
.job_failure_schedule
|
||||
.iter()
|
||||
.any(|(r, name_sub)| *r <= round && job_id.job_name.contains(name_sub.as_str()));
|
||||
|
||||
let passed = !should_fail;
|
||||
|
||||
if passed {
|
||||
if let Some(pipeline) = pipelines.get_mut(&job_id.pipeline_id) {
|
||||
pipeline.set_job_status(&job_id.job_name, JobStatus::Passed);
|
||||
}
|
||||
} else {
|
||||
if let Some(pipeline) = pipelines.get_mut(&job_id.pipeline_id) {
|
||||
pipeline.set_job_status(
|
||||
&job_id.job_name,
|
||||
JobStatus::Failed {
|
||||
reason: "command failed".into(),
|
||||
},
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::JobCompleted {
|
||||
job_id: job_id.clone(),
|
||||
passed,
|
||||
},
|
||||
));
|
||||
|
||||
// Terminate instance.
|
||||
active_instances.retain(|id| id != &instance_id);
|
||||
terminated_instances.push(instance_id.clone());
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::InstanceTerminated {
|
||||
instance_id: instance_id.clone(),
|
||||
},
|
||||
));
|
||||
}
|
||||
|
||||
// 6. Advance all pipelines: schedule newly-eligible jobs.
|
||||
let pipeline_ids: Vec<PipelineId> = pipelines.keys().copied().collect();
|
||||
for pid in pipeline_ids {
|
||||
let eligible = pipelines[&pid].eligible_jobs();
|
||||
for job_name in eligible {
|
||||
let job_id = JobId {
|
||||
pipeline_id: pid,
|
||||
job_name: job_name.clone(),
|
||||
};
|
||||
let docker_required = pipelines[&pid].jobs[&job_name].definition.docker;
|
||||
|
||||
let request = ProvisionRequest {
|
||||
job_id: job_id.clone(),
|
||||
instance_spec: swactor_ci::InstanceSpec {
|
||||
docker_required,
|
||||
..Default::default()
|
||||
},
|
||||
};
|
||||
|
||||
if provisioner_online {
|
||||
if let Some(pipeline) = pipelines.get_mut(&pid) {
|
||||
if let Some(job) = pipeline.jobs.get_mut(&job_name) {
|
||||
job.status = JobStatus::Provisioning;
|
||||
}
|
||||
}
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::ProvisionRequested { job_id },
|
||||
));
|
||||
pending_provisions.push(PendingProvision {
|
||||
request,
|
||||
started_round: round,
|
||||
});
|
||||
} else {
|
||||
if let Some(pipeline) = pipelines.get_mut(&pid) {
|
||||
if let Some(job) = pipeline.jobs.get_mut(&job_name) {
|
||||
job.status = JobStatus::WaitingForProvisioner;
|
||||
}
|
||||
}
|
||||
queued_provisions.push(request);
|
||||
}
|
||||
}
|
||||
|
||||
// Emit final status updates for terminal pipelines.
|
||||
if let Some(pipeline) = pipelines.get(&pid) {
|
||||
if pipeline.status.is_terminal() {
|
||||
// Check if we already emitted a terminal status for this pipeline.
|
||||
let context = format!("ci/{}", pipeline.pipeline_name);
|
||||
let already_emitted = all_status_updates.iter().any(|u| {
|
||||
u.context == context
|
||||
&& u.commit_sha == pipeline.commit_sha
|
||||
&& (u.state == "success" || u.state == "failure" || u.state == "error")
|
||||
});
|
||||
if !already_emitted {
|
||||
all_status_updates.push(StatusUpdate {
|
||||
repo_owner: pipeline.repo_owner.clone(),
|
||||
repo_name: pipeline.repo_name.clone(),
|
||||
commit_sha: pipeline.commit_sha.clone(),
|
||||
state: pipeline.status.forgejo_state().into(),
|
||||
context,
|
||||
description: format!(
|
||||
"Pipeline '{}' {}",
|
||||
pipeline.pipeline_name,
|
||||
pipeline.status.forgejo_state()
|
||||
),
|
||||
});
|
||||
|
||||
// Emit per-job skipped events.
|
||||
for (_name, job) in &pipeline.jobs {
|
||||
if job.status == JobStatus::Skipped {
|
||||
events.push((
|
||||
round,
|
||||
CiSimEvent::JobSkipped {
|
||||
job_id: job.job_id.clone(),
|
||||
},
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 7. Snapshot.
|
||||
let completed_count = pipelines.values().filter(|p| p.status.is_terminal()).count();
|
||||
snapshots.push(CiSimSnapshot {
|
||||
active_pipelines: pipelines.len() - completed_count,
|
||||
completed_pipelines: completed_count,
|
||||
active_provisions: pending_provisions.len(),
|
||||
active_jobs: running_jobs.len(),
|
||||
provisioner_online,
|
||||
active_instances: active_instances.len(),
|
||||
});
|
||||
}
|
||||
|
||||
CiSimTrace {
|
||||
name: config.name,
|
||||
events,
|
||||
snapshots,
|
||||
status_updates: all_status_updates,
|
||||
num_rounds: config.num_rounds,
|
||||
final_pipelines: pipelines.into_values().collect(),
|
||||
provisioned_instances,
|
||||
terminated_instances,
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Properties ─────────────────────────────────────────────────────────────
|
||||
|
||||
/// Every webhook eventually produces a terminal Forgejo status (success/failure/error).
|
||||
pub fn check_all_webhooks_terminate(trace: &CiSimTrace) -> bool {
|
||||
let webhook_commits: Vec<&str> = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter_map(|(_, e)| match e {
|
||||
CiSimEvent::WebhookReceived { commit_sha } => Some(commit_sha.as_str()),
|
||||
_ => None,
|
||||
})
|
||||
.collect();
|
||||
|
||||
for sha in webhook_commits {
|
||||
let has_terminal = trace.status_updates.iter().any(|u| {
|
||||
u.commit_sha == sha && (u.state == "success" || u.state == "failure" || u.state == "error")
|
||||
});
|
||||
if !has_terminal {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Job DAG ordering is always respected: no job runs before its `needs`.
|
||||
pub fn check_dag_ordering(trace: &CiSimTrace) -> bool {
|
||||
// Build a map of (pipeline_id, job_name) → round when started.
|
||||
let mut started: HashMap<(u64, &str), usize> = HashMap::new();
|
||||
let mut completed: HashMap<(u64, &str), usize> = HashMap::new();
|
||||
|
||||
for (round, event) in &trace.events {
|
||||
match event {
|
||||
CiSimEvent::JobStarted { job_id } => {
|
||||
started.insert(
|
||||
(job_id.pipeline_id.0, job_id.job_name.as_str()),
|
||||
*round,
|
||||
);
|
||||
}
|
||||
CiSimEvent::JobCompleted { job_id, .. } => {
|
||||
completed.insert(
|
||||
(job_id.pipeline_id.0, job_id.job_name.as_str()),
|
||||
*round,
|
||||
);
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
// For each pipeline, check that if job B needs job A, then A completed before B started.
|
||||
for pipeline in &trace.final_pipelines {
|
||||
for (name, job) in &pipeline.jobs {
|
||||
if let Some(&start_round) = started.get(&(pipeline.pipeline_id.0, name.as_str())) {
|
||||
for dep in &job.definition.needs {
|
||||
if let Some(&dep_complete_round) =
|
||||
completed.get(&(pipeline.pipeline_id.0, dep.as_str()))
|
||||
{
|
||||
if dep_complete_round > start_round {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Every provisioned instance is eventually terminated (no resource leaks).
|
||||
pub fn check_no_instance_leaks(trace: &CiSimTrace) -> bool {
|
||||
// Every instance that was provisioned should also be terminated.
|
||||
for instance_id in &trace.provisioned_instances {
|
||||
if !trace.terminated_instances.contains(instance_id) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Coordinator state is bounded: active pipeline count doesn't grow unboundedly.
|
||||
pub fn check_bounded_state(trace: &CiSimTrace, max_active: usize) -> bool {
|
||||
trace
|
||||
.snapshots
|
||||
.iter()
|
||||
.all(|s| s.active_pipelines <= max_active)
|
||||
}
|
||||
|
|
@ -9,3 +9,6 @@ pub mod gossip;
|
|||
|
||||
#[cfg(feature = "dashboard")]
|
||||
pub mod dashboard;
|
||||
|
||||
#[cfg(feature = "ci")]
|
||||
pub mod ci;
|
||||
|
|
|
|||
275
crates/simulation/tests/ci_properties.rs
Normal file
275
crates/simulation/tests/ci_properties.rs
Normal file
|
|
@ -0,0 +1,275 @@
|
|||
//! Property-based tests for the CI simulation.
|
||||
//!
|
||||
//! These verify invariants that should hold across all possible simulation configurations.
|
||||
|
||||
use swactor_ci::{EventType, WebhookEvent};
|
||||
use simulation::ci::sim::{self, CiSimConfig};
|
||||
|
||||
fn simple_yaml() -> String {
|
||||
r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
fmt:
|
||||
run: cargo fmt -- --check
|
||||
test:
|
||||
needs: [fmt]
|
||||
run: cargo test
|
||||
"#
|
||||
.into()
|
||||
}
|
||||
|
||||
fn push(branch: &str, sha: &str) -> WebhookEvent {
|
||||
WebhookEvent {
|
||||
event_type: EventType::Push,
|
||||
repo_owner: "user".into(),
|
||||
repo_name: "repo".into(),
|
||||
branch: branch.into(),
|
||||
commit_sha: sha.into(),
|
||||
tag: None,
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Property: Every webhook produces a terminal status ─────────────────────
|
||||
|
||||
#[test]
|
||||
fn property_all_webhooks_terminate_single() {
|
||||
let config = CiSimConfig {
|
||||
name: "prop-single".into(),
|
||||
num_rounds: 40,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: vec![(1, push("main", "sha-1"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = sim::run_simulation(config);
|
||||
assert!(
|
||||
sim::check_all_webhooks_terminate(&trace),
|
||||
"single webhook must terminate"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn property_all_webhooks_terminate_burst() {
|
||||
// Burst of webhooks all at once.
|
||||
let config = CiSimConfig {
|
||||
name: "prop-burst".into(),
|
||||
num_rounds: 60,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: (1..=5)
|
||||
.map(|i| (1, push("main", &format!("sha-burst-{i}"))))
|
||||
.collect(),
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = sim::run_simulation(config);
|
||||
assert!(
|
||||
sim::check_all_webhooks_terminate(&trace),
|
||||
"burst of 5 webhooks must all terminate"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn property_all_webhooks_terminate_staggered() {
|
||||
// Webhooks spread across rounds.
|
||||
let config = CiSimConfig {
|
||||
name: "prop-staggered".into(),
|
||||
num_rounds: 60,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: vec![
|
||||
(1, push("main", "sha-s1")),
|
||||
(5, push("main", "sha-s2")),
|
||||
(10, push("main", "sha-s3")),
|
||||
(15, push("main", "sha-s4")),
|
||||
],
|
||||
provision_latency: 2,
|
||||
job_duration: 3,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = sim::run_simulation(config);
|
||||
assert!(
|
||||
sim::check_all_webhooks_terminate(&trace),
|
||||
"staggered webhooks must all terminate"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Property: DAG ordering is always respected ─────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn property_dag_ordering_always_respected() {
|
||||
// Deep chain: a → b → c → d
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
deep:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
a:
|
||||
run: echo a
|
||||
b:
|
||||
needs: [a]
|
||||
run: echo b
|
||||
c:
|
||||
needs: [b]
|
||||
run: echo c
|
||||
d:
|
||||
needs: [c]
|
||||
run: echo d
|
||||
"#;
|
||||
let config = CiSimConfig {
|
||||
name: "prop-dag-deep".into(),
|
||||
num_rounds: 40,
|
||||
ci_yaml: yaml.into(),
|
||||
webhook_schedule: vec![(1, push("main", "sha-dag"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = sim::run_simulation(config);
|
||||
assert!(
|
||||
sim::check_dag_ordering(&trace),
|
||||
"deep DAG ordering must be respected"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn property_dag_ordering_diamond() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
diamond:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
root:
|
||||
run: echo root
|
||||
left:
|
||||
needs: [root]
|
||||
run: echo left
|
||||
right:
|
||||
needs: [root]
|
||||
run: echo right
|
||||
merge:
|
||||
needs: [left, right]
|
||||
run: echo merge
|
||||
"#;
|
||||
let config = CiSimConfig {
|
||||
name: "prop-dag-diamond".into(),
|
||||
num_rounds: 40,
|
||||
ci_yaml: yaml.into(),
|
||||
webhook_schedule: vec![(1, push("main", "sha-diamond"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = sim::run_simulation(config);
|
||||
assert!(
|
||||
sim::check_dag_ordering(&trace),
|
||||
"diamond DAG ordering must be respected"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Property: No instance leaks ────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn property_no_instance_leaks_under_failures() {
|
||||
let config = CiSimConfig {
|
||||
name: "prop-no-leaks-fail".into(),
|
||||
num_rounds: 40,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: vec![
|
||||
(1, push("main", "sha-leak1")),
|
||||
(3, push("main", "sha-leak2")),
|
||||
],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
job_failure_schedule: vec![(0, "fmt".into())],
|
||||
..Default::default()
|
||||
};
|
||||
let trace = sim::run_simulation(config);
|
||||
assert!(
|
||||
sim::check_no_instance_leaks(&trace),
|
||||
"no instance leaks even when jobs fail"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn property_no_instance_leaks_under_interruption() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test
|
||||
"#;
|
||||
let config = CiSimConfig {
|
||||
name: "prop-no-leaks-interrupt".into(),
|
||||
num_rounds: 40,
|
||||
ci_yaml: yaml.into(),
|
||||
webhook_schedule: vec![(1, push("main", "sha-int"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 5,
|
||||
instance_interrupt_schedule: vec![(4, "test".into())],
|
||||
..Default::default()
|
||||
};
|
||||
let trace = sim::run_simulation(config);
|
||||
assert!(
|
||||
sim::check_no_instance_leaks(&trace),
|
||||
"interrupted instances must be terminated"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Property: Bounded state ────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn property_bounded_state_under_rapid_pushes() {
|
||||
let config = CiSimConfig {
|
||||
name: "prop-bounded".into(),
|
||||
num_rounds: 100,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: (1..=20)
|
||||
.map(|i| (i, push("main", &format!("sha-rapid-{i}"))))
|
||||
.collect(),
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = sim::run_simulation(config);
|
||||
|
||||
// With 20 pushes, each triggering 1 pipeline with 2 jobs, we should never
|
||||
// have more than 20 active pipelines at once (and in practice much fewer).
|
||||
assert!(
|
||||
sim::check_bounded_state(&trace, 20),
|
||||
"active pipelines should be bounded"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Property: Provisioner offline doesn't lose work ────────────────────────
|
||||
|
||||
#[test]
|
||||
fn property_provisioner_offline_eventually_resolves() {
|
||||
let config = CiSimConfig {
|
||||
name: "prop-offline-resolve".into(),
|
||||
num_rounds: 60,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: vec![(3, push("main", "sha-offline"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
provisioner_offline_schedule: vec![(1, 15)],
|
||||
..Default::default()
|
||||
};
|
||||
let trace = sim::run_simulation(config);
|
||||
assert!(
|
||||
sim::check_all_webhooks_terminate(&trace),
|
||||
"webhooks during provisioner outage must still terminate"
|
||||
);
|
||||
}
|
||||
368
crates/simulation/tests/ci_scenarios.rs
Normal file
368
crates/simulation/tests/ci_scenarios.rs
Normal file
|
|
@ -0,0 +1,368 @@
|
|||
//! Scenario tests for the CI simulation.
|
||||
//!
|
||||
//! Each test tells a story: set up a scenario, run the simulation, verify outcomes.
|
||||
|
||||
use swactor_ci::{EventType, WebhookEvent};
|
||||
use simulation::ci::sim::{self, CiSimConfig, CiSimEvent};
|
||||
|
||||
fn basic_ci_yaml() -> String {
|
||||
r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
exclude: ["master"]
|
||||
jobs:
|
||||
fmt:
|
||||
run: cargo fmt -- --check
|
||||
clippy:
|
||||
run: cargo clippy
|
||||
test:
|
||||
needs: [fmt, clippy]
|
||||
run: cargo test
|
||||
full:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["master"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test --all-features
|
||||
timeout: 600
|
||||
bench:
|
||||
needs: [test]
|
||||
run: cargo bench
|
||||
"#
|
||||
.into()
|
||||
}
|
||||
|
||||
fn push_event(branch: &str, sha: &str) -> WebhookEvent {
|
||||
WebhookEvent {
|
||||
event_type: EventType::Push,
|
||||
repo_owner: "user".into(),
|
||||
repo_name: "repo".into(),
|
||||
branch: branch.into(),
|
||||
commit_sha: sha.into(),
|
||||
tag: None,
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Scenario: Single push triggers correct pipeline ────────────────────────
|
||||
|
||||
#[test]
|
||||
fn single_push_to_feature_branch_triggers_check_pipeline() {
|
||||
let config = CiSimConfig {
|
||||
name: "single-push-feature".into(),
|
||||
num_rounds: 30,
|
||||
ci_yaml: basic_ci_yaml(),
|
||||
webhook_schedule: vec![(1, push_event("feature-x", "abc123"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = sim::run_simulation(config);
|
||||
|
||||
// A pipeline was created.
|
||||
let pipeline_created = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter(|(_, e)| matches!(e, CiSimEvent::PipelineCreated { .. }))
|
||||
.count();
|
||||
assert_eq!(pipeline_created, 1, "exactly one pipeline should be created");
|
||||
|
||||
// The pipeline name should be "check" (not "full", since branch is not master).
|
||||
let check_created = trace.events.iter().any(|(_, e)| {
|
||||
matches!(e, CiSimEvent::PipelineCreated { name, .. } if name == "check")
|
||||
});
|
||||
assert!(check_created, "pipeline 'check' should be created");
|
||||
|
||||
// All jobs eventually complete (fmt, clippy, test).
|
||||
let completed_jobs: Vec<_> = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter_map(|(_, e)| match e {
|
||||
CiSimEvent::JobCompleted { job_id, passed } => Some((job_id.job_name.clone(), *passed)),
|
||||
_ => None,
|
||||
})
|
||||
.collect();
|
||||
|
||||
assert_eq!(completed_jobs.len(), 3, "all 3 jobs should complete");
|
||||
assert!(
|
||||
completed_jobs.iter().all(|(_, passed)| *passed),
|
||||
"all jobs should pass"
|
||||
);
|
||||
|
||||
// A terminal Forgejo status is emitted.
|
||||
assert!(
|
||||
sim::check_all_webhooks_terminate(&trace),
|
||||
"webhook should produce terminal status"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Scenario: Push to master triggers full pipeline ────────────────────────
|
||||
|
||||
#[test]
|
||||
fn push_to_master_triggers_full_pipeline() {
|
||||
let config = CiSimConfig {
|
||||
name: "push-master".into(),
|
||||
num_rounds: 30,
|
||||
ci_yaml: basic_ci_yaml(),
|
||||
webhook_schedule: vec![(1, push_event("master", "def456"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = sim::run_simulation(config);
|
||||
|
||||
let full_created = trace.events.iter().any(|(_, e)| {
|
||||
matches!(e, CiSimEvent::PipelineCreated { name, .. } if name == "full")
|
||||
});
|
||||
assert!(full_created, "pipeline 'full' should be created");
|
||||
|
||||
// Both test and bench should eventually complete.
|
||||
let completed: Vec<String> = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter_map(|(_, e)| match e {
|
||||
CiSimEvent::JobCompleted { job_id, .. } => Some(job_id.job_name.clone()),
|
||||
_ => None,
|
||||
})
|
||||
.collect();
|
||||
|
||||
assert!(completed.contains(&"test".to_string()), "test should complete");
|
||||
assert!(completed.contains(&"bench".to_string()), "bench should complete");
|
||||
}
|
||||
|
||||
// ─── Scenario: Jobs execute in DAG order ────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn jobs_execute_in_dag_order() {
|
||||
let config = CiSimConfig {
|
||||
name: "dag-order".into(),
|
||||
num_rounds: 30,
|
||||
ci_yaml: basic_ci_yaml(),
|
||||
webhook_schedule: vec![(1, push_event("feature-y", "aaa111"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = sim::run_simulation(config);
|
||||
|
||||
// test should start after fmt and clippy complete.
|
||||
assert!(
|
||||
sim::check_dag_ordering(&trace),
|
||||
"DAG ordering must be respected"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Scenario: Job failure skips dependents ─────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn job_failure_skips_downstream_dependents() {
|
||||
let config = CiSimConfig {
|
||||
name: "failure-skip".into(),
|
||||
num_rounds: 30,
|
||||
ci_yaml: basic_ci_yaml(),
|
||||
webhook_schedule: vec![(1, push_event("feature-z", "bbb222"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
// fmt will fail, so test (which needs fmt) should be skipped.
|
||||
job_failure_schedule: vec![(0, "fmt".into())],
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = sim::run_simulation(config);
|
||||
|
||||
// test should be skipped.
|
||||
let test_skipped = trace.events.iter().any(|(_, e)| {
|
||||
matches!(e, CiSimEvent::JobSkipped { job_id } if job_id.job_name == "test")
|
||||
});
|
||||
assert!(test_skipped, "test job should be skipped when fmt fails");
|
||||
|
||||
// Pipeline should be marked as failed.
|
||||
let pipeline_failed = trace.status_updates.iter().any(|u| {
|
||||
u.context == "ci/check" && u.state == "failure"
|
||||
});
|
||||
assert!(pipeline_failed, "pipeline should report failure status");
|
||||
}
|
||||
|
||||
// ─── Scenario: Parallel pushes execute independently ────────────────────────
|
||||
|
||||
#[test]
|
||||
fn parallel_pushes_execute_independently() {
|
||||
let config = CiSimConfig {
|
||||
name: "parallel-pushes".into(),
|
||||
num_rounds: 40,
|
||||
ci_yaml: basic_ci_yaml(),
|
||||
webhook_schedule: vec![
|
||||
(1, push_event("feature-a", "ccc333")),
|
||||
(1, push_event("feature-b", "ddd444")),
|
||||
],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = sim::run_simulation(config);
|
||||
|
||||
// Two pipelines should be created.
|
||||
let pipeline_count = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter(|(_, e)| matches!(e, CiSimEvent::PipelineCreated { .. }))
|
||||
.count();
|
||||
assert_eq!(pipeline_count, 2, "two pipelines should be created");
|
||||
|
||||
// Both should have terminal status.
|
||||
assert!(
|
||||
sim::check_all_webhooks_terminate(&trace),
|
||||
"both webhooks should produce terminal statuses"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Scenario: Provisioner goes offline, jobs queue and resume ──────────────
|
||||
|
||||
#[test]
|
||||
fn provisioner_offline_queues_then_resumes() {
|
||||
let config = CiSimConfig {
|
||||
name: "provisioner-offline".into(),
|
||||
num_rounds: 50,
|
||||
ci_yaml: basic_ci_yaml(),
|
||||
// Push at round 3, provisioner offline rounds 1-10.
|
||||
webhook_schedule: vec![(3, push_event("feature-q", "eee555"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
provisioner_offline_schedule: vec![(1, 10)],
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = sim::run_simulation(config);
|
||||
|
||||
// Provisioner went offline and came back.
|
||||
let went_offline = trace
|
||||
.events
|
||||
.iter()
|
||||
.any(|(_, e)| matches!(e, CiSimEvent::ProvisionerWentOffline));
|
||||
let came_online = trace
|
||||
.events
|
||||
.iter()
|
||||
.any(|(_, e)| matches!(e, CiSimEvent::ProvisionerCameOnline));
|
||||
assert!(went_offline, "provisioner should go offline");
|
||||
assert!(came_online, "provisioner should come back online");
|
||||
|
||||
// Despite the outage, all jobs should eventually complete.
|
||||
assert!(
|
||||
sim::check_all_webhooks_terminate(&trace),
|
||||
"pipeline should complete after provisioner returns"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Scenario: Spot instance interrupted mid-job ────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn spot_instance_interruption_reports_failure() {
|
||||
// Use a simpler pipeline so we have a clear target to interrupt.
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test
|
||||
"#;
|
||||
let config = CiSimConfig {
|
||||
name: "spot-interrupt".into(),
|
||||
num_rounds: 30,
|
||||
ci_yaml: yaml.into(),
|
||||
webhook_schedule: vec![(1, push_event("main", "fff666"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 5,
|
||||
// Interrupt the test job at round 4 (while it's still running).
|
||||
instance_interrupt_schedule: vec![(4, "test".into())],
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = sim::run_simulation(config);
|
||||
|
||||
// The pipeline should be terminal (failed due to interruption).
|
||||
let has_failure_status = trace
|
||||
.status_updates
|
||||
.iter()
|
||||
.any(|u| u.state == "failure" || u.state == "error");
|
||||
assert!(
|
||||
has_failure_status,
|
||||
"interrupted job should produce a failure status"
|
||||
);
|
||||
|
||||
// The instance should be terminated.
|
||||
assert!(
|
||||
sim::check_no_instance_leaks(&trace),
|
||||
"interrupted instance should be terminated"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Scenario: All jobs pass → pipeline success ─────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn all_jobs_pass_marks_pipeline_success() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
lint:
|
||||
run: cargo clippy
|
||||
test:
|
||||
needs: [lint]
|
||||
run: cargo test
|
||||
"#;
|
||||
let config = CiSimConfig {
|
||||
name: "all-pass".into(),
|
||||
num_rounds: 30,
|
||||
ci_yaml: yaml.into(),
|
||||
webhook_schedule: vec![(1, push_event("main", "ggg777"))],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = sim::run_simulation(config);
|
||||
|
||||
let has_success = trace
|
||||
.status_updates
|
||||
.iter()
|
||||
.any(|u| u.state == "success" && u.context == "ci/check");
|
||||
assert!(has_success, "pipeline should be marked as success");
|
||||
}
|
||||
|
||||
// ─── Scenario: Every provisioned instance is terminated ─────────────────────
|
||||
|
||||
#[test]
|
||||
fn no_instance_resource_leaks() {
|
||||
let config = CiSimConfig {
|
||||
name: "no-leaks".into(),
|
||||
num_rounds: 40,
|
||||
ci_yaml: basic_ci_yaml(),
|
||||
webhook_schedule: vec![
|
||||
(1, push_event("feature-1", "h1")),
|
||||
(5, push_event("feature-2", "h2")),
|
||||
],
|
||||
provision_latency: 1,
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = sim::run_simulation(config);
|
||||
|
||||
assert!(
|
||||
sim::check_no_instance_leaks(&trace),
|
||||
"all provisioned instances must be terminated"
|
||||
);
|
||||
}
|
||||
256
crates/simulation/tests/local_ci_properties.rs
Normal file
256
crates/simulation/tests/local_ci_properties.rs
Normal file
|
|
@ -0,0 +1,256 @@
|
|||
//! Property-based tests for the local CI simulation.
|
||||
//!
|
||||
//! These verify invariants that should hold across all possible simulation configurations.
|
||||
|
||||
use simulation::ci::local_sim::{self, LocalSimConfig};
|
||||
use swactor_ci::{EventType, WebhookEvent};
|
||||
|
||||
fn simple_yaml() -> String {
|
||||
r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
fmt:
|
||||
run: cargo fmt -- --check
|
||||
test:
|
||||
needs: [fmt]
|
||||
run: cargo test
|
||||
"#
|
||||
.into()
|
||||
}
|
||||
|
||||
fn push(branch: &str, sha: &str) -> WebhookEvent {
|
||||
WebhookEvent {
|
||||
event_type: EventType::Push,
|
||||
repo_owner: "user".into(),
|
||||
repo_name: "repo".into(),
|
||||
branch: branch.into(),
|
||||
commit_sha: sha.into(),
|
||||
tag: None,
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Property: One-at-a-time ────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn property_one_at_a_time_single_push() {
|
||||
let config = LocalSimConfig {
|
||||
name: "prop-1at1-single".into(),
|
||||
num_rounds: 30,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: vec![(1, push("main", "sha-1"))],
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = local_sim::run_simulation(config);
|
||||
assert!(
|
||||
local_sim::check_one_at_a_time(&trace),
|
||||
"at most one job running at a time"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn property_one_at_a_time_burst() {
|
||||
let config = LocalSimConfig {
|
||||
name: "prop-1at1-burst".into(),
|
||||
num_rounds: 100,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: (1..=10)
|
||||
.map(|i| (1, push(&format!("branch-{i}"), &format!("sha-{i}"))))
|
||||
.collect(),
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = local_sim::run_simulation(config);
|
||||
assert!(
|
||||
local_sim::check_one_at_a_time(&trace),
|
||||
"at most one job running at a time under burst"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn property_one_at_a_time_staggered() {
|
||||
let config = LocalSimConfig {
|
||||
name: "prop-1at1-stagger".into(),
|
||||
num_rounds: 80,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: vec![
|
||||
(1, push("a", "sha-a")),
|
||||
(3, push("b", "sha-b")),
|
||||
(5, push("c", "sha-c")),
|
||||
(7, push("d", "sha-d")),
|
||||
],
|
||||
job_duration: 3,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = local_sim::run_simulation(config);
|
||||
assert!(
|
||||
local_sim::check_one_at_a_time(&trace),
|
||||
"at most one job running at a time under stagger"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Property: Supersede correctness ────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn property_superseded_pipelines_never_run() {
|
||||
let config = LocalSimConfig {
|
||||
name: "prop-supersede".into(),
|
||||
num_rounds: 60,
|
||||
ci_yaml: simple_yaml(),
|
||||
// Same branch, rapid pushes while jobs are long.
|
||||
webhook_schedule: (1..=8)
|
||||
.map(|i| (i, push("feature", &format!("sha-ss-{i}"))))
|
||||
.collect(),
|
||||
job_duration: 4,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = local_sim::run_simulation(config);
|
||||
assert!(
|
||||
local_sim::check_superseded_no_running(&trace),
|
||||
"superseded pipelines should never have a running job"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Property: Termination ──────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn property_all_non_superseded_terminate() {
|
||||
let config = LocalSimConfig {
|
||||
name: "prop-terminate".into(),
|
||||
num_rounds: 100,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: vec![
|
||||
(1, push("a", "sha-t1")),
|
||||
(2, push("b", "sha-t2")),
|
||||
(3, push("a", "sha-t3")),
|
||||
(10, push("c", "sha-t4")),
|
||||
],
|
||||
job_duration: 3,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = local_sim::run_simulation(config);
|
||||
assert!(
|
||||
local_sim::check_termination(&trace),
|
||||
"all non-superseded pipelines must reach terminal status"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn property_all_webhooks_terminate() {
|
||||
let config = LocalSimConfig {
|
||||
name: "prop-wh-terminate".into(),
|
||||
num_rounds: 100,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: (1..=5)
|
||||
.map(|i| (i * 3, push("main", &format!("sha-wh-{i}"))))
|
||||
.collect(),
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = local_sim::run_simulation(config);
|
||||
assert!(
|
||||
local_sim::check_all_webhooks_terminate(&trace),
|
||||
"every webhook must produce a terminal status"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Property: DAG ordering ─────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn property_dag_ordering_deep_chain() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
deep:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
a:
|
||||
run: echo a
|
||||
b:
|
||||
needs: [a]
|
||||
run: echo b
|
||||
c:
|
||||
needs: [b]
|
||||
run: echo c
|
||||
d:
|
||||
needs: [c]
|
||||
run: echo d
|
||||
"#;
|
||||
let config = LocalSimConfig {
|
||||
name: "prop-dag-deep".into(),
|
||||
num_rounds: 50,
|
||||
ci_yaml: yaml.into(),
|
||||
webhook_schedule: vec![(1, push("main", "sha-dag"))],
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = local_sim::run_simulation(config);
|
||||
assert!(
|
||||
local_sim::check_dag_ordering(&trace),
|
||||
"deep DAG ordering must be respected"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn property_dag_ordering_diamond() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
diamond:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
root:
|
||||
run: echo root
|
||||
left:
|
||||
needs: [root]
|
||||
run: echo left
|
||||
right:
|
||||
needs: [root]
|
||||
run: echo right
|
||||
merge:
|
||||
needs: [left, right]
|
||||
run: echo merge
|
||||
"#;
|
||||
let config = LocalSimConfig {
|
||||
name: "prop-dag-diamond".into(),
|
||||
num_rounds: 50,
|
||||
ci_yaml: yaml.into(),
|
||||
webhook_schedule: vec![(1, push("main", "sha-dia"))],
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = local_sim::run_simulation(config);
|
||||
assert!(
|
||||
local_sim::check_dag_ordering(&trace),
|
||||
"diamond DAG ordering must be respected"
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Property: FIFO across branches ─────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn property_fifo_across_branches() {
|
||||
let config = LocalSimConfig {
|
||||
name: "prop-fifo".into(),
|
||||
num_rounds: 80,
|
||||
ci_yaml: simple_yaml(),
|
||||
webhook_schedule: vec![
|
||||
(1, push("a", "sha-f1")),
|
||||
(2, push("b", "sha-f2")),
|
||||
(3, push("c", "sha-f3")),
|
||||
],
|
||||
job_duration: 3,
|
||||
..Default::default()
|
||||
};
|
||||
let trace = local_sim::run_simulation(config);
|
||||
assert!(
|
||||
local_sim::check_fifo_order(&trace),
|
||||
"branches must execute in FIFO queue order"
|
||||
);
|
||||
}
|
||||
324
crates/simulation/tests/local_ci_scenarios.rs
Normal file
324
crates/simulation/tests/local_ci_scenarios.rs
Normal file
|
|
@ -0,0 +1,324 @@
|
|||
//! Scenario tests for the local CI simulation.
|
||||
//!
|
||||
//! Each test tells a story: set up a scenario, run the simulation, verify outcomes.
|
||||
|
||||
use simulation::ci::local_sim::{self, LocalSimConfig, LocalSimEvent};
|
||||
use swactor_ci::{EventType, WebhookEvent};
|
||||
|
||||
fn basic_ci_yaml() -> String {
|
||||
r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
exclude: ["master"]
|
||||
jobs:
|
||||
fmt:
|
||||
run: cargo fmt -- --check
|
||||
clippy:
|
||||
run: cargo clippy
|
||||
test:
|
||||
needs: [fmt, clippy]
|
||||
run: cargo test
|
||||
full:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["master"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test --all-features
|
||||
timeout: 600
|
||||
bench:
|
||||
needs: [test]
|
||||
run: cargo bench
|
||||
"#
|
||||
.into()
|
||||
}
|
||||
|
||||
fn push_event(branch: &str, sha: &str) -> WebhookEvent {
|
||||
WebhookEvent {
|
||||
event_type: EventType::Push,
|
||||
repo_owner: "user".into(),
|
||||
repo_name: "repo".into(),
|
||||
branch: branch.into(),
|
||||
commit_sha: sha.into(),
|
||||
tag: None,
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Scenario: Single push → single job → passes ────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn single_push_runs_and_completes() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test
|
||||
"#;
|
||||
let config = LocalSimConfig {
|
||||
name: "single-push".into(),
|
||||
num_rounds: 20,
|
||||
ci_yaml: yaml.into(),
|
||||
webhook_schedule: vec![(1, push_event("main", "sha1"))],
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = local_sim::run_simulation(config);
|
||||
|
||||
let created = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter(|(_, e)| matches!(e, LocalSimEvent::PipelineCreated { .. }))
|
||||
.count();
|
||||
assert_eq!(created, 1);
|
||||
|
||||
let completed = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter(|(_, e)| matches!(e, LocalSimEvent::JobCompleted { passed: true, .. }))
|
||||
.count();
|
||||
assert_eq!(completed, 1);
|
||||
|
||||
assert!(local_sim::check_all_webhooks_terminate(&trace));
|
||||
}
|
||||
|
||||
// ─── Scenario: Push A, push A again while queued → only latest runs ─────────
|
||||
|
||||
#[test]
|
||||
fn supersede_queued_same_branch() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test
|
||||
"#;
|
||||
let config = LocalSimConfig {
|
||||
name: "supersede-queued".into(),
|
||||
num_rounds: 30,
|
||||
ci_yaml: yaml.into(),
|
||||
// Push A at round 1 occupies the runner. Push A' at round 2 queues.
|
||||
// Push A'' at round 3 should supersede A'.
|
||||
webhook_schedule: vec![
|
||||
(1, push_event("feature", "sha-a1")),
|
||||
(2, push_event("feature", "sha-a2")),
|
||||
(3, push_event("feature", "sha-a3")),
|
||||
],
|
||||
job_duration: 5,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = local_sim::run_simulation(config);
|
||||
|
||||
// sha-a2 should be superseded.
|
||||
let superseded = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter(|(_, e)| matches!(e, LocalSimEvent::PipelineSuperseded { .. }))
|
||||
.count();
|
||||
assert!(superseded >= 1, "at least one pipeline should be superseded");
|
||||
|
||||
// sha-a2 should have an "error" status (superseded).
|
||||
let a2_error = trace
|
||||
.status_updates
|
||||
.iter()
|
||||
.any(|u| u.commit_sha == "sha-a2" && u.state == "error");
|
||||
assert!(a2_error, "superseded pipeline should report error status");
|
||||
|
||||
// sha-a1 and sha-a3 should both reach terminal status.
|
||||
let a1_terminal = trace
|
||||
.status_updates
|
||||
.iter()
|
||||
.any(|u| u.commit_sha == "sha-a1" && (u.state == "success" || u.state == "failure"));
|
||||
let a3_terminal = trace
|
||||
.status_updates
|
||||
.iter()
|
||||
.any(|u| u.commit_sha == "sha-a3" && (u.state == "success" || u.state == "failure"));
|
||||
assert!(a1_terminal, "first push should complete");
|
||||
assert!(a3_terminal, "latest push should complete");
|
||||
|
||||
assert!(local_sim::check_superseded_no_running(&trace));
|
||||
}
|
||||
|
||||
// ─── Scenario: Push A, push A while running → running completes, new queues ─
|
||||
|
||||
#[test]
|
||||
fn push_while_running_does_not_supersede_active() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test
|
||||
"#;
|
||||
let config = LocalSimConfig {
|
||||
name: "no-supersede-active".into(),
|
||||
num_rounds: 30,
|
||||
ci_yaml: yaml.into(),
|
||||
// Push A at round 1 starts running immediately.
|
||||
// Push A' at round 2 should queue (not cancel the running job).
|
||||
webhook_schedule: vec![
|
||||
(1, push_event("feature", "sha-run1")),
|
||||
(2, push_event("feature", "sha-run2")),
|
||||
],
|
||||
job_duration: 4,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = local_sim::run_simulation(config);
|
||||
|
||||
// Both should reach terminal status.
|
||||
let run1_terminal = trace
|
||||
.status_updates
|
||||
.iter()
|
||||
.any(|u| u.commit_sha == "sha-run1" && u.state == "success");
|
||||
let run2_terminal = trace
|
||||
.status_updates
|
||||
.iter()
|
||||
.any(|u| u.commit_sha == "sha-run2" && u.state == "success");
|
||||
|
||||
assert!(run1_terminal, "running pipeline should complete normally");
|
||||
assert!(run2_terminal, "queued pipeline should run after");
|
||||
|
||||
assert!(local_sim::check_one_at_a_time(&trace));
|
||||
}
|
||||
|
||||
// ─── Scenario: Push A, push B → both run in FIFO order ─────────────────────
|
||||
|
||||
#[test]
|
||||
fn different_branches_run_fifo() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
check:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
test:
|
||||
run: cargo test
|
||||
"#;
|
||||
let config = LocalSimConfig {
|
||||
name: "fifo-branches".into(),
|
||||
num_rounds: 30,
|
||||
ci_yaml: yaml.into(),
|
||||
webhook_schedule: vec![
|
||||
(1, push_event("feature-a", "sha-fa")),
|
||||
(1, push_event("feature-b", "sha-fb")),
|
||||
],
|
||||
job_duration: 3,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = local_sim::run_simulation(config);
|
||||
|
||||
// Both should complete.
|
||||
assert!(local_sim::check_all_webhooks_terminate(&trace));
|
||||
|
||||
// FIFO order respected.
|
||||
assert!(local_sim::check_fifo_order(&trace));
|
||||
|
||||
// One at a time.
|
||||
assert!(local_sim::check_one_at_a_time(&trace));
|
||||
}
|
||||
|
||||
// ─── Scenario: Job failure → dependents skipped → next pipeline starts ──────
|
||||
|
||||
#[test]
|
||||
fn job_failure_skips_dependents_and_advances() {
|
||||
let config = LocalSimConfig {
|
||||
name: "failure-skip-advance".into(),
|
||||
num_rounds: 40,
|
||||
ci_yaml: basic_ci_yaml(),
|
||||
webhook_schedule: vec![
|
||||
(1, push_event("feature-x", "sha-fail")),
|
||||
(2, push_event("feature-y", "sha-next")),
|
||||
],
|
||||
job_duration: 2,
|
||||
// fmt fails.
|
||||
job_failure_schedule: vec![(0, "fmt".into())],
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = local_sim::run_simulation(config);
|
||||
|
||||
// test should be skipped in feature-x pipeline (needs fmt which fails).
|
||||
let test_skipped = trace.events.iter().any(|(_, e)| {
|
||||
matches!(e, LocalSimEvent::JobSkipped { job_id } if job_id.job_name == "test")
|
||||
});
|
||||
assert!(test_skipped, "test should be skipped when fmt fails");
|
||||
|
||||
// Both pipelines should reach terminal.
|
||||
assert!(local_sim::check_all_webhooks_terminate(&trace));
|
||||
|
||||
// One at a time.
|
||||
assert!(local_sim::check_one_at_a_time(&trace));
|
||||
}
|
||||
|
||||
// ─── Scenario: Diamond DAG → jobs serialize respecting deps ─────────────────
|
||||
|
||||
#[test]
|
||||
fn diamond_dag_serialized_with_deps() {
|
||||
let yaml = r#"
|
||||
pipelines:
|
||||
diamond:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
root:
|
||||
run: echo root
|
||||
left:
|
||||
needs: [root]
|
||||
run: echo left
|
||||
right:
|
||||
needs: [root]
|
||||
run: echo right
|
||||
merge:
|
||||
needs: [left, right]
|
||||
run: echo merge
|
||||
"#;
|
||||
let config = LocalSimConfig {
|
||||
name: "diamond-dag".into(),
|
||||
num_rounds: 50,
|
||||
ci_yaml: yaml.into(),
|
||||
webhook_schedule: vec![(1, push_event("main", "sha-diamond"))],
|
||||
job_duration: 2,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let trace = local_sim::run_simulation(config);
|
||||
|
||||
// All 4 jobs should complete.
|
||||
let completed = trace
|
||||
.events
|
||||
.iter()
|
||||
.filter(|(_, e)| matches!(e, LocalSimEvent::JobCompleted { .. }))
|
||||
.count();
|
||||
assert_eq!(completed, 4, "all 4 diamond jobs should complete");
|
||||
|
||||
// DAG ordering respected.
|
||||
assert!(local_sim::check_dag_ordering(&trace));
|
||||
|
||||
// One at a time (serial).
|
||||
assert!(local_sim::check_one_at_a_time(&trace));
|
||||
|
||||
// Pipeline should succeed.
|
||||
let success = trace
|
||||
.status_updates
|
||||
.iter()
|
||||
.any(|u| u.state == "success" && u.context == "ci/diamond");
|
||||
assert!(success, "diamond pipeline should succeed");
|
||||
}
|
||||
520
docs/development_history/CI_RELAY.md
Normal file
520
docs/development_history/CI_RELAY.md
Normal file
|
|
@ -0,0 +1,520 @@
|
|||
# CI Webhook Relay via Iroh — Development History
|
||||
|
||||
> Covers the implementation of `ci-relay` and the iroh webhook receiver in
|
||||
> `local-runner`, enabling Forgejo webhooks to reach a NAT'd CI runner via
|
||||
> iroh's QUIC transport with automatic NAT traversal.
|
||||
>
|
||||
> ~3 files created · ~2 files modified · ~350 insertions
|
||||
>
|
||||
> *Branch: `spot-instance`*
|
||||
|
||||
---
|
||||
|
||||
## Table of Contents
|
||||
|
||||
1. [Problem & Motivation](#1-problem--motivation)
|
||||
2. [Architecture](#2-architecture)
|
||||
3. [What Was Built](#3-what-was-built)
|
||||
4. [ci-relay Binary](#4-ci-relay-binary)
|
||||
5. [local-runner Iroh Receiver](#5-local-runner-iroh-receiver)
|
||||
6. [Wire Protocol](#6-wire-protocol)
|
||||
7. [Connection Flow](#7-connection-flow)
|
||||
8. [Design Decisions & Tradeoffs](#8-design-decisions--tradeoffs)
|
||||
9. [Manual Testing Guide](#9-manual-testing-guide)
|
||||
10. [Known Gaps & Future Improvements](#10-known-gaps--future-improvements)
|
||||
|
||||
---
|
||||
|
||||
## 1. Problem & Motivation
|
||||
|
||||
The CI runner (`local-runner`) was designed for same-LAN usage: Forgejo sends
|
||||
webhooks over HTTP to the runner's listen port. In the real deployment:
|
||||
|
||||
- **Forgejo** runs on a VPS (`zachery.lol` / `139.59.195.69`)
|
||||
- **CI runner** runs on a Thinkpad at home (`192.168.1.102`), behind NAT
|
||||
|
||||
The VPS cannot reach the Thinkpad directly — no inbound port is open, no
|
||||
static IP, no UPnP. Traditional solutions (SSH reverse tunnel, VPN, port
|
||||
forwarding on router) all require ongoing configuration and are fragile.
|
||||
|
||||
iroh is already integrated in swactor's distribution layer (`iroh_driver.rs`)
|
||||
for SWIM protocol traffic. It provides QUIC connections with automatic NAT
|
||||
traversal via relay servers — exactly what's needed to bridge the webhook gap.
|
||||
|
||||
### Why Not Just SSH Tunnel?
|
||||
|
||||
An SSH tunnel (`ssh -R 8787:localhost:8787 zachery.lol`) would work, but:
|
||||
|
||||
- Tunnels drop on network changes (laptop suspend, WiFi roaming)
|
||||
- Requires autossh or systemd to keep alive
|
||||
- Another moving part to debug when CI stops working
|
||||
- Doesn't reuse any existing infrastructure
|
||||
|
||||
iroh handles reconnection, relay fallback, and NAT traversal automatically.
|
||||
The implementation reuses the same tagged-message-over-QUIC-stream pattern
|
||||
already proven in `iroh_driver.rs`.
|
||||
|
||||
---
|
||||
|
||||
## 2. Architecture
|
||||
|
||||
```
|
||||
┌─────────────────────────────┐ ┌──────────────────────────────────┐
|
||||
│ VPS (zachery.lol) │ │ Thinkpad (192.168.1.102) │
|
||||
│ │ iroh │ │
|
||||
│ Forgejo ──webhook──► Relay ├───────►│ local-runner │
|
||||
│ :8787 │ QUIC │ (coordinator, runner, reporter) │
|
||||
│ │ │ │
|
||||
└─────────────────────────────┘ └──────────────────────────────────┘
|
||||
```
|
||||
|
||||
**VPS side** — `ci-relay` binary:
|
||||
- HTTP listener receives webhook POSTs from Forgejo (localhost only)
|
||||
- iroh endpoint accepts the runner's inbound connection
|
||||
- Forwards parsed `WebhookEvent` payloads over iroh uni streams
|
||||
|
||||
**Thinkpad side** — `local-runner` with `--relay-node-id`:
|
||||
- Connects to the VPS relay's iroh endpoint on startup
|
||||
- Receives `WebhookEvent` over iroh uni streams
|
||||
- Feeds events into `LocalCoordinator` via existing `Webhook` message
|
||||
- Status updates go directly Thinkpad → Forgejo API over HTTPS (no relay needed)
|
||||
|
||||
The relay is intentionally minimal — it's a bridge, not a CI component. All CI
|
||||
logic stays in `local-runner`.
|
||||
|
||||
---
|
||||
|
||||
## 3. What Was Built
|
||||
|
||||
| Component | Location | Nature |
|
||||
|-----------|----------|--------|
|
||||
| ci-relay binary | `crates/ci-relay/Cargo.toml`, `src/main.rs` | **New** — VPS webhook relay |
|
||||
| Iroh receiver | `crates/local-runner/src/main.rs` | **Modified** — iroh webhook source |
|
||||
| Dependencies | `crates/local-runner/Cargo.toml` | **Modified** — added iroh, tokio, serde_json |
|
||||
| Workspace | `Cargo.toml` | **Modified** — added ci-relay to members |
|
||||
|
||||
---
|
||||
|
||||
## 4. ci-relay Binary
|
||||
|
||||
### `crates/ci-relay/src/main.rs`
|
||||
|
||||
The relay runs two subsystems on a single process:
|
||||
|
||||
1. **iroh acceptor** (tokio task): accepts inbound connections from the runner,
|
||||
caches the most recent one in `Arc<TokioMutex<Option<Connection>>>`
|
||||
2. **HTTP listener** (main thread, blocking `tiny_http`): receives Forgejo
|
||||
webhook POSTs, verifies HMAC, parses event, forwards over iroh
|
||||
|
||||
### Webhook Handling
|
||||
|
||||
Reuses the same verification and parsing logic as `webhook_server.rs`:
|
||||
|
||||
- HMAC-SHA256 verification via `X-Forgejo-Signature` header (skippable with empty secret)
|
||||
- Event type from `X-Forgejo-Event` header: `push` → `Push`, `create` → `Tag`, `pull_request` → `Merge`
|
||||
- JSON parsing via `parse_webhook_json()` (re-exported from `swactor-ci`)
|
||||
|
||||
The relay uses `parse_webhook_json` directly rather than duplicating parsing
|
||||
logic. This keeps webhook interpretation consistent between HTTP and iroh paths.
|
||||
|
||||
### Forwarding
|
||||
|
||||
On webhook receipt, the relay:
|
||||
1. Serializes the `WebhookEvent` to JSON
|
||||
2. Opens a unidirectional QUIC stream on the cached connection
|
||||
3. Writes the tagged message (`ci::WebhookEvent` tag + JSON payload)
|
||||
4. Finishes the stream
|
||||
|
||||
If no runner is connected, the relay returns HTTP 502 to Forgejo. Forgejo will
|
||||
retry the webhook per its configured retry policy.
|
||||
|
||||
### CLI
|
||||
|
||||
```
|
||||
ci-relay [OPTIONS]
|
||||
|
||||
Options:
|
||||
--port <PORT> HTTP port for Forgejo webhooks [default: 8787]
|
||||
--secret <SECRET> HMAC-SHA256 secret [default: "" (no verification)]
|
||||
```
|
||||
|
||||
On startup, the relay prints its iroh Node ID — this is the value the runner
|
||||
needs for `--relay-node-id`.
|
||||
|
||||
---
|
||||
|
||||
## 5. local-runner Iroh Receiver
|
||||
|
||||
### New CLI Flag
|
||||
|
||||
```
|
||||
--relay-node-id <HEX> Iroh Node ID of the VPS ci-relay
|
||||
```
|
||||
|
||||
When `--relay-node-id` is provided:
|
||||
- The HTTP webhook listener is **not started** (no port conflict, no exposure)
|
||||
- An `iroh-receiver` thread starts instead
|
||||
|
||||
When omitted, behavior is unchanged — the HTTP listener starts on `--port`
|
||||
as before.
|
||||
|
||||
### `start_iroh_receiver()`
|
||||
|
||||
Spawns a dedicated thread (`iroh-receiver`) with its own single-threaded tokio
|
||||
runtime:
|
||||
|
||||
1. Creates an iroh `Endpoint` with ALPN `b"swactor/ci/1"`
|
||||
2. Connects to the relay's `PublicKey` (parsed from the hex flag)
|
||||
3. Enters a receive loop:
|
||||
- `conn.accept_uni()` with 1-second timeout
|
||||
- On stream: reads tagged message, deserializes `WebhookEvent`
|
||||
- Sends `LocalCoordinatorMsg::Webhook(event)` to the coordinator via the swactor runtime
|
||||
- On timeout: checks the `stop` flag (for graceful shutdown via Ctrl-C)
|
||||
- On connection error: breaks and exits
|
||||
|
||||
The thread respects the same `AtomicBool` stop flag as the main loop, so
|
||||
Ctrl-C cleanly shuts down both the swactor runtime and the iroh connection.
|
||||
|
||||
---
|
||||
|
||||
## 6. Wire Protocol
|
||||
|
||||
### ALPN
|
||||
|
||||
```rust
|
||||
const CI_ALPN: &[u8] = b"swactor/ci/1";
|
||||
```
|
||||
|
||||
Distinct from SWIM traffic (`b"swactor/swim/1"`). This allows both protocols
|
||||
to coexist on the same iroh endpoint in the future if needed.
|
||||
|
||||
### Frame Format
|
||||
|
||||
Same tagged-message format as `iroh_driver.rs`:
|
||||
|
||||
```
|
||||
[4 bytes: tag_len (big-endian u32)]
|
||||
[tag_len bytes: tag string]
|
||||
[remaining bytes: payload]
|
||||
```
|
||||
|
||||
For webhook events:
|
||||
- Tag: `"ci::WebhookEvent"` (17 bytes)
|
||||
- Payload: JSON-serialized `WebhookEvent`
|
||||
|
||||
### Transport
|
||||
|
||||
Each webhook is one unidirectional QUIC stream. The relay opens the stream,
|
||||
writes the tagged message, and finishes. The runner reads the message and the
|
||||
stream closes. No persistent framing or multiplexing needed — QUIC streams
|
||||
are lightweight.
|
||||
|
||||
---
|
||||
|
||||
## 7. Connection Flow
|
||||
|
||||
```
|
||||
1. VPS starts ci-relay
|
||||
→ iroh Endpoint binds
|
||||
→ prints Node ID (ed25519 public key, hex)
|
||||
→ HTTP listener starts on --port
|
||||
→ waits for runner connection
|
||||
|
||||
2. Thinkpad starts local-runner --relay-node-id <hex>
|
||||
→ iroh Endpoint binds
|
||||
→ connects to relay's PublicKey
|
||||
→ iroh handles NAT traversal (direct or via relay server)
|
||||
→ relay logs "Runner connected: <runner-node-id>"
|
||||
|
||||
3. Forgejo sends webhook POST to localhost:8787 on VPS
|
||||
→ relay verifies HMAC, parses event
|
||||
→ relay opens uni stream on cached connection
|
||||
→ writes tagged WebhookEvent
|
||||
→ runner receives, deserializes, dispatches to coordinator
|
||||
|
||||
4. Coordinator triggers pipeline
|
||||
→ StatusReporter posts status to Forgejo API directly
|
||||
(Thinkpad → zachery.lol over HTTPS, no relay involvement)
|
||||
```
|
||||
|
||||
The iroh connection is initiated by the runner (outbound from NAT), so no port
|
||||
forwarding is needed. iroh's relay servers handle the initial rendezvous, then
|
||||
attempt direct QUIC hole-punching for subsequent traffic.
|
||||
|
||||
---
|
||||
|
||||
## 8. Design Decisions & Tradeoffs
|
||||
|
||||
### 8.1 Separate Binary vs. Library Module
|
||||
|
||||
**Choice**: `ci-relay` is a standalone binary, not a module in `swactor-ci`.
|
||||
|
||||
**Why**: The relay runs on the VPS, which doesn't need swactor's runtime,
|
||||
actors, or any CI execution logic. A small binary with minimal dependencies
|
||||
deploys easily. It only depends on `swactor-ci` for `parse_webhook_json` and
|
||||
the `WebhookEvent`/`EventType` types.
|
||||
|
||||
**Tradeoff**: Two binaries to build and deploy instead of one. Acceptable
|
||||
given they run on different machines.
|
||||
|
||||
### 8.2 Runner Connects to Relay (Not Vice Versa)
|
||||
|
||||
**Choice**: The runner initiates the iroh connection to the relay.
|
||||
|
||||
**Why**: The runner is behind NAT. iroh can traverse NAT for established
|
||||
connections, but the initial rendezvous requires at least one side to be
|
||||
reachable. The VPS relay has a public IP and gets a stable relay URL from iroh's
|
||||
infrastructure. The runner connects outbound, which always works regardless of
|
||||
NAT type.
|
||||
|
||||
### 8.3 Single Cached Connection (Not Connection Pool)
|
||||
|
||||
**Choice**: The relay caches exactly one runner connection in
|
||||
`Arc<TokioMutex<Option<Connection>>>`.
|
||||
|
||||
**Why**: There's one runner. If a new connection arrives (e.g., runner
|
||||
restarts), it replaces the old one. No pool management needed.
|
||||
|
||||
**Tradeoff**: If multiple runners were needed, this would need a map. For
|
||||
single-runner use, the simplicity is worth it.
|
||||
|
||||
### 8.4 Own Tokio Runtime Per Thread
|
||||
|
||||
**Choice**: The iroh-receiver thread creates its own single-threaded tokio
|
||||
runtime rather than sharing the swactor runtime or the main thread's runtime.
|
||||
|
||||
**Why**: swactor's runtime is not tokio — it's a custom actor scheduler. The
|
||||
iroh receiver needs async for QUIC operations. A dedicated single-threaded
|
||||
runtime keeps the iroh I/O isolated from actor scheduling. Same pattern as
|
||||
`IrohDriver` in the distribution layer (which owns a multi-thread runtime).
|
||||
|
||||
### 8.5 HTTP 502 When No Runner Connected
|
||||
|
||||
**Choice**: If Forgejo sends a webhook but no runner is connected, the relay
|
||||
returns HTTP 502 (Bad Gateway).
|
||||
|
||||
**Why**: 502 tells Forgejo the upstream is unavailable. Forgejo will retry
|
||||
the webhook according to its retry policy. This is better than 200 (silently
|
||||
dropping) or 500 (suggesting a relay bug). When the runner reconnects, the
|
||||
next webhook will succeed.
|
||||
|
||||
---
|
||||
|
||||
## 9. Manual Testing Guide
|
||||
|
||||
### Prerequisites
|
||||
|
||||
Build both binaries:
|
||||
|
||||
```bash
|
||||
cargo build -p ci-relay -p local-runner
|
||||
```
|
||||
|
||||
### 9.1 Local Smoke Test (Single Machine)
|
||||
|
||||
This tests the full relay path without needing two machines or Forgejo.
|
||||
|
||||
**Terminal 1 — Start the relay:**
|
||||
|
||||
```bash
|
||||
./target/debug/ci-relay --port 9787
|
||||
```
|
||||
|
||||
Output:
|
||||
```
|
||||
ci-relay started
|
||||
Iroh Node ID: <NODE_ID_HEX>
|
||||
Webhook HTTP: http://0.0.0.0:9787
|
||||
|
||||
Waiting for runner to connect...
|
||||
Listening for webhooks...
|
||||
```
|
||||
|
||||
Copy the Node ID.
|
||||
|
||||
**Terminal 2 — Start the runner:**
|
||||
|
||||
You need a `.ci.yml` file. Create a minimal one:
|
||||
|
||||
```yaml
|
||||
# /tmp/test-ci.yml
|
||||
pipelines:
|
||||
test:
|
||||
triggers:
|
||||
- event: push
|
||||
branches: ["*"]
|
||||
jobs:
|
||||
hello:
|
||||
run: echo "hello from CI"
|
||||
```
|
||||
|
||||
Then start:
|
||||
|
||||
```bash
|
||||
./target/debug/local-runner \
|
||||
--relay-node-id <NODE_ID_HEX> \
|
||||
--yaml /tmp/test-ci.yml \
|
||||
--work-dir /tmp/ci-work-test
|
||||
```
|
||||
|
||||
You should see:
|
||||
```
|
||||
Iroh local ID: <RUNNER_ID>
|
||||
Connecting to relay <NODE_ID>...
|
||||
Connected to relay!
|
||||
Local CI runner started
|
||||
Webhook: via iroh relay
|
||||
```
|
||||
|
||||
And in Terminal 1:
|
||||
```
|
||||
Runner connected: <RUNNER_ID>
|
||||
```
|
||||
|
||||
**Terminal 3 — Send a fake webhook:**
|
||||
|
||||
```bash
|
||||
curl -X POST http://localhost:9787 \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "X-Forgejo-Event: push" \
|
||||
-d '{
|
||||
"ref": "refs/heads/main",
|
||||
"after": "abc123def456789012345678901234567890abcd",
|
||||
"repository": {
|
||||
"name": "test-repo",
|
||||
"owner": { "login": "testuser" }
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
Expected output:
|
||||
|
||||
- **curl** returns: `ok`
|
||||
- **Terminal 1** (relay):
|
||||
```
|
||||
webhook: abc123de main on testuser/test-repo
|
||||
→ forwarded to runner
|
||||
```
|
||||
- **Terminal 2** (runner):
|
||||
```
|
||||
iroh: received webhook abc123de on main
|
||||
```
|
||||
|
||||
The runner will also try to post status to Forgejo and log URL errors (since
|
||||
we didn't pass `--forgejo-url`) — that's expected and confirms the event
|
||||
reached the coordinator.
|
||||
|
||||
### 9.2 HMAC Verification Test
|
||||
|
||||
Start the relay with a secret:
|
||||
|
||||
```bash
|
||||
./target/debug/ci-relay --port 9787 --secret mysecret
|
||||
```
|
||||
|
||||
**Without signature — should be rejected (401):**
|
||||
|
||||
```bash
|
||||
curl -v -X POST http://localhost:9787 \
|
||||
-H "X-Forgejo-Event: push" \
|
||||
-d '{"ref":"refs/heads/main","after":"abc123","repository":{"name":"r","owner":{"login":"u"}}}'
|
||||
```
|
||||
|
||||
**With correct signature:**
|
||||
|
||||
```bash
|
||||
# Compute HMAC-SHA256
|
||||
BODY='{"ref":"refs/heads/main","after":"abc123","repository":{"name":"r","owner":{"login":"u"}}}'
|
||||
SIG=$(echo -n "$BODY" | openssl dgst -sha256 -hmac "mysecret" | awk '{print $2}')
|
||||
|
||||
curl -X POST http://localhost:9787 \
|
||||
-H "X-Forgejo-Event: push" \
|
||||
-H "X-Forgejo-Signature: $SIG" \
|
||||
-d "$BODY"
|
||||
```
|
||||
|
||||
Should return `ok` and forward to the runner.
|
||||
|
||||
### 9.3 Runner Reconnection Test
|
||||
|
||||
1. Start relay and runner as in 9.1
|
||||
2. Kill the runner (Ctrl-C in Terminal 2)
|
||||
3. Restart the runner with the same `--relay-node-id`
|
||||
4. The relay should log `Runner connected: <ID>` again
|
||||
5. Send another webhook — it should flow through
|
||||
|
||||
### 9.4 No Runner Connected Test
|
||||
|
||||
1. Start the relay only (no runner)
|
||||
2. Send a webhook via curl
|
||||
3. Should get HTTP 502 and relay logs: `forward failed: no runner connected`
|
||||
|
||||
### 9.5 Full End-to-End with Forgejo
|
||||
|
||||
For a real deployment:
|
||||
|
||||
**On VPS:**
|
||||
|
||||
```bash
|
||||
./ci-relay --port 8787 --secret <your-webhook-secret>
|
||||
```
|
||||
|
||||
**On Thinkpad:**
|
||||
|
||||
```bash
|
||||
./local-runner \
|
||||
--relay-node-id <NODE_ID_FROM_VPS> \
|
||||
--forgejo-url https://zachery.lol \
|
||||
--forgejo-token <your-forgejo-api-token> \
|
||||
--yaml .ci.yml \
|
||||
--work-dir ~/ci-work \
|
||||
--repo-url https://zachery.lol/<owner>/<repo>.git
|
||||
```
|
||||
|
||||
**In Forgejo (repo settings → Webhooks):**
|
||||
|
||||
- Target URL: `http://localhost:8787`
|
||||
- Secret: `<your-webhook-secret>`
|
||||
- Events: Push, Create (tags), Pull Request
|
||||
|
||||
Push a commit and watch:
|
||||
1. Relay logs the webhook and forwards it
|
||||
2. Runner logs the received event and starts a pipeline
|
||||
3. Forgejo shows commit status checks (pending → success/failure)
|
||||
|
||||
### 9.6 Inspecting Iroh Connectivity
|
||||
|
||||
Both binaries print their iroh Node ID on startup. To verify they're using
|
||||
relay servers (expected when both are behind NAT or on different networks),
|
||||
look for connection timing:
|
||||
|
||||
- **Fast connection (~1-3s)**: direct QUIC hole-punch succeeded
|
||||
- **Slower connection (~5-10s)**: using iroh relay server fallback
|
||||
|
||||
If connection hangs indefinitely, check that both machines have internet
|
||||
access and can reach iroh's relay servers (`https://relay.iroh.network`).
|
||||
|
||||
---
|
||||
|
||||
## 10. Known Gaps & Future Improvements
|
||||
|
||||
| Gap | Effort | Impact | Notes |
|
||||
|-----|--------|--------|-------|
|
||||
| Reconnection on runner side | Small | High | If the iroh connection drops mid-operation, the runner currently exits the receive loop. Should retry with backoff. |
|
||||
| Multiple runner support | Medium | Medium | Relay caches one connection. For running CI on multiple machines, need a connection map keyed by runner identity. |
|
||||
| Health check / heartbeat | Small | Medium | Neither side detects a silently dead connection until the next webhook. A periodic ping would surface stale connections faster. |
|
||||
| Relay authentication | Small | Medium | Any iroh endpoint can connect to the relay. Should verify the runner's public key against an allowlist. |
|
||||
| Binary size | Small | Low | ci-relay pulls in `swactor-ci` (which includes all CI types). A slimmer dependency with just `WebhookEvent` + `parse_webhook_json` would reduce the VPS binary. |
|
||||
| Logging | Small | Low | Both binaries use `eprintln!`. Structured logging (tracing) would help in production. |
|
||||
|
||||
---
|
||||
|
||||
## Files Created/Modified
|
||||
|
||||
| Action | File | Purpose |
|
||||
|--------|------|---------|
|
||||
| Created | `crates/ci-relay/Cargo.toml` | Relay binary manifest |
|
||||
| Created | `crates/ci-relay/src/main.rs` | Webhook relay: HTTP → iroh |
|
||||
| Modified | `crates/local-runner/Cargo.toml` | Added iroh, tokio, serde_json deps |
|
||||
| Modified | `crates/local-runner/src/main.rs` | Added `--relay-node-id` flag and iroh receiver |
|
||||
| Modified | `Cargo.toml` (workspace root) | Added ci-relay to workspace members |
|
||||
Loading…
Reference in a new issue