mirror of
https://github.com/dotnet/skills.git
synced 2026-09-20 09:49:54 +08:00
36222bf32d
* feat(evaluation): add custom agent coverage Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): address agent review feedback Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): reject linked fixture sources Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): preserve agent result invariants Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): fail closed on agent errors Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): preserve completion regressions Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): preserve nested command quotes Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): harden native agent evidence Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): honor declared agent layout Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): resolve declared agent sources Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): secure agent path discovery Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): reject linked dependencies Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): centralize path safety checks Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): diagnose ambiguous dependencies Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): reject linked allowed roots Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): preserve skill agent isolation Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): normalize dashboard evidence Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): preserve agent gate semantics Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): fail closed on incomplete evidence Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): preserve completion evidence Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): reject overflowing durations Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): stage verified plugin skills Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): block shell network access Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): reject linked MCP config files Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): trust manual dispatch path safety Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): keep agent plugin activation diagnostic Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): count failed tool completions Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> * fix(evaluation): synchronize agent event capture Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --------- Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com>
101 lines
3.0 KiB
YAML
101 lines
3.0 KiB
YAML
name: evaluation-workflow-tests
|
|
|
|
on:
|
|
pull_request:
|
|
paths:
|
|
- ".github/workflows/evaluation.yml"
|
|
- ".github/workflows/evaluation-run.yml"
|
|
- ".github/workflows/evaluation-workflow-tests.yml"
|
|
- "eng/evaluation-tools/**"
|
|
- "eng/evaluation/test_token_failover.py"
|
|
- "eng/evaluation/path-safety.ps1"
|
|
- "eng/dashboard/**"
|
|
- "eng/vally-adapter/**"
|
|
- "eng/skill-validator/src/**"
|
|
push:
|
|
branches: [main]
|
|
paths:
|
|
- ".github/workflows/evaluation.yml"
|
|
- ".github/workflows/evaluation-run.yml"
|
|
- ".github/workflows/evaluation-workflow-tests.yml"
|
|
- "eng/evaluation-tools/**"
|
|
- "eng/evaluation/test_token_failover.py"
|
|
- "eng/evaluation/path-safety.ps1"
|
|
- "eng/dashboard/**"
|
|
- "eng/vally-adapter/**"
|
|
- "eng/skill-validator/src/**"
|
|
workflow_dispatch:
|
|
|
|
permissions:
|
|
contents: read
|
|
|
|
jobs:
|
|
vally-adapter:
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 10
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v6
|
|
with:
|
|
persist-credentials: false
|
|
|
|
- name: Setup Node.js
|
|
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
|
with:
|
|
node-version: 22
|
|
|
|
- name: Run adapter fault-injection and report tests
|
|
run: node --test eng/vally-adapter/*.test.mjs eng/dashboard/*.test.js
|
|
|
|
evaluation-tools:
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 10
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v6
|
|
with:
|
|
persist-credentials: false
|
|
|
|
- name: Setup Node.js
|
|
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
|
with:
|
|
node-version: 22
|
|
|
|
- name: Install evaluation tools
|
|
run: |
|
|
printf '' > "$RUNNER_TEMP/evaluation.npmrc"
|
|
npm ci \
|
|
--prefix eng/evaluation-tools \
|
|
--userconfig "$RUNNER_TEMP/evaluation.npmrc" \
|
|
--registry https://registry.npmjs.org/
|
|
|
|
- name: Smoke test evaluation tools
|
|
working-directory: eng/evaluation-tools
|
|
run: |
|
|
node_modules/.bin/vally --version
|
|
node vally.mjs --version
|
|
node_modules/.bin/copilot --version
|
|
node --input-type=module -e "import.meta.resolve('@github/copilot-linux-x64/sdk')"
|
|
|
|
- name: Test SDK startup ordering without model calls
|
|
run: node --test eng/evaluation-tools/*.test.mjs
|
|
|
|
token-failover:
|
|
runs-on: ubuntu-latest
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v6
|
|
with:
|
|
persist-credentials: false
|
|
|
|
- name: Set up Python
|
|
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
|
with:
|
|
python-version: "3.12"
|
|
|
|
- name: Install PyYAML
|
|
run: python -m pip install --quiet pyyaml
|
|
|
|
- name: Test evaluation workflow behavior
|
|
run: python eng/evaluation/test_token_failover.py
|