Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions .github/dependabot.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
version: 2
updates:
- package-ecosystem: "github-actions"
directory: "/"
schedule:
interval: "weekly"
open-pull-requests-limit: 5
commit-message:
prefix: "ci"
3 changes: 3 additions & 0 deletions .github/requirements-ci.in
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
pytest>=8
ruff>=0.6
vlabs-sdk==0.0.2
76 changes: 76 additions & 0 deletions .github/requirements-ci.lock
Original file line number Diff line number Diff line change
@@ -0,0 +1,76 @@
#
# This file is autogenerated by pip-compile with Python 3.12
# by the following command:
#
# pip-compile --generate-hashes --output-file=vlabs-examples/.github/requirements-ci.lock --strip-extras vlabs-examples/.github/requirements-ci.in
#
annotated-doc==0.0.4 \
--hash=sha256:571ac1dc6991c450b25a9c2d84a3705e2ae7a53467b5d111c24fa8baabbed320 \
--hash=sha256:fbcda96e87e9c92ad167c2e53839e57503ecfda18804ea28102353485033faa4
# via typer
iniconfig==2.3.0 \
--hash=sha256:c76315c77db068650d49c5b56314774a7804df16fee4402c1f19d6d15d8c4730 \
--hash=sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12
# via pytest
markdown-it-py==4.2.0 \
--hash=sha256:04a21681d6fbb623de53f6f364d352309d4094dd4194040a10fd51833e418d49 \
--hash=sha256:9f7ebbcd14fe59494226453aed97c1070d83f8d24b6fc3a3bcf9a38092641c4a
# via rich
mdurl==0.1.2 \
--hash=sha256:84008a41e51615a49fc9966191ff91509e3c40b939176e643fd50a5c2196b8f8 \
--hash=sha256:bb413d29f5eea38f31dd4754dd7377d4465116fb207585f97bf925588687c1ba
# via markdown-it-py
packaging==26.2 \
--hash=sha256:5fc45236b9446107ff2415ce77c807cee2862cb6fac22b8a73826d0693b0980e \
--hash=sha256:ff452ff5a3e828ce110190feff1178bb1f2ea2281fa2075aadb987c2fb221661
# via pytest
pluggy==1.6.0 \
--hash=sha256:7dcc130b76258d33b90f61b658791dede3486c3e6bfb003ee5c9bfb396dd22f3 \
--hash=sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746
# via pytest
pygments==2.20.0 \
--hash=sha256:6757cd03768053ff99f3039c1a36d6c0aa0b263438fcab17520b30a303a82b5f \
--hash=sha256:81a9e26dd42fd28a23a2d169d86d7ac03b46e2f8b59ed4698fb4785f946d0176
# via
# pytest
# rich
pytest==9.1.1 \
--hash=sha256:1088fbde8f2b49d95a549a195707afa7a76a3ce9bcadc26b6d71f0ffda5fe313 \
--hash=sha256:37a86b45efb9a47a61a36449063e8e18d0cab3161329fc099eb21783169c4f0c
# via -r vlabs-examples/.github/requirements-ci.in
rich==15.0.0 \
--hash=sha256:33bd4ef74232fb73fe9279a257718407f169c09b78a87ad3d296f548e27de0bb \
--hash=sha256:edd07a4824c6b40189fb7ac9bc4c52536e9780fbbfbddf6f1e2502c31b068c36
# via typer
ruff==0.15.21 \
--hash=sha256:00eca240af5789fec6fe7df74c088cc1f9644ed83027113468efba7c92b94075 \
--hash=sha256:01d65b4831c6b2a4ba8ee6faa84049d44d982b7a706e622c4094c509e51673be \
--hash=sha256:01f8d5be84823c172b389e123174f781f9daf86d6c58719d603f941932195cdd \
--hash=sha256:0f212c5d7d54c01bbfe6dcab02b724a39300f3e34ed7acbe995ccb320a2c58bd \
--hash=sha256:16d090c0740916594157e75b80d666eab8e78083b39b3b0e1d698f4670a17b86 \
--hash=sha256:262ab31557a75141325e32d3357f3597645a7f084e732b6b054dde428ecd9341 \
--hash=sha256:2c5a913a589120ce67933d5d05fd6ddbcc2481c6a054980ee767f7414c72b4fd \
--hash=sha256:3a10e74757dd65004d779b73e2f3c5210156d9980b41224d50d2ebcf1db51e67 \
--hash=sha256:5ef04b681d02ad4dc9620f00f83ac5c22f652d0e9a9cfe431d219b16ad5ccc41 \
--hash=sha256:63ea0e965e5d73c90e95b2434beeafc70820536717f561b32ab6e777cb9bdf5d \
--hash=sha256:659c4e7a4212f83306045ec7c5e5a356d16d9a6ef4ae0c7a4d872914fc655d9d \
--hash=sha256:6e83115d4b9377c1cbc13abf0e051f069fab0ef815ea0504a8a008cee24dd0a8 \
--hash=sha256:9e866eab611a5f959d36df2d10e446973a3610bc42b0c15b31dc27977d59c233 \
--hash=sha256:bab0905d2f29e0d9fbc3c373ed23db0095edaa3f71f1f4f519ec15134d9e85c8 \
--hash=sha256:d0cfc841c572283c36548f82664a54ce6565567f1b0d5b4cf2caac693d8b7500 \
--hash=sha256:d4b8d9a2f0f12b816b50447f6eccb9f4bb01a6b82c86b50fb3b5354b458dc6d3 \
--hash=sha256:e6312e41bc96791299614995ea3a977c5857c3b5662b1ecef6755b02b87cb646 \
--hash=sha256:e89bc93c0d3803ba870b55c29671bad9dc6d94bb1eb181b056b52eb05b52854f
# via -r vlabs-examples/.github/requirements-ci.in
shellingham==1.5.4 \
--hash=sha256:7ecfff8f2fd72616f7481040475a65b2bf8af90a56c89140852d1120324e8686 \
--hash=sha256:8dbca0739d487e5bd35ab3ca4b36e11c4078f3a234bfce294b0a0291363404de
# via typer
typer==0.26.8 \
--hash=sha256:3512ca79ac5c11113414b36e80281b872884477722440691c89d1112e321a49c \
--hash=sha256:c244a6bd558886fe3f8780efb6bdd28bb9aff005a94eedebaa5cb32926fe2f7e
# via vlabs-sdk
vlabs-sdk==0.0.2 \
--hash=sha256:b6d5836ac5319c24f1141db07d6cf5b8ad93fb5c4181e088149da5ac7241a516 \
--hash=sha256:f4e839591021b189b0442a1830aeab0c821d61f5f5bdd9e222e21868ae73bb2c
# via -r vlabs-examples/.github/requirements-ci.in
15 changes: 9 additions & 6 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -2,16 +2,19 @@ name: CI
on:
push: { branches: [main] }
pull_request: { branches: [main] }
permissions:
contents: read
jobs:
run-examples:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1
with: { persist-credentials: false }
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with: { python-version: "3.12" }
- name: install public SDK surface (dummy provider only)
run: |
pip install "vlabs-sdk @ git+https://github.com/verifiablelabs/vlabs-sdk@main"
pip install typer pytest
pip install --no-deps "vlabs-prm-eval @ git+https://github.com/verifiablelabs/vlabs-sdk@main#subdirectory=tools/vlabs-prm-eval"
- run: pytest tests -q
python -m pip install --disable-pip-version-check --no-deps pip==26.1.2
python -m pip install --require-hashes -r .github/requirements-ci.lock
- run: python -m ruff check examples tests
- run: python -m pytest tests -q
10 changes: 5 additions & 5 deletions PROVENANCE.md
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
# Provenance

Clean import (no history rewrite) from `verifiablelabs/verifiable-labs-envs`
at commit `762b44e8019af3e89c55bba0f88e9157bb50c5c3` (main). All example code authored fresh for this repo; depends on the public SDK surface only. Everything synthetic.
Clean import (no history rewrite) from the archived legacy workspace at commit
`762b44e8019af3e89c55bba0f88e9157bb50c5c3`. All example code was authored
fresh for this repository, depends only on the public SDK surface, and uses
synthetic inputs.

The source monorepo remains canonical until the split flips; this mirror is
refreshed by the migration tooling documented in
`verifiable-labs-private/docs/ops/github-repo-split-migration.md`.
`vlabs-examples` is now canonical for these public examples.
4 changes: 2 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -56,8 +56,8 @@ private engine internals; those are never published.
## Formal scope

Selected mathematical properties behind the contamination-resistant promotion
gate are machine-verified in Lean 4. The implementation is property-tested
against the formal specification.
gate are machine-verified in Lean 4. A hand-maintained Python mirror has property
tests derived from selected definitions; no mechanized code-to-proof parity is claimed.

## License

Expand Down
2 changes: 1 addition & 1 deletion examples/cards/clean_new_accept.json
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
"model_id": "qwen-2.5-1.5b-grpo-clean-accept",
"vgs": 0.78,
"contamination_risk": 0.08,
"clean_vgs": 0.62,
"clean_vgs": 0.6776,
"public_score": 0.85,
"hidden_score": 0.78,
"ood_score": 0.72,
Expand Down
2 changes: 1 addition & 1 deletion examples/cards/clean_old.json
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
"model_id": "qwen-2.5-1.5b-base-clean",
"vgs": 0.70,
"contamination_risk": 0.10,
"clean_vgs": 0.50,
"clean_vgs": 0.58,
"public_score": 0.80,
"hidden_score": 0.70,
"ood_score": 0.70,
Expand Down
4 changes: 2 additions & 2 deletions examples/cards/clean_reject_dcr.json
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
{
"model_id": "qwen-2.5-1.5b-grpo-clean-reject-dcr",
"vgs": 0.78,
"contamination_risk": 0.5,
"clean_vgs": 0.62,
"contamination_risk": 0.13,
"clean_vgs": 0.6136,
"public_score": 0.85,
"hidden_score": 0.78,
"ood_score": 0.72,
Expand Down
18 changes: 11 additions & 7 deletions examples/demo/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -31,12 +31,13 @@ but is weaker on **hidden / OOD** transfer (`hidden_score` 0.68,
`ood_score` 0.66). Two candidates then ask to be promoted:

- **`candidate.json`** improves clean verified-generalization score
(`clean_vgs` 0.50 → 0.63) with no regression in contamination risk, hack
(`clean_vgs` 0.58 → 0.6868) with no regression in contamination risk, hack
risk, calibration, OOD, cost, or latency → **ACCEPT**.
- **`candidate_overfit.json`** has the **highest public score** of all (0.92)
— but it got there by memorising the visible set: contamination risk jumps
(0.10 → 0.34) and OOD transfer drops (0.66 → 0.62). The gate **REJECT**s it
and names exactly why: `ood_regressed`, `dcr_increased`.
and names exactly why: `clean_vgs_not_improved`, `ood_regressed`, and
`dcr_increased`.

That contrast is the whole point: **a higher public score is not a promotion.**
The clean gate only accepts a change that *truly generalizes*.
Expand All @@ -48,7 +49,7 @@ The clean gate only accepts a change that *truly generalizes*.

condition old new budget OK
-------------------------------- ---------- ---------- ---------- --
clean_vgs >= +tau 0.5000 0.5500 0.0100 OK
clean_vgs >= +tau 0.5800 0.3448 0.0100 !!
hack_risk <= +eps_h 0.1000 0.1100 0.0200 OK
calibration >= -eps_c 0.9000 0.9000 0.0200 OK
ood_score >= -eps_o 0.6600 0.6200 0.0200 !!
Expand All @@ -58,6 +59,7 @@ latency <= +eps_l 1.0000 1.0000 0.5000 OK
regression flag False False False OK

Reasons:
- clean_vgs_not_improved
- ood_regressed
- dcr_increased
```
Expand All @@ -84,8 +86,10 @@ a partial promotion when a change is a net improvement but carries a watch-item
[`sample_assurance_card_redacted.json`](sample_assurance_card_redacted.json),
which records a `LIMITED_ROLLOUT` decision with reason `ood_regressed`.

`clean_score = raw * (1 - dcr)` — contamination directly discounts the score,
which is why a memorised public win cannot buy a promotion.
`clean_vgs = raw_vgs * (1 - dcr) - beta * dcr` — contamination directly
discounts and penalizes the score, which is why a memorised public win cannot
buy a promotion. The CLI recomputes this value instead of trusting the derived
field supplied by a card.

## What this does NOT show

Expand All @@ -109,5 +113,5 @@ which is why a memorised public win cannot buy a promotion.
## Formal scope

Selected mathematical properties behind the contamination-resistant promotion
gate are machine-verified in Lean 4. The implementation is property-tested
against the formal specification.
gate are machine-verified in Lean 4. A hand-maintained Python mirror has property
tests derived from selected definitions; no mechanized code-to-proof parity is claimed.
2 changes: 1 addition & 1 deletion examples/demo/baseline.json
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
"model_id": "refund-agent-baseline",
"vgs": 0.70,
"contamination_risk": 0.10,
"clean_vgs": 0.50,
"clean_vgs": 0.58,
"public_score": 0.80,
"hidden_score": 0.68,
"ood_score": 0.66,
Expand Down
2 changes: 1 addition & 1 deletion examples/demo/candidate.json
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
"model_id": "refund-agent-candidate",
"vgs": 0.79,
"contamination_risk": 0.08,
"clean_vgs": 0.63,
"clean_vgs": 0.6868,
"public_score": 0.86,
"hidden_score": 0.79,
"ood_score": 0.73,
Expand Down
2 changes: 1 addition & 1 deletion examples/demo/candidate_overfit.json
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
"model_id": "refund-agent-candidate-overfit",
"vgs": 0.78,
"contamination_risk": 0.34,
"clean_vgs": 0.55,
"clean_vgs": 0.3448,
"public_score": 0.92,
"hidden_score": 0.69,
"ood_score": 0.62,
Expand Down
5 changes: 3 additions & 2 deletions examples/demo/expected_output.txt
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@

condition old new budget OK
-------------------------------- ---------- ---------- ---------- --
clean_vgs >= +tau 0.5000 0.6300 0.0100 OK
clean_vgs >= +tau 0.5800 0.6868 0.0100 OK
hack_risk <= +eps_h 0.1000 0.0800 0.0200 OK
calibration >= -eps_c 0.9000 0.9200 0.0200 OK
ood_score >= -eps_o 0.6600 0.7300 0.0200 OK
Expand All @@ -21,7 +21,7 @@ regression flag False False False OK

condition old new budget OK
-------------------------------- ---------- ---------- ---------- --
clean_vgs >= +tau 0.5000 0.5500 0.0100 OK
clean_vgs >= +tau 0.5800 0.3448 0.0100 !!
hack_risk <= +eps_h 0.1000 0.1100 0.0200 OK
calibration >= -eps_c 0.9000 0.9000 0.0200 OK
ood_score >= -eps_o 0.6600 0.6200 0.0200 !!
Expand All @@ -31,5 +31,6 @@ latency <= +eps_l 1.0000 1.0000 0.5000 OK
regression flag False False False OK

Reasons:
- clean_vgs_not_improved
- ood_regressed
- dcr_increased
30 changes: 22 additions & 8 deletions examples/demo/sample_assurance_card_redacted.json
Original file line number Diff line number Diff line change
@@ -1,14 +1,28 @@
{
"_comment": "ILLUSTRATIVE EXAMPLE — synthetic numbers, fake IDs, fields redacted as they would be for a real customer. Not a real evaluation.",
"card_version": "v2",
"run_id": "run_redacted_xxxx",
"org": "REDACTED",
"agent": "REDACTED",
"scores": { "public": 0.77, "hidden": 0.70, "ood": 0.66, "adversarial": 0.58 },
"contamination": { "dcr": 0.05 },
"clean_vgs": { "baseline": 0.55, "candidate": 0.59 },
"agent_id": "REDACTED",
"baseline_id": "REDACTED",
"candidate_id": "REDACTED",
"decision": "LIMITED_ROLLOUT",
"raw_vgs": 0.65,
"dcr": 0.05,
"clean_vgs": 0.5925,
"public_score": 0.77,
"hidden_score": 0.70,
"ood_score": 0.66,
"generalization_gap": 0.07,
"gate": { "outcome": "LIMITED_ROLLOUT", "reasons": ["ood_regressed"] },
"reject_reasons": ["ood_regressed"],
"redaction_status": "redacted_public_safe",
"formal_claim": "Selected mathematical properties behind the contamination-resistant promotion gate are machine-verified in Lean 4. The implementation is property-tested against the formal specification."
"hf_public_safe": true,
"formal_claim": "Selected mathematical properties behind the contamination-resistant promotion gate are machine-verified in Lean 4. A hand-maintained Python mirror has property tests derived from selected definitions; no mechanized code-to-proof parity is claimed.",
"formal_scope": "Selected mathematical properties behind the contamination-resistant promotion gate are machine-verified in Lean 4. A hand-maintained Python mirror has property tests derived from selected definitions; no mechanized code-to-proof parity is claimed.",
"metadata": {
"illustrative": true,
"comment": "Synthetic numbers, fake IDs, and redacted fields; not a real evaluation.",
"org": "REDACTED",
"baseline_clean_vgs": 0.55,
"adversarial_score": 0.58,
"clean_vgs_beta": 0.5
}
}
Loading
Loading