Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/pull_request_template.md
Original file line number Diff line number Diff line change
Expand Up @@ -40,7 +40,7 @@ Maintainer live check: no | yes

Surface: N/A | provider | browser | gateway | channel | release

Maintainer-only note: contributors are not expected to provide secrets or run credentialed live checks. Maintainers may run `Live Release E2E` for provider, browser, gateway, channel, or release smoke coverage.
Maintainer-only note: contributors are not expected to provide secrets or run credentialed live checks. Maintainers may manually run `LLM Smoke`, `Live Search E2E`, or `Live Telegram Smoke` for model API connectivity, search providers, or Telegram delivery. These workflows do not provide full Agent or Gateway model E2E coverage.

## Safety

Expand Down
31 changes: 9 additions & 22 deletions .github/scripts/windows_test_assignments.json
Original file line number Diff line number Diff line change
Expand Up @@ -136,7 +136,6 @@
"tests/test_execution_status_contract.py",
"tests/test_git_runtime.py",
"tests/test_github_issue_link_sync.py",
"tests/test_long_task_fault_proxy.py",
"tests/test_lossless_toml.py",
"tests/test_memory/test_profile_import.py",
"tests/test_meta_skill_openclaw_lifestyle_comparison.py",
Expand Down Expand Up @@ -225,7 +224,6 @@
"tests/test_sandbox/test_windows_default_runner.py",
"tests/test_scripts/test_check_treatment_delivery.py",
"tests/test_scripts/test_exp_quarantine.py",
"tests/test_scripts/test_live_meta_skill_creator_e2e.py",
"tests/test_skill_pdf_toolkit.py",
"tests/test_skills/test_awesome_webpage_meta_skill.py",
"tests/test_skills/test_clarify_schema_protocol.py",
Expand Down Expand Up @@ -722,10 +720,6 @@
"tests/test_identity/test_user_profile_prompt.py",
"tests/test_identity/test_workspace_injection_scan.py",
"tests/test_inference_postprocess_rules.py",
"tests/test_live_long_task_case_driver.py",
"tests/test_live_mixed_provider_gateway.py",
"tests/test_live_multi_provider_matrix.py",
"tests/test_live_provider_profile_gateway_e2e.py",
"tests/test_mcp_server/test_bridge.py",
"tests/test_mcp_server/test_gateway_client.py",
"tests/test_memory/test_store_vec_extension_cleanup.py",
Expand Down Expand Up @@ -838,7 +832,6 @@
"tests/test_scheduler/test_webhook_delivery.py",
"tests/test_scripts/test_analyze_dashscope_payload_risk.py",
"tests/test_scripts/test_check_docker_image_lock.py",
"tests/test_scripts/test_live_meta_soft_activation_e2e.py",
"tests/test_scripts/test_meta_trigger_accuracy_script.py",
"tests/test_scripts/test_refresh_models_dev_snapshot.py",
"tests/test_scripts/test_replay_finalize_gate.py",
Expand Down Expand Up @@ -1317,10 +1310,9 @@
"tests/test_engine/turn_runner/test_stream_consumer_stage_unit.py",
"tests/test_engine/turn_runner/test_turn_finalizer_paused_im_layout.py",
"tests/test_envelope_policy_deny_cap.py",
"tests/test_gateway/test_offline_document_workbench_e2e.py",
"tests/test_gemini_thought_signature.py",
"tests/test_install_scripts.py",
"tests/test_live_artifact_prompt_annotations_e2e.py",
"tests/test_live_long_task_release_gate.py",
"tests/test_llm_ensemble_config.py",
"tests/test_mcp/test_sse_client.py",
"tests/test_mcp_server/test_cli.py",
Expand Down Expand Up @@ -1496,30 +1488,25 @@
"reason": "Reviewed affinity exception for an environment-independent test; rebalance the measured Windows critical path using three comparable successful runs."
},
{
"path": "tests/test_live_long_task_case_driver.py",
"from": "gateway-sqlite",
"to": "recovery-migration",
"reason": "Rebalance the measured Windows critical path using three comparable successful runs."
},
{
"path": "tests/test_live_multi_provider_matrix.py",
"path": "tests/test_observability/test_bundle.py",
"from": "gateway-sqlite",
"to": "recovery-migration",
"reason": "Rebalance the measured Windows critical path using three comparable successful runs."
"affinity_exception": true,
"reason": "Retain the reviewed environment-independent affinity exception and balance the remaining historical Windows weights after retiring model E2E harnesses."
},
{
"path": "tests/test_observability/test_bundle.py",
"path": "tests/test_persistence/test_router_decision_writer.py",
"from": "gateway-sqlite",
"to": "desktop-installer-contracts",
"to": "core",
"affinity_exception": true,
"reason": "Reviewed affinity exception for an environment-independent test; rebalance the measured Windows critical path using three comparable successful runs."
},
{
"path": "tests/test_persistence/test_router_decision_writer.py",
"path": "tests/test_persistence/test_turn_error_writer.py",
"from": "gateway-sqlite",
"to": "core",
"to": "recovery-migration",
"affinity_exception": true,
"reason": "Reviewed affinity exception for an environment-independent test; rebalance the measured Windows critical path using three comparable successful runs."
"reason": "Balance the remaining historical Windows weights after retiring model E2E harnesses; these synthetic SQLite tests need no shard-specific setup."
},
{
"path": "tests/test_sandbox/test_windows_default_capability.py",
Expand Down
10 changes: 1 addition & 9 deletions .github/scripts/windows_test_durations.json
Original file line number Diff line number Diff line change
Expand Up @@ -791,16 +791,10 @@
"tests/test_identity/test_workspace_injection_scan.py": 0.067,
"tests/test_inference_postprocess_rules.py": 0.076,
"tests/test_install_scripts.py": 6.582,
"tests/test_live_artifact_prompt_annotations_e2e.py": 0.01,
"tests/test_live_long_task_case_driver.py": 54.424,
"tests/test_live_long_task_release_gate.py": 0.23,
"tests/test_live_mixed_provider_gateway.py": 0.2,
"tests/test_live_multi_provider_matrix.py": 13.55,
"tests/test_live_provider_profile_gateway_e2e.py": 0.131,
"tests/test_gateway/test_offline_document_workbench_e2e.py": 0.01,
"tests/test_live_provider_profile_smoke.py": 0.065,
"tests/test_live_tokenrhythm_billing_audit.py": 0.146,
"tests/test_llm_ensemble_config.py": 1.845,
"tests/test_long_task_fault_proxy.py": 4.571,
"tests/test_lossless_toml.py": 0.094,
"tests/test_mcp/test_discovery_lifecycle.py": 0.203,
"tests/test_mcp/test_sse_client.py": 0.832,
Expand Down Expand Up @@ -1177,8 +1171,6 @@
"tests/test_scripts/test_check_treatment_delivery.py": 0.073,
"tests/test_scripts/test_exp_ledger.py": 0.01,
"tests/test_scripts/test_exp_quarantine.py": 0.01,
"tests/test_scripts/test_live_meta_skill_creator_e2e.py": 5.755,
"tests/test_scripts/test_live_meta_soft_activation_e2e.py": 1.914,
"tests/test_scripts/test_meta_skill_validation_matrix.py": 0.03,
"tests/test_scripts/test_meta_trigger_accuracy_script.py": 0.016,
"tests/test_scripts/test_prestage_release_to_oss.py": 12.831,
Expand Down
23 changes: 3 additions & 20 deletions .github/workflows/live-release-e2e.yml
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
name: Live Release E2E
name: Live Telegram Smoke

"on":
workflow_dispatch:
Expand All @@ -18,16 +18,14 @@ concurrency:

jobs:
live-release-e2e:
name: Maintainer live release gates
name: Maintainer Telegram channel smoke
if: ${{ inputs.run_telegram_channel }}
runs-on: ubuntu-latest
timeout-minutes: 30

env:
PYTHONPATH: ${{ github.workspace }}
OPENSQUILLA_TURN_CALL_LOG: "0"
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
OPENROUTER_BASE_URL: ${{ vars.OPENROUTER_BASE_URL || 'https://openrouter.ai/api/v1' }}
LLM_TEST_MODEL: ${{ vars.LLM_TEST_MODEL || 'openai/gpt-4o-mini' }}
OPENSQUILLA_LIVE_TELEGRAM_BOT_TOKEN: ${{ secrets.OPENSQUILLA_LIVE_TELEGRAM_BOT_TOKEN }}
OPENSQUILLA_LIVE_TELEGRAM_CHAT_ID: ${{ secrets.OPENSQUILLA_LIVE_TELEGRAM_CHAT_ID }}

Expand All @@ -52,16 +50,7 @@ jobs:
- name: Install dependencies
run: uv sync --extra dev --extra recommended --frozen

- name: Fail if OpenRouter secret is missing
shell: bash
run: |
if [ -z "$OPENROUTER_API_KEY" ]; then
echo "OPENROUTER_API_KEY GitHub secret is required"
exit 1
fi

- name: Fail if Telegram secrets are missing when channel smoke is enabled
if: ${{ inputs.run_telegram_channel }}
shell: bash
run: |
if [ -z "$OPENSQUILLA_LIVE_TELEGRAM_BOT_TOKEN" ]; then
Expand All @@ -73,13 +62,7 @@ jobs:
exit 1
fi

- name: Run gateway LLM e2e
env:
OPENSQUILLA_GATEWAY_LLM_E2E: "1"
run: uv run pytest tests/functional/test_gateway_llm_e2e.py -q -s

- name: Run live Telegram channel smoke
if: ${{ inputs.run_telegram_channel }}
env:
OPENSQUILLA_LIVE_TELEGRAM_E2E: "1"
run: uv run pytest tests/functional/test_live_channel_telegram_smoke.py -q -s
10 changes: 1 addition & 9 deletions .github/workflows/live-search-e2e.yml
Original file line number Diff line number Diff line change
Expand Up @@ -18,17 +18,14 @@ concurrency:

jobs:
live-search-e2e:
name: Live search provider and agent gates
name: Live search provider gates
runs-on: ubuntu-latest
timeout-minutes: 30

env:
PYTHONPATH: ${{ github.workspace }}
OPENSQUILLA_TURN_CALL_LOG: "0"
OPENSQUILLA_LIVE_SEARCH: "1"
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
OPENROUTER_BASE_URL: ${{ vars.OPENROUTER_BASE_URL || 'https://openrouter.ai/api/v1' }}
LLM_TEST_MODEL: ${{ vars.LLM_TEST_MODEL || 'openai/gpt-4o-mini' }}
TAVILY_API_KEY: ${{ secrets.TAVILY_API_KEY }}
BRAVE_SEARCH_API_KEY: ${{ secrets.BRAVE_SEARCH_API_KEY }}
EXA_API_KEY: ${{ secrets.EXA_API_KEY }}
Expand Down Expand Up @@ -58,10 +55,6 @@ jobs:
- name: Fail if required live search secrets are missing
shell: bash
run: |
if [ -z "$OPENROUTER_API_KEY" ]; then
echo "OPENROUTER_API_KEY GitHub secret is required"
exit 1
fi
if [ -z "$TAVILY_API_KEY" ]; then
echo "TAVILY_API_KEY GitHub secret is required"
exit 1
Expand All @@ -86,7 +79,6 @@ jobs:
run: |
timeout 180s uv run pytest \
tests/live/test_search_retrieval_live.py \
tests/live/test_web_search_agent_e2e.py \
-m live_search \
-q -s

Expand Down
6 changes: 5 additions & 1 deletion CONTRIBUTING.md
Original file line number Diff line number Diff line change
Expand Up @@ -85,7 +85,11 @@ Default tests must be offline, deterministic, credential-free, and safe for fork

Add or update public regression tests for behavior changes and bug fixes. Prefer focused unit or integration tests unless the behavior crosses the gateway, browser UI, provider, or channel boundary.

Live checks are maintainer-only gates. The `Live Release E2E` workflow covers real provider, browser, and optional channel smoke tests with GitHub secrets and explicit opt-in inputs.
Live checks are maintainer-only gates. The manually triggered `LLM Smoke`,
`Live Search E2E`, and `Live Telegram Smoke` workflows cover model API
connectivity, search providers, and Telegram delivery respectively. They use
GitHub secrets and explicit opt-in inputs; they do not provide full Agent or
Gateway model E2E coverage.

## Private Materials

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -50,7 +50,7 @@ await run(
'run',
'pytest',
'-q',
'tests/test_live_artifact_prompt_annotations_e2e.py::test_owned_gateway_html_workbench_lifecycle_is_offline_and_immutable',
'tests/test_gateway/test_offline_document_workbench_e2e.py::test_owned_gateway_html_workbench_lifecycle_is_offline_and_immutable',
],
)

Expand Down
5 changes: 1 addition & 4 deletions docs/authoring/meta-skills.md
Original file line number Diff line number Diff line change
Expand Up @@ -438,10 +438,7 @@ Before sharing or enabling a MetaSkill:
reflect the workflow's true side effects.
7. If supporting `meta_skill.auto_trigger = true`, run deterministic trigger
checks with `scripts/meta_trigger_accuracy.py`.
8. If supporting `meta_skill.auto_trigger = true`, run model-decision soft
activation checks with
`scripts/live_meta_soft_activation_e2e.py --env-file /path/to/.env`.
9. For generated skills, inspect the Web UI proposal detail and its auto-enable
8. For generated skills, inspect the Web UI proposal detail and its auto-enable
audit before accepting or enabling.

## Troubleshooting
Expand Down
71 changes: 2 additions & 69 deletions docs/features/prompt-annotation-editing.md
Original file line number Diff line number Diff line change
Expand Up @@ -377,82 +377,15 @@ write. A discarded or interrupted candidate creates no revision. It is
credential-free and requires Electron foreground focus;
a locked or background-only macOS session fails the gate.

### Live provider certification

Keep provider credentials only in the process environment or an ignored local
environment file. Never put a key in a command, fixture, report, or committed
configuration. A sanitized provider/Gateway prerequisite can be run with:

```sh
uv run python scripts/live_provider_profile_gateway_e2e.py \
--providers tokenrhythm \
--output "${TMPDIR:?}/opensquilla-provider-gateway.json"
```

That script certifies provider transport and accounting; it is not by itself
PromptAnnotation certification. For every release that changes this path, an
isolated owned Gateway and Desktop build must additionally pass this live matrix:

The dedicated certification boundary can be checked with:

```sh
uv run python scripts/live_artifact_prompt_annotations_e2e.py \
--output "${TMPDIR:?}/opensquilla-prompt-annotations.json" \
--confirm-live-cost \
--confirm-rotated-key \
--execute-live-matrix
```

Without `--execute-live-matrix`, the command is an intentional zero-call dry
run that writes `certification=incomplete`. With the flag, the isolated worker
runs the owned Gateway/provider path and requires the ten-tool document-agent surface.
It does not replace the separate real-Electron selection gate.
The fixed Direct `glm-5.2` source-fallback case must exercise a repair loop:
`document_inspect → document_read → document_patch →
document_browser_inspect → document_browser_screenshot` (or a bounded browser
action) `→ document_patch → document_browser_inspect → document_finish(commit)
→ tools=[]` finalization. The model chooses whether to take each observation,
repair, or discard step; the harness checks this representative sequence but
does not impose a PromptAnnotation-specific iteration count. The live harness
therefore uses a 120-second per-case deadline; this is a task deadline, not a
provider retry. A B5 Ensemble case uses the configured proposer/Aggregator
rounds while keeping proposer and finalizer tools empty. The outcome-finalization
round is required Agent work; it must not be removed or treated as a free call.
The approved autonomous-loop matrix therefore reserves 42 baseline physical
provider calls, allows a bounded worst-case reservation of 63, and hard-stops
raw provider traffic at 64. Preview, browser action, `document_finish`, and
tools-disabled finalizer requests are all included in that physical count;
none is deducted or treated as free work.
A build is not release-ready until the following matrix is
verified end to end:

- Direct: one source-fallback insertion and a two-annotation semantic batch
using fixed model `glm-5.2`;
- Router: a single selection at `c2`, a structural batch at `c3`, and no
fallback below the effective floor;
- Ensemble: the full configured lineup, zero executable proposer tool calls,
an Aggregator-owned candidate loop, and one final `document_finish(commit)`;
- rejection cases for stale head, cross-session draft, DOM mismatch, and visual
selection, each with zero provider calls;
- exactly one revision and one change set for every successful batch, plus a
successful whole-turn revert.

Store only case name, mode/tier/model, tool name and count, content hashes, and
boolean results. Scan the report and temporary directory for credentials and
delete the temporary data after review. If this feature-specific live matrix
has not been completed, set the release override for
`artifactPromptAnnotations` to `false`.

## Release, rollback, and maintenance

For each release:

1. Run the offline, packaged-wheel, Web UI, and real Electron suites against
the release candidate.
2. Complete the live Direct/Router/Ensemble matrix with an isolated profile.
3. Canary the exact Desktop build and watch sanitized audit events,
2. Canary the exact Desktop build and watch sanitized audit events,
annotation remap outcomes, validation failures, and orphan cleanup.
4. Keep the default enabled only while the one-turn/one-change-set, zero-call
3. Keep the default enabled only while the one-turn/one-change-set, zero-call
rejection, and Aggregator-only mutation invariants remain true.

For an incident, apply the explicit `artifactPromptAnnotations: false` override
Expand Down
Loading
Loading