From 8b024b4c5cb8f2b73fcf2341276cca94c3ac5175 Mon Sep 17 00:00:00 2001 From: "Andrei G." Date: Sun, 4 Oct 2026 14:42:02 +0200 Subject: [PATCH 1/2] feat(skill): port neighbor-project CI protocols into continuous-improvement, bump v1.50.0 Unchanged-HEAD cycle fallback, coverage-status format, cycle start sync, mandatory P0-P4 labels at filing, spec-and-issue same pass, symptom filing, arch audit minimums, process self-improvement loop, and optional benchmark, CVE sweep, fork-aware dedup and competitor gap checks. --- .claude-plugin/marketplace.json | 4 +- README.md | 2 +- rust-code/.claude-plugin/plugin.json | 2 +- rust-code/CHANGELOG.md | 19 +++++ rust-code/README.md | 2 +- rust-code/agents/06-rust-code-reviewer.md | 2 +- rust-code/agents/14-rust-researcher.md | 2 + rust-code/agents/15-rust-arch-analyst.md | 2 + rust-code/agents/16-rust-security-analyst.md | 2 + rust-code/skills/arch-inspect/SKILL.md | 21 +++++- .../skills/continuous-improvement/SKILL.md | 40 ++++++++++- .../references/issue-management.md | 40 ++++++++++- .../references/research-protocol.md | 29 ++++++++ .../references/sdd-integration.md | 6 ++ .../references/testing-methodology.md | 72 ++++++++++++++++++- rust-code/skills/live-testing/SKILL.md | 19 +++-- .../references/issue-management.md | 40 ++++++++++- .../references/sdd-integration.md | 6 ++ .../references/testing-methodology.md | 72 ++++++++++++++++++- rust-code/skills/research-protocol/SKILL.md | 17 ++++- .../references/issue-management.md | 40 ++++++++++- .../references/research-protocol.md | 29 ++++++++ .../references/sdd-integration.md | 6 ++ rust-code/skills/security-audit/SKILL.md | 17 ++++- 24 files changed, 454 insertions(+), 37 deletions(-) diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 68eb318..d953a4a 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -6,14 +6,14 @@ }, "metadata": { "description": "Professional Rust development plugin: specialist agents with LSP integration, team orchestration, and peer-to-peer communication", - "version": "1.49.0" + "version": "1.50.0" }, "plugins": [ { "name": "rust-agents", "source": "./rust-code", "description": "Rust development agents and skills for Claude Code: 15 specialist agents (architect, developer, testing, performance, security, reviewer, CI/CD, debugger, critic, SDD, live-tester, tech-writer, researcher, architecture and security analysts), team-develop and team-debug agent-team orchestration, a read-only continuous-improvement cycle that also runs headless, Rust 1.89-1.99 API reference, handoff protocol, spec-driven development pipeline, and rust-analyzer LSP integration.", - "version": "1.49.0", + "version": "1.50.0", "author": { "name": "Andrei G", "email": "andrei.g@my.com" diff --git a/README.md b/README.md index ed9a680..5e21ed1 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,7 @@ This repository contains plugins that extend Claude Code's capabilities with spe ### Rust Agents Plugin (`rust-code`) -[![Version](https://img.shields.io/badge/version-1.49.0-blue)](./rust-code) +[![Version](https://img.shields.io/badge/version-1.50.0-blue)](./rust-code) [![License](https://img.shields.io/badge/license-MIT-green)](./rust-code/LICENSE) A comprehensive collection of specialized Rust development agents covering the entire Rust development lifecycle. diff --git a/rust-code/.claude-plugin/plugin.json b/rust-code/.claude-plugin/plugin.json index 110f791..56b75bc 100644 --- a/rust-code/.claude-plugin/plugin.json +++ b/rust-code/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "rust-agents", - "version": "1.49.0", + "version": "1.50.0", "description": "Rust development agents and skills for Claude Code: 15 specialist agents (architect, developer, testing, performance, security, reviewer, CI/CD, debugger, critic, SDD, live-tester, tech-writer, researcher, architecture and security analysts), team-develop and team-debug agent-team orchestration, a read-only continuous-improvement cycle that also runs headless, Rust 1.89-1.99 API reference, handoff protocol, spec-driven development pipeline, and rust-analyzer LSP integration.", "author": { "name": "Andrei G", diff --git a/rust-code/CHANGELOG.md b/rust-code/CHANGELOG.md index 17df40c..b59eecc 100644 --- a/rust-code/CHANGELOG.md +++ b/rust-code/CHANGELOG.md @@ -5,6 +5,25 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [1.50.0] - 2026-10-04 + +### Added + +- `continuous-improvement`: unchanged-HEAD cycle mode, cycle-start sync, `head:` journal frontmatter, archive-aware numbering. (#PR) +- `live-testing`: permanent `coverage-status.md` master table and process self-improvement loop. (#PR) +- `research-protocol`: delta check on unchanged HEAD and dependency functionality coverage. (#PR) +- `arch-inspect`: mandatory DRY, type-safety, modern-API, and MSRV checks every cycle. (#PR) +- `research-protocol`: optional monthly CVE class sweep and competitor-gap analysis. (#PR) +- `live-testing`: optional benchmark comparison and live drift gate. (#PR) +- `security-audit`: optional scope structure (trust boundaries, accepted risks, sensitive assets). (#PR) + +### Changed + +- CI skills and agents: literal `P0`-`P4` label required at issue creation. (#PR) +- CI skills: spec and implementation issue filed in one pass; symptoms filed without a root cause. (#PR) +- `security-audit`: P0 findings go to private reporting first when supported. (#PR) +- `live-testing`: `Tested` coverage rows are re-verified when dependencies move. (#PR) + ## [1.49.0] - 2026-10-01 ### Added diff --git a/rust-code/README.md b/rust-code/README.md index 6e0096a..f217900 100644 --- a/rust-code/README.md +++ b/rust-code/README.md @@ -1,6 +1,6 @@ # Rust Agents Plugin -[![Version](https://img.shields.io/badge/version-1.49.0-blue)](https://github.com/bug-ops/claude-plugins) +[![Version](https://img.shields.io/badge/version-1.50.0-blue)](https://github.com/bug-ops/claude-plugins) [![License](https://img.shields.io/badge/license-MIT-green)](LICENSE) [![Rust Edition](https://img.shields.io/badge/rust-Edition%202024-orange)](https://doc.rust-lang.org/edition-guide/rust-2024/) diff --git a/rust-code/agents/06-rust-code-reviewer.md b/rust-code/agents/06-rust-code-reviewer.md index b2b9561..82d1367 100644 --- a/rust-code/agents/06-rust-code-reviewer.md +++ b/rust-code/agents/06-rust-code-reviewer.md @@ -195,7 +195,7 @@ Found during code review of PR # / commit . ## Priority " \ - --label "tech-debt" # or "bug", "enhancement" — whichever fits + --label ",tech-debt" # priority label required; category may instead be "bug" or "enhancement" ``` Report the created issue URLs in your review summary so the author can reference them. diff --git a/rust-code/agents/14-rust-researcher.md b/rust-code/agents/14-rust-researcher.md index 9951356..d1ccf1b 100644 --- a/rust-code/agents/14-rust-researcher.md +++ b/rust-code/agents/14-rust-researcher.md @@ -52,6 +52,8 @@ Follow the phase sequence from the `research-protocol` skill. Summary: For P0–P2 bugs, enhancements, and all research findings: spawn the `sdd` agent first (`Agent(subagent_type: "rust-agents:sdd")`) to produce a spec before filing the issue. See the SDD integration protocol in the skill references. +Labeling, the unchanged-HEAD delta check, dependency functionality coverage, and the optional competitor-gap and CVE sweeps are defined in the `research-protocol` skill. + # Research Knowledge Base Maintain these files in `.local/testing/`: diff --git a/rust-code/agents/15-rust-arch-analyst.md b/rust-code/agents/15-rust-arch-analyst.md index 0e8d370..4caf037 100644 --- a/rust-code/agents/15-rust-arch-analyst.md +++ b/rust-code/agents/15-rust-arch-analyst.md @@ -23,4 +23,6 @@ You are not designing new architecture — you are auditing what exists. Every f If the `Skill` tool is not available in your session, the skills listed in your frontmatter are already preloaded — continue with their content and do not treat the missing call as a failure. +The every-cycle checks (DRY, type safety, modern APIs, MSRV), unchanged-HEAD module selection, and issue labeling are defined in `arch-inspect`. + Before finishing: write handoff and return frontmatter per the handoff protocol. diff --git a/rust-code/agents/16-rust-security-analyst.md b/rust-code/agents/16-rust-security-analyst.md index 6488d9d..b2a34ad 100644 --- a/rust-code/agents/16-rust-security-analyst.md +++ b/rust-code/agents/16-rust-security-analyst.md @@ -21,4 +21,6 @@ You are not implementing security fixes — you are finding what is exploitable If the `Skill` tool is not available in your session, the skills listed in your frontmatter are already preloaded — continue with their content and do not treat the missing call as a failure. +Unchanged-HEAD module selection, issue labeling, and the P0 private-reporting branch are defined in `security-audit`. + Before finishing: write handoff and return frontmatter per the handoff protocol, including the Security Review section from the audit protocol. diff --git a/rust-code/skills/arch-inspect/SKILL.md b/rust-code/skills/arch-inspect/SKILL.md index 26b57a1..eb025ef 100644 --- a/rust-code/skills/arch-inspect/SKILL.md +++ b/rust-code/skills/arch-inspect/SKILL.md @@ -39,6 +39,17 @@ The agent runs in the background; report its result when the task notification a **Type safety is the primary defense against entire classes of bugs.** Every invariant expressible in the type system is a bug that cannot exist at runtime. Audit this first and treat violations as the highest priority findings. +## Every-Cycle Requirements + +These checks run explicitly on every cycle, including cycles where HEAD is unchanged since the last audit. Never report zero findings by default; state what was checked. + +- **DRY** — grep for duplicated logic and types (near-identical match arms, repeated validation, parallel structs for one shape) before concluding there are none. +- **Type safety** — re-check newtype coverage for ids and unit-bearing fields, and look for id-shaped fields added since the last audit; do not mark a standing gap as covered without re-checking. +- **Modern APIs** — use `rust-modern-apis` and cite the specific API and file/line where a newer stable API replaces a manual workaround. +- **MSRV impact** — for each modern-API finding, say whether the declared `rust-version` already covers it. If it needs a higher MSRV, say so in the finding: an MSRV bump is a breaking change and part of the fix's cost. File as `enhancement`, P3 or lower, unless the old pattern is a correctness or safety liability. + +**Unchanged HEAD** — do not re-audit the subsystem you audited last. Read prior `journal/ci-*.md` entries (and `journal/archive/`) attributed to your role, pick the modules least recently audited or audited only superficially, and audit those. Record the modules covered in the handoff. + Gather evidence with the toolchain before reading by hand — the tools enumerate what a manual pass misses: ```bash @@ -211,6 +222,8 @@ cargo tree --duplicates **Swallowed errors** — `let _ = fallible()`, `.ok()` discarding a `Result`, or a `match` arm that logs nothing leave failures invisible. Every discarded error needs a log line or a justification comment. +**New work paths without spans** (optional, when the project requires tracing) — a newly added path that does meaningful work (I/O, network call, database query, CPU-heavy transform) and ships without a `tracing` span is invisible to trace analysis; file it as P3, or P2 on a hot path. + **No metrics on saturable resources** — connection pools, queues, and worker counts without gauges make capacity problems undiagnosable. Note absence for services; libraries may expose hooks instead. --- @@ -241,12 +254,12 @@ For each finding, assess priority: - **P2** — structural debt that multiplies as the codebase grows: DRY violations, missing type abstractions, untestable design, crate boundary violations, missing timeouts - **P3** — maintainability and readability: API naming, comment quality, function length -File a GitHub issue for every P1 and P2 finding. Batch multiple P3 findings of the same kind into one issue: +File a GitHub issue for every P1 and P2 finding. Batch multiple P3 findings of the same kind into one issue. The literal priority label is set at creation (create the label first if missing; never use another priority scheme); finding symptoms count even without a proven root cause: ```bash gh issue create \ --title "" \ - --label "architecture,code-quality" \ + --label ",architecture,code-quality" \ --body "$(cat <<'EOF' ## Finding @@ -282,8 +295,10 @@ Write your handoff with an **Architecture Review** section: ## Architecture Review ### Summary -- Findings: (P1: N, P2: N, P3: N) +- Findings: (P1: N, P2: N, P3: N, P4: N) - Issues filed: +- Modules audited: +- Every-cycle checks: DRY | type safety | modern APIs | MSRV ### Findings by Category | Category | Count | Top Issue | diff --git a/rust-code/skills/continuous-improvement/SKILL.md b/rust-code/skills/continuous-improvement/SKILL.md index 449bbbc..47f4671 100644 --- a/rust-code/skills/continuous-improvement/SKILL.md +++ b/rust-code/skills/continuous-improvement/SKILL.md @@ -3,7 +3,7 @@ name: continuous-improvement description: "Orchestrate a continuous improvement cycle: spawn rust-live-tester for live testing, rust-researcher for dependency monitoring and research, rust-arch-analyst for code quality and architecture review, and rust-security-analyst for vulnerability scanning. Read-only; aggregates findings into a cycle journal. Runs as an agent team when CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1, otherwise (including headless `claude -p`, /loop, /schedule) as background subagents." when_to_use: "'run a CI cycle', 'continuous improvement', 'live-test the latest changes', 'monitor dependencies', 'audit architecture and security', 'what should we improve next'." argument-hint: "[testing|research|dependencies|parity|arch|security|full]" -allowed-tools: Bash(printenv *), Bash(ls *), Bash(test *), Bash(echo *), Bash(grep *), Bash(sort *), Bash(tail *) +allowed-tools: Bash(printenv *), Bash(ls *), Bash(test *), Bash(echo *), Bash(grep *), Bash(sort *), Bash(tail *), Bash(dirname *), Bash(git rev-parse *) --- # Continuous Improvement Orchestrator @@ -32,6 +32,8 @@ Run a continuous improvement cycle for the current Rust project by coordinating 1. **NEVER modify source code** in this orchestrator session — delegate all execution to agents 2. **NEVER run live tests, research, or security scans directly** — spawn the appropriate agent 3. All spawned agents are read-only with respect to source code; they only write to `.local/` +4. CI sessions never fix anything: every finding, including a symptom without a known root cause, becomes an issue, and fixes run in a separate team session (`/rust-agents:team-develop`) +5. `.local/testing/` artifacts live under the **main repository root**, even when the cycle runs from a git worktree; handoffs stay in the current directory's `.local/handoff/` ## Project-Specific Rules @@ -43,7 +45,7 @@ If the preflight shows `.claude/rules/continuous-improvement.md` as present, pas - Task tools flag: !`printenv CLAUDE_CODE_ENABLE_TODO_TOOLS || echo unset` - Cargo.toml: !`test -f Cargo.toml && echo present || echo missing` - Project CI rules: !`test -f .claude/rules/continuous-improvement.md && echo present || echo absent` -- Last cycle journal: !`ls .local/testing/journal/ 2>/dev/null | grep -E '^ci-[0-9]{3}\.md$' | sort | tail -1 | grep . || echo none` +- Last cycle journal: !`ls "$(dirname "$(git rev-parse --path-format=absolute --git-common-dir)")/.local/testing/journal/" "$(dirname "$(git rev-parse --path-format=absolute --git-common-dir)")/.local/testing/journal/archive/" 2>/dev/null | grep -E '^ci-[0-9]{3}\.md$' | sort | tail -1 | grep . || echo none` STOP if Cargo.toml is `missing`. @@ -67,7 +69,23 @@ ToolSearch("select:TaskCreate,TaskUpdate,TaskList,TaskGet") If the Task tools are not found: Claude Code 2.1.233+ omits them on current models unless `CLAUDE_CODE_ENABLE_TODO_TOOLS=1` is set (e.g. in the `env` block of `settings.json`). Tell the user, then continue in **message-based fallback**: skip every TaskCreate/TaskUpdate call in this workflow, drop the Task Management section from the spawn template, and track each agent's completion by its handoff message. -Determine the next cycle number from the preflight "Last cycle journal" value: `none` means `001`; otherwise increment by one. Create `.local/testing/journal/` if it does not exist. +Determine the next cycle number from the preflight "Last cycle journal" value (it resolves the main repository root and covers `journal/archive/`): `none` means `001`; otherwise increment by one. Create `.local/testing/journal/` under the main repository root if it does not exist. A legacy `coverage.md` or `journal.md` is migrated by `rust-live-tester` per the [Testing Methodology](references/testing-methodology.md#coverage-status-file). + +### Cycle start + +Before spawning, sync and decide the cycle mode: + +1. `git pull origin main` (or the project's default branch) and read the new commits. +2. Record `git rev-parse HEAD` as `{head}` and read the previous HEAD from the last journal's `head:` frontmatter or the last handoff. +3. Equal HEADs mean **unchanged-HEAD mode**: the cycle is never skipped. Pass the mode and both SHAs to every agent in its prompt. + +| Role | Unchanged-HEAD behavior | +|---|---| +| `rust-live-tester` | Tests the stalest `Untested`/`Partial` rows and `Tested` rows whose dependencies moved | +| `rust-arch-analyst`, `rust-security-analyst` | Audit the modules least recently covered, found in prior `journal/ci-*.md` entries by that role | +| `rust-researcher` | Delta check only | + +A re-verification with zero findings is a valid cycle outcome. ### Agent outcomes @@ -85,6 +103,7 @@ Create `.local/testing/journal/ci-NNN.md`: --- cycle: NNN date: YYYY-MM-DD +head: focus: team: --- @@ -151,9 +170,13 @@ You are operating as a teammate in this session's CI cycle team. 2. TaskUpdate(status: "in_progress") when starting 3. TaskUpdate(status: "completed") when done +## Cycle Mode +HEAD: {changed|unchanged} (previous {previous-sha}, current {head}). Follow the Unchanged HEAD rules in your protocol when unchanged; never skip the cycle. + ## Journal Append each finding as a new row in the Findings table of `{journal-path}`: `| N | | | <P0-P4> | #<issue> | <spec-path or —> |` +Record the issue number and the spec path of a finding together in the same row. Follow the Priority Label rules of your protocol. ## Communication - Send results to the lead: SendMessage(to: "team-lead", message: "...", summary: "...") @@ -226,6 +249,7 @@ Agent({ Run a full architecture and code quality audit of this project. This is a READ-ONLY analysis pass — do NOT modify source files. Use the audit checklist in your agent definition. +Explicitly cover DRY, type safety, modern APIs (rust-modern-apis) and MSRV impact every cycle, including unchanged-HEAD cycles; do not report zero findings by default. Project-specific rules: <paste .claude/rules/continuous-improvement.md if it exists, else omit> Write your handoff with an Architecture Review section listing all findings and filed issue URLs." }) @@ -292,9 +316,19 @@ Aggregate results from agent messages and complete the remaining sections of `{j - Issues filed: <links> - Top security risk: <one-sentence summary; note if immediate rotation/patch is required> +### Process Retrospective + +- <methodology only: what worked, what failed and is dropped, techniques to promote to a playbook — appended to `.local/testing/process-notes.md`> + ### Next Cycle Priorities - <top 3 items based on Findings table> ``` +Then close the cycle (the orchestrator touches only `.local/` and issue labels, never source code): + +1. Append the retrospective to `.local/testing/process-notes.md` per the [Process Self-Improvement Loop](references/testing-methodology.md#process-self-improvement-loop). +2. If a bug category recurs three or more times across cycle journals, list it under Next Cycle Priorities and file one `testing-infra` issue yourself proposing a structural fix or automated regression test. After the agents finish, the orchestrator is the single permitted filer. +3. Run the label spot-check from [Issue Management](references/issue-management.md#priority-label-mandatory); the orchestrator may add a missing `P0`-`P4` label with `gh issue edit`. + Print `{journal-path}` to the console so the user can locate the cycle record. diff --git a/rust-code/skills/continuous-improvement/references/issue-management.md b/rust-code/skills/continuous-improvement/references/issue-management.md index 0d1e26f..37f8b7f 100644 --- a/rust-code/skills/continuous-improvement/references/issue-management.md +++ b/rust-code/skills/continuous-improvement/references/issue-management.md @@ -15,6 +15,18 @@ When an anomaly is found during testing, classify it by severity and map to a pr | Low | P3 | Cosmetic, edge case unlikely in practice | File for backlog | | Nice-to-have | P4 | Research ideas, future enhancements | File with `research` label | +## Priority Label (Mandatory) + +- Every issue carries exactly one label named literally `P0`, `P1`, `P2`, `P3`, or `P4`, passed via `--label` in the `gh issue create` call. Never file first and label later. +- Do not use or recreate any other priority scheme (`priority: high`, `severity/critical`); such labels are legacy. +- If a `P0`-`P4` label is missing in the repository, create it before filing: `gh label create "P0" --color b60205 --description "Critical: broken core, data loss, security"`. +- An existing issue touched during a cycle that lacks a priority label gets the correct one in the same pass. +- Spot-check compliance; the output must be empty: + ```bash + gh issue list --state open --limit 100 --json number,title,labels \ + --jq '.[] | select(([.labels[].name] | map(select(test("^P[0-4]$"))) | length) == 0)' + ``` + ## Filing Protocol For each anomaly: @@ -23,7 +35,7 @@ For each anomaly: 2. **Document** — exact steps, configuration used, relevant log excerpts 3. **Classify** — assign severity per table above 4. **Spec** — for P0–P2 and all enhancements/research: spawn `sdd` agent before filing. - See [SDD Integration](sdd-integration.md) for threshold rules and invocation template. + See [SDD Integration](sdd-integration.md) for threshold rules and invocation template. File the implementation issue in the same pass. Save spec to `.local/specs/<NNN>-<slug>/spec.md`. 5. **Check duplicates** — search existing issues before filing: ```bash @@ -33,12 +45,32 @@ For each anomaly: - Clear, descriptive title - Reproduction steps (numbered) - Expected vs actual behavior - - Priority label (P0-P4) + - Priority label (P0-P4), set at creation (see Priority Label above) - Category label (bug, enhancement, research, etc.) - Relevant log excerpts or debug output - If a spec was created: `Spec: .local/specs/<NNN>-<slug>/spec.md` 7. **Link** related issues when patterns emerge (e.g., multiple issues from same root cause) +## Read-Only Sessions and Symptoms + +- CI sessions never change source code, not even one-liners. Fixes go to a separate team session (`/rust-agents:team-develop`), guided by the filed issue. +- File an issue for every symptom, even when the root cause is unknown. Symptom, evidence, suspected area, and reproduction steps make a complete report; root-cause work belongs to the fix session. + +## Security Findings + +For a P0 security finding, first check whether the repository supports private reporting (`SECURITY.md` names a private channel, or GitHub private vulnerability reporting is enabled). If it does, report through that channel. Otherwise file a public issue labeled `P0`. + +## Optional Conventions + +Apply when the repository uses them. + +- **Label budget** — at most 5 labels per issue: priority + category (`bug`, `enhancement`, `research`) + up to 3 area labels. If a fitting label does not exist, create it rather than misclassify; if more than 5 seem necessary, split the issue. +- **Fork-aware dedup** — in a fork, also search the upstream repository's issues and PRs, open and closed, before filing. An open upstream match: file locally and link it. A merged upstream fix: record a sync gap instead of a new issue. Wrap every upstream reference in backticks so GitHub does not cross-link it. + ```bash + gh issue list --repo <upstream> --state all --search "<keywords>" --json number,title,state,url + gh pr list --repo <upstream> --state all --search "<keywords>" --json number,title,state,mergedAt,url + ``` + ## Issue Template ``` @@ -65,6 +97,8 @@ For each anomaly: [Relevant excerpts] ``` +Write issue text as a human reviewer would: never mention an AI assistant or agent tooling in a title, body, or comment. + ## Issue Triage Rules - Issues with `wontfix` or `duplicate` labels are skipped in future cycles @@ -76,4 +110,4 @@ For each anomaly: Record both positive and negative results in the current cycle journal file (`.local/testing/journal/ci-NNN.md`): - **Positive results** (feature works correctly, expected behavior confirmed) are equally important — they confirm stability and prevent redundant retesting -- A feature marked `Tested` with positive results gives confidence to skip it in the next cycle unless its code changes +- A feature marked `Tested` with positive results gives confidence, but it is re-verified when its code or its dependencies move diff --git a/rust-code/skills/continuous-improvement/references/research-protocol.md b/rust-code/skills/continuous-improvement/references/research-protocol.md index 0233aa3..a529a33 100644 --- a/rust-code/skills/continuous-improvement/references/research-protocol.md +++ b/rust-code/skills/continuous-improvement/references/research-protocol.md @@ -1,5 +1,9 @@ # Research & Monitoring Protocol +## Delta Check on Unchanged HEAD + +Dependency and parity state does not accumulate untested surface the way code does. When HEAD is unchanged since the last cycle, run a delta check only: new advisories, new releases of key dependencies and reference projects, and new activity since the dates in `competitive-parity.md`. Do not widen scope or repeat earlier scans. + ## Research & Innovation Proactively search for new techniques relevant to the project's domain: @@ -67,6 +71,10 @@ Include in the issue: Monitor changelogs for breaking changes in key dependencies. When a key dep releases a major version, assess migration effort and file an appropriately prioritized issue. +## Dependency Functionality Coverage + +Beyond version health, ask whether each major dependency is used to its full capability. For every major crate in the dependency tree, read its docs and changelog for features the project could use but does not (built-in validation, cancellation or timeout primitives, structured tracing, zero-copy deserialization, derive macros that replace hand-rolled code). File a research or enhancement issue (not P0-P1) when an unused capability would measurably improve robustness, simplify code, or resolve an open issue. Do not file for stylistic preference. + ## Competitive Parity Monitoring ### When to Run a Parity Scan @@ -127,3 +135,24 @@ Maintain `.local/testing/playbooks/competitive-parity.md` as a living document: - One row per reference project with last-checked version and date - Table of known gaps: feature / project(s) that have it / backing research / issue link / status - Update after every parity scan + +### Competitor Gap Analysis (optional) + +For projects with direct competitors rather than reference libraries: + +1. Compare each competitor's tool set, capabilities, configuration UX, and transport or protocol support with this project. +2. Check for an existing issue, then file each gap with the `competitor-gap` label (plus priority and `enhancement`), naming the competitor and the feature. +3. Record competitor, date, and gaps (or "no gaps") in the cycle journal. +4. A competitor is exhausted when a comparison yields no new gaps and every earlier gap is filed, fixed, or `wontfix`. +5. When all competitors are exhausted, search for a new analogue (GitHub, crates.io, npm, PyPI, registries, directories) that is actively maintained and overlaps in function. Add it to the competitor list in the project's CI rules and run the analysis on it immediately. If none qualifies, note the search date and queries in the journal and retry later. + +## CVE and Vulnerability-Class Sweep (optional, monthly) + +Find vulnerability classes before they are reported against this project by watching where equivalent code has already been hit. Run it as its own monthly pass, separate from the parity scan: parity gaps are missing features, CVE gaps are missing defenses, and they need different severity handling. + +1. Search advisories (GitHub Advisory Database, NVD, RUSTSEC) published since the last sweep for the key dependencies and reference libraries. +2. For each advisory, identify the vulnerability class, not just the specific bug. +3. Map the class onto this project's own pipeline: does an existing validation already close the vector? +4. Confirm with a regression test or live scenario that reproduces the analogous attack; never assume coverage from a description. +5. Protected: add the regression test if missing, citing the CVE or GHSA id, and record the mitigation in the journal. +6. Unprotected: file a P0 security issue per Security Findings in [Issue Management](issue-management.md). diff --git a/rust-code/skills/continuous-improvement/references/sdd-integration.md b/rust-code/skills/continuous-improvement/references/sdd-integration.md index 49996d7..915487b 100644 --- a/rust-code/skills/continuous-improvement/references/sdd-integration.md +++ b/rust-code/skills/continuous-improvement/references/sdd-integration.md @@ -125,6 +125,12 @@ The CI analyst MUST: --- +## Spec Authorship + +The originating agent never drafts a spec inline: delegate to `sdd` and use the path it returns. Once the spec exists, file its implementation issue in the same pass (do not defer it to a later cycle) with the spec path in the issue body. Record the issue number and the spec path together in the same journal Findings row (`Issue` and `Spec` columns). + +--- + ## Issue Body Template (with spec) ``` diff --git a/rust-code/skills/continuous-improvement/references/testing-methodology.md b/rust-code/skills/continuous-improvement/references/testing-methodology.md index 8b75597..4855a8e 100644 --- a/rust-code/skills/continuous-improvement/references/testing-methodology.md +++ b/rust-code/skills/continuous-improvement/references/testing-methodology.md @@ -10,10 +10,27 @@ Unit and integration tests run in CI automatically and verify isolated code path - Performance degradation under realistic load - Feature interactions that span multiple subsystems +## Cycle Start + +Before any testing, in this order: + +1. Sync with the remote: `git pull origin main` (or the project's default branch). +2. Read the new commits and the files they touched to scope what changed. +3. Add an `Untested` row to `coverage-status.md` for every new feature that has no row yet. +4. Compare `git rev-parse HEAD` with the HEAD recorded in the last handoff or cycle journal (`head:` frontmatter). Equal HEADs mean the cycle runs in unchanged-HEAD mode. + +## Unchanged HEAD + +Never skip the cycle and never replay the same regression scenarios. Redirect effort to stale coverage: + +1. Collect the `Untested` and `Partial` rows of `coverage-status.md`, oldest `Last session` first, and live-test the top 3-5. +2. Re-verify every `Tested` row whose dependencies moved since its `Last session` date (`Cargo.lock` history, dependency-bump commits, the researcher's findings). A `Tested` status tied to old dependency versions is stale. +3. Update each touched row in place, even when re-verification finds nothing; a clean re-verification is a valid result. + ## Testing Priority Order 1. **New functionality first** — features added or changed in the most recent PRs. New code has the least real-world exercise and is the most likely source of regressions -2. **Regression testing second** — features not tested in a long time (check `coverage-status.md`; components marked `Untested` or last tested more than 2 milestones ago) +2. **Regression testing second** — features not tested in a long time (check `coverage-status.md`; components marked `Untested`, last tested more than 2 milestones ago, or whose dependencies moved since) 3. **Everything else** — research, dependency updates, tooling improvements Transition to research only when all recently-changed components are verified and no `Untested` critical components remain. @@ -31,6 +48,38 @@ Transition to research only when all recently-changed components are verified an 4. Only after confirming no critical anomalies — continue the cycle - Small isolated changes (docs, cosmetic fixes, config-only) may skip live testing if covered by CI +## Coverage Status File + +`.local/testing/coverage-status.md` is a permanent master table: one row per feature or subsystem, always the current state. Rows are updated in place. No per-cycle headers, no round logs, no history; history lives in `journal/ci-NNN.md`. + +| Component | Status | Last session | Version | Issues | Result | +|---|---|---|---|---|---| +| Feature / subsystem | Tested / Partial / Untested / Blocked / Removed | CI-NNN (YYYY-MM-DD) | vX.Y.Z | #NNN or — | One-line outcome or blocker | + +- **Tested** — all primary scenarios verified live, no known gaps +- **Partial** — happy path verified, edge or secondary scenarios remain +- **Untested** — never live-tested, or reset after a significant change +- **Blocked** — cannot be tested; state the blocker in Result +- **Removed** — feature left the codebase; keep the row with the last version + +Maintenance rules: + +1. Add an `Untested` row before a feature PR merges. +2. Reset a row to `Untested` when its code changes significantly. +3. Update status and result right after the session that tested it; do not batch updates. +4. One row per logical feature, not per PR; merge related PRs into one issue list. +5. Never delete the file or its rows. + +A feature PR should ideally ship a playbook under `playbooks/` and its coverage row (recommended, not enforced). + +Legacy layout: if the project has `coverage.md` or a chronological `journal.md`, migrate once. Rename `coverage.md` to `coverage-status.md` and convert it to the table above; move `journal.md` entries into `journal/ci-NNN.md` files. + +## Artifact Location + +Write `.local/testing/` artifacts (playbooks, coverage status, journal, process notes) under the **main repository root**, even when running from a git worktree. They are shared project knowledge and must not live inside a disposable worktree. Handoffs stay in the current directory's `.local/handoff/`. + +When `journal/` grows unwieldy, move old `ci-NNN.md` files to `journal/archive/`; never delete them. Cycle numbering counts both directories: the next number is the highest `ci-NNN` across `journal/` and `journal/archive/`, plus one. + ## Project Discovery Before testing, understand the project: @@ -108,7 +157,26 @@ Actively expand testing approaches: | Boundary | Extreme config values, disabled features, minimal resources | | End-to-end | Simulate real user sessions and evaluate holistically | -When a technique proves effective, formalize it into a playbook in `.local/testing/playbooks/`. When a technique fails, document why in `process-notes.md` and stop using it. +Proven techniques and failures feed the loop below. + +## Process Self-Improvement Loop + +After every session, evaluate the testing process as well as the product: + +1. Append a short retrospective to `.local/testing/process-notes.md`: methodology only (what was tried, whether it worked, what to try next). Findings go to issues and the journal. +2. A technique that proved effective becomes a playbook in `.local/testing/playbooks/`. +3. A technique that failed or added no value is documented in `process-notes.md` with the reason, then dropped. +4. Every bug gets its minimal reproduction in `.local/testing/regressions.md`; re-run the file after significant changes. +5. The same category of bug three or more times calls for a structural fix or an automated regression test; file an issue proposing it. +6. Improvements that need code changes (better diagnostics, test hooks, fixtures) are filed with the `testing-infra` label. Helper scripts and fixtures needed for live testing are written directly in `.local/testing/`. +7. Prune outdated entries in `process-notes.md` and `regressions.md` periodically. + +## Optional Checks + +Run when the project qualifies; skip silently otherwise. + +- **Benchmarks (when `benches/` exists)** — run the benchmark suite each cycle, including unchanged-HEAD cycles, and compare with the previous cycle's recorded numbers. A regression above 10% is a P1 finding; smaller ones are P2. Record the numbers in the journal even when unchanged. +- **Live drift gate** — when the project depends on an external API or registry, run one cheap live call each cycle (a one-line `curl` or client invocation) and compare status and response shape with what the code expects. A mismatch is a finding even when CI fixtures pass. ## Test Environment Hygiene diff --git a/rust-code/skills/live-testing/SKILL.md b/rust-code/skills/live-testing/SKILL.md index 9abb11f..58ef916 100644 --- a/rust-code/skills/live-testing/SKILL.md +++ b/rust-code/skills/live-testing/SKILL.md @@ -36,8 +36,8 @@ If `.claude/skills/verify/SKILL.md` exists (created by `/rust-agents:init-projec ## Hard Rules 1. **NEVER modify source code** — not even one-liners -2. **ALL findings become GitHub issues** — fixes happen in separate sessions -3. **You MAY write ONLY to `.local/testing/`** — journal, coverage status, playbooks, debug logs +2. **ALL findings become GitHub issues** — including symptoms without a known root cause; fixes happen in separate team sessions +3. **You MAY write ONLY to `.local/testing/`** — journal, coverage status, playbooks, debug logs — under the main repository root, even from a git worktree ## Phase 1: Sync @@ -47,7 +47,8 @@ git pull origin main - Review new commits to identify what changed since the last cycle - Examine changed files to understand scope -- Update `.local/testing/coverage-status.md` — mark changed components as `Untested` +- Update `.local/testing/coverage-status.md` — add `Untested` rows for new features, reset rows of significantly changed components to `Untested` +- Compare `git rev-parse HEAD` with the HEAD recorded in the last handoff or journal; equal means unchanged-HEAD mode (see Phase 3) - Prioritize testing changed functionality first ## Phase 2: Project Discovery @@ -69,6 +70,10 @@ Before testing, understand the project: 3. Known-tricky scenarios from `regressions.md` 4. Cross-interface consistency (if project has multiple I/O modes) +**Unchanged HEAD:** never skip the cycle; follow [Testing Methodology](references/testing-methodology.md#unchanged-head). + +**Optional checks** (benchmarks, live drift gate): see [Testing Methodology](references/testing-methodology.md#optional-checks). + **Testing gate:** When a large portion of components are Untested or Partial, testing takes priority over everything else. Do not proceed to Phase 4 until critical components are verified. **After each test session, review:** @@ -86,8 +91,8 @@ For each anomaly found: 3. **Classify** — P0 (critical) to P4 (nice-to-have); see [Issue Management](references/issue-management.md) 4. **Spec** — for P0–P2: spawn `sdd` agent before filing; see [SDD Integration](references/sdd-integration.md) 5. **Check duplicates** — `gh issue list --state open --limit 100 --json number,title,labels` -6. **File** — `gh issue create` with priority + category labels, reproduction steps, evidence -7. **Record** — append finding row to the cycle journal (path passed via `{journal-path}` in team prompt), update `coverage-status.md` +6. **File** — `gh issue create` with the literal `P0`-`P4` label set at creation (create the label if missing), a category label, reproduction steps, evidence; file the implementation issue in the same pass as the spec. A symptom without a root cause is still filed +7. **Record** — append finding row (issue and spec together) to the cycle journal (path passed via `{journal-path}` in team prompt), update `coverage-status.md` ## Phase 5: Cross-Interface Consistency @@ -101,7 +106,7 @@ If the project supports multiple interfaces (CLI, TUI, web, API, bots): Before finishing: -1. Update `.local/testing/coverage-status.md` for all components touched -2. Append session retrospective to `.local/testing/process-notes.md` +1. Update `.local/testing/coverage-status.md` for all components touched (rows in place, no cycle headers) +2. Run the [Process Self-Improvement Loop](references/testing-methodology.md#process-self-improvement-loop) 3. Print a summary: features tested, issues filed, coverage changes 4. Write handoff with **Testing Results** section listing all filed issue URLs diff --git a/rust-code/skills/live-testing/references/issue-management.md b/rust-code/skills/live-testing/references/issue-management.md index 9f8a327..7706140 100644 --- a/rust-code/skills/live-testing/references/issue-management.md +++ b/rust-code/skills/live-testing/references/issue-management.md @@ -15,6 +15,18 @@ When an anomaly is found during testing, classify it by severity and map to a pr | Low | P3 | Cosmetic, edge case unlikely in practice | File for backlog | | Nice-to-have | P4 | Research ideas, future enhancements | File with `research` label | +## Priority Label (Mandatory) + +- Every issue carries exactly one label named literally `P0`, `P1`, `P2`, `P3`, or `P4`, passed via `--label` in the `gh issue create` call. Never file first and label later. +- Do not use or recreate any other priority scheme (`priority: high`, `severity/critical`); such labels are legacy. +- If a `P0`-`P4` label is missing in the repository, create it before filing: `gh label create "P0" --color b60205 --description "Critical: broken core, data loss, security"`. +- An existing issue touched during a cycle that lacks a priority label gets the correct one in the same pass. +- Spot-check compliance; the output must be empty: + ```bash + gh issue list --state open --limit 100 --json number,title,labels \ + --jq '.[] | select(([.labels[].name] | map(select(test("^P[0-4]$"))) | length) == 0)' + ``` + ## Filing Protocol For each anomaly: @@ -23,7 +35,7 @@ For each anomaly: 2. **Document** — exact steps, configuration used, relevant log excerpts 3. **Classify** — assign severity per table above 4. **Spec** — for P0–P2 and all enhancements/research: spawn `sdd` agent before filing. - See [SDD Integration](sdd-integration.md) for threshold rules and invocation template. + See [SDD Integration](sdd-integration.md) for threshold rules and invocation template. File the implementation issue in the same pass. Save spec to `.local/specs/<NNN>-<slug>/spec.md`. 5. **Check duplicates** — search existing issues before filing: ```bash @@ -33,12 +45,32 @@ For each anomaly: - Clear, descriptive title - Reproduction steps (numbered) - Expected vs actual behavior - - Priority label (P0-P4) + - Priority label (P0-P4), set at creation (see Priority Label above) - Category label (bug, enhancement, research, etc.) - Relevant log excerpts or debug output - If a spec was created: `Spec: .local/specs/<NNN>-<slug>/spec.md` 7. **Link** related issues when patterns emerge (e.g., multiple issues from same root cause) +## Read-Only Sessions and Symptoms + +- CI sessions never change source code, not even one-liners. Fixes go to a separate team session (`/rust-agents:team-develop`), guided by the filed issue. +- File an issue for every symptom, even when the root cause is unknown. Symptom, evidence, suspected area, and reproduction steps make a complete report; root-cause work belongs to the fix session. + +## Security Findings + +For a P0 security finding, first check whether the repository supports private reporting (`SECURITY.md` names a private channel, or GitHub private vulnerability reporting is enabled). If it does, report through that channel. Otherwise file a public issue labeled `P0`. + +## Optional Conventions + +Apply when the repository uses them. + +- **Label budget** — at most 5 labels per issue: priority + category (`bug`, `enhancement`, `research`) + up to 3 area labels. If a fitting label does not exist, create it rather than misclassify; if more than 5 seem necessary, split the issue. +- **Fork-aware dedup** — in a fork, also search the upstream repository's issues and PRs, open and closed, before filing. An open upstream match: file locally and link it. A merged upstream fix: record a sync gap instead of a new issue. Wrap every upstream reference in backticks so GitHub does not cross-link it. + ```bash + gh issue list --repo <upstream> --state all --search "<keywords>" --json number,title,state,url + gh pr list --repo <upstream> --state all --search "<keywords>" --json number,title,state,mergedAt,url + ``` + ## Issue Template ``` @@ -65,6 +97,8 @@ For each anomaly: [Relevant excerpts] ``` +Write issue text as a human reviewer would: never mention an AI assistant or agent tooling in a title, body, or comment. + ## Issue Triage Rules - Issues with `wontfix` or `duplicate` labels are skipped in future cycles @@ -76,4 +110,4 @@ For each anomaly: Record both positive and negative results in the testing journal: - **Positive results** (feature works correctly, expected behavior confirmed) are equally important — they confirm stability and prevent redundant retesting -- A feature marked `Tested` with positive results gives confidence to skip it in the next cycle unless its code changes +- A feature marked `Tested` with positive results gives confidence, but it is re-verified when its code or its dependencies move diff --git a/rust-code/skills/live-testing/references/sdd-integration.md b/rust-code/skills/live-testing/references/sdd-integration.md index 8cad24f..04fc666 100644 --- a/rust-code/skills/live-testing/references/sdd-integration.md +++ b/rust-code/skills/live-testing/references/sdd-integration.md @@ -125,6 +125,12 @@ The CI analyst MUST: --- +## Spec Authorship + +The originating agent never drafts a spec inline: delegate to `sdd` and use the path it returns. Once the spec exists, file its implementation issue in the same pass (do not defer it to a later cycle) with the spec path in the issue body. Record the issue number and the spec path together in the same journal Findings row (`Issue` and `Spec` columns). + +--- + ## Issue Body Template (with spec) ``` diff --git a/rust-code/skills/live-testing/references/testing-methodology.md b/rust-code/skills/live-testing/references/testing-methodology.md index 8b75597..4855a8e 100644 --- a/rust-code/skills/live-testing/references/testing-methodology.md +++ b/rust-code/skills/live-testing/references/testing-methodology.md @@ -10,10 +10,27 @@ Unit and integration tests run in CI automatically and verify isolated code path - Performance degradation under realistic load - Feature interactions that span multiple subsystems +## Cycle Start + +Before any testing, in this order: + +1. Sync with the remote: `git pull origin main` (or the project's default branch). +2. Read the new commits and the files they touched to scope what changed. +3. Add an `Untested` row to `coverage-status.md` for every new feature that has no row yet. +4. Compare `git rev-parse HEAD` with the HEAD recorded in the last handoff or cycle journal (`head:` frontmatter). Equal HEADs mean the cycle runs in unchanged-HEAD mode. + +## Unchanged HEAD + +Never skip the cycle and never replay the same regression scenarios. Redirect effort to stale coverage: + +1. Collect the `Untested` and `Partial` rows of `coverage-status.md`, oldest `Last session` first, and live-test the top 3-5. +2. Re-verify every `Tested` row whose dependencies moved since its `Last session` date (`Cargo.lock` history, dependency-bump commits, the researcher's findings). A `Tested` status tied to old dependency versions is stale. +3. Update each touched row in place, even when re-verification finds nothing; a clean re-verification is a valid result. + ## Testing Priority Order 1. **New functionality first** — features added or changed in the most recent PRs. New code has the least real-world exercise and is the most likely source of regressions -2. **Regression testing second** — features not tested in a long time (check `coverage-status.md`; components marked `Untested` or last tested more than 2 milestones ago) +2. **Regression testing second** — features not tested in a long time (check `coverage-status.md`; components marked `Untested`, last tested more than 2 milestones ago, or whose dependencies moved since) 3. **Everything else** — research, dependency updates, tooling improvements Transition to research only when all recently-changed components are verified and no `Untested` critical components remain. @@ -31,6 +48,38 @@ Transition to research only when all recently-changed components are verified an 4. Only after confirming no critical anomalies — continue the cycle - Small isolated changes (docs, cosmetic fixes, config-only) may skip live testing if covered by CI +## Coverage Status File + +`.local/testing/coverage-status.md` is a permanent master table: one row per feature or subsystem, always the current state. Rows are updated in place. No per-cycle headers, no round logs, no history; history lives in `journal/ci-NNN.md`. + +| Component | Status | Last session | Version | Issues | Result | +|---|---|---|---|---|---| +| Feature / subsystem | Tested / Partial / Untested / Blocked / Removed | CI-NNN (YYYY-MM-DD) | vX.Y.Z | #NNN or — | One-line outcome or blocker | + +- **Tested** — all primary scenarios verified live, no known gaps +- **Partial** — happy path verified, edge or secondary scenarios remain +- **Untested** — never live-tested, or reset after a significant change +- **Blocked** — cannot be tested; state the blocker in Result +- **Removed** — feature left the codebase; keep the row with the last version + +Maintenance rules: + +1. Add an `Untested` row before a feature PR merges. +2. Reset a row to `Untested` when its code changes significantly. +3. Update status and result right after the session that tested it; do not batch updates. +4. One row per logical feature, not per PR; merge related PRs into one issue list. +5. Never delete the file or its rows. + +A feature PR should ideally ship a playbook under `playbooks/` and its coverage row (recommended, not enforced). + +Legacy layout: if the project has `coverage.md` or a chronological `journal.md`, migrate once. Rename `coverage.md` to `coverage-status.md` and convert it to the table above; move `journal.md` entries into `journal/ci-NNN.md` files. + +## Artifact Location + +Write `.local/testing/` artifacts (playbooks, coverage status, journal, process notes) under the **main repository root**, even when running from a git worktree. They are shared project knowledge and must not live inside a disposable worktree. Handoffs stay in the current directory's `.local/handoff/`. + +When `journal/` grows unwieldy, move old `ci-NNN.md` files to `journal/archive/`; never delete them. Cycle numbering counts both directories: the next number is the highest `ci-NNN` across `journal/` and `journal/archive/`, plus one. + ## Project Discovery Before testing, understand the project: @@ -108,7 +157,26 @@ Actively expand testing approaches: | Boundary | Extreme config values, disabled features, minimal resources | | End-to-end | Simulate real user sessions and evaluate holistically | -When a technique proves effective, formalize it into a playbook in `.local/testing/playbooks/`. When a technique fails, document why in `process-notes.md` and stop using it. +Proven techniques and failures feed the loop below. + +## Process Self-Improvement Loop + +After every session, evaluate the testing process as well as the product: + +1. Append a short retrospective to `.local/testing/process-notes.md`: methodology only (what was tried, whether it worked, what to try next). Findings go to issues and the journal. +2. A technique that proved effective becomes a playbook in `.local/testing/playbooks/`. +3. A technique that failed or added no value is documented in `process-notes.md` with the reason, then dropped. +4. Every bug gets its minimal reproduction in `.local/testing/regressions.md`; re-run the file after significant changes. +5. The same category of bug three or more times calls for a structural fix or an automated regression test; file an issue proposing it. +6. Improvements that need code changes (better diagnostics, test hooks, fixtures) are filed with the `testing-infra` label. Helper scripts and fixtures needed for live testing are written directly in `.local/testing/`. +7. Prune outdated entries in `process-notes.md` and `regressions.md` periodically. + +## Optional Checks + +Run when the project qualifies; skip silently otherwise. + +- **Benchmarks (when `benches/` exists)** — run the benchmark suite each cycle, including unchanged-HEAD cycles, and compare with the previous cycle's recorded numbers. A regression above 10% is a P1 finding; smaller ones are P2. Record the numbers in the journal even when unchanged. +- **Live drift gate** — when the project depends on an external API or registry, run one cheap live call each cycle (a one-line `curl` or client invocation) and compare status and response shape with what the code expects. A mismatch is a finding even when CI fixtures pass. ## Test Environment Hygiene diff --git a/rust-code/skills/research-protocol/SKILL.md b/rust-code/skills/research-protocol/SKILL.md index edc74f7..ca7472f 100644 --- a/rust-code/skills/research-protocol/SKILL.md +++ b/rust-code/skills/research-protocol/SKILL.md @@ -33,7 +33,9 @@ Read all reference files before starting: 1. **NEVER modify source code** — not even `Cargo.toml` dependency versions 2. **ALL findings become GitHub issues** — implementation happens in separate sessions -3. **You MAY write ONLY to `.local/testing/` and `.local/specs/`** +3. **You MAY write ONLY to `.local/testing/` and `.local/specs/`** — under the main repository root, even from a git worktree + +**Unchanged HEAD:** when the cycle prompt says HEAD is unchanged, run a delta check only. See [Research Protocol](references/research-protocol.md#delta-check-on-unchanged-head). ## Phase 1: Dependency Monitoring (`dependencies`, `full`) @@ -56,11 +58,13 @@ Search for new techniques relevant to the project's domain: - Ecosystem evolution: new crates, deprecated dependencies, emerging standards - Tooling improvements: profiling, debugging, testing frameworks +Also treat dependency functionality coverage as a research question: does a major dependency offer capabilities the project does not use? See [Research Protocol](references/research-protocol.md#dependency-functionality-coverage). + For each finding: 1. Assess: impact on project quality vs implementation complexity 2. Spawn `sdd` agent to produce a spec (all research findings require a spec) 3. Check for duplicate issues before filing -4. File research issue with source links, implementation sketch, spec path +4. File research issue with source links, implementation sketch, spec path, and a priority label at creation (see [Issue Management](references/issue-management.md#priority-label-mandatory)); file it in the same pass as the spec ## Phase 3: Competitive Parity (`parity`, `full`) @@ -74,10 +78,17 @@ Parity gap severity: Update `.local/testing/playbooks/competitive-parity.md` after each scan. +## Optional Phases + +When applicable to the project (see [Research Protocol](references/research-protocol.md)): + +- **Competitor gap analysis** — `competitor-gap` label, exhaustion criterion, discovery of new competitors +- **CVE and vulnerability-class sweep** — monthly, separate from the parity scan; P0 if the class is unprotected + ## Session Exit Before finishing: -1. Append session retrospective to `.local/testing/process-notes.md` +1. Append a methodology-only retrospective to `.local/testing/process-notes.md`; a technique that proved effective becomes a playbook, a failed one is documented and dropped 2. Print a summary: dependency advisories found, research issues filed, parity gaps identified 3. Write handoff with **Research Results** section listing all filed issue URLs and spec paths diff --git a/rust-code/skills/research-protocol/references/issue-management.md b/rust-code/skills/research-protocol/references/issue-management.md index 9f8a327..7706140 100644 --- a/rust-code/skills/research-protocol/references/issue-management.md +++ b/rust-code/skills/research-protocol/references/issue-management.md @@ -15,6 +15,18 @@ When an anomaly is found during testing, classify it by severity and map to a pr | Low | P3 | Cosmetic, edge case unlikely in practice | File for backlog | | Nice-to-have | P4 | Research ideas, future enhancements | File with `research` label | +## Priority Label (Mandatory) + +- Every issue carries exactly one label named literally `P0`, `P1`, `P2`, `P3`, or `P4`, passed via `--label` in the `gh issue create` call. Never file first and label later. +- Do not use or recreate any other priority scheme (`priority: high`, `severity/critical`); such labels are legacy. +- If a `P0`-`P4` label is missing in the repository, create it before filing: `gh label create "P0" --color b60205 --description "Critical: broken core, data loss, security"`. +- An existing issue touched during a cycle that lacks a priority label gets the correct one in the same pass. +- Spot-check compliance; the output must be empty: + ```bash + gh issue list --state open --limit 100 --json number,title,labels \ + --jq '.[] | select(([.labels[].name] | map(select(test("^P[0-4]$"))) | length) == 0)' + ``` + ## Filing Protocol For each anomaly: @@ -23,7 +35,7 @@ For each anomaly: 2. **Document** — exact steps, configuration used, relevant log excerpts 3. **Classify** — assign severity per table above 4. **Spec** — for P0–P2 and all enhancements/research: spawn `sdd` agent before filing. - See [SDD Integration](sdd-integration.md) for threshold rules and invocation template. + See [SDD Integration](sdd-integration.md) for threshold rules and invocation template. File the implementation issue in the same pass. Save spec to `.local/specs/<NNN>-<slug>/spec.md`. 5. **Check duplicates** — search existing issues before filing: ```bash @@ -33,12 +45,32 @@ For each anomaly: - Clear, descriptive title - Reproduction steps (numbered) - Expected vs actual behavior - - Priority label (P0-P4) + - Priority label (P0-P4), set at creation (see Priority Label above) - Category label (bug, enhancement, research, etc.) - Relevant log excerpts or debug output - If a spec was created: `Spec: .local/specs/<NNN>-<slug>/spec.md` 7. **Link** related issues when patterns emerge (e.g., multiple issues from same root cause) +## Read-Only Sessions and Symptoms + +- CI sessions never change source code, not even one-liners. Fixes go to a separate team session (`/rust-agents:team-develop`), guided by the filed issue. +- File an issue for every symptom, even when the root cause is unknown. Symptom, evidence, suspected area, and reproduction steps make a complete report; root-cause work belongs to the fix session. + +## Security Findings + +For a P0 security finding, first check whether the repository supports private reporting (`SECURITY.md` names a private channel, or GitHub private vulnerability reporting is enabled). If it does, report through that channel. Otherwise file a public issue labeled `P0`. + +## Optional Conventions + +Apply when the repository uses them. + +- **Label budget** — at most 5 labels per issue: priority + category (`bug`, `enhancement`, `research`) + up to 3 area labels. If a fitting label does not exist, create it rather than misclassify; if more than 5 seem necessary, split the issue. +- **Fork-aware dedup** — in a fork, also search the upstream repository's issues and PRs, open and closed, before filing. An open upstream match: file locally and link it. A merged upstream fix: record a sync gap instead of a new issue. Wrap every upstream reference in backticks so GitHub does not cross-link it. + ```bash + gh issue list --repo <upstream> --state all --search "<keywords>" --json number,title,state,url + gh pr list --repo <upstream> --state all --search "<keywords>" --json number,title,state,mergedAt,url + ``` + ## Issue Template ``` @@ -65,6 +97,8 @@ For each anomaly: [Relevant excerpts] ``` +Write issue text as a human reviewer would: never mention an AI assistant or agent tooling in a title, body, or comment. + ## Issue Triage Rules - Issues with `wontfix` or `duplicate` labels are skipped in future cycles @@ -76,4 +110,4 @@ For each anomaly: Record both positive and negative results in the testing journal: - **Positive results** (feature works correctly, expected behavior confirmed) are equally important — they confirm stability and prevent redundant retesting -- A feature marked `Tested` with positive results gives confidence to skip it in the next cycle unless its code changes +- A feature marked `Tested` with positive results gives confidence, but it is re-verified when its code or its dependencies move diff --git a/rust-code/skills/research-protocol/references/research-protocol.md b/rust-code/skills/research-protocol/references/research-protocol.md index 0233aa3..a529a33 100644 --- a/rust-code/skills/research-protocol/references/research-protocol.md +++ b/rust-code/skills/research-protocol/references/research-protocol.md @@ -1,5 +1,9 @@ # Research & Monitoring Protocol +## Delta Check on Unchanged HEAD + +Dependency and parity state does not accumulate untested surface the way code does. When HEAD is unchanged since the last cycle, run a delta check only: new advisories, new releases of key dependencies and reference projects, and new activity since the dates in `competitive-parity.md`. Do not widen scope or repeat earlier scans. + ## Research & Innovation Proactively search for new techniques relevant to the project's domain: @@ -67,6 +71,10 @@ Include in the issue: Monitor changelogs for breaking changes in key dependencies. When a key dep releases a major version, assess migration effort and file an appropriately prioritized issue. +## Dependency Functionality Coverage + +Beyond version health, ask whether each major dependency is used to its full capability. For every major crate in the dependency tree, read its docs and changelog for features the project could use but does not (built-in validation, cancellation or timeout primitives, structured tracing, zero-copy deserialization, derive macros that replace hand-rolled code). File a research or enhancement issue (not P0-P1) when an unused capability would measurably improve robustness, simplify code, or resolve an open issue. Do not file for stylistic preference. + ## Competitive Parity Monitoring ### When to Run a Parity Scan @@ -127,3 +135,24 @@ Maintain `.local/testing/playbooks/competitive-parity.md` as a living document: - One row per reference project with last-checked version and date - Table of known gaps: feature / project(s) that have it / backing research / issue link / status - Update after every parity scan + +### Competitor Gap Analysis (optional) + +For projects with direct competitors rather than reference libraries: + +1. Compare each competitor's tool set, capabilities, configuration UX, and transport or protocol support with this project. +2. Check for an existing issue, then file each gap with the `competitor-gap` label (plus priority and `enhancement`), naming the competitor and the feature. +3. Record competitor, date, and gaps (or "no gaps") in the cycle journal. +4. A competitor is exhausted when a comparison yields no new gaps and every earlier gap is filed, fixed, or `wontfix`. +5. When all competitors are exhausted, search for a new analogue (GitHub, crates.io, npm, PyPI, registries, directories) that is actively maintained and overlaps in function. Add it to the competitor list in the project's CI rules and run the analysis on it immediately. If none qualifies, note the search date and queries in the journal and retry later. + +## CVE and Vulnerability-Class Sweep (optional, monthly) + +Find vulnerability classes before they are reported against this project by watching where equivalent code has already been hit. Run it as its own monthly pass, separate from the parity scan: parity gaps are missing features, CVE gaps are missing defenses, and they need different severity handling. + +1. Search advisories (GitHub Advisory Database, NVD, RUSTSEC) published since the last sweep for the key dependencies and reference libraries. +2. For each advisory, identify the vulnerability class, not just the specific bug. +3. Map the class onto this project's own pipeline: does an existing validation already close the vector? +4. Confirm with a regression test or live scenario that reproduces the analogous attack; never assume coverage from a description. +5. Protected: add the regression test if missing, citing the CVE or GHSA id, and record the mitigation in the journal. +6. Unprotected: file a P0 security issue per Security Findings in [Issue Management](issue-management.md). diff --git a/rust-code/skills/research-protocol/references/sdd-integration.md b/rust-code/skills/research-protocol/references/sdd-integration.md index 8cad24f..04fc666 100644 --- a/rust-code/skills/research-protocol/references/sdd-integration.md +++ b/rust-code/skills/research-protocol/references/sdd-integration.md @@ -125,6 +125,12 @@ The CI analyst MUST: --- +## Spec Authorship + +The originating agent never drafts a spec inline: delegate to `sdd` and use the path it returns. Once the spec exists, file its implementation issue in the same pass (do not defer it to a later cycle) with the spec path in the issue body. Record the issue number and the spec path together in the same journal Findings row (`Issue` and `Spec` columns). + +--- + ## Issue Body Template (with spec) ``` diff --git a/rust-code/skills/security-audit/SKILL.md b/rust-code/skills/security-audit/SKILL.md index cac8a25..6077af2 100644 --- a/rust-code/skills/security-audit/SKILL.md +++ b/rust-code/skills/security-audit/SKILL.md @@ -46,6 +46,16 @@ Pattern searches only find what matches a pattern. Before the category passes, b Reference frameworks for completeness checks: OWASP Top 10, CWE Top 25, the ANSSI secure Rust guidelines, and the Rustonomicon for `unsafe`. When a finding maps to a CWE, cite it. +**Unchanged HEAD** — never skip the audit and never re-audit the module you covered last. Read prior `journal/ci-*.md` entries (and `journal/archive/`) attributed to your role, pick the modules least recently audited, and audit those. Record the modules covered in the handoff. + +**Scope structure (optional)** — when the project's CI rules define one, judge findings against it: + +- **Trust boundaries** — where untrusted input enters; a finding is real only when an input on a listed boundary reaches the sink. +- **Accepted risks** — re-verify each cycle that the accepted condition still holds (for example, no `unsafe` exists); do not re-file it. A change in the condition is a new finding. +- **Sensitive assets** — what an attacker would target; weight severity by reach to these. +- **Minimum floor** — a project checklist is a floor, re-derived against the code every cycle (new modules, dependencies, and entry points expand it), never a list to rubber-stamp. +- **CI-gated scanners** — when CI already runs a scanner on every push (for example `cargo audit`), a clean local run adds no signal; spend the effort on what CI does not cover (unsafe review, injection, crypto misuse, supply-chain trust). + --- ## 0. Attack Surface Map @@ -306,12 +316,14 @@ Assign a severity and map it to the cycle-journal priority column: | **Medium** | `P2` | Real but bounded: panic-DoS on request path, weak crypto, advisory in a non-critical path, missing authz on a low-value action | | **Low** | `P3` | Hardening / defense-in-depth: unmaintained deps without a known advisory, missing zeroize, license policy drift | -File a GitHub issue for every Critical, High, and Medium finding. Batch same-kind Low findings into one issue. +File a GitHub issue for every Critical, High, and Medium finding. Batch same-kind Low findings into one issue. The literal `P0`-`P3` label is set at creation (create it if missing; never use another priority scheme). Findings without a proven root cause are still filed as symptoms. + +**P0 branch** — before filing a Critical finding publicly, check whether the repository supports private reporting (`SECURITY.md` routes to a private channel, or GitHub private vulnerability reporting is enabled). If so, use that channel; otherwise file a public issue labeled `P0`. ```bash gh issue create \ --title "<severity>: <concise vulnerability title>" \ - --label "security,vulnerability" \ + --label "<P0|P1|P2|P3>,security,vulnerability" \ --body "$(cat <<'EOF' ## Vulnerability <what is wrong> @@ -355,6 +367,7 @@ Write your handoff with a **Security Review** section: - Findings: <N total> (Critical: N, High: N, Medium: N, Low: N) - Issues filed: <links> - `cargo audit`: <N advisories> | `cargo deny`: <pass/fail> | `gitleaks`: <clean/N hits> +- Modules audited: <list; feeds least-recently-audited selection next cycle> ### Findings by Category | Category | Count | Top Issue | From 600da0f2aeef95dab1d649da897dccd3523f923c Mon Sep 17 00:00:00 2001 From: "Andrei G." <k05h31@gmail.com> Date: Sun, 4 Oct 2026 14:42:15 +0200 Subject: [PATCH 2/2] docs(changelog): link 1.50.0 entries to PR #13 --- rust-code/CHANGELOG.md | 22 +++++++++++----------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/rust-code/CHANGELOG.md b/rust-code/CHANGELOG.md index b59eecc..7d408c5 100644 --- a/rust-code/CHANGELOG.md +++ b/rust-code/CHANGELOG.md @@ -9,20 +9,20 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added -- `continuous-improvement`: unchanged-HEAD cycle mode, cycle-start sync, `head:` journal frontmatter, archive-aware numbering. (#PR) -- `live-testing`: permanent `coverage-status.md` master table and process self-improvement loop. (#PR) -- `research-protocol`: delta check on unchanged HEAD and dependency functionality coverage. (#PR) -- `arch-inspect`: mandatory DRY, type-safety, modern-API, and MSRV checks every cycle. (#PR) -- `research-protocol`: optional monthly CVE class sweep and competitor-gap analysis. (#PR) -- `live-testing`: optional benchmark comparison and live drift gate. (#PR) -- `security-audit`: optional scope structure (trust boundaries, accepted risks, sensitive assets). (#PR) +- `continuous-improvement`: unchanged-HEAD cycle mode, cycle-start sync, `head:` journal frontmatter, archive-aware numbering. (#13) +- `live-testing`: permanent `coverage-status.md` master table and process self-improvement loop. (#13) +- `research-protocol`: delta check on unchanged HEAD and dependency functionality coverage. (#13) +- `arch-inspect`: mandatory DRY, type-safety, modern-API, and MSRV checks every cycle. (#13) +- `research-protocol`: optional monthly CVE class sweep and competitor-gap analysis. (#13) +- `live-testing`: optional benchmark comparison and live drift gate. (#13) +- `security-audit`: optional scope structure (trust boundaries, accepted risks, sensitive assets). (#13) ### Changed -- CI skills and agents: literal `P0`-`P4` label required at issue creation. (#PR) -- CI skills: spec and implementation issue filed in one pass; symptoms filed without a root cause. (#PR) -- `security-audit`: P0 findings go to private reporting first when supported. (#PR) -- `live-testing`: `Tested` coverage rows are re-verified when dependencies move. (#PR) +- CI skills and agents: literal `P0`-`P4` label required at issue creation. (#13) +- CI skills: spec and implementation issue filed in one pass; symptoms filed without a root cause. (#13) +- `security-audit`: P0 findings go to private reporting first when supported. (#13) +- `live-testing`: `Tested` coverage rows are re-verified when dependencies move. (#13) ## [1.49.0] - 2026-10-01