From ea2db6c5ea656909174583f89d5204f3d8741982 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Fri, 9 Oct 2026 06:12:46 +0000 Subject: [PATCH] chore: refresh upstream skill catalog --- .../skills/astro-developer/manifest.json | 2 +- .../skills/csharp-expert/SKILL.md | 350 ++++++++++++++++ .../skills/csharp-expert/manifest.json | 5 + .../references/dotnet-skills-marketplace.md | 145 +++++++ .../agents/test-migration/AGENT.md | 9 +- .../migrate-mstest-v1v2-to-v3/manifest.json | 2 +- .../migrate-mstest-v3-to-v4/manifest.json | 2 +- .../migrate-nunit-to-mstest/manifest.json | 2 +- .../migrate-vstest-to-mtp/manifest.json | 2 +- .../migrate-xunit-to-mstest/manifest.json | 2 +- .../migrate-xunit-to-xunit-v3/manifest.json | 2 +- .../agents/code-testing-implementer/AGENT.md | 8 +- .../agents/code-testing-tester/AGENT.md | 2 +- .../AGENT.md | 86 +++- .../agents/test-quality-auditor/AGENT.md | 15 +- .../agents/testability-migration/AGENT.md | 34 +- .../skills/assertion-quality/SKILL.md | 6 +- .../skills/assertion-quality/manifest.json | 2 +- .../skills/code-testing-agent/SKILL.md | 375 +---------------- .../skills/code-testing-agent/manifest.json | 2 +- .../extensions/dotnet-examples.md | 2 +- .../extensions/powershell.md | 2 +- .../extensions/python-examples.md | 2 +- .../extensions/typescript-examples.md | 2 +- .../extensions/typescript.md | 2 +- .../code-testing-extensions/manifest.json | 2 +- .../skills/code-testing/SKILL.md | 381 ++++++++++++++++++ .../skills/code-testing/manifest.json | 5 + .../unit-test-generation.prompt.md | 0 .../skills/coverage-analysis/SKILL.md | 2 +- .../skills/coverage-analysis/manifest.json | 2 +- .../skills/crap-score/SKILL.md | 2 +- .../skills/crap-score/manifest.json | 2 +- .../detect-static-dependencies/manifest.json | 2 +- .../skills/filter-syntax/manifest.json | 2 +- .../find-untested-sources/manifest.json | 2 +- .../manifest.json | 2 +- .../skills/grade-tests/SKILL.md | 131 +++++- .../skills/grade-tests/manifest.json | 2 +- .../migrate-static-to-wrapper/manifest.json | 2 +- .../skills/mtp-hot-reload/manifest.json | 2 +- .../skills/platform-detection/manifest.json | 2 +- .../skills/run-tests/manifest.json | 2 +- .../scaffold-dotnet-test-project/SKILL.md | 4 +- .../manifest.json | 2 +- .../test-analysis-extensions/manifest.json | 2 +- .../skills/test-anti-patterns/SKILL.md | 4 +- .../skills/test-anti-patterns/manifest.json | 2 +- .../skills/test-gap-analysis/SKILL.md | 40 +- .../skills/test-gap-analysis/manifest.json | 2 +- .../references/mutation-catalog.md | 11 +- .../references/per-test-read-only.md | 106 +++++ .../skills/test-smell-detection/manifest.json | 2 +- .../skills/test-tagging/SKILL.md | 4 +- .../skills/test-tagging/manifest.json | 2 +- .../skills/testability-obstacle/SKILL.md | 2 +- .../skills/testability-obstacle/manifest.json | 2 +- .../skills/writing-mstest-tests/SKILL.md | 4 +- .../skills/writing-mstest-tests/manifest.json | 2 +- .../manifest.json | 2 +- .../skills/create-skill-test/SKILL.md | 33 +- .../references/writing-for-baseline-delta.md | 4 +- .../astro/packages/astro/package.json | 3 +- .../.agents/skills/create-skill-test/SKILL.md | 33 +- .../references/writing-for-baseline-delta.md | 4 +- .../.claude-plugin/plugin.json | 2 +- .../.codex-plugin/plugin.json | 2 +- .../plugins/dotnet-test-migration/README.md | 7 +- .../agents/test-migration.agent.md | 9 +- .../plugins/dotnet-test-migration/plugin.json | 2 +- .../dotnet-test/.claude-plugin/plugin.json | 4 +- .../dotnet-test/.codex-plugin/plugin.json | 2 +- .../plugins/dotnet-test/README.md | 97 +++-- .../agents/code-testing-implementer.agent.md | 8 +- .../agents/code-testing-tester.agent.md | 2 +- ...erator.agent.md => test-engineer.agent.md} | 86 +++- .../agents/test-quality-auditor.agent.md | 15 +- .../agents/testability-migration.agent.md | 34 +- .../plugins/dotnet-test/plugin.json | 4 +- .../skills/assertion-quality/SKILL.md | 6 +- .../skills/code-testing-agent/SKILL.md | 375 +---------------- .../extensions/dotnet-examples.md | 2 +- .../extensions/powershell.md | 2 +- .../extensions/python-examples.md | 2 +- .../extensions/typescript-examples.md | 2 +- .../extensions/typescript.md | 2 +- .../dotnet-test/skills/code-testing/SKILL.md | 381 ++++++++++++++++++ .../unit-test-generation.prompt.md | 0 .../skills/coverage-analysis/SKILL.md | 2 +- .../dotnet-test/skills/crap-score/SKILL.md | 2 +- .../dotnet-test/skills/grade-tests/SKILL.md | 131 +++++- .../scaffold-dotnet-test-project/SKILL.md | 4 +- .../skills/test-anti-patterns/SKILL.md | 4 +- .../skills/test-gap-analysis/SKILL.md | 40 +- .../references/mutation-catalog.md | 11 +- .../references/per-test-read-only.md | 106 +++++ .../dotnet-test/skills/test-tagging/SKILL.md | 4 +- .../skills/testability-obstacle/SKILL.md | 2 +- .../skills/writing-mstest-tests/SKILL.md | 4 +- .../plugins/dotnet-test/version.json | 2 +- .../dotnet-skills/plugins/dotnet/README.md | 25 +- .../dotnet/skills/csharp-expert/SKILL.md | 350 ++++++++++++++++ .../references/dotnet-skills-marketplace.md | 145 +++++++ external-sources/vendir.lock.yml | 12 +- 104 files changed, 2738 insertions(+), 1030 deletions(-) create mode 100644 catalog/Platform/Official-DotNet/skills/csharp-expert/SKILL.md create mode 100644 catalog/Platform/Official-DotNet/skills/csharp-expert/manifest.json create mode 100644 catalog/Platform/Official-DotNet/skills/csharp-expert/references/dotnet-skills-marketplace.md rename catalog/Testing/Official-DotNet-Test/agents/{code-testing-generator => test-engineer}/AGENT.md (86%) create mode 100644 catalog/Testing/Official-DotNet-Test/skills/code-testing/SKILL.md create mode 100644 catalog/Testing/Official-DotNet-Test/skills/code-testing/manifest.json rename catalog/Testing/Official-DotNet-Test/skills/{code-testing-agent => code-testing}/unit-test-generation.prompt.md (100%) create mode 100644 catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/references/per-test-read-only.md rename external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/{code-testing-generator.agent.md => test-engineer.agent.md} (86%) create mode 100644 external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing/SKILL.md rename external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/{code-testing-agent => code-testing}/unit-test-generation.prompt.md (100%) create mode 100644 external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/references/per-test-read-only.md create mode 100644 external-sources/upstreams/dotnet-skills/plugins/dotnet/skills/csharp-expert/SKILL.md create mode 100644 external-sources/upstreams/dotnet-skills/plugins/dotnet/skills/csharp-expert/references/dotnet-skills-marketplace.md diff --git a/catalog/Frameworks/Official-Astro/skills/astro-developer/manifest.json b/catalog/Frameworks/Official-Astro/skills/astro-developer/manifest.json index c4e0ffac..bf85ad3d 100644 --- a/catalog/Frameworks/Official-Astro/skills/astro-developer/manifest.json +++ b/catalog/Frameworks/Official-Astro/skills/astro-developer/manifest.json @@ -1,5 +1,5 @@ { - "version": "7.3.7", + "version": "7.3.8", "category": "Web", "compatibility": "Requires the withastro/astro monorepo." } diff --git a/catalog/Platform/Official-DotNet/skills/csharp-expert/SKILL.md b/catalog/Platform/Official-DotNet/skills/csharp-expert/SKILL.md new file mode 100644 index 00000000..fd0fed9d --- /dev/null +++ b/catalog/Platform/Official-DotNet/skills/csharp-expert/SKILL.md @@ -0,0 +1,350 @@ +--- +name: csharp-expert +description: >- + Route C# and .NET requests to the exact installed specialist or smallest dotnet/skills + marketplace plugin. USE FOR: "which specialist should own this" or "how do I add the skill" + requests involving ASP.NET Core endpoints, Blazor, MAUI binding, Windows Forms specialist + selection or installation, EF Core queries, test or framework migration, runtime + CPU/allocation evidence, file-based C#, editor/compiler defects, a surviving MSBuild + `.binlog`, or a plugin missing from `/skills`; also use for C# semantics + when no narrower specialist exists. DO NOT USE FOR: requests that already name the exact + installed specialist to invoke, or work unrelated to C# or .NET. +license: MIT +--- + +# C# Expert + +## Purpose + +Act as the front door for C# and .NET work. Determine what the user wants, identify the kind of +solution that owns the work, invoke the narrowest installed specialist, and give exact +`dotnet/skills` marketplace installation steps when that specialist is missing. Keep direct C# +language guidance as the fallback, not the default. + +## Routing Contract + +1. Classify the requested outcome from the prompt. +2. Inspect the smallest set of repository files needed to identify the solution type. +3. Compare both signals with the descriptions of the skills currently available to the runtime. +4. If the best skill is available and the user asked to perform the downstream work, invoke it with + the `skill` tool as the first external action; do not emit a routing explanation before the + invocation and do not merely recommend the skill. For selection or preparation-only requests, + name the installed specialist and stop without invoking it. +5. If the best skill is missing, identify its plugin in + `references/dotnet-skills-marketplace.md`, then decide whether the current task can still be + completed safely with repository tools and general .NET knowledge. +6. Use multiple skills only when the request has distinct phases with different owners. +7. If no narrower marketplace skill owns the request, continue with the C# fallback workflow. + +Routing is not task completion. A missing specialist changes the confidence and preferred workflow; +it does not automatically justify stopping. Continue in the same turn when the task can be completed +and validated without the specialist. Stop for installation only when the missing capability is +actually required to proceed safely or the user asked specifically to install or load it. + +Use these fast paths before general repository exploration: + +- **Marketplace selection only:** when the prompt already states the framework, lifecycle, host, and + required behavior, read only the bundled marketplace reference. Do not inspect the fixture, + repository, GitHub, or plugin source. Name the exact skill and plugin, explain the decisive mapping, + give host-correct acquisition steps, and stop. +- **Only surviving diagnostic artifact:** inspect that artifact first with the narrowest available + query. For a supplied `.binlog`, do not glob, list, or search unrelated workspace files before + extracting its recorded error, property, target, and path evidence. + +Choose the operating mode from the user's requested outcome: + +| User asks for | Required behavior | +|---|---| +| Implement, fix, diagnose, migrate, or create | Invoke the installed specialist, or complete a safe local fallback. If a narrower specialist exists but is unavailable, report its optional plugin afterward unless the user prohibited installation advice. | +| Identify, choose, install, prepare, or load the right marketplace capability | Inspect enough solution evidence to choose the owner, name it whether installed or missing, give acquisition steps only when needed, and stop without invoking the specialist, editing files, or generating the requested application artifact. | +| Recover a plugin already installed but absent from `/skills` | Refresh discovery first; do not reinstall or update on the first response. | + +Choose one owner per phase. Do not expose internal routing ceremony or turn the answer into a menu. + +## Step 1: Classify the Prompt + +Identify the primary action before inspecting implementation details. + +| Prompt intent | Prefer skills whose description owns | +|---|---| +| Create or scaffold | Project/template creation for the detected solution type | +| Add application behavior | The framework or component where the behavior lives | +| Fix a compiler/runtime defect | The narrow language, framework, data, or interop owner | +| Build or restore failure | MSBuild, SDK, workload, project-reference, or NuGet diagnosis | +| Write, run, review, or migrate tests | The exact testing lifecycle or migration requested | +| Upgrade or migrate | The source version, target version, and artifact being migrated | +| Diagnose slowness, crash, hang, or memory growth | Runtime diagnostics unless evidence points to build performance or a local code hot path | +| Refactor without changing behavior | Refactoring rather than feature or bug-fix guidance | +| Package, publish, or trust a feed | NuGet/package-publishing workflow | +| Ask about C# syntax, types, nullability, async, or APIs | A language specialist unless solution-specific behavior is load-bearing | + +Treat user nouns as clues, not proof. "Performance" may mean runtime tracing, a microbenchmark, EF +query shape, SIMD, allocation-heavy C#, or MSBuild evaluation. "API" may mean ASP.NET Core, a public +library contract, or an external service client. + +When a deployed .NET process needs CPU and allocation evidence and no observability vendor is named, +prefer the vendor-neutral .NET runtime diagnostics route. Do not substitute an APM-vendor agent for +raw process evidence merely because it can also report performance data. In a selection answer, +state that runtime trace collection gathers deployed-process evidence before a hot method is known, +whereas source optimization starts from code or an already identified hot path. + +## Step 2: Detect the Solution Type + +Inspect only likely manifests and nearby owning files. Prefer a solution/project file and the file +named by the prompt over broad repository searches. + +Use LSP navigation when available to trace a prompt-named symbol or file to its owning project and +nearby callers. Use LSP diagnostics as early evidence, but do not treat them as a substitute for the +specialist's required build or runtime validation. + +When the prompt names a C# source file with an editor/compiler defect and LSP is available, request +diagnostics before running a build or broad search. Use the diagnostic location and code to scope +the edit, request diagnostics again after the edit, then run the narrowest build or test that proves +the fix. + +| Evidence | Solution or concern | +|---|---| +| `Microsoft.NET.Sdk.Web`, controllers, endpoints, middleware, OpenAPI | ASP.NET Core | +| `.razor`, `AddRazorComponents`, Blazor bootstrapping | Blazor | +| `true`, `MauiProgram`, XAML pages | .NET MAUI | +| `true`, `Form`, designer files | Windows Forms | +| EF Core package references, `DbContext`, migrations | .NET data / EF Core | +| `true`, test SDK/framework packages | .NET testing | +| `Directory.Build.*`, custom targets/tasks, `.binlog`, evaluation errors | MSBuild | +| `Directory.Packages.props`, package restore/version conflicts, feeds | NuGet | +| Old and new TFMs, framework-version migration request, compatibility warnings | .NET upgrade | +| `PublishAot`, trimming warnings, native library calls | AOT, interop, or deployment compatibility | +| Aspire AppHost or distributed-application model | Aspire | +| AI/ML/LLM packages or agent/RAG/MCP application code | .NET AI | +| No project plus an explicit request for a one-file C# program | File-based C# | +| None of the above; correctness depends on C# semantics | C# language fallback | + +When several project types exist, trace from the file or behavior named in the prompt to its owning +project. Do not route the whole solution from the first `.csproj` found. + +## Step 3: Match the Skill and Marketplace Plugin + +Use the runtime-provided available-skill names and descriptions as the source of truth. Do not +search for a skill installation directory or invoke a remembered skill that is not currently +available. + +Rank candidates in this order: + +1. An exact transformation or lifecycle skill, such as a specific test migration, framework + conversion, template operation, query optimization, or diagnostic collection workflow. +2. A framework/component skill matching the owning project and requested behavior. +3. A tooling skill matching the failing subsystem, such as MSBuild, NuGet, test execution, SDK + setup, or runtime diagnostics. +4. A cross-cutting specialist matching the actual mechanism, such as interop, vectorization, + serialization, AOT, or microbenchmarking. +5. `csharp-refactoring` for behavior-preserving structural change. +6. The C# language fallback below when no narrower available skill owns the work. + +Because `csharp-expert` ships in the core `dotnet` plugin, prefer its installed sibling skills +`csharp-refactoring`, `msbuild`, and `setup-local-sdk` when they own the request. Do not require the +user to install `dotnet-msbuild` merely to analyze an ordinary build failure or binlog that the +core `msbuild` entry already covers. + +The most specific noun is not always the owner. Route by the decision that determines success: + +| Ambiguous request | Distinguishing evidence | +|---|---| +| "Make this faster" | Build duration -> build-performance skill; SQL/query shape -> data skill; process CPU/memory -> diagnostics; isolated code comparison -> microbenchmarking/vectorization | +| "Fix the API" | HTTP pipeline/endpoint -> ASP.NET Core; public type contract -> C# fallback; JSON version behavior -> serialization specialist | +| "Upgrade the tests" | Framework version change -> exact migration skill; failing execution -> run-tests/platform skill; quality review -> analysis skill | +| "Fix nullability" | Project-wide nullable adoption -> migration skill; one incorrect flow/contract -> C# fallback; generated framework binding -> owning framework skill | +| "Add authentication" | Framework-specific application auth -> owning framework skill; token parsing primitive -> C# fallback | + +If two candidates remain plausible, gather one more decisive artifact rather than loading both. + +After selecting the capability: + +1. If its skill appears in the runtime's available-skill catalog, invoke it immediately only when + the operating mode requires downstream implementation. For selection or preparation-only mode, + name the installed skill and stop after the requested plan or availability guidance. +2. If it does not appear, open `references/dotnet-skills-marketplace.md` and map the capability or + skill name to the owning marketplace plugin. +3. Recommend the smallest plugin that contains the needed skill. Do not install every .NET plugin. +4. Follow the missing-skill workflow below. Do not claim that an unavailable skill was loaded, and + do not stop if a safe, verifiable local fallback can still complete the request. + +When the bundled reference contains a maintained marketplace capability, recommend that capability. +Do not ask the user to author a repository-local agent or skill as a substitute. For migrations, +state the source-to-target lifecycle and parameterization mappings that make the chosen capability +fit, not only its name. + +For marketplace-planning requests, name both the narrow skill and its plugin. Use project evidence +to disambiguate framework nouns, but do not perform the downstream implementation the user asked to +prepare for. + +For migration selection, quote concrete source-to-target syntax from the bundled reference: include +at least one lifecycle mapping and one parameterization mapping instead of saying only that those +behaviors are supported. + +Use the bundled marketplace reference as the authoritative lookup. Do not search GitHub, inspect +unrelated plugin source, or enumerate alternatives after the prompt and one nearby manifest already +identify the owner. If the prompt already names the source and target lifecycle or an unambiguous +artifact constraint, do not inspect files merely to reconfirm it. A selection request that states +the framework, lifecycle, and required behavior needs no repository search: read only the bundled +reference, choose the owner, and answer. Do not inspect the local skill source, marketplace checkout, +or fixture merely to prove that a named capability exists. + +Answer in four compact parts: + +1. Exact skill name and plugin; never substitute a generic capability label when the bundled + reference contains an exact route. +2. One sentence matching the decisive behavior or artifact evidence. Use the user's concrete + mechanism: N+1/database round trips for repeated EF related-data queries; deployed-process + CPU/allocation collection before a known hot path for runtime tracing; source and target TFM + plus compatibility work for upgrades. +3. Host-correct install steps. +4. Restart/discovery verification, when the host requires it. + +For a selection answer, completeness beats extra exploration. Read the bundled reference once, +then answer. Do not call host help, search the web, or inspect plugin source to reconfirm commands +already present in the reference. + +Do not add a `Route:` header in marketplace-planning mode; lead with the capability and plugin. +Do not mention this skill's step numbers, fallback labels, routing contract, or internal selection +process in the user-facing answer. + +## Step 4: Obtain a Missing Skill + +For GitHub Copilot CLI or Claude Code, give these exact commands with the selected plugin substituted: + +```text +/plugin marketplace add dotnet/skills +/plugin install @dotnet-agent-skills +``` + +When installation is the next step, require: + +```text +Restart the host, run `/skills` to confirm the specialist is available, and rerun the request. +``` + +When the task can proceed without the specialist: + +1. State the missing specialist and reduced coverage in one concise sentence. +2. Complete the requested work now using the repository, standard .NET tooling, and the fallback + rules that match the task. +3. Validate the result as narrowly as possible. +4. Put optional installation guidance after the result. Do not ask whether to proceed, defer the + implementation, or make the user repeat the request. + +If the user explicitly says not to recommend or discuss installation, omit the missing-plugin +sentence and all acquisition guidance. Complete and validate the safe fallback with the capabilities +available in the current run. + +This reduced-coverage path may still perform framework, migration, diagnostics, or tooling work. +Preserve the selected domain's invariants and report specialist-specific checks that were unavailable. + +Rules: + +- The marketplace name is `dotnet-agent-skills`; the source repository is `dotnet/skills`. +- The install target is the plugin name, not the individual skill name. +- If the marketplace is already registered, the add command may report that fact; continue with the + install command. +- Slash commands are host actions. Present them exactly; do not run shell commands that pretend to + install a Copilot or Claude plugin. +- Do not pretend to continue with the unavailable specialist workflow. Use an explicit local + fallback when the task remains safely achievable. +- If the plugin is installed but the skill is absent, ask the user to restart and check `/skills` + before recommending a reinstall. +- For Codex CLI, VS Code, Cursor, or individual-skill installation, use the host-specific commands in + `references/dotnet-skills-marketplace.md`. +- If installation is impossible, declined, or not necessary for the immediate task, state the + reduced coverage and use the safest local fallback when it can still satisfy the request. +- Never trade away implementation or validation merely to produce installation instructions. + +### Installed but not discovered + +When the user says the plugin is already installed but its skills are absent: + +1. Trust the stated installed state unless repository evidence directly contradicts it. +2. Start by acknowledging that the plugin is installed and discovery is stale. +3. Tell the user to restart or reload the host, then run `/skills`. +4. Name the expected skill so discovery can be verified. +5. Stop there on the first response. Do not emit marketplace-add, install, update, shell-level + plugin-management, `/skills reload`, or invented explicit-invocation commands. + +Only after the user reports that restart plus `/skills` still fails should the next response move to +host-specific update or reinstall diagnostics. + +## Step 5: Compose Skills Deliberately + +Use a sequence only when phases are independently owned. Examples: + +- Scaffold a project, then author a framework-specific component. +- Collect a trace, then analyze the captured performance evidence. +- Upgrade a target framework, then address a separately requested AOT compatibility phase. +- Detect the test platform, then run tests with the correct filter syntax. + +Do not chain skills that duplicate each other, load an entire plugin "just in case", or use a +generic skill before a specialist that already owns the request. After a specialist is loaded, +follow its workflow and boundaries. + +If one or more phase specialists are missing but ordinary `dotnet` commands and repository edits +can complete the phases, execute the phases in order and report the optional plugins afterward. +Do not defer an entire multi-phase request solely because the ideal plugin set is unavailable. + +## Step 6: C# Language Fallback + +Use this only when no narrower available skill matches and the load-bearing problem is C# language +or runtime semantics. + +1. Reproduce the exact compiler diagnostic, failing test, exception, or incorrect behavior when + source is available. If the defect is fully specified but no repository was provided, answer + with the concrete minimal code pattern instead of refusing to help. +2. Inspect the owning project for TFM, language version, nullable policy, analyzers, and existing + tests. +3. Preserve public signatures, serialization shape, ownership, cancellation, disposal, and + multi-target behavior unless the request explicitly changes them. +4. Implement the smallest complete fix through the affected call path. +5. Check LSP diagnostics when available, then build the narrowest affected project and run focused + tests or the executable path that proves the original symptom is gone. + +For a marketplace-planning request whose correct route is this fallback, say: `No additional +marketplace plugin is required; the loaded csharp-expert skill owns this C# semantic fix.` Do not +claim that no skill or specialist is involved. + +Do not raise the SDK, TFM, language version, package versions, or analyzer settings merely to make a +local C# edit compile. Do not edit generated files. Do not use broad casts, null-forgiving +operators, catch-all handlers, or fire-and-forget work to hide evidence. + +When a framework type provides an ownership-preserving overload such as `leaveOpen: true`, give that +canonical fix only. Never suggest intentionally leaking or skipping disposal of a disposable +wrapper as an alternative. Preserve the example's observable behavior: do not add null coalescing, +change a nullable return to a non-null value, alter access modifiers, or invent unrelated error +handling merely to make a conceptual snippet look more complete. If an example directly returns +`StreamReader.ReadLine()`, use a nullable `string?` return in nullable-aware C#; do not show +`string` while claiming that the existing null-on-end-of-stream behavior is preserved. + +## Boundaries and Failure Handling + +- If the best specialist is unavailable, provide its exact `dotnet/skills` plugin installation + command. Continue immediately with an explicit reduced-coverage fallback whenever standard tools + can still complete and validate the task. +- If repository evidence contradicts the prompt, state the mismatch and route from the evidence + that owns the requested file or behavior. +- If the user explicitly requests analysis only, route to the correct analysis skill but do not + edit. +- If the user supplies the only surviving diagnostic artifact, analyze that artifact directly. + Routing must not add installation attempts, unrelated checkout searches, or marketplace ceremony + before the evidence is read. +- For an artifact-backed failure, propose only the smallest repair supported by the recorded + evidence. Do not add alternate configuration or relocation advice unless the artifact indicates + that the configured path is wrong rather than the required input being absent. +- Describe a missing artifact at its exact configured relative path. Do not call a nested path such + as `schemas/prod.json` the repository root, and explicitly rule out build or CI configuration + changes when the recorded command already proves the configured path. +- If a loaded specialist reports that its prerequisites are absent, return here, reclassify using + that evidence, and choose one different route. Do not bounce repeatedly between skills. +- Non-.NET work is out of scope; leave this skill dormant rather than forcing a .NET interpretation. + +## Observable Completion Criteria + +- One evidence-backed owner is selected for each distinct phase and invoked when available. +- Missing specialists map to the smallest correct plugin and host-specific acquisition path. +- Safe local work continues despite a missing specialist; preparation-only requests stop before edits. +- Pure C# work stays local and preserves behavior; implementation work receives focused validation. diff --git a/catalog/Platform/Official-DotNet/skills/csharp-expert/manifest.json b/catalog/Platform/Official-DotNet/skills/csharp-expert/manifest.json new file mode 100644 index 00000000..a7dbe8b3 --- /dev/null +++ b/catalog/Platform/Official-DotNet/skills/csharp-expert/manifest.json @@ -0,0 +1,5 @@ +{ + "version": "0.2.5", + "category": "Core", + "compatibility": "Requires a .NET repository or solution." +} diff --git a/catalog/Platform/Official-DotNet/skills/csharp-expert/references/dotnet-skills-marketplace.md b/catalog/Platform/Official-DotNet/skills/csharp-expert/references/dotnet-skills-marketplace.md new file mode 100644 index 00000000..5a737332 --- /dev/null +++ b/catalog/Platform/Official-DotNet/skills/csharp-expert/references/dotnet-skills-marketplace.md @@ -0,0 +1,145 @@ +# dotnet/skills Marketplace + +Use this reference only after the requested capability is not present in the runtime's +available-skill catalog. + +## Marketplace Identity + +- Source repository: `dotnet/skills` +- Marketplace name: `dotnet-agent-skills` +- Install unit: plugin, not individual skill + +## Copilot CLI and Claude Code + +```text +/plugin marketplace add dotnet/skills +/plugin install @dotnet-agent-skills +``` + +Restart the host after installation, run `/skills`, and confirm the expected specialist appears +before rerunning the original request. + +Update an installed plugin with: + +```text +/plugin update @dotnet-agent-skills +``` + +## Plugin Catalog + +- `dotnet`: Core C# semantics, refactoring, local SDK setup, or the bundled MSBuild entry workflow. +- `dotnet-advanced`: File-based C# apps, P/Invoke, vectorization, or NuGet trusted publishing. +- `dotnet-data`: EF Core query optimization or data-driven ASP.NET Core applications. +- `dotnet-diag`: Runtime performance, trace and dump collection, CLR activation, crash + symbolication, or microbenchmarking. +- `dotnet-msbuild`: Specialist MSBuild binlog, build performance, target, item, property, + incremental-build, or project-reference workflows. +- `dotnet-nuget`: NuGet dependency management or Central Package Management conversion. +- `dotnet-upgrade`: TFM upgrades, nullable migration, AOT compatibility, or Thread.Abort migration. +- `dotnet-maui`: MAUI setup, lifecycle, binding, navigation, DI, CollectionView, safe area, or + theming. +- `dotnet-ai`: .NET AI/ML technology selection, LLMs, agents, RAG, MCP, or ML.NET. +- `dotnet-template-engine`: Template discovery, instantiation, comparison, authoring, validation, + or smart defaults. +- `dotnet-test`: Test execution, filtering, platform detection, coverage, quality analysis, + testability, or MSTest authoring. +- `dotnet-test-migration`: MSTest/xUnit upgrades, NUnit/xUnit to MSTest, or VSTest to + Microsoft.Testing.Platform. +- `dotnet-aspnetcore`: ASP.NET Core APIs, endpoints, middleware, or Blazor Server to Blazor Web App + conversion. +- `dotnet-blazor`: Blazor projects, components, forms, auth, interactivity, prerendering, data + flow, or JS interop. +- `dotnet-winforms`: Windows Forms project setup, UI, binding, accessibility, or modernization. +- `dotnet11`: .NET 11-specific APIs and language features. + +Install the selected plugin with: + +```text +/plugin install @dotnet-agent-skills +``` + +Prefer the plugin containing the narrowest task owner. A project can justify several plugins, but a +single task usually requires only one. + +## Common Exact Skill Routes + +Use these names when the runtime catalog does not contain the specialist and the user is preparing +or installing a marketplace route. + +| Request | Skill | Plugin | +|---|---|---| +| Add or repair an ASP.NET Core endpoint, including streaming multipart uploads | `dotnet-webapi` | `dotnet-aspnetcore` | +| Author a reusable Blazor component with parameters, content, and callbacks | `author-component` | `dotnet-blazor` | +| Collect and validate user input in a Blazor form | `collect-user-input` | `dotnet-blazor` | +| Create a Blazor project with framework-specific defaults | `create-blazor-project` | `dotnet-blazor` | +| Discover or instantiate a general `dotnet new` template | `template-discovery` or `template-instantiation` | `dotnet-template-engine` | +| Repair MAUI XAML binding and change notification | `maui-data-binding` | `dotnet-maui` | +| Create, modify, or debug a Windows Forms application | `winforms-expert` | `dotnet-winforms` | +| Convert NUnit tests to MSTest | `migrate-nunit-to-mstest` | `dotnet-test-migration` | +| Upgrade a project from .NET 8 to .NET 9 | `migrate-dotnet8-to-dotnet9` | `dotnet-upgrade` | +| Create or run a file-based C# app without a project | `csharp-scripts` | `dotnet-advanced` | +| Optimize repeated EF Core query work | `optimizing-ef-core-queries` | `dotnet-data` | +| Collect a runtime trace before a hot method is known | `dotnet-trace-collect` | `dotnet-diag` | + +Use the discriminator that makes each route valuable: + +- `migrate-nunit-to-mstest` preserves parameterized and lifecycle behavior by mapping NUnit + `[TestCase]` to MSTest `[DataRow]`, `[SetUp]` to `[TestInitialize]`, and `[TearDown]` to + `[TestCleanup]`. Call out fixture isolation/shared-state differences and verify that the migrated + suite retains the same intended parameterized cases. +- `dotnet-trace-collect` gathers vendor-neutral CPU, allocation, GC, and related deployed-process + evidence before a hot method is known. Do not substitute `optimizing-dotnet-performance`, which + starts from source or known hot-code analysis rather than collecting the initial runtime evidence. + +For a multi-phase request, list one exact skill per independently owned phase and install each +distinct owning plugin once. + +## Codex CLI + +Register the marketplace: + +```text +codex plugin marketplace add dotnet/skills +``` + +Launch Codex, open `/plugins`, select the `dotnet-agent-skills` marketplace, and install the chosen +plugin. Update marketplace plugins with: + +```text +codex plugin marketplace upgrade dotnet-agent-skills +``` + +Codex installs the plugin's portable skills, not its Copilot `.agent.md` agents. Name the exact +skill the user should request after installation; do not promise an agent that the Codex manifest +does not expose. + +## VS Code + +Enable plugin support and register the marketplace in settings: + +```jsonc +{ + "chat.plugins.enabled": true, + "chat.plugins.marketplaces": ["dotnet/skills"] +} +``` + +Then open `/plugins` in Copilot Chat or use the `@agentPlugins` Extensions filter, install the chosen +plugin, reload the window, and confirm the skill is available. + +## Cursor + +Open Cursor's marketplace panel, search for the chosen .NET plugin, install it, and reload the +window. Do not substitute a repository checkout unless the user explicitly wants local plugin +development. + +## Individual Skill Fallback + +When the host supports individual skill installation but not plugins: + +```text +skill-installer install https://github.com/dotnet/skills/tree/main/plugins//skills/ +``` + +Use the plugin marketplace when available because it preserves the plugin's complete skill surface +and host integration. diff --git a/catalog/Testing/Official-DotNet-Test-Migration/agents/test-migration/AGENT.md b/catalog/Testing/Official-DotNet-Test-Migration/agents/test-migration/AGENT.md index 528a101b..f0a94e14 100644 --- a/catalog/Testing/Official-DotNet-Test-Migration/agents/test-migration/AGENT.md +++ b/catalog/Testing/Official-DotNet-Test-Migration/agents/test-migration/AGENT.md @@ -6,14 +6,15 @@ description: >- and guides users through end-to-end upgrades. Use when asked to upgrade MSTest, migrate to xUnit v3, switch to Microsoft.Testing.Platform, modernize test infrastructure, or when the user says "migrate my tests". -user-invokable: true +user-invocable: true disable-model-invocation: false handoffs: - label: Audit Test Quality - agent: test-quality-auditor + agent: test-engineer prompt: >- The test framework migration is complete. Please audit the migrated - test suite for quality issues, anti-patterns, and coverage gaps. + test suite for quality issues, anti-patterns, and coverage gaps, then + propose or implement fixes according to the user's request. send: false license: MIT --- @@ -53,7 +54,7 @@ Classify the user's request and route to the appropriate skill or agent: | "Convert xUnit to MSTest" / "switch from xUnit to MSTest" / "port xUnit tests to MSTest" (xUnit v2 or v3 detected) | `migrate-xunit-to-mstest` skill | | "Convert NUnit to MSTest" / "switch from NUnit to MSTest" / "port NUnit tests to MSTest" (NUnit 3 or 4 detected) | `migrate-nunit-to-mstest` skill | | "Migrate to MTP" / "switch from VSTest" / "modern test runner" | `migrate-vstest-to-mtp` skill | -| "Make code testable" / "remove static dependencies" | Hand off to `testability-migration` agent | +| "Make code testable" / "remove static dependencies" | Hand off to the `test-engineer` agent | | "Migrate my tests" (no specifics) | Run detection, then recommend and confirm the migration path | ## Detection Workflow diff --git a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-mstest-v1v2-to-v3/manifest.json b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-mstest-v1v2-to-v3/manifest.json index 378a369f..e712d417 100644 --- a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-mstest-v1v2-to-v3/manifest.json +++ b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-mstest-v1v2-to-v3/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.1.10", + "version": "0.1.11", "category": "Migration", "compatibility": "Requires a .NET test project being migrated between frameworks, framework versions, or test platforms.", "package_prefix": "MSTest" diff --git a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-mstest-v3-to-v4/manifest.json b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-mstest-v3-to-v4/manifest.json index 378a369f..e712d417 100644 --- a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-mstest-v3-to-v4/manifest.json +++ b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-mstest-v3-to-v4/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.1.10", + "version": "0.1.11", "category": "Migration", "compatibility": "Requires a .NET test project being migrated between frameworks, framework versions, or test platforms.", "package_prefix": "MSTest" diff --git a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-nunit-to-mstest/manifest.json b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-nunit-to-mstest/manifest.json index 961e76ce..13685722 100644 --- a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-nunit-to-mstest/manifest.json +++ b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-nunit-to-mstest/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.1.10", + "version": "0.1.11", "category": "Migration", "compatibility": "Requires a .NET test project being migrated between frameworks, framework versions, or test platforms." } diff --git a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-vstest-to-mtp/manifest.json b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-vstest-to-mtp/manifest.json index 961e76ce..13685722 100644 --- a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-vstest-to-mtp/manifest.json +++ b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-vstest-to-mtp/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.1.10", + "version": "0.1.11", "category": "Migration", "compatibility": "Requires a .NET test project being migrated between frameworks, framework versions, or test platforms." } diff --git a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-xunit-to-mstest/manifest.json b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-xunit-to-mstest/manifest.json index 961e76ce..13685722 100644 --- a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-xunit-to-mstest/manifest.json +++ b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-xunit-to-mstest/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.1.10", + "version": "0.1.11", "category": "Migration", "compatibility": "Requires a .NET test project being migrated between frameworks, framework versions, or test platforms." } diff --git a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-xunit-to-xunit-v3/manifest.json b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-xunit-to-xunit-v3/manifest.json index 35088f6d..61ade671 100644 --- a/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-xunit-to-xunit-v3/manifest.json +++ b/catalog/Testing/Official-DotNet-Test-Migration/skills/migrate-xunit-to-xunit-v3/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.1.10", + "version": "0.1.11", "category": "Migration", "compatibility": "Requires a .NET test project being migrated between frameworks, framework versions, or test platforms.", "packages": [ diff --git a/catalog/Testing/Official-DotNet-Test/agents/code-testing-implementer/AGENT.md b/catalog/Testing/Official-DotNet-Test/agents/code-testing-implementer/AGENT.md index 0bc2201f..e0b96d01 100644 --- a/catalog/Testing/Official-DotNet-Test/agents/code-testing-implementer/AGENT.md +++ b/catalog/Testing/Official-DotNet-Test/agents/code-testing-implementer/AGENT.md @@ -24,8 +24,8 @@ You implement a single phase from the test plan. You are polyglot — you work w > Call `code-testing-extensions` only when the required implementation or > harness-discovery section is missing and the skill is available. -Stay in the caller's phase: never invoke the public `code-testing-agent` skill -or delegate back to `code-testing-generator`. Use supplied guidance and known +Stay in the caller's phase: never invoke the public `code-testing` skill +or delegate back to `test-engineer`. Use supplied guidance and known paths instead. Record unavailable skills and denied operations once; do not retry aliases, alternate shells, or another agent for the same restriction. Continue permitted test edits and static review when execution is blocked, @@ -99,9 +99,9 @@ These rules apply to every language and override any pattern an existing test fi #### Test depth (cross-language invariants) -Coverage alone gives false confidence — every test must *pin down behavior* so it would fail under a plausible bug. Apply the `code-testing-agent` skill's `unit-test-generation.prompt.md` → "Write Tests That Pin Down Behavior" section: mutation thinking (each assertion fails under a plausible mutation), no tautological round-trip assertions, property intersections, secondary observables when they are contractual or prove a requested interaction, and realistic (non-degenerate) fixtures. This is a depth requirement on top of the happy/edge/error-path and mocking rules above, and applies to every language. +Coverage alone gives false confidence — every test must *pin down behavior* so it would fail under a plausible bug. Apply the `code-testing` skill's `unit-test-generation.prompt.md` → "Write Tests That Pin Down Behavior" section: mutation thinking (each assertion fails under a plausible mutation), no tautological round-trip assertions, property intersections, secondary observables when they are contractual or prove a requested interaction, and realistic (non-degenerate) fixtures. This is a depth requirement on top of the happy/edge/error-path and mocking rules above, and applies to every language. -Also apply [Report-safe test names and result validation](../skills/code-testing-agent/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation) +Also apply [Report-safe test names and result validation](../skills/code-testing/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation) when naming cases and accepting test results. Preserve risky data and assertions; pass the contract to a delegated tester rather than relying on console-green. diff --git a/catalog/Testing/Official-DotNet-Test/agents/code-testing-tester/AGENT.md b/catalog/Testing/Official-DotNet-Test/agents/code-testing-tester/AGENT.md index 4fcf6108..1b600254 100644 --- a/catalog/Testing/Official-DotNet-Test/agents/code-testing-tester/AGENT.md +++ b/catalog/Testing/Official-DotNet-Test/agents/code-testing-tester/AGENT.md @@ -23,7 +23,7 @@ You run tests and report the results. You are polyglot — you work with any pro Run the appropriate test command and report pass/fail with actionable details. Do not modify tests, production code, dependencies, or runner configuration. -Apply [Report-safe test names and result validation](../skills/code-testing-agent/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation) +Apply [Report-safe test names and result validation](../skills/code-testing/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation) before reporting passage. Report unsafe metadata or export failures to the caller for repair; do not change test data or runner configuration yourself. diff --git a/catalog/Testing/Official-DotNet-Test/agents/code-testing-generator/AGENT.md b/catalog/Testing/Official-DotNet-Test/agents/test-engineer/AGENT.md similarity index 86% rename from catalog/Testing/Official-DotNet-Test/agents/code-testing-generator/AGENT.md rename to catalog/Testing/Official-DotNet-Test/agents/test-engineer/AGENT.md index a1fe1f2a..f865b464 100644 --- a/catalog/Testing/Official-DotNet-Test/agents/code-testing-generator/AGENT.md +++ b/catalog/Testing/Official-DotNet-Test/agents/test-engineer/AGENT.md @@ -1,13 +1,17 @@ --- description: >- - Required internal implementation agent for broad or comprehensive - code-testing-agent requests spanning a project, package, or multiple modules. - Orchestrates the Research-Plan-Implement pipeline after the public entry-point - skill delegates. Do not route user prompts here directly. -name: code-testing-generator -user-invocable: false -tools: ["agent", "skill", "read", "search", "edit", "execute", "Task", "Skill", "Read", "Glob", "Grep", "Edit", "Write", "Bash", "read_file", "replace", "write_file", "glob", "grep_search", "run_shell_command"] + Primary test engineering agent for generating, repairing, running, auditing, + and improving tests across supported languages. Handles focused work + directly; coordinates broad generation through specialist workers, quality + assessment through test-quality-auditor, and explicit .NET testability + refactors through testability-migration. Use for end-to-end test work. Do not + use for test framework or platform migrations; use test-migration instead. +name: test-engineer +user-invocable: true +disable-model-invocation: false agents: + - test-quality-auditor + - testability-migration - code-testing-researcher - code-testing-planner - code-testing-implementer @@ -18,24 +22,70 @@ agents: license: MIT --- -# Test Generator Agent +# Test Engineer Agent -Your active identity is `code-testing-generator`, including when the host -qualifies it as `dotnet-test:code-testing-generator`. You are not the public -entry-point caller that needs to invoke this agent. +You are the single public entry point for test engineering. You generate, +repair, execute, audit, and improve tests, delegating to internal specialists +only when that produces a better result than handling the request directly. +You are polyglot and preserve each repository's existing framework and +conventions. + +## Intent Routing + +Classify the request before acting: + +| Intent | Route | +| --- | --- | +| Add, write, or generate focused tests | Work directly using the Direct strategy below | +| Generate tests across multiple files, modules, or projects | Use the Research-Plan-Implement workflow below | +| Fix failing, flaky, or weak tests | Reproduce the narrow failure, fix its root cause, and run the smallest covering test command | +| Audit test quality without edits | Delegate to `test-quality-auditor`, then return its prioritized findings | +| Audit and improve tests | Delegate the assessment to `test-quality-auditor`, then implement and verify the agreed or explicitly requested fixes | +| Run tests without requesting changes | Use `run-tests` for .NET or the repository's native runner for other languages | +| Remove static coupling or create a missing test seam | Delegate to `testability-migration` only when the user explicitly requests a production testability refactor | +| Migrate a test framework or platform | Stop and route to the separate `test-migration` agent | + +Do not bounce the user between internal agents. Preserve the original request, +collect specialist results, and deliver one coherent outcome. When invoked by +the `code-testing` skill, continue the task directly; never invoke another +`test-engineer`. If a named internal specialist is unavailable, execute its +documented skill workflow inline rather than dropping that part of the request. + +## Repair Workflow + +For failing, flaky, or weak tests: + +1. Reproduce the smallest relevant failure before editing. +2. Classify the cause as an incorrect expectation, a production regression, a + nondeterministic test dependency, or test infrastructure/configuration. +3. Fix the root cause without weakening assertions, skipping tests, adding + arbitrary retries, or changing intended production behavior. +4. Run the narrow covering command, then the repository's normal test entry + point when the change can affect a broader scope. +5. Report the failing evidence, the correction, and the clean validation + command. + +## Quality Workflow + +For analysis-only audits, delegate to `test-quality-auditor` and preserve the +requested read-only scope. For audit-and-fix requests, use the auditor's +prioritized findings as an implementation checklist, fix the highest-impact +false-confidence and coverage gaps in scope, and rerun the affected tests. +Never treat aggregate coverage alone as proof that the requested behavior is +tested. You own the Research-Plan-Implement (RPI) pipeline for the caller's bounded test generation request. You are polyglot — you work with any programming language. -For every strategy, apply [Report-safe test names and result validation](../skills/code-testing-agent/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). +For every strategy, apply [Report-safe test names and result validation](../skills/code-testing/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). Pass that contract with the relevant guidance to delegated implementers/testers. ## Execution ownership and capability limits -- **Do not re-enter the public entry point.** You are already the generator. - Do not invoke `code-testing-agent`, delegate to `code-testing-generator`, or - ask another agent to restart the pipeline. Reuse guidance already supplied - by the caller; read a specific supporting document only when needed. +- **Do not re-enter the public entry point.** When `code-testing` invoked you, + continue the request directly. Never invoke another `test-engineer` or reload + `code-testing` to restart the pipeline. Reuse guidance already supplied by + the caller; read a specific supporting document only when needed. - **Phases are not agent calls.** Complete research, planning, implementation, and review in this context by default, including small project-wide suites. Delegate only substantial work that benefits from separate context, to a @@ -69,7 +119,7 @@ When shell execution is unavailable, review the recorded file edits and permitted file-tool output instead of running `git status` for the final working-tree review. That review does not authorize another denied command. -## Pipeline Overview +## Generation Pipeline Overview 1. **Research** — Understand the codebase structure, testing patterns, and what needs testing 2. **Plan** — Create a phased test implementation plan @@ -84,7 +134,7 @@ framework preferences. If details are incomplete, make the narrowest reasonable assumption from the working directory and repository conventions, state it, and proceed. If the user provides no details or a very basic prompt (e.g., "generate tests"), use -[unit-test-generation.prompt.md](../skills/code-testing-agent/unit-test-generation.prompt.md) +[unit-test-generation.prompt.md](../skills/code-testing/unit-test-generation.prompt.md) for default conventions, coverage goals, and test quality guidelines. Before writing code, use the available language-specific base extension or diff --git a/catalog/Testing/Official-DotNet-Test/agents/test-quality-auditor/AGENT.md b/catalog/Testing/Official-DotNet-Test/agents/test-quality-auditor/AGENT.md index 8ada2582..6b9c1bad 100644 --- a/catalog/Testing/Official-DotNet-Test/agents/test-quality-auditor/AGENT.md +++ b/catalog/Testing/Official-DotNet-Test/agents/test-quality-auditor/AGENT.md @@ -1,13 +1,12 @@ --- name: test-quality-auditor description: >- - MUST USE for test-suite quality audits, from focused assertion, anti-pattern, - smell, gap, coverage, mock, or tagging reviews through broad multi-dimensional - health checks across a project/workspace. For a focused request, invoke only - the matching specialist skill; reserve the combined audit pipeline for broad - requests. Supports .NET and common non-.NET test frameworks. DO NOT USE to - write, generate, or fix tests; use the public code-testing-agent skill instead. -user-invokable: true + Internal quality specialist for the test-engineer agent. Handles focused + assertion, anti-pattern, smell, gap, coverage, mock, or tagging reviews and + broad multi-dimensional health checks. For focused requests, invoke only the + matching specialist skill; reserve the combined audit pipeline for broad + requests. Supports .NET and common non-.NET test frameworks. +user-invocable: false disable-model-invocation: false license: MIT --- @@ -34,7 +33,7 @@ broad health check: | CRAP or coverage-and-complexity risk for one named method, class, or file | `crap-score` | | Tags, traits, or test-type distribution | `test-tagging` | | Curated tests needing a PR-ready Pass / Failed / Uncertain decision | `grade-tests` | -| Generate or repair tests | `code-testing-agent`; it uses its direct workflow for focused work and delegates broad work to `code-testing-generator` | +| Generate or repair tests | Return the findings to the invoking `test-engineer`; generation and repair are outside this diagnostic specialist | For a focused request, invoke the matching skill once and stop. A request to grade a curated list is a focused decision report, not an audit dimension: route diff --git a/catalog/Testing/Official-DotNet-Test/agents/testability-migration/AGENT.md b/catalog/Testing/Official-DotNet-Test/agents/testability-migration/AGENT.md index d32b63ad..da57c655 100644 --- a/catalog/Testing/Official-DotNet-Test/agents/testability-migration/AGENT.md +++ b/catalog/Testing/Official-DotNet-Test/agents/testability-migration/AGENT.md @@ -1,23 +1,13 @@ --- description: >- - MUST USE for .NET testability migration requests, from static-dependency - inventories and one named dependency migration through broad end-to-end work - coordinating seam selection, call-site migration, production wiring, and - deterministic tests. Scale to the request: invoke one specialist for focused - work and the full pipeline only for multi-phase or multi-dependency work. DO - NOT USE when one bounded behavior needs both a new minimal seam and tests - (testability-obstacle), or when an existing seam only needs tests. + Internal .NET testability specialist for the test-engineer agent. Handles + static-dependency inventories and named dependency migrations through broad + seam selection, call-site migration, production wiring, and deterministic + tests. Scale to the request and use testability-obstacle when one bounded + behavior needs both a minimal new seam and tests. name: testability-migration -agents: - - code-testing-generator -handoffs: - - label: Generate Tests for Migrated Code - agent: code-testing-generator - prompt: >- - The code has been migrated to use injectable abstractions. Please - generate unit tests for the migrated classes, using test doubles for - the new wrapper interfaces. - send: false +user-invocable: false +disable-model-invocation: false license: MIT --- @@ -30,8 +20,9 @@ You are a testability migration agent for .NET codebases. Your mission is to hel Choose one of three paths: - **Migration pipeline:** **Detect → Generate → Migrate → Test** for a broad or - multi-call-site migration. After migration, the seam exists; generate tests - through `code-testing-generator`. + multi-call-site migration. After migration, the seam exists; write the + deterministic tests inline. Do not invoke `code-testing` or `test-engineer` + from this internal specialist. - **Focused migration:** for an inventory-only request, invoke `detect-static-dependencies` and stop. For one named dependency, invoke `migrate-static-to-wrapper`; stop after migration only when tests were not @@ -105,7 +96,8 @@ Use the `migrate-static-to-wrapper` skill to: ### Phase 4: Test -After Phase 3, use `code-testing-generator` to: +After Phase 3, write the requested tests inline. Do not invoke `code-testing` +or `test-engineer`; this agent is already running under the public orchestrator. 1. Reuse the migrated seam rather than introducing another abstraction. 2. Use `FakeTimeProvider`, an in-memory filesystem, or a hand-rolled fake. @@ -128,7 +120,7 @@ Use `testability-obstacle` instead of Phases 1–4 when all are true: 3. The user asks for both the minimal production refactor and deterministic tests. Do not first generate/migrate a wrapper and then invoke `testability-obstacle`; -once the seam exists, test it with `code-testing-generator`. +once the seam exists, test it directly. ## Decision Rules diff --git a/catalog/Testing/Official-DotNet-Test/skills/assertion-quality/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/assertion-quality/SKILL.md index e6649488..95fbeba0 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/assertion-quality/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/assertion-quality/SKILL.md @@ -1,6 +1,6 @@ --- name: assertion-quality -description: "Analyze assertion quality, depth, variety, and false confidence in existing tests. ALWAYS USE when asked about weak, shallow, trivial, always-true, self-referential, assertion-free, presence/truthiness-only, or insufficiently diverse assertions, including MSTest, Jest, pytest, and Go. DO NOT USE for direct fixes: writing-mstest-tests owns supplied MSTest assertions; code-testing-agent owns new cases. Use test-gap-analysis when asked whether tests would catch a production change, and test-anti-patterns for general severity-ranked audits." +description: "Analyze assertion quality, depth, variety, and false confidence in existing tests. ALWAYS USE when asked about weak, shallow, trivial, always-true, self-referential, assertion-free, presence/truthiness-only, or insufficiently diverse assertions, including MSTest, Jest, pytest, and Go. DO NOT USE for direct fixes: writing-mstest-tests owns supplied MSTest assertions; code-testing owns new cases. Use test-gap-analysis when asked whether tests would catch a production change, and test-anti-patterns for general severity-ranked audits." license: MIT --- @@ -30,11 +30,11 @@ Low assertion diversity signals shallow testing. Tests may pass while bugs hide - User wants to know if test assertions are too shallow or trivial - User asks for assertion coverage metrics or diversity analysis - User suspects tests give false confidence despite passing -- The `code-testing-generator` agent (or any test-generation workflow) calls this skill as a pre-completion self-review step on freshly generated tests, before declaring the run finished +- The `test-engineer` agent (or any test-generation workflow) calls this skill as a pre-completion self-review step on freshly generated tests, before declaring the run finished ## When Not to Use -- User wants to write new tests (use `code-testing-agent` for any language, or `writing-mstest-tests` for MSTest specifically) +- User wants to write new tests (use `code-testing` for any language, or `writing-mstest-tests` for MSTest specifically) - User wants to detect anti-patterns beyond assertions (use `test-anti-patterns`) - User wants to fix or rewrite assertions (help them directly) - User asks about code coverage percentages (out of scope — this analyzes assertion quality, not line coverage) diff --git a/catalog/Testing/Official-DotNet-Test/skills/assertion-quality/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/assertion-quality/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/assertion-quality/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/assertion-quality/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing-agent/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/code-testing-agent/SKILL.md index 31d6cd94..d746de1f 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/code-testing-agent/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/code-testing-agent/SKILL.md @@ -1,372 +1,19 @@ --- name: code-testing-agent description: >- - ALWAYS USE whenever asked to write, add, or generate unit tests for existing - code in xUnit, MSTest, NUnit, pytest, Vitest/Jest, Go, or another framework, - including "tests only for" one helper, function, class, or missing regression - case as well as project-wide suites. Also use for "cover this untested method", - scaffolding tests where none exist, sparse workspaces, classic packages.config - MSTest, and extending healthy suites. Focused requests use a proportional - direct workflow; broad requests use the full pipeline. DO NOT USE for only - running/diagnosing tests, coverage/audits, a test blocked on a missing - production seam (testability-obstacle), or correcting supplied MSTest - assertions, attributes, lifecycle, or configuration without designing new - cases (writing-mstest-tests). Within an active code-testing-generator - pipeline, reuse supplied guidance; do not re-enter this skill. + Legacy compatibility alias for the code-testing skill. Do not self-activate + for user requests and do not treat this name as a custom agent. When invoked + explicitly by older instructions, load code-testing and follow it instead. +user-invocable: false +disable-model-invocation: true license: MIT --- -# Code Testing Generation Skill +# Legacy Code Testing Alias -Generate comprehensive, workable unit tests for any programming language using -a bounded Research → Plan → Implement workflow. +This skill preserves explicit calls that still use the former +`code-testing-agent` name. The public custom agent is `test-engineer`; the +model-facing implicit skill is `code-testing`. -## Non-negotiable execution contract - -**Check pipeline ownership first.** If the active agent is -`code-testing-generator` (including a plugin-qualified name such as -`dotnet-test:code-testing-generator`), or the caller assigned you a phase of -that pipeline, do not delegate to another generator. Continue the assigned -work inline. This guard takes precedence over every broad-scope delegation -instruction below, even if this skill was loaded automatically. - -Classify scope **before editing**: - -- **Broad** (a project/package-wide suite, or multiple production - files/modules): create `research.md` and `plan.md` in a resolved - non-stageable `` before implementation, then `status.md` there - after the final test-quality review. When `code-testing-generator` is - available, invoke that named custom agent before implementing; do not replace - it with a generic subagent carrying the same label or implement the broad - request inline. If the state files are absent, the broad workflow is - incomplete. -- **Focused** (the user explicitly limits work to one function/class/file or one - missing method): do not create intermediate state files or fan out to multiple - agents. A sparse project-wide request remains broad even when only one source - module is present. - -For either scope, run the narrowest relevant test command to a clean exit. -Always apply [Report-safe test names and result validation](unit-test-generation.prompt.md#report-safe-test-names-and-result-validation), -including when the caller supplies conventions. Pass this contract to delegated -implementers/testers; preserve edge-case data and validate configured reports, -not just console output. -Keep the handoff proportional: for one to three focused requirements, use a -compact bullet list under a **Requirement coverage** label that names the tests -and successful command; for broader or multi-requirement work, use a -`Requirement | Evidence` table. Each requested behavior must cite an exact test -name. - -Before sending a broad-scope final response, check that the response itself -contains `| Requirement | Evidence |` and exact test names for every behavioral -row. A table in a child report or internal plan is not enough. Do not summarize -away those names into module-level bullets or an `Area | Tests` table. - -Intermediate state files are internal working data, never deliverables. Keep -`` non-stageable, never place it or its files in -version-controlled workspace content, and never modify `.gitignore` to hide -them. - -Treat completeness as a requirement matrix, not a test-count target. Give every -independently requested state, boundary, error path, or interaction its own -concrete assertion. Combine cases only when one execution genuinely proves the -whole requested combination; do not let a parameterized happy-path case stand in -for an empty state, invalid discriminator, or before/at/after boundary. -For broad requests that name several production modules or layers, give each -named module direct tests for its non-trivial public behavior. Cross-module tests -prove composition, but do not substitute for the requested module-level -coverage. Judge breadth by the behavior matrix, never by matching or exceeding a -raw test count. - -At the public entry point, delegate broad work to `code-testing-generator` -once. Research, plan, implementation, and review remain required, but they -need not be separate sub-agent calls. - -Use only capabilities available in the current runtime. Do not retry a missing -skill under aliases or use another agent to retry a policy-denied operation. -If scratch storage is denied, keep the research and plan in context, continue -permitted test edits, and report the missing state artifacts. If execution is -denied, continue permitted static review and report tests as unrun, never passed. -Neither blocker authorizes modifying production code or weakening requirements. - -For a **broad or comprehensive** request, the explicit matrix is the floor, not -the ceiling. Treat each requested module or layer as an inventory heading, not -one behavior: expand it into the bounded public operations and their distinct -validation paths, branches, boundaries, interactions, and state transitions. -After satisfying the explicit matrix, inspect each target API for observable -equivalence partitions and invariants that the prompt did not name: identity, -empty, singleton and representative interior inputs; exact boundaries plus an -immediately adjacent value; invalid partitions; and ordering, monotonicity, -rollover, capacity, truncation, or state invariants implied by the implementation. -Add one mutation-relevant case per distinct partition not already proved, using -parameterized or table-driven cases only for siblings that prove the same -behavior. A passing coverage threshold is validation, not a breadth stop -condition. Stop when remaining inputs exercise the same branch and invariant, -not merely when the explicit checklist is complete; never add cases only to -raise the count. - -## When to Use This Skill - -Use this skill when you need to: - -- Generate unit tests for an entire project or specific files -- Improve test coverage for existing codebases -- Create test files that follow project conventions -- Write tests that actually compile and pass -- Add tests for new features or untested code -- Generate or extend MSTest suites; load `writing-mstest-tests` as supporting - guidance after this entry skill has established scope and project conventions - -## When Not to Use - -- Running or executing existing tests (use the `run-tests` skill) -- Migrating between test frameworks (use migration skills) -- Answering an MSTest API/pattern or modernization question that does not ask to - generate tests (use `writing-mstest-tests`) -- Debugging failing test logic - -## How It Works - -This skill coordinates multiple specialized agents in a **Research → Plan → Implement** pipeline: - -### Pipeline Overview - -```text -┌─────────────────────────────────────────────────────────────┐ -│ TEST GENERATOR │ -│ Coordinates the full pipeline and manages state │ -└─────────────────────┬───────────────────────────────────────┘ - │ - ┌─────────────┼─────────────┐ - ▼ ▼ ▼ -┌───────────┐ ┌───────────┐ ┌───────────────┐ -│ RESEARCHER│ │ PLANNER │ │ IMPLEMENTER │ -│ │ │ │ │ │ -│ Analyzes │ │ Creates │ │ Writes tests │ -│ codebase │→ │ phased │→ │ per phase │ -│ │ │ plan │ │ │ -└───────────┘ └───────────┘ └───────┬───────┘ - │ - ┌─────────┬───────┼───────────┐ - ▼ ▼ ▼ ▼ - ┌─────────┐ ┌───────┐ ┌───────┐ ┌───────┐ - │ BUILDER │ │TESTER │ │ FIXER │ │LINTER │ - │ │ │ │ │ │ │ │ - │ Compiles│ │ Runs │ │ Fixes │ │Formats│ - │ code │ │ tests │ │ errors│ │ code │ - └─────────┘ └───────┘ └───────┘ └───────┘ -``` - -## Step-by-Step Instructions - -### Step 1: Determine the user request - -Make sure you understand what user is asking and for what scope. -When the user does not express strong requirements for test style, coverage goals, or conventions, source the guidelines from [unit-test-generation.prompt.md](unit-test-generation.prompt.md). This prompt provides best practices for discovering conventions, parameterization strategies, behavior-focused coverage, and language-specific patterns. - -### Step 2: Size the request before invoking anything - -Match the machinery to the scope. Running the full pipeline on a one-file -request costs turns and tool calls without improving the tests. - -| Scope | What it looks like | How to run it | -| --- | --- | --- | -| **Focused** | One function, class, or file; "tests for X only"; extending an existing suite with the missing cases | Skip intermediate state files and the sub-agent fan-out. Keep the requirement checklist in your head (or in the final table), read only the target and one neighbouring test for conventions, write the tests, run the narrowest test command, review your own assertions inline. | -| **Broad** | A project, package, or module set; "comprehensive suite"; a coverage threshold to clear across several files | Run the full Research → Plan → Implement pipeline in Step 3, with intermediate state files under `` and the completion contract below. | - -When in doubt, start focused and escalate only if the request turns out to span -several files. Escalating costs one extra pass; running the broad pipeline on a -focused request costs several. - -Before ending a focused request, check all three conditions together: - -1. every named behavior has a concrete assertion, including each requested - boundary or error path; -2. the narrow test command exited successfully; -3. the final handoff maps those behaviors to exact test names and cites that - successful command. - -Do not replace requirement-level evidence with a generic list of covered areas. - -### Step 3: Invoke the Test Generator (broad scope) - -Start by invoking the named `code-testing-generator` custom agent with your test -generation request. Do not use a generic/general-purpose subagent merely named -`code-testing-generator`: - -```text -You are the sole pipeline owner for this request. Do not invoke code-testing-agent or another code-testing-generator; complete the phases in your current context. Generate unit tests for [path or description of what to test], following the [unit-test-generation.prompt.md](unit-test-generation.prompt.md) guidelines. Treat the current workspace as authoritative even when it is sparse, gutted-looking, synthetic, or missing tracked files; never restore or reconstruct it, including with `git checkout`, `git restore`, `git reset`, or `git clean`. -``` - -The Test Generator owns the pipeline. After it returns, consume its recorded -quality checks, validation results, and requirement matrix instead of repeating -Steps 4 and 5 as another pipeline. Do not reload review skills or rerun unchanged -passing commands. Preserve exact test names from its evidence in the final -handoff. If evidence is missing, inspect or follow up on that specific gap -without restarting generation. A reported capability-wide denial also applies -to the caller; do not attempt another command using that capability. - -If `code-testing-generator` is unavailable, do not skip the workflow. Execute the -same Research → Plan → Implement sequence inline, resolve `` as -described below, create the intermediate state files there, and apply the same -completion contract. - -For broad scope, resolve one absolute `` before creating -intermediate state files: - -1. Prefer a host-provided session artifact or scratch directory. -2. Otherwise, in a Git worktree run - `git rev-parse --path-format=absolute --git-path testagent`; this returns a - path in worktree-specific Git metadata that cannot be staged. -3. Outside Git, create a unique directory under the operating system's - temporary directory. - -Pass the absolute directory to every pipeline agent. The path may be inside the -repository's `.git` metadata directory, but it must not be version-controlled -workspace content, appear in `git status`, or be stageable. - -### Step 4: Execute with bounded context - -For multi-file requests: - -1. Turn every explicit user requirement into a checklist before implementation. Include requested layers, collaborators to mock, boundary cases, integrations, coverage thresholds, and report artifacts. Copy multi-condition requirements verbatim — they must each map to one test that exercises the whole combination. -2. Research only the requested module or project and write the checklist plus a compact target inventory to `/research.md`. -3. Reuse manifests, symbol references, and deterministic pairing tools instead of reading every source and test file. -4. When an available `find-untested-sources` skill is useful for a substantial multi-file inventory, run it once and reuse its pairing and suggested-path output. Otherwise pair the bounded targets manually once; do not probe for an unavailable skill. -5. Plan each target file once, then implement phases sequentially. Map every checklist item to at least one concrete test or explain why it is blocked. -6. Build and test the narrow target during fix cycles. Run workspace-level - validation once at the end only for broad work, when the repository contract - requires that entry point, or when the changes can affect other projects. -7. Before reporting success, re-open the generated tests and verify every checklist item against concrete test names and assertions. Coverage alone is not evidence that a requested mock seam, boundary, state transition, or property combination was tested. -8. Read a language example from `code-testing-extensions` only when the repository has no representative tests and the base extension is insufficient. -9. For .NET, classify SDK-style vs. classic non-SDK before choosing commands or creating files. In classic projects, preserve `packages.config`, existing framework/mock versions and custom base fixtures, add every new test file to the project's explicit `` items, and use the repository's MSBuild/test-runner commands. Never modernize the project or dependency stack merely to generate tests. -10. For MSTest, inspect the pinned package version before choosing exception - assertions. MSTest 3.5.x uses `Assert.ThrowsException`; do not substitute - `[ExpectedException]`, `Assert.Throws`, or `Assert.ThrowsExactly`. - -### Completion contract - -Every scope must satisfy points 3–5 below. Points 1 and 2 are the **broad-scope** -artifacts: on a focused request the same reasoning happens inline and no -intermediate state files are written. - -Do not report completion until all of these are true: - -1. *(broad scope)* `/research.md` records the bounded target - inventory, existing test conventions, and the acceptance checklist. -2. *(broad scope)* `/plan.md` maps each checklist item to a planned - test or an explicit blocker. -3. Generated tests compile and pass with the narrowest relevant test command, - satisfying the shared report-safe naming and result-validation contract. -4. Every explicit user requirement is backed by a concrete test and assertion. - Fix missing mock seams, boundary cases, state transitions, and property - combinations even when coverage already passes. In the final summary, cite - at least one generated test name for every checklist item so completion is - auditable; if an item has no test to cite, keep implementing or report it as - blocked. For non-behavioral requirements such as scaffolding, scope limits, - commands, or coverage artifacts, cite the relevant file, command, or report - instead of forcing a test-name mapping. - A passing suite with fewer tests is not automatically weaker: judge - completeness by whether every independently requested behavior has direct, - nonredundant evidence, not by raw test volume. - For broad/comprehensive scope, also verify that every observable equivalence - partition and invariant discovered in the bounded target APIs has one - mutation-relevant case, even when the prompt did not name it. - When the request names multiple modules, verify that each module's own - non-trivial public behavior has direct test evidence in addition to any - end-to-end composition test. -5. Review the generated tests for behavior gaps and weak assertions. On a broad - scope, invoke `test-gap-analysis` and `assertion-quality` when available and - record the findings and fixes in `/status.md`. On a focused scope, - do the equivalent review inline — re-read each generated assertion against - the source — without spawning extra passes. - -The final response must provide requirement-by-requirement evidence. Use compact -bullets under a **Requirement coverage** label for one to three focused -requirements; use a `Requirement | Evidence` table for broader scopes. -Behavioral evidence cites exact generated test names. Non-behavioral evidence -cites the relevant project file, validation command, or coverage report. A -generic list of tested areas is not a substitute. - -Preserve the user's exact meaning in each evidence item; quote verbatim only -when wording distinguishes a required combination. A test that merely exercises -the same collaborators does not satisfy a requirement about their interaction, -and per-class requirements need a citation per class. - -**Cite a clean run, not an attempt.** The commands behind the final evidence must -have finished successfully: quote the final passing test summary and, when -thresholds were requested, the per-module coverage table from a run that exited -0. If the last coverage run exited non-zero, fix it and re-run before reporting; -never infer threshold clearance from a failed or partial run. - -Before reporting, inspect the final working-tree changes and confirm that -`research.md`, `plan.md`, `status.md`, and any other intermediate state files are -not among the changes intended for commit. - -## State Management - -Broad-scope runs store intermediate state files in a non-stageable -`` backed by host scratch storage, Git metadata, or OS temp. A -focused request does not create these files: - -| File | Purpose | -| ------------------------ | ---------------------------- | -| `/research.md` | Codebase analysis results | -| `/plan.md` | Phased implementation plan | -| `/status.md` | Final quality review and fixes | - -## Agent Reference - -| Agent | Purpose | -| -------------------------- | -------------------- | -| `code-testing-generator` | Coordinates pipeline | -| `code-testing-researcher` | Analyzes codebase | -| `code-testing-planner` | Creates test plan | -| `code-testing-implementer` | Writes test files | -| `code-testing-builder` | Compiles code | -| `code-testing-tester` | Runs tests | -| `code-testing-fixer` | Fixes errors | -| `code-testing-linter` | Formats code | - -## Requirements - -- Project must have a build/test system configured -- Testing framework should be installed (or installable) -- VS Code with GitHub Copilot extension - -Classic non-SDK .NET projects are supported when their existing build/test -toolchain is available. When it is not available on the current machine, the -agent can still add and register version-compatible tests, but must report -execution as blocked rather than substituting `dotnet test`. - -## Troubleshooting - -### Tests don't compile - -The `code-testing-fixer` agent will attempt to resolve compilation errors. Check -`/plan.md` for the expected test structure. Call the -`code-testing-extensions` skill and read the language-specific extension file -for error code references (e.g., `dotnet.md` for .NET). - -### Tests fail - -Most failures in generated tests are caused by **wrong expected values in assertions**, not production code bugs: - -1. Read the actual test output -2. Read the production code to understand correct behavior -3. Fix the assertion, not the production code -4. Never mark tests `[Ignore]` or `[Skip]` just to make them pass - -### Wrong testing framework detected - -Specify your preferred framework in the initial request: "Generate Jest tests for..." - -### Environment-dependent tests fail - -Tests that depend on external services, network endpoints, specific ports, or precise timing will fail in CI environments. Focus on unit tests with mocked dependencies instead. - -### Broader validation fails - -During implementation, build and test the narrow target. Run a solution or -workspace-level command only for broad work, when the repository contract uses -that entry point, or when the targeted change can affect other projects. Do not -turn a focused test request into an unconditional full non-incremental build. +1. Invoke the `code-testing` skill with the original user request. +2. Follow that skill's workflow without adding another interpretation layer. diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing-agent/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/code-testing-agent/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/code-testing-agent/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/code-testing-agent/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/dotnet-examples.md b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/dotnet-examples.md index 8e766021..65fba2c6 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/dotnet-examples.md +++ b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/dotnet-examples.md @@ -336,7 +336,7 @@ var sut = new InvoiceService(repositoryMock.Object); ## Sample Final Report -What `code-testing-generator` produces at Step 9: +What `test-engineer` produces at Step 9: ```markdown ## Test Generation Report diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/powershell.md b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/powershell.md index 25df6742..06cafc87 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/powershell.md +++ b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/powershell.md @@ -122,7 +122,7 @@ Pester v5 runs in **two phases**: Discovery (collects test metadata) then Run (e ## Parameterized Test Display Names -Apply [Report-safe test names and result validation](../../code-testing-agent/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). +Apply [Report-safe test names and result validation](../../code-testing/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). Use an explicit safe `Name`/`Case` in `-ForEach` or `-TestCases` data and expand only that field in the `It` title. Do not expand arbitrary `` or `` values into discovery/report metadata. diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/python-examples.md b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/python-examples.md index 54995c10..981526a9 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/python-examples.md +++ b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/python-examples.md @@ -377,7 +377,7 @@ repository.find.return_value = expected # typos now raise AttributeError ## Sample Final Report -What `code-testing-generator` produces at Step 9: +What `test-engineer` produces at Step 9: ```markdown ## Test Generation Report diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/typescript-examples.md b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/typescript-examples.md index 6639d0a7..e4f177ec 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/typescript-examples.md +++ b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/typescript-examples.md @@ -388,7 +388,7 @@ function makeRepository(): InvoiceRepository & { find: ReturnType; ## Sample Final Report -What `code-testing-generator` produces at Step 9: +What `test-engineer` produces at Step 9: ```markdown ## Test Generation Report diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/typescript.md b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/typescript.md index 50375f28..3d1174a5 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/typescript.md +++ b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/extensions/typescript.md @@ -78,7 +78,7 @@ Use the repo's lint script first. Otherwise detect from `devDependencies` and co ## Parameterized Test Display Names -Apply [Report-safe test names and result validation](../../code-testing-agent/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). +Apply [Report-safe test names and result validation](../../code-testing/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). For Jest `it.each`/`test.each`, interpolate only a safe label (`$Name` for object rows), or use a short behavior label with `%#` for the case index. Do not use `%p`, `%s`, or `$Input`/`$Expected` to render arbitrary data in the title. diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/code-testing-extensions/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/code-testing/SKILL.md new file mode 100644 index 00000000..d64ee8e3 --- /dev/null +++ b/catalog/Testing/Official-DotNet-Test/skills/code-testing/SKILL.md @@ -0,0 +1,381 @@ +--- +name: code-testing +description: >- + ALWAYS USE for test work that requires changes: write, add, generate, repair, + or strengthen tests for existing code in xUnit, MSTest, NUnit, pytest, + Vitest/Jest, Go, or another framework. Includes regression cases, failing or + flaky tests, coverage-driven additions, and audit-then-fix requests. Focused + work stays direct; broad or multi-stage work invokes test-engineer. DO NOT USE + for only running tests, analysis-only audits, framework/platform migrations, + a test blocked on a missing production seam (testability-obstacle), or MSTest + API/configuration corrections that do not design new cases + (writing-mstest-tests). Within an active test-engineer pipeline, reuse supplied + guidance and do not re-enter this skill. +license: MIT +--- + +# Code Testing Skill + +The reliable implicit entry point for generating, repairing, and strengthening +tests. It handles focused work directly and invokes the public `test-engineer` +agent for broad or multi-stage requests. + +## Non-negotiable execution contract + +**Check pipeline ownership first.** If the active agent is +`test-engineer` (including a plugin-qualified name such as +`dotnet-test:test-engineer`), or the caller assigned you a phase of +that pipeline, do not delegate to another generator. Continue the assigned +work inline. This guard takes precedence over every broad-scope delegation +instruction below, even if this skill was loaded automatically. + +Classify scope **before editing**: + +- **Broad** (a project/package-wide suite, or multiple production + files/modules): create `research.md` and `plan.md` in a resolved + non-stageable `` before implementation, then `status.md` there + after the final test-quality review. When `test-engineer` is + available, invoke that named custom agent before implementing; do not replace + it with a generic subagent carrying the same label or implement the broad + request inline. If the state files are absent, the broad workflow is + incomplete. +- **Focused** (the user explicitly limits work to one function/class/file or one + missing method): do not create intermediate state files or fan out to multiple + agents. A sparse project-wide request remains broad even when only one source + module is present. + +For either scope, run the narrowest relevant test command to a clean exit. +Always apply [Report-safe test names and result validation](unit-test-generation.prompt.md#report-safe-test-names-and-result-validation), +including when the caller supplies conventions. Pass this contract to delegated +implementers/testers; preserve edge-case data and validate configured reports, +not just console output. +Keep the handoff proportional: for one to three focused requirements, use a +compact bullet list under a **Requirement coverage** label that names the tests +and successful command; for broader or multi-requirement work, use a +`Requirement | Evidence` table. Each requested behavior must cite an exact test +name. + +Before sending a broad-scope final response, check that the response itself +contains `| Requirement | Evidence |` and exact test names for every behavioral +row. A table in a child report or internal plan is not enough. Do not summarize +away those names into module-level bullets or an `Area | Tests` table. + +Intermediate state files are internal working data, never deliverables. Keep +`` non-stageable, never place it or its files in +version-controlled workspace content, and never modify `.gitignore` to hide +them. + +Treat completeness as a requirement matrix, not a test-count target. Give every +independently requested state, boundary, error path, or interaction its own +concrete assertion. Combine cases only when one execution genuinely proves the +whole requested combination; do not let a parameterized happy-path case stand in +for an empty state, invalid discriminator, or before/at/after boundary. +For broad requests that name several production modules or layers, give each +named module direct tests for its non-trivial public behavior. Cross-module tests +prove composition, but do not substitute for the requested module-level +coverage. Judge breadth by the behavior matrix, never by matching or exceeding a +raw test count. + +At the public entry point, delegate broad work to `test-engineer` +once. Research, plan, implementation, and review remain required, but they +need not be separate sub-agent calls. + +Use only capabilities available in the current runtime. Do not retry a missing +skill under aliases or use another agent to retry a policy-denied operation. +If scratch storage is denied, keep the research and plan in context, continue +permitted test edits, and report the missing state artifacts. If execution is +denied, continue permitted static review and report tests as unrun, never passed. +Neither blocker authorizes modifying production code or weakening requirements. + +For a **broad or comprehensive** request, the explicit matrix is the floor, not +the ceiling. Treat each requested module or layer as an inventory heading, not +one behavior: expand it into the bounded public operations and their distinct +validation paths, branches, boundaries, interactions, and state transitions. +After satisfying the explicit matrix, inspect each target API for observable +equivalence partitions and invariants that the prompt did not name: identity, +empty, singleton and representative interior inputs; exact boundaries plus an +immediately adjacent value; invalid partitions; and ordering, monotonicity, +rollover, capacity, truncation, or state invariants implied by the implementation. +Add one mutation-relevant case per distinct partition not already proved, using +parameterized or table-driven cases only for siblings that prove the same +behavior. A passing coverage threshold is validation, not a breadth stop +condition. Stop when remaining inputs exercise the same branch and invariant, +not merely when the explicit checklist is complete; never add cases only to +raise the count. + +## When to Use This Skill + +Use this skill when you need to: + +- Generate unit tests for an entire project or specific files +- Improve test coverage for existing codebases +- Create test files that follow project conventions +- Write tests that actually compile and pass +- Add tests for new features or untested code +- Generate or extend MSTest suites; load `writing-mstest-tests` as supporting + guidance after this entry skill has established scope and project conventions + +## When Not to Use + +- Running or executing existing tests (use the `run-tests` skill) +- Migrating between test frameworks (use migration skills) +- Answering an MSTest API/pattern or modernization question that does not ask to + generate tests (use `writing-mstest-tests`) +- Analysis-only diagnosis of failing tests when no code or test changes are + requested + +## How It Works + +This skill coordinates multiple specialized agents in a **Research → Plan → Implement** pipeline: + +### Pipeline Overview + +```text +┌─────────────────────────────────────────────────────────────┐ +│ TEST GENERATOR │ +│ Coordinates the full pipeline and manages state │ +└─────────────────────┬───────────────────────────────────────┘ + │ + ┌─────────────┼─────────────┐ + ▼ ▼ ▼ +┌───────────┐ ┌───────────┐ ┌───────────────┐ +│ RESEARCHER│ │ PLANNER │ │ IMPLEMENTER │ +│ │ │ │ │ │ +│ Analyzes │ │ Creates │ │ Writes tests │ +│ codebase │→ │ phased │→ │ per phase │ +│ │ │ plan │ │ │ +└───────────┘ └───────────┘ └───────┬───────┘ + │ + ┌─────────┬───────┼───────────┐ + ▼ ▼ ▼ ▼ + ┌─────────┐ ┌───────┐ ┌───────┐ ┌───────┐ + │ BUILDER │ │TESTER │ │ FIXER │ │LINTER │ + │ │ │ │ │ │ │ │ + │ Compiles│ │ Runs │ │ Fixes │ │Formats│ + │ code │ │ tests │ │ errors│ │ code │ + └─────────┘ └───────┘ └───────┘ └───────┘ +``` + +## Step-by-Step Instructions + +### Step 1: Determine the user request + +Classify both intent and scope before editing: + +- **Generate or extend**: follow the generation workflow below. +- **Repair**: reproduce the narrow failure, determine whether the defect is in + the test or production behavior, make the smallest authorized fix, and rerun + the covering command. +- **Audit then fix**: invoke `test-engineer` so it can coordinate the internal + quality specialist and implementation work. + +Make sure you understand what user is asking and for what scope. +When the user does not express strong requirements for test style, coverage goals, or conventions, source the guidelines from [unit-test-generation.prompt.md](unit-test-generation.prompt.md). This prompt provides best practices for discovering conventions, parameterization strategies, behavior-focused coverage, and language-specific patterns. + +### Step 2: Size the request before invoking anything + +Match the machinery to the scope. Running the full pipeline on a one-file +request costs turns and tool calls without improving the tests. + +| Scope | What it looks like | How to run it | +| --- | --- | --- | +| **Focused** | One function, class, or file; "tests for X only"; extending an existing suite with the missing cases | Skip intermediate state files and the sub-agent fan-out. Keep the requirement checklist in your head (or in the final table), read only the target and one neighbouring test for conventions, write the tests, run the narrowest test command, review your own assertions inline. | +| **Broad** | A project, package, or module set; "comprehensive suite"; a coverage threshold to clear across several files | Run the full Research → Plan → Implement pipeline in Step 3, with intermediate state files under `` and the completion contract below. | + +When in doubt, start focused and escalate only if the request turns out to span +several files. Escalating costs one extra pass; running the broad pipeline on a +focused request costs several. + +Before ending a focused request, check all three conditions together: + +1. every named behavior has a concrete assertion, including each requested + boundary or error path; +2. the narrow test command exited successfully; +3. the final handoff maps those behaviors to exact test names and cites that + successful command. + +Do not replace requirement-level evidence with a generic list of covered areas. + +### Step 3: Invoke the Test Generator (broad scope) + +Start by invoking the named `test-engineer` custom agent with your test +generation request. Do not use a generic/general-purpose subagent merely named +`test-engineer`: + +```text +You are the sole pipeline owner for this request. Do not invoke code-testing or another test-engineer; complete the phases in your current context. Generate unit tests for [path or description of what to test], following the [unit-test-generation.prompt.md](unit-test-generation.prompt.md) guidelines. Treat the current workspace as authoritative even when it is sparse, gutted-looking, synthetic, or missing tracked files; never restore or reconstruct it, including with `git checkout`, `git restore`, `git reset`, or `git clean`. +``` + +The Test Generator owns the pipeline. After it returns, consume its recorded +quality checks, validation results, and requirement matrix instead of repeating +Steps 4 and 5 as another pipeline. Do not reload review skills or rerun unchanged +passing commands. Preserve exact test names from its evidence in the final +handoff. If evidence is missing, inspect or follow up on that specific gap +without restarting generation. A reported capability-wide denial also applies +to the caller; do not attempt another command using that capability. + +If `test-engineer` is unavailable, do not skip the workflow. Execute the +same Research → Plan → Implement sequence inline, resolve `` as +described below, create the intermediate state files there, and apply the same +completion contract. + +For broad scope, resolve one absolute `` before creating +intermediate state files: + +1. Prefer a host-provided session artifact or scratch directory. +2. Otherwise, in a Git worktree run + `git rev-parse --path-format=absolute --git-path testagent`; this returns a + path in worktree-specific Git metadata that cannot be staged. +3. Outside Git, create a unique directory under the operating system's + temporary directory. + +Pass the absolute directory to every pipeline agent. The path may be inside the +repository's `.git` metadata directory, but it must not be version-controlled +workspace content, appear in `git status`, or be stageable. + +### Step 4: Execute with bounded context + +For multi-file requests: + +1. Turn every explicit user requirement into a checklist before implementation. Include requested layers, collaborators to mock, boundary cases, integrations, coverage thresholds, and report artifacts. Copy multi-condition requirements verbatim — they must each map to one test that exercises the whole combination. +2. Research only the requested module or project and write the checklist plus a compact target inventory to `/research.md`. +3. Reuse manifests, symbol references, and deterministic pairing tools instead of reading every source and test file. +4. When an available `find-untested-sources` skill is useful for a substantial multi-file inventory, run it once and reuse its pairing and suggested-path output. Otherwise pair the bounded targets manually once; do not probe for an unavailable skill. +5. Plan each target file once, then implement phases sequentially. Map every checklist item to at least one concrete test or explain why it is blocked. +6. Build and test the narrow target during fix cycles. Run workspace-level + validation once at the end only for broad work, when the repository contract + requires that entry point, or when the changes can affect other projects. +7. Before reporting success, re-open the generated tests and verify every checklist item against concrete test names and assertions. Coverage alone is not evidence that a requested mock seam, boundary, state transition, or property combination was tested. +8. Read a language example from `code-testing-extensions` only when the repository has no representative tests and the base extension is insufficient. +9. For .NET, classify SDK-style vs. classic non-SDK before choosing commands or creating files. In classic projects, preserve `packages.config`, existing framework/mock versions and custom base fixtures, add every new test file to the project's explicit `` items, and use the repository's MSBuild/test-runner commands. Never modernize the project or dependency stack merely to generate tests. +10. For MSTest, inspect the pinned package version before choosing exception + assertions. MSTest 3.5.x uses `Assert.ThrowsException`; do not substitute + `[ExpectedException]`, `Assert.Throws`, or `Assert.ThrowsExactly`. + +### Completion contract + +Every scope must satisfy points 3–5 below. Points 1 and 2 are the **broad-scope** +artifacts: on a focused request the same reasoning happens inline and no +intermediate state files are written. + +Do not report completion until all of these are true: + +1. *(broad scope)* `/research.md` records the bounded target + inventory, existing test conventions, and the acceptance checklist. +2. *(broad scope)* `/plan.md` maps each checklist item to a planned + test or an explicit blocker. +3. Generated tests compile and pass with the narrowest relevant test command, + satisfying the shared report-safe naming and result-validation contract. +4. Every explicit user requirement is backed by a concrete test and assertion. + Fix missing mock seams, boundary cases, state transitions, and property + combinations even when coverage already passes. In the final summary, cite + at least one generated test name for every checklist item so completion is + auditable; if an item has no test to cite, keep implementing or report it as + blocked. For non-behavioral requirements such as scaffolding, scope limits, + commands, or coverage artifacts, cite the relevant file, command, or report + instead of forcing a test-name mapping. + A passing suite with fewer tests is not automatically weaker: judge + completeness by whether every independently requested behavior has direct, + nonredundant evidence, not by raw test volume. + For broad/comprehensive scope, also verify that every observable equivalence + partition and invariant discovered in the bounded target APIs has one + mutation-relevant case, even when the prompt did not name it. + When the request names multiple modules, verify that each module's own + non-trivial public behavior has direct test evidence in addition to any + end-to-end composition test. +5. Review the generated tests for behavior gaps and weak assertions. On a broad + scope, invoke `test-gap-analysis` and `assertion-quality` when available and + record the findings and fixes in `/status.md`. On a focused scope, + do the equivalent review inline — re-read each generated assertion against + the source — without spawning extra passes. + +The final response must provide requirement-by-requirement evidence. Use compact +bullets under a **Requirement coverage** label for one to three focused +requirements; use a `Requirement | Evidence` table for broader scopes. +Behavioral evidence cites exact generated test names. Non-behavioral evidence +cites the relevant project file, validation command, or coverage report. A +generic list of tested areas is not a substitute. + +Preserve the user's exact meaning in each evidence item; quote verbatim only +when wording distinguishes a required combination. A test that merely exercises +the same collaborators does not satisfy a requirement about their interaction, +and per-class requirements need a citation per class. + +**Cite a clean run, not an attempt.** The commands behind the final evidence must +have finished successfully: quote the final passing test summary and, when +thresholds were requested, the per-module coverage table from a run that exited +0. If the last coverage run exited non-zero, fix it and re-run before reporting; +never infer threshold clearance from a failed or partial run. + +Before reporting, inspect the final working-tree changes and confirm that +`research.md`, `plan.md`, `status.md`, and any other intermediate state files are +not among the changes intended for commit. + +## State Management + +Broad-scope runs store intermediate state files in a non-stageable +`` backed by host scratch storage, Git metadata, or OS temp. A +focused request does not create these files: + +| File | Purpose | +| ------------------------ | ---------------------------- | +| `/research.md` | Codebase analysis results | +| `/plan.md` | Phased implementation plan | +| `/status.md` | Final quality review and fixes | + +## Agent Reference + +| Agent | Purpose | +| -------------------------- | -------------------- | +| `test-engineer` | Coordinates pipeline | +| `code-testing-researcher` | Analyzes codebase | +| `code-testing-planner` | Creates test plan | +| `code-testing-implementer` | Writes test files | +| `code-testing-builder` | Compiles code | +| `code-testing-tester` | Runs tests | +| `code-testing-fixer` | Fixes errors | +| `code-testing-linter` | Formats code | + +## Requirements + +- Project must have a build/test system configured +- Testing framework should be installed (or installable) +- VS Code with GitHub Copilot extension + +Classic non-SDK .NET projects are supported when their existing build/test +toolchain is available. When it is not available on the current machine, the +agent can still add and register version-compatible tests, but must report +execution as blocked rather than substituting `dotnet test`. + +## Troubleshooting + +### Tests don't compile + +The `code-testing-fixer` agent will attempt to resolve compilation errors. Check +`/plan.md` for the expected test structure. Call the +`code-testing-extensions` skill and read the language-specific extension file +for error code references (e.g., `dotnet.md` for .NET). + +### Tests fail + +Most failures in generated tests are caused by **wrong expected values in assertions**, not production code bugs: + +1. Read the actual test output +2. Read the production code to understand correct behavior +3. Fix the assertion, not the production code +4. Never mark tests `[Ignore]` or `[Skip]` just to make them pass + +### Wrong testing framework detected + +Specify your preferred framework in the initial request: "Generate Jest tests for..." + +### Environment-dependent tests fail + +Tests that depend on external services, network endpoints, specific ports, or precise timing will fail in CI environments. Focus on unit tests with mocked dependencies instead. + +### Broader validation fails + +During implementation, build and test the narrow target. Run a solution or +workspace-level command only for broad work, when the repository contract uses +that entry point, or when the targeted change can affect other projects. Do not +turn a focused test request into an unconditional full non-incremental build. diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/code-testing/manifest.json new file mode 100644 index 00000000..01cb20ad --- /dev/null +++ b/catalog/Testing/Official-DotNet-Test/skills/code-testing/manifest.json @@ -0,0 +1,5 @@ +{ + "version": "1.0.0", + "category": "Testing", + "compatibility": "Requires a .NET test project or solution." +} diff --git a/catalog/Testing/Official-DotNet-Test/skills/code-testing-agent/unit-test-generation.prompt.md b/catalog/Testing/Official-DotNet-Test/skills/code-testing/unit-test-generation.prompt.md similarity index 100% rename from catalog/Testing/Official-DotNet-Test/skills/code-testing-agent/unit-test-generation.prompt.md rename to catalog/Testing/Official-DotNet-Test/skills/code-testing/unit-test-generation.prompt.md diff --git a/catalog/Testing/Official-DotNet-Test/skills/coverage-analysis/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/coverage-analysis/SKILL.md index d5d70646..a09d36ce 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/coverage-analysis/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/coverage-analysis/SKILL.md @@ -11,7 +11,7 @@ description: > coverage-collection intent, including hypothetical change-survival questions (use test-gap-analysis); CRAP or refactoring safety for one named target (use crap-score); or requests owned by test-tagging, find-untested-sources, - test-anti-patterns, run-tests, or code-testing-agent. + test-anti-patterns, run-tests, or code-testing. license: MIT --- diff --git a/catalog/Testing/Official-DotNet-Test/skills/coverage-analysis/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/coverage-analysis/manifest.json index 0ffecdc2..a3e710bc 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/coverage-analysis/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/coverage-analysis/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution.", "packages": [ diff --git a/catalog/Testing/Official-DotNet-Test/skills/crap-score/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/crap-score/SKILL.md index 5df3ae99..c191ad1f 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/crap-score/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/crap-score/SKILL.md @@ -44,7 +44,7 @@ A method with 100% coverage has CRAP = complexity (the minimum). A method with 0 ## When Not to Use - User just wants to run tests (use `run-tests` skill) -- User wants to write new tests (use `code-testing-agent`) +- User wants to write new tests (use `code-testing`) - User only wants a coverage percentage without complexity analysis - User wants project-wide coverage/CRAP analysis or priorities (use `coverage-analysis`) diff --git a/catalog/Testing/Official-DotNet-Test/skills/crap-score/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/crap-score/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/crap-score/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/crap-score/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/detect-static-dependencies/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/detect-static-dependencies/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/detect-static-dependencies/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/detect-static-dependencies/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/filter-syntax/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/filter-syntax/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/filter-syntax/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/filter-syntax/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/find-untested-sources/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/find-untested-sources/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/find-untested-sources/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/find-untested-sources/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/generate-testability-wrappers/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/generate-testability-wrappers/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/generate-testability-wrappers/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/generate-testability-wrappers/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/grade-tests/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/grade-tests/SKILL.md index a7f2b5a5..a2ec68ee 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/grade-tests/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/grade-tests/SKILL.md @@ -1,13 +1,15 @@ --- name: grade-tests description: > - Assess a curated list of tests and produce a PR-ready table with a primary - Pass, Failed, Uncertain, or Not applicable result plus A-F quality detail for - every resolved test; Uncertain and Not applicable omit the grade. USE FOR new - or modified tests supplied as methods, bodies, file spans, or a bounded PR - diff. Polyglot: .NET, Python, TS/JS, Java, Go, Ruby, Rust, Swift, Kotlin, - PowerShell, C++. DO NOT USE FOR: suite-wide audits (use test-quality-auditor - or test-anti-patterns), writing or fixing tests, or measuring coverage. + Grade a curated list of individual tests for readiness, A-F quality, and + concrete improvements. ALWAYS USE FOR: grade tests, review only a named test, + per-test readiness decisions, or quality bands for supplied methods, bodies, + file spans, or bounded PR diffs, including existing tests. Produce a PR-ready + Pass, Failed, Uncertain, or Not applicable table; unresolved or empty scopes + omit the grade. Compose read-only per-test mutation evidence when available. + Polyglot: .NET, Python, TS/JS, Java, Go, Ruby, Rust, Swift, Kotlin, + PowerShell, C++. DO NOT USE FOR: suite-wide audits (test-engineer or + test-anti-patterns), writing or fixing tests, or measuring coverage. license: MIT --- @@ -20,7 +22,22 @@ diagnostic information. The skill **does not discover tests on its own** — the caller (typically a PR automation workflow or a human reviewer holding a specific list) provides the tests or a bounded diff to assess. -> **Language-specific guidance**: Call the `test-analysis-extensions` skill +After Step 0 admits a bounded scope, enforce these grading invariants: + +- With production context, **load `test-gap-analysis` by name once before + scoring**, in `per-test-read-only` caller context. Read its owned composition + reference; do not compute mutation evidence from this grading rubric or run + its standalone workflow. Report N/A / unverified only when the dependency, + reference, or required context is actually unavailable. +- Any reported mutation inference must use **Likely killed (inferred)** or + **Candidate survivor (unverified)**, even when explained in prose. +- Apply only the rubric below, not extra heuristics such as a duplicate-test + penalty. Do not use sibling tests to alter the individual assessment. +- A **B** quality grade does not require a Failed result; a complete focused + test may have no actionable change. + +> **Language-specific guidance**: If the caller supplies the matching bundled +> extension file path, read it directly. Otherwise call `test-analysis-extensions` > to discover available extension files, then read the file matching the > target codebase's language and framework (e.g., `extensions/dotnet.md`, > `extensions/python.md`, `extensions/typescript.md`, `extensions/go.md`). @@ -48,8 +65,8 @@ quality and severity. - The caller wants a full suite audit or comparative metrics — use `test-anti-patterns` (pragmatic) or `test-smell-detection` (formal) and - let the `test-quality-auditor` agent orchestrate. -- The caller wants to *write* new tests — use `code-testing-generator` + let the `test-engineer` agent orchestrate its internal quality specialist. +- The caller wants to *write* new tests — use `test-engineer` (any language) or `writing-mstest-tests` (MSTest specifically). - The caller wants to measure code coverage or CRAP scores — use `coverage-analysis` or `crap-score` (.NET only). @@ -64,7 +81,8 @@ quality and severity. |-------|----------|-------------| | Test methods | Yes | A scope to grade. Provide one of: (a) an explicit list of test method names (fully-qualified, e.g. `Namespace.ClassName.TestMethodName`); (b) one or more file paths plus an explicit instruction to grade every test declared in those files; or (c) a diff hunk / PR identifier whose changed tests should be graded. File paths are recommended but optional when method names are unambiguous in the workspace. Ambiguous requests like *"grade my tests"* with no scope are rejected up-front (see Step 0); this skill is for curated input and does not auto-grade an entire workspace. | | Test bodies / spans | Recommended | The exact source lines for each test method. If omitted, read them from the listed files. | -| Production code | No | The code under test, for judging whether assertions cover the meaningful behaviors. When unavailable, mark relevant findings as "Unverified" rather than guessing. | +| Production code | No | The code under test, for judging whether assertions cover the claimed behavior. When unavailable, mark the mutation assessment N/A / unverified rather than guessing or deducting. | +| Language reference | No | A host-supplied path to the matching bundled `test-analysis-extensions` file. Read it directly instead of invoking its reference-only loader; do not substitute unverified framework guidance. | | Diff context | No | When grading PR changes, the unified diff for each test method helps focus on what actually changed. | ### Step 0: Validate the input @@ -80,7 +98,7 @@ If the request is ambiguous (e.g., *"Grade my tests"*, *"Are these tests any good?"* with no scope, *"Review the test suite"*), **do not load extensions, do not read files, and do not grade anything**. Reply with a short message asking the caller to provide an explicit list / file(s) / -diff, and optionally point them at `test-quality-auditor` agent or +diff, and optionally point them at the `test-engineer` agent or `test-anti-patterns` skill for full-suite analysis. Stop there. If a valid bounded scope resolves to zero eligible tests, return @@ -92,7 +110,8 @@ If a valid bounded scope resolves to zero eligible tests, return Identify the target codebase's language and test framework from the file extensions and the test method markers in the provided list. Call the -`test-analysis-extensions` skill and read the matching extension file (e.g., +`test-analysis-extensions` skill unless the caller already supplied the matching +bundled extension file path. In either case, read that extension file (e.g., `extensions/dotnet.md` for MSTest/xUnit/NUnit/TUnit, `extensions/python.md` for pytest, `extensions/typescript.md` for Jest/Vitest, `extensions/go.md` for the standard `testing` package). If the input contains tests from @@ -112,7 +131,42 @@ For each entry in the input list: invent a body to grade. A missing requested method requires human review; it is not the same as a valid scope containing no tests. -### Step 3: Score each resolved test +**Composition checkpoint:** for resolved tests with available production +context, load `test-gap-analysis` now, once for the batch, with +`per-test-read-only` assessment context. Complete its owned reference assessment +before Step 3. Do not skip this load just because a body-level weakness already +seems obvious; a locally invented mutation explanation is not composition. + +### Step 3: Assess the claimed behavior and score each resolved test + +Keep grading read-only: no build/test runs, mutation execution, file edits, +tool installation, broad suite discovery, or agent delegation. Resolve only +the supplied tests, their relevant fixtures/helpers, and the production call +chain needed for their claims. + +Use the inline `test-gap-analysis` assessment from Step 2's checkpoint; +do not load it a second time. Supply each test's +identifier/body, relevant setup/helpers, claimed behavior, assertion semantics, +and available source. Its composition dispatch loads the owned read-only +reference rather than its standalone baseline/verification workflow. Consume +its per-test evidence; do not duplicate its mutation catalog here or invoke an +audit/generation agent. +Convey mode and inputs as assessment context using the host's supported caller +instructions. If the loader accepts only a skill name, load `test-gap-analysis` +by name only; do not invent tool arguments or a mode-specific skill name. + +If the skill/reference or production context is unavailable, record +`Pseudo-mutation: N/A / unverified — ` and continue normal body-level +grading. This is not a grade deduction or, by itself, an Uncertain result. +Do not search installation directories or substitute a mutation runner. + +Assess only what each test claims: do not borrow another test's assertions, +or demand unrelated branches, outputs, or scenarios. An observable survivor +can support an existing Assertion strength category when it proves that the +test does not verify its claimed outcome; do not introduce mutation points, +weights, ceilings, or an automatic survivor penalty. Apply the existing rubric +normally, including weaknesses it classifies in both Assertion and Anti-pattern +dimensions; do not add another deduction for the same mutation evidence. Start every test at grade **A (score band 90–100)**, then apply deductions strictly for **observable issues** in the captured body. Do **not** deduct @@ -138,7 +192,7 @@ assertion in the test body. Score from highest to lowest: |-----------|---------| | **A** | At least one meaningful value assertion (equality / structural / exception / state) plus, where appropriate, additional checks (negative, type, collection contents). Mock-call verifications (`Verify`, `toHaveBeenCalledWith`, `Should -Invoke`) and bare assertion forms (pytest `assert`, Go `if got != want { t.Errorf(...) }`, Rust `assert!()`) count as real assertions. | | **B** | One clear meaningful assertion that verifies the behavior under test. | -| **C** | Only trivial assertions (single `IsNotNull` / `toBeDefined` / `assert x is not None`), or assertions that check a single field while the operation produces a richer result. | +| **C** | Only trivial assertions (single `IsNotNull` / `toBeDefined` / `assert x is not None`), or assertions that leave a meaningful part of the test's claimed result unchecked. A focused single-field claim does not require unrelated fields. | | **D** | One self-referential / tautological assertion (`Assert.AreEqual(x, x)`, `assert dto.name == dto.name`, round-trip identity without a non-trivial input), or broad exception assertions (`Assert.ThrowsException`). | | **F** | No assertions at all; **all** assertions are always-true literals (`Assert.IsTrue(true)`, `assert True`, `expect(true).toBe(true)`) — these verify nothing and are equivalent to having no assertions; or all assertions are silently un-awaited (e.g., `expect(promise).resolves.toBe(x)` without `await`/`return`, async TUnit/xUnit `Assert.ThrowsAsync` without `await`, pytest-asyncio with un-awaited coroutine). | @@ -284,10 +338,27 @@ uncertainty. ### Step 5: Build the note Use one sentence (target ≤ 120 characters) for the most important reason: -`No issues found.`, `Only checks IsNotNull; add value verification.`, or +`No issues found.`, `Only checks IsNotNull; receipt contents are unverified.`, or `Method body could not be resolved; human review is required.` Do not invent a weakness to justify a grade or Failed result. +Keep the action in a separate **How to improve** field. For each Failed test, +name the smallest useful input, assertion, or fixture change and its expected +outcome, grounded in the body, source, or an explicit contract. For example, +`Replace self-comparison with Assert.AreEqual(60m, account.Balance).`, not +`Improve assertions`; `Remove Console.WriteLine after Deposit(25m).`, not +`Clean up`. Prioritize the highest-impact distinct finding, and include other +actionable findings only when they require a different change. + +For a behavioral gap, use the distinguishing witness and original/mutant +observations from the shared assessment; check the expected result against +the unmodified source. If essential context is missing, name the evidence +needed instead of inventing an expected value. Pass gets `None`; Uncertain +gets a concrete evidence-resolution step, not a speculative test rewrite. +A rubric-only deduction is not proof of a behavioral gap or an actionable +improvement: a focused **B / Pass** may need no change. **A / Failed** still +needs its concrete action, such as removing debug output. + ### Step 6: Report Produce two sections. @@ -302,13 +373,22 @@ empty scope and omit the table. #### 2. Per-test table ```markdown -| Test | Result | Quality | Notes | -|------|--------|---------|-------| -| `Namespace.ClassName.Test_Method_Condition_Expected` | Pass | A (90–100) | No issues found. | -| `Namespace.ClassName.Test_Other` | Failed | C (70–79) | Only `IsNotNull`; add value verification. | -| `Namespace.ClassName.Test_Missing` | Uncertain | — | Method body could not be resolved; human review is required. | +| Test | Result | Quality | Notes | How to improve | +|------|--------|---------|-------|----------------| +| `Namespace.ClassName.Test_Method_Condition_Expected` | Pass | B (80–89) | One complete value assertion. | None | +| `Namespace.ClassName.Withdraw_SufficientFunds` | Failed | D (60–69) | Balance is compared with itself. | Replace self-comparison with `Assert.AreEqual(60m, account.Balance)` after withdrawing 40m from 100m. | +| `Namespace.ClassName.Test_Missing` | Uncertain | — | Method body could not be resolved; human review is required. | Supply the method body and its referenced fixture. | ``` +Keep these two report sections and the original Test/Result/Quality/Notes +fields. When mutation evidence explains a finding or the caller requests +detail, append a compact per-test **Pseudo-mutation evidence** block inside +the per-test section: change, witness, original/mutant observations, relevant +assertion, and classification. Static results are **Likely killed (inferred)** +or **Candidate survivor (unverified)**, never executed Killed/Survived or +empirical killed/total counts. State missing-context N/A / unverified once +per shared limitation. Do not repeat the improvement table in prose. + **Caps and ordering**: - If the table would exceed **50 rows**, show Failed tests first, then Uncertain tests, then a sample of Pass tests. Wrap overflow in a collapsed @@ -329,6 +409,13 @@ prefix each section with the language name and framework. - [ ] Uncertain is an evidence gap; Not applicable is a valid empty scope. - [ ] Every grade is justified by at least one observable signal in the captured body — no speculative deductions. +- [ ] Every Failed row has a concrete, evidence-backed How to improve action; + Pass rows have no invented weakness, even when the quality grade is B. +- [ ] Mutation assessment stayed read-only and per-test; unavailable context + was not penalized, equivalents were excluded, and static labels/counts + were not presented as executed evidence. +- [ ] Verified observable findings inform existing categories without a + duplicate deduction or any change to scoring weights and ceilings. - [ ] Trivial-assertion tests are flagged only when the **only** assertion is trivial (a null check before a meaningful assertion is not trivial). - [ ] Exception-only tests are not penalized for low assertion count. @@ -362,6 +449,8 @@ prefix each section with the language name and framework. | Using a fake-precise score (e.g., 87/100) | Use the score band only — 90–100, 80–89, 70–79, 60–69, 0–59. | | Spilling a 500-row table into a PR comment | Apply the row cap from Step 6; collapse extras into `
`. | | Re-reporting an existing finding three times under different categories | Pick the most fitting category and report once. | +| Giving a weak test credit for a sibling's assertions | Use only the current test and helpers/fixtures it executes. | +| Turning pseudo-mutation composition into a suite audit | Pass explicit per-test-read-only mode; no runs, edits, broad discovery, or agent recursion. | | Inventing weaknesses for A-grade tests to make the note "balanced" | If a test is clean, the note may simply read `No issues found.` | | Mapping status from grade or comments | Fail only for actionable improvements; a B can Pass and an A can Fail. | | Confusing Uncertain and Not applicable | Evidence gaps are Uncertain; a valid empty scope is Not applicable. | diff --git a/catalog/Testing/Official-DotNet-Test/skills/grade-tests/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/grade-tests/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/grade-tests/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/grade-tests/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/migrate-static-to-wrapper/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/migrate-static-to-wrapper/manifest.json index cee18a89..f23ea3c7 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/migrate-static-to-wrapper/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/migrate-static-to-wrapper/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Migration", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/mtp-hot-reload/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/mtp-hot-reload/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/mtp-hot-reload/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/mtp-hot-reload/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/platform-detection/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/platform-detection/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/platform-detection/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/platform-detection/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/run-tests/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/run-tests/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/run-tests/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/run-tests/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/scaffold-dotnet-test-project/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/scaffold-dotnet-test-project/SKILL.md index 48e1ee48..8f2ef17c 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/scaffold-dotnet-test-project/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/scaffold-dotnet-test-project/SKILL.md @@ -7,7 +7,7 @@ description: >- include, or repair a test project. Handles "tests pass directly but CI discovers zero", exact solution wiring, xUnit/NUnit/MSTest, and central packages. DO NOT USE to only author tests in an already-wired project - (code-testing-agent), run tests, migrate, or correct MSTest syntax/configuration + (code-testing), run tests, migrate, or correct MSTest syntax/configuration without changing project or CI files (writing-mstest-tests). license: MIT metadata: @@ -52,7 +52,7 @@ Inspect the repository before editing, then choose exactly one path: | No suitable test project | Create one bounded project, reference the production project, and register it | Create a project per source project | | Test project exists but lacks the required `ProjectReference` | Add only that reference and verify direct plus entry-point execution | Scaffold another project or rewrite tests | | Test project passes directly but is absent from `.sln`, `.slnx`, or `.slnf` | Register the existing project in the exact entry point CI uses | Recreate the project or switch solution formats | -| Suitable project, reference, and requested entry point are already correct | Leave the workspace unchanged; use `code-testing-agent` if test methods are requested | Normalize or replace working files | +| Suitable project, reference, and requested entry point are already correct | Leave the workspace unchanged; use `code-testing` if test methods are requested | Normalize or replace working files | An existing project is suitable when its target framework can reference the production project and its purpose matches the requested layer. A different diff --git a/catalog/Testing/Official-DotNet-Test/skills/scaffold-dotnet-test-project/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/scaffold-dotnet-test-project/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/scaffold-dotnet-test-project/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/scaffold-dotnet-test-project/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/test-analysis-extensions/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/test-analysis-extensions/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/test-analysis-extensions/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/test-analysis-extensions/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/test-anti-patterns/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/test-anti-patterns/SKILL.md index f28abadc..bc375a64 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/test-anti-patterns/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/test-anti-patterns/SKILL.md @@ -6,7 +6,7 @@ description: > assertions, swallowed/broad exceptions, flaky/order-dependent tests, duplication, or magic values. Polyglot. DO NOT USE for direct edits: writing-mstest-tests owns supplied MSTest assertions/attributes/lifecycle; - code-testing-agent owns new tests. Exclude running tests, migration, assertion + code-testing owns new tests. Exclude running tests, migration, assertion metrics (assertion-quality), raw .NET coverage collection (run-tests), non-.NET coverage collection/analysis (native tooling), project-wide .NET coverage/CRAP (coverage-analysis), named-target .NET CRAP @@ -34,7 +34,7 @@ Quick, pragmatic analysis of test code in any supported language for anti-patter ## When Not to Use -- User wants to write new tests from scratch (use `code-testing-agent`) +- User wants to write new tests from scratch (use `code-testing`) - User wants direct implementation fixes rather than a diagnostic review (use the relevant write/edit skill) - User asks to fix swapped `Assert.AreEqual` argument order in MSTest (use `writing-mstest-tests`) - User asks to convert MSTest `DynamicData` from `IEnumerable` to `ValueTuple` (use `writing-mstest-tests`) diff --git a/catalog/Testing/Official-DotNet-Test/skills/test-anti-patterns/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/test-anti-patterns/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/test-anti-patterns/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/test-anti-patterns/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/SKILL.md index 365b8ee2..b024da0a 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/SKILL.md @@ -4,14 +4,14 @@ description: >- Pseudo-mutation analysis ONLY: answer whether tests would catch a bug if production code changed, which meaningful changes would still pass, or which caller-visible mutations existing assertions would miss; verify candidates - when requested, then optionally close verified gaps. Activate for behavioral - blind spots or missing edge cases tied to production behavior. Polyglot. DO - NOT USE FOR: suite organization, taxonomy, + when requested, then optionally close verified gaps. Includes explicit + read-only per-test composition for grading. Activate for behavioral blind + spots tied to production behavior. Polyglot. DO NOT USE FOR: suite taxonomy, metadata, or distribution reports (test-tagging); .NET line-vs-branch or Cobertura interpretation, arithmetic, plateaus, project-wide coverage gaps, or coverage-backed test/CRAP priorities (coverage-analysis; use native coverage tooling outside .NET); named-target CRAP (crap-score); new suites - (code-testing-agent); assertion/smell audits; or mutation tools. + (code-testing); assertion/smell audits; or mutation tools. license: MIT --- @@ -21,6 +21,22 @@ Answer one question: **which caller-visible production behaviors could change without an existing test failing?** Mutation reasoning is a probe, not the goal. Inventory public outcomes first, then verify only credible gaps. +## Composition dispatch + +Before entering the decision flow, check the caller's mode. An explicit +`per-test-read-only` request (including composition from `grade-tests`) uses +only [references/per-test-read-only.md](references/per-test-read-only.md). +This mode is assessment context conveyed through the host's supported caller +instructions, not a new skill-loader parameter or a mode-specific skill name. +Read that reference and return its per-test assessment inline; **do not enter +the standalone workflow below**. Loading this mode does not authorize a +baseline run, mutation execution, file edits, suite discovery, or delegation. +If the bundled reference is unavailable, allow at most one listing of the known +`references/` directory, then return **N/A / unverified** with the reason. + +Requests without this mode retain the standalone analysis, verification, and +test-addition paths below. + ## Decision flow ### 1. Set scope @@ -35,7 +51,7 @@ search misses, inspect the current directory broadly before asking for paths. | Explicit survivor verification | Inventory all requested outcomes; execute one representative observable candidate for each distinct high-risk outcome under verification, then classify it as **Survived** or **Killed** | | Explicit exhaustive audit | Read [references/mutation-catalog.md](references/mutation-catalog.md) and classify all meaningful candidates | | Add tests to an existing suite | Analyze first; add tests only for verified survivors or demonstrated no-coverage outcomes | -| Create a new suite | Stop and use `code-testing-agent` | +| Create a new suite | Stop and use `code-testing` | When the request names a risk, turn it into a one-line public-outcome allowlist before reading code. An outcome is not in scope merely because the same method writes it. @@ -204,6 +220,7 @@ calculate a score. |---|---| | **Likely killed** | An existing assertion observes the changed outcome | | **Candidate survivor (unverified)** | Observable change appears unasserted; not executed | +| **Killed** | Exact observable mutation executed and a relevant assertion failed | | **Survived** | Exact observable mutation executed and tests stayed green | | **No coverage** | No test reaches the public outcome; report the missing branch without inventing a survivor | | **Equivalent** | No public observation changes; omit from findings | @@ -222,8 +239,10 @@ requested test addition. 1. Apply one candidate and confirm the diff changes exactly one intended expression. -2. Run the narrowest covering test: green means **Survived**, red means - **Killed**, for that edit only. +2. Run the narrowest covering test and confirm tests executed: green means + **Survived**; a failure at a relevant assertion means **Killed**, for that + edit only. Build, discovery, infrastructure, or unrelated failures leave + the candidate **unverified**, not Killed. 3. Revert immediately and confirm the clean source/test baseline. 4. After a green run, re-check the public counterfactual. Execution proves the suite missed the edit, not that the edit changes behavior; drop inert or @@ -291,9 +310,10 @@ For focused or small analysis, return: Do not repeat the table in prose or report discarded mutants, tool chronology, or in-flight reasoning. -For an exhaustive audit, add counts for Killed / Survived / No coverage / -Equivalent and group findings by risk. Count only executed or definitively -classified candidates. +For an exhaustive audit, separate executed **Killed / Survived** counts from +static **Likely killed / Candidate survivor (unverified)** classifications and +**No coverage / Equivalent** inventory counts. Never include static reasoning +in an empirical killed/total ratio. Group findings by risk. For test additions, name the tests added, the verified mutations they kill, and the successful final command. diff --git a/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/references/mutation-catalog.md b/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/references/mutation-catalog.md index 78d6d943..0df14817 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/references/mutation-catalog.md +++ b/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/references/mutation-catalog.md @@ -1,8 +1,9 @@ # Mutation Candidate Catalog Read this reference only for an explicitly exhaustive audit or when a language's -mutation semantics are unfamiliar. For focused analysis, use the smaller -risk-ranked table in `SKILL.md`. +mutation semantics are unfamiliar, or for the applicable categories in +[per-test read-only composition](per-test-read-only.md). Reading the catalog +does not authorize the exhaustive execution procedure below. ## Candidate categories @@ -55,5 +56,7 @@ Exclude: 5. Execute every candidate that might be reported as Survived. 6. After a green run, re-check that the mutation is publicly observable. 7. Revert after each run and confirm the clean baseline at the end. -8. Count only executed or definitively killed/equivalent candidates in the - mutation totals; disclose any omitted scope. +8. Separate executed Killed/Survived totals from inferred Likely killed, + unverified candidates, and equivalent inventory classifications. Never turn + static classifications into empirical killed/total claims; disclose omitted + scope. diff --git a/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/references/per-test-read-only.md b/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/references/per-test-read-only.md new file mode 100644 index 00000000..a19bc7f2 --- /dev/null +++ b/catalog/Testing/Official-DotNet-Test/skills/test-gap-analysis/references/per-test-read-only.md @@ -0,0 +1,106 @@ +# Per-Test Read-Only Assessment + +This is the composition contract for `mode: per-test-read-only`, not a suite +audit or a mutation runner. `test-gap-analysis` owns this methodology and the +[mutation catalog](mutation-catalog.md); consumers own decisions and scoring. + +## Inputs and execution boundary + +The caller supplies resolved test identifiers/bodies, relevant setup and +helpers, language/framework assertion semantics, and available production +context. A batch may contain several tests, but assess each independently. + +- Read only those tests, their fixtures/helpers, and the production call chain + needed to explain their claimed outcomes. No broad suite discovery. +- Do not run tests, build/restore, execute/import production code, install + tooling, apply mutations, edit files, invoke agents, or recurse into another + grading/audit workflow. A read of this reference is not execution permission. +- Missing source, an unresolved call chain, or unsupported assertion semantics + yields **N/A / unverified** for the affected assessment, with the reason. + Assess any remaining known behavior normally; absence is not a weakness. +- Do not replace the caller's report with a suite-wide Strong/Mixed/Weak verdict, + a mutation score, or a coverage dashboard. + +## Assessment + +1. **Bind the claim.** State the behavior this individual test promises from + its name, arranged inputs, act, assertions, and any explicit contract. + Resolve a contradiction as an evidence gap instead of inventing intent. + An exception-only test owns that invalid-input outcome, not happy paths; + a price-only test does not own an unrelated flag or formatting field. + Do not demand every method branch, adjacent scenario, or sibling test case. +2. **Trace observations.** Map each supplied input/sequence through the public + caller to its return value, exception, state, or external side effect and + the assertion that observes it. Include relevant fixture cleanup/assertions + and helpers executed by this test, but never borrow another test's checks + or deduct because another test is absent. +3. **Probe meaningful changes.** Use the applicable categories in the owned + mutation catalog without enumerating every operator. Fix all arguments when + comparing original and mutant. First replay the change mentally on this + test's existing inputs and assertions. A changed observation that makes a + relevant assertion fail is **Likely killed (inferred)**, even if the check + is indirect or the return value is not asserted directly. +4. **Prove an actionable survivor.** For a change the current test appears to + miss, supply a distinguishing witness within its claimed behavior: + `input/sequence -> original observation -> mutant observation`. + Explain why the current test's assertions still pass, and how the smallest + input/assertion/fixture change would expose the difference. If a new input + is needed, distinguish that proposed witness from the current test input; + do not pretend an assertion on the proposed witness already exists. + A witness from an unrelated behavior is not a finding against this test. +5. **Filter equivalence and uncertainty.** Omit non-compiling changes, + impossible inputs, equivalent guards/boundaries, private representation + changes with no public effect, and duplicate syntax variants. In particular, + removing a guard that falls through to the same public exception is not a + survivor; `<` to `<=` at a floor is equivalent when both return the floor. + Do not require exception message/parameter metadata unless it is an + established contract. If original and mutant cannot be distinguished from + available context, mark the assessment unverified rather than deducting. +6. **Return evidence, not grades.** Report only credible distinct findings and + relevant protected behavior. An empty findings list is valid; do not invent + a mutant, improvement, count, or denominator to fill the report. + +## Return contract + +For each requested test, return: + +| Field | Content | +|---|---| +| Test / claim | Stable identifier and the individual behavior assessed | +| Availability | Assessed, or N/A / unverified with the missing evidence | +| Change / witness | Exact existing expression/condition/side effect changed; fixed input or sequence | +| Observations | Original versus mutant caller-visible outcomes on that same witness | +| Detection evidence | This test's relevant assertion and why it would fail or still pass; distinguish current inputs from a proposed witness | +| Classification | Likely killed (inferred), or Candidate survivor (unverified) | +| Smallest improvement | Concrete input, expected assertion, or fixture change for each candidate survivor; none when protected | + +These fields form an evidence ledger, not a mandatory extra table in the +consumer's final response. Preserve source/test locations when available. +Keep evidence proportional to the claim; stop once the claim is protected or +no credible observable candidate remains. + +Only already-supplied execution evidence for this exact test, source revision, +and mutation may use **Killed (executed)** or **Survived (executed)**. A killed +result requires a relevant assertion failure, not a build/runner failure. +Keep such evidence separate from static classifications. Never label a static +assessment Killed/Survived alone or report empirical killed/total counts, +percentages, or a mutation score from source reasoning. + +## Examples + +- A test withdraws 40 from a balance of 100 and compares the balance with + itself. Omitting the decrement leaves 100 instead of 60; the self-comparison + still passes. **Candidate survivor (unverified)**; replace that comparison + with the literal expected balance 60. Do not credit a sibling balance test. +- A test claims to verify the correct standard fee calculation, but checks + only that the cost is positive. A fee change yields 12 instead of 10 for + its arranged order while that check still passes. Recommend equality to 10 + for that order, not generic "stronger assertions". If the test deliberately + claims only positivity, that same change is not a gap against its claim; + do not require an exact fee or unrelated cancellation coverage. +- A test asserts the promised exception type for one invalid input. Removing + its guard still throws that same type later and changes no contracted side + effect. Omit the equivalent mutation; do not demand message assertions. +- A focused test pins the result at a threshold. An inclusive-to-exclusive + change alters that asserted result: **Likely killed (inferred)**. Do not + deduct for not testing a different scenario, or call this an executed kill. diff --git a/catalog/Testing/Official-DotNet-Test/skills/test-smell-detection/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/test-smell-detection/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/test-smell-detection/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/test-smell-detection/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/test-tagging/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/test-tagging/SKILL.md index c403a91f..13755a2d 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/test-tagging/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/test-tagging/SKILL.md @@ -8,7 +8,7 @@ description: > by test type, or tag then verify the project builds. Read bodies when names mislead. Apply canonical attributes; otherwise report only. DO NOT USE FOR: requests owned by test-anti-patterns, coverage-analysis, crap-score, - test-gap-analysis, code-testing-agent, or migration skills. + test-gap-analysis, code-testing, or migration skills. license: MIT --- @@ -29,7 +29,7 @@ Analyze an existing test suite in any supported language and apply a standardize ## When Not to Use -- Writing new tests from scratch (use `code-testing-agent` for any language, or `writing-mstest-tests` for MSTest) +- Writing new tests from scratch (use `code-testing` for any language, or `writing-mstest-tests` for MSTest) - Running or filtering tests (use `run-tests` for .NET; equivalent native runners elsewhere) - Migrating between test frameworks - General quality, smell, flakiness, or assertion audits (use `test-anti-patterns` or the matching analysis skill) diff --git a/catalog/Testing/Official-DotNet-Test/skills/test-tagging/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/test-tagging/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/test-tagging/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/test-tagging/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/testability-obstacle/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/testability-obstacle/SKILL.md index 54f9d254..93ace48a 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/testability-obstacle/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/testability-obstacle/SKILL.md @@ -30,7 +30,7 @@ redesign adjacent code. ## When Not to Use - The dependency is already injected or passed as an argument. Write tests with - a fake through the existing seam using `code-testing-agent`. + a fake through the existing seam using `code-testing`. - The user wants a repository-wide testability audit. Use `detect-static-dependencies`. - The user wants wrappers generated but not call sites/tests changed. Use diff --git a/catalog/Testing/Official-DotNet-Test/skills/testability-obstacle/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/testability-obstacle/manifest.json index dec12fb9..01cb20ad 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/testability-obstacle/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/testability-obstacle/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution." } diff --git a/catalog/Testing/Official-DotNet-Test/skills/writing-mstest-tests/SKILL.md b/catalog/Testing/Official-DotNet-Test/skills/writing-mstest-tests/SKILL.md index e6c8794f..080854b6 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/writing-mstest-tests/SKILL.md +++ b/catalog/Testing/Official-DotNet-Test/skills/writing-mstest-tests/SKILL.md @@ -9,7 +9,7 @@ description: > identity, exception, hard-cast, and object[] checks; TestContext/lifecycle; timeout/cancellation; OS/CI conditions, retry, cleanup, parallelization, MSTest.Sdk project setup, and MSTESTxxxx. Honor the installed - MSTest version. DO NOT USE to design new test cases (code-testing-agent), + MSTest version. DO NOT USE to design new test cases (code-testing), perform report-only audits, create project files rather than explain MSTest setup, run tests, migrate frameworks, or handle non-MSTest/non-.NET code. license: MIT @@ -76,7 +76,7 @@ permissions or the task's scope. ## Response Guidelines - **Specific API or pattern questions** (assertions, data-driven, lifecycle): Jump directly to the relevant workflow step. Do not follow the full workflow. -- **Generate new tests from scratch**: Hand off to `code-testing-agent`; use this +- **Generate new tests from scratch**: Hand off to `code-testing`; use this skill only as supporting MSTest API/version guidance. - **Review and fix existing tests**: Fix only the issues present. Do not add unrelated improvements. - **Assertion transformations**: Show the corrected call, then state the semantic diff --git a/catalog/Testing/Official-DotNet-Test/skills/writing-mstest-tests/manifest.json b/catalog/Testing/Official-DotNet-Test/skills/writing-mstest-tests/manifest.json index d48ab94a..e39e927d 100644 --- a/catalog/Testing/Official-DotNet-Test/skills/writing-mstest-tests/manifest.json +++ b/catalog/Testing/Official-DotNet-Test/skills/writing-mstest-tests/manifest.json @@ -1,5 +1,5 @@ { - "version": "0.2.23", + "version": "1.0.0", "category": "Testing", "compatibility": "Requires a .NET test project or solution.", "package_prefix": "MSTest" diff --git a/catalog/Tools/Official-DotNet-Create-Skill-Test/manifest.json b/catalog/Tools/Official-DotNet-Create-Skill-Test/manifest.json index dd809894..23011fb2 100644 --- a/catalog/Tools/Official-DotNet-Create-Skill-Test/manifest.json +++ b/catalog/Tools/Official-DotNet-Create-Skill-Test/manifest.json @@ -1,7 +1,7 @@ { "name": "create-skill-test", "title": "Official .NET skills: Skill evaluations", - "description": "Scaffolds eval.yaml evaluation specs for agent skills in the dotnet/skills repository. Use when creating skill tests, writing evaluation stimuli, defining graders and rubrics, sizing an eval for statistical power, or setting up test fixture files. Handles the Vally eval.yaml schema, fixture organization, and overfitting avoidance. Do not use for running or debugging existing evals (use improve-skill-quality) nor for skills authoring (use create-skill).", + "description": "Scaffolds eval.yaml evaluation specs for skills, custom agents, and redistributable gh-aw workflow packages in the dotnet/skills repository. Use when creating skill or workflow-package tests, writing evaluation stimuli, defining graders and rubrics, sizing an eval for statistical power, or setting up test fixture files. Handles the Vally eval.yaml schema, fixture organization, and overfitting avoidance. Do not use for running or debugging existing evals (use improve-skill-quality) nor for skills authoring (use create-skill).", "links": { "repository": "https://github.com/dotnet/skills", "docs": "https://github.com/dotnet/skills/tree/main/.agents/skills/create-skill-test" diff --git a/catalog/Tools/Official-DotNet-Create-Skill-Test/skills/create-skill-test/SKILL.md b/catalog/Tools/Official-DotNet-Create-Skill-Test/skills/create-skill-test/SKILL.md index b5eaa5b0..34640846 100644 --- a/catalog/Tools/Official-DotNet-Create-Skill-Test/skills/create-skill-test/SKILL.md +++ b/catalog/Tools/Official-DotNet-Create-Skill-Test/skills/create-skill-test/SKILL.md @@ -1,17 +1,17 @@ --- name: create-skill-test -description: Scaffolds eval.yaml evaluation specs for agent skills in the dotnet/skills repository. Use when creating skill tests, writing evaluation stimuli, defining graders and rubrics, sizing an eval for statistical power, or setting up test fixture files. Handles the Vally eval.yaml schema, fixture organization, and overfitting avoidance. Do not use for running or debugging existing evals (use improve-skill-quality) nor for skills authoring (use create-skill). +description: Scaffolds eval.yaml evaluation specs for skills, custom agents, and redistributable gh-aw workflow packages in the dotnet/skills repository. Use when creating skill or workflow-package tests, writing evaluation stimuli, defining graders and rubrics, sizing an eval for statistical power, or setting up test fixture files. Handles the Vally eval.yaml schema, fixture organization, and overfitting avoidance. Do not use for running or debugging existing evals (use improve-skill-quality) nor for skills authoring (use create-skill). --- # Create Skill Test -Scaffold an evaluation spec (`eval.yaml`) for a skill or agent so it conforms to the Vally schema, +Scaffold an evaluation spec (`eval.yaml`) for a skill, agent, or workflow package so it conforms to the Vally schema, passes `skill-validator check` and `check_eval_quality.py`, is powerful enough to return a verdict, and does not overfit to the skill's own wording. ## When to Use -- Creating a new `eval.yaml` for a skill or agent +- Creating a new `eval.yaml` for a skill, agent, or workflow package - Adding stimuli to an existing eval - Sizing an eval so the pass gate can actually be reached - Setting up or repairing fixture files alongside an eval @@ -61,6 +61,7 @@ Then locate the target and test directory: ```text tests///eval.yaml # skills tests//agent./eval.yaml # agents (the agent. prefix disambiguates) +tests/agentic-workflows//eval.yaml # redistributable gh-aw packages ``` Verify the target exists at `plugins//skills//SKILL.md` or @@ -328,6 +329,32 @@ incompatible project type, wrong framework version, prerequisite absent. > unexpected isolated activation blocks a pass. `expect_activation: false` **alone** is the repo > convention. +### Workflow-package scenarios + +For a package target, verify `agentic-workflows//aw.yml`, then read its +entry workflow, local imports, and bundled agents. The native SDK lane evaluates +their real prompt bodies and installed resources against offline fixtures. +Specify collector outputs, revision/tracking evidence, and service responses as +fixture inputs; propose terminal actions in `result.json` rather than pretending +to publish through live GitHub or safe-output tools. Assert the structured result +with deterministic graders. Do not place expected answers in agent-readable +fixtures or staged grader scripts; pass expected values through grader argv. + +Prompt expressions are rendered from a flat `workflow-context.json` fixture, +whose keys are exact trimmed expressions and values are strings. Missing context +fails setup. A workflow that correctly chooses noop is still expected-active +decision evidence, not `expect_activation: false` routing evidence. Include +normal, partial, stale, incompatible, missing-evidence, and multi-module cases +where applicable. Keep compilation, helper execution, and actual consumer +publication tests separate: this lane is labeled `workflow-prompt-sdk`, not +end-to-end Actions execution. + +```powershell +dotnet run --project eng/skill-validator/src/SkillValidator.csproj -- evaluate ` + agentic-workflows//aw.yml ` + --tests-dir tests/agentic-workflows --runs 1 --verdict-warn-only +``` + Guard rubrics verify three things: **recognition** (why it does not apply), **restraint** (no workflow, no file changes, no installs), **redirection** (the correct next step). diff --git a/catalog/Tools/Official-DotNet-Improve-Skill-Quality/skills/improve-skill-quality/references/writing-for-baseline-delta.md b/catalog/Tools/Official-DotNet-Improve-Skill-Quality/skills/improve-skill-quality/references/writing-for-baseline-delta.md index b722dce3..7ffd60d7 100644 --- a/catalog/Tools/Official-DotNet-Improve-Skill-Quality/skills/improve-skill-quality/references/writing-for-baseline-delta.md +++ b/catalog/Tools/Official-DotNet-Improve-Skill-Quality/skills/improve-skill-quality/references/writing-for-baseline-delta.md @@ -91,7 +91,7 @@ A one-cell mapping steered frontier models into a behavior change: `TimeProvider before claiming success. `migrate-static-to-wrapper` lost trials for claiming "Build succeeded" after a restore failure; -`code-testing-agent` had to be told to cite a clean run. (PR #945) +`code-testing` had to be told to cite a clean run. (PR #945) ## 10. Prove already-correct inputs are left alone @@ -112,7 +112,7 @@ references read only when needed — cost down, contract unchanged. (PR #971) **Rule:** Do not run a full research → plan → implement pipeline for one function. -`code-testing-agent` was split into focused and broad paths so a single-function request skips +`code-testing` was split into focused and broad paths so a single-function request skips `.testagent/` artifacts and extra passes. (PR #971) ## 13. Structure beats verbosity diff --git a/external-sources/upstreams/astro/packages/astro/package.json b/external-sources/upstreams/astro/packages/astro/package.json index 1d0f9e02..58c8ce0c 100644 --- a/external-sources/upstreams/astro/packages/astro/package.json +++ b/external-sources/upstreams/astro/packages/astro/package.json @@ -1,6 +1,6 @@ { "name": "astro", - "version": "7.3.7", + "version": "7.3.8", "description": "Astro is a modern site builder with web best practices, performance, and DX front-of-mind.", "type": "module", "author": "withastro", @@ -142,7 +142,6 @@ "diff": "^9.0.0", "dset": "^3.1.4", "es-module-lexer": "^3.0.2", - "esbuild": "^0.28.0", "find-proc": "0.2.0", "flattie": "^1.1.1", "fontace": "~0.4.1", diff --git a/external-sources/upstreams/dotnet-skills/.agents/skills/create-skill-test/SKILL.md b/external-sources/upstreams/dotnet-skills/.agents/skills/create-skill-test/SKILL.md index b5eaa5b0..34640846 100644 --- a/external-sources/upstreams/dotnet-skills/.agents/skills/create-skill-test/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/.agents/skills/create-skill-test/SKILL.md @@ -1,17 +1,17 @@ --- name: create-skill-test -description: Scaffolds eval.yaml evaluation specs for agent skills in the dotnet/skills repository. Use when creating skill tests, writing evaluation stimuli, defining graders and rubrics, sizing an eval for statistical power, or setting up test fixture files. Handles the Vally eval.yaml schema, fixture organization, and overfitting avoidance. Do not use for running or debugging existing evals (use improve-skill-quality) nor for skills authoring (use create-skill). +description: Scaffolds eval.yaml evaluation specs for skills, custom agents, and redistributable gh-aw workflow packages in the dotnet/skills repository. Use when creating skill or workflow-package tests, writing evaluation stimuli, defining graders and rubrics, sizing an eval for statistical power, or setting up test fixture files. Handles the Vally eval.yaml schema, fixture organization, and overfitting avoidance. Do not use for running or debugging existing evals (use improve-skill-quality) nor for skills authoring (use create-skill). --- # Create Skill Test -Scaffold an evaluation spec (`eval.yaml`) for a skill or agent so it conforms to the Vally schema, +Scaffold an evaluation spec (`eval.yaml`) for a skill, agent, or workflow package so it conforms to the Vally schema, passes `skill-validator check` and `check_eval_quality.py`, is powerful enough to return a verdict, and does not overfit to the skill's own wording. ## When to Use -- Creating a new `eval.yaml` for a skill or agent +- Creating a new `eval.yaml` for a skill, agent, or workflow package - Adding stimuli to an existing eval - Sizing an eval so the pass gate can actually be reached - Setting up or repairing fixture files alongside an eval @@ -61,6 +61,7 @@ Then locate the target and test directory: ```text tests///eval.yaml # skills tests//agent./eval.yaml # agents (the agent. prefix disambiguates) +tests/agentic-workflows//eval.yaml # redistributable gh-aw packages ``` Verify the target exists at `plugins//skills//SKILL.md` or @@ -328,6 +329,32 @@ incompatible project type, wrong framework version, prerequisite absent. > unexpected isolated activation blocks a pass. `expect_activation: false` **alone** is the repo > convention. +### Workflow-package scenarios + +For a package target, verify `agentic-workflows//aw.yml`, then read its +entry workflow, local imports, and bundled agents. The native SDK lane evaluates +their real prompt bodies and installed resources against offline fixtures. +Specify collector outputs, revision/tracking evidence, and service responses as +fixture inputs; propose terminal actions in `result.json` rather than pretending +to publish through live GitHub or safe-output tools. Assert the structured result +with deterministic graders. Do not place expected answers in agent-readable +fixtures or staged grader scripts; pass expected values through grader argv. + +Prompt expressions are rendered from a flat `workflow-context.json` fixture, +whose keys are exact trimmed expressions and values are strings. Missing context +fails setup. A workflow that correctly chooses noop is still expected-active +decision evidence, not `expect_activation: false` routing evidence. Include +normal, partial, stale, incompatible, missing-evidence, and multi-module cases +where applicable. Keep compilation, helper execution, and actual consumer +publication tests separate: this lane is labeled `workflow-prompt-sdk`, not +end-to-end Actions execution. + +```powershell +dotnet run --project eng/skill-validator/src/SkillValidator.csproj -- evaluate ` + agentic-workflows//aw.yml ` + --tests-dir tests/agentic-workflows --runs 1 --verdict-warn-only +``` + Guard rubrics verify three things: **recognition** (why it does not apply), **restraint** (no workflow, no file changes, no installs), **redirection** (the correct next step). diff --git a/external-sources/upstreams/dotnet-skills/.agents/skills/improve-skill-quality/references/writing-for-baseline-delta.md b/external-sources/upstreams/dotnet-skills/.agents/skills/improve-skill-quality/references/writing-for-baseline-delta.md index b722dce3..7ffd60d7 100644 --- a/external-sources/upstreams/dotnet-skills/.agents/skills/improve-skill-quality/references/writing-for-baseline-delta.md +++ b/external-sources/upstreams/dotnet-skills/.agents/skills/improve-skill-quality/references/writing-for-baseline-delta.md @@ -91,7 +91,7 @@ A one-cell mapping steered frontier models into a behavior change: `TimeProvider before claiming success. `migrate-static-to-wrapper` lost trials for claiming "Build succeeded" after a restore failure; -`code-testing-agent` had to be told to cite a clean run. (PR #945) +`code-testing` had to be told to cite a clean run. (PR #945) ## 10. Prove already-correct inputs are left alone @@ -112,7 +112,7 @@ references read only when needed — cost down, contract unchanged. (PR #971) **Rule:** Do not run a full research → plan → implement pipeline for one function. -`code-testing-agent` was split into focused and broad paths so a single-function request skips +`code-testing` was split into focused and broad paths so a single-function request skips `.testagent/` artifacts and extra passes. (PR #971) ## 13. Structure beats verbosity diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/.claude-plugin/plugin.json b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/.claude-plugin/plugin.json index ab40a94b..3e2cdfde 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/.claude-plugin/plugin.json +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "dotnet-test-migration", - "version": "0.1.10", + "version": "0.1.11", "description": "Skills and an orchestrator agent for migrating .NET test frameworks and platforms: MSTest and xUnit version upgrades, xUnit/NUnit-to-MSTest conversion, and VSTest to Microsoft.Testing.Platform.", "skills": ["./skills/"], "agents": [ diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/.codex-plugin/plugin.json b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/.codex-plugin/plugin.json index 6e782354..f4004a83 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/.codex-plugin/plugin.json +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "dotnet-test-migration", - "version": "0.1.10", + "version": "0.1.11", "description": "Skills for migrating .NET test frameworks and platforms: MSTest and xUnit version upgrades, xUnit/NUnit-to-MSTest conversion, and VSTest to Microsoft.Testing.Platform.", "skills": ["./skills/"] } diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/README.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/README.md index ee766946..057ecd41 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/README.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/README.md @@ -33,7 +33,12 @@ migration skills, but not this agent or its static handoff. ## Related plugins -- **dotnet-test** — running, generating, and analyzing tests; testability improvement; coverage. The migration skills here reference shared `dotnet-test` skills by name (e.g., `platform-detection` for framework/platform detection, `writing-mstest-tests` for idiomatic MSTest polish, and `run-tests` for verification). Install `dotnet-test` alongside this plugin to get the full workflow. +- **dotnet-test** — running, generating, repairing, and analyzing tests; + testability improvement; coverage; and the `test-engineer` handoff used after + migrations. The migration skills here reference shared `dotnet-test` skills + by name (for example `platform-detection`, `writing-mstest-tests`, and + `run-tests`). Install `dotnet-test` alongside this plugin to get the full + workflow. ## Prerequisites diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/agents/test-migration.agent.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/agents/test-migration.agent.md index 528a101b..f0a94e14 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/agents/test-migration.agent.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/agents/test-migration.agent.md @@ -6,14 +6,15 @@ description: >- and guides users through end-to-end upgrades. Use when asked to upgrade MSTest, migrate to xUnit v3, switch to Microsoft.Testing.Platform, modernize test infrastructure, or when the user says "migrate my tests". -user-invokable: true +user-invocable: true disable-model-invocation: false handoffs: - label: Audit Test Quality - agent: test-quality-auditor + agent: test-engineer prompt: >- The test framework migration is complete. Please audit the migrated - test suite for quality issues, anti-patterns, and coverage gaps. + test suite for quality issues, anti-patterns, and coverage gaps, then + propose or implement fixes according to the user's request. send: false license: MIT --- @@ -53,7 +54,7 @@ Classify the user's request and route to the appropriate skill or agent: | "Convert xUnit to MSTest" / "switch from xUnit to MSTest" / "port xUnit tests to MSTest" (xUnit v2 or v3 detected) | `migrate-xunit-to-mstest` skill | | "Convert NUnit to MSTest" / "switch from NUnit to MSTest" / "port NUnit tests to MSTest" (NUnit 3 or 4 detected) | `migrate-nunit-to-mstest` skill | | "Migrate to MTP" / "switch from VSTest" / "modern test runner" | `migrate-vstest-to-mtp` skill | -| "Make code testable" / "remove static dependencies" | Hand off to `testability-migration` agent | +| "Make code testable" / "remove static dependencies" | Hand off to the `test-engineer` agent | | "Migrate my tests" (no specifics) | Run detection, then recommend and confirm the migration path | ## Detection Workflow diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/plugin.json b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/plugin.json index ab40a94b..3e2cdfde 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/plugin.json +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test-migration/plugin.json @@ -1,6 +1,6 @@ { "name": "dotnet-test-migration", - "version": "0.1.10", + "version": "0.1.11", "description": "Skills and an orchestrator agent for migrating .NET test frameworks and platforms: MSTest and xUnit version upgrades, xUnit/NUnit-to-MSTest conversion, and VSTest to Microsoft.Testing.Platform.", "skills": ["./skills/"], "agents": [ diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/.claude-plugin/plugin.json b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/.claude-plugin/plugin.json index 84622731..d41265fb 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/.claude-plugin/plugin.json +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/.claude-plugin/plugin.json @@ -1,10 +1,10 @@ { "name": "dotnet-test", - "version": "0.2.23", + "version": "1.0.0", "description": "Skills for running, generating, analyzing, and improving .NET tests: test execution, filtering, platform detection, coverage, testability, and MSTest workflows.", "skills": ["./skills/"], "agents": [ - "./agents/code-testing-generator.agent.md", + "./agents/test-engineer.agent.md", "./agents/code-testing-researcher.agent.md", "./agents/code-testing-planner.agent.md", "./agents/code-testing-implementer.agent.md", diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/.codex-plugin/plugin.json b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/.codex-plugin/plugin.json index 1bf79f57..d28eeb56 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/.codex-plugin/plugin.json +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "dotnet-test", - "version": "0.2.23", + "version": "1.0.0", "description": "Skills for running, generating, analyzing, and improving .NET tests: test execution, filtering, platform detection, coverage, testability, and MSTest workflows.", "skills": ["./skills/"] } diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/README.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/README.md index 90bab32b..fe02d20e 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/README.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/README.md @@ -1,13 +1,20 @@ # dotnet-test -Skills and GitHub Copilot custom agents for running, generating, analyzing, and improving tests. Originally built for .NET (MSTest, xUnit, NUnit, TUnit) and platforms (VSTest, Microsoft.Testing.Platform); the test-generation pipeline and the six test-analysis skills (anti-patterns, smells, assertion quality, gap analysis, tagging, grade tests) plus the `test-quality-auditor` agent are **polyglot** and also work with Python (pytest/unittest), TypeScript/JavaScript (Jest/Vitest/Mocha/Jasmine/node:test), Java (JUnit 4/5/TestNG), Go (testing/testify), Ruby (RSpec/Minitest), Rust (built-in/proptest), Swift (XCTest/Swift Testing), Kotlin (JUnit/Kotest), PowerShell (Pester), and C++ (GoogleTest/Catch2/doctest/Boost.Test). +Skills and a GitHub Copilot `test-engineer` agent for running, generating, +repairing, analyzing, and improving tests. Originally built for .NET (MSTest, +xUnit, NUnit, TUnit) and platforms (VSTest, Microsoft.Testing.Platform), the +test-engineering workflows are **polyglot** and also work with Python +(pytest/unittest), TypeScript/JavaScript (Jest/Vitest/Mocha/Jasmine/node:test), +Java (JUnit 4/5/TestNG), Go (testing/testify), Ruby (RSpec/Minitest), Rust +(built-in/proptest), Swift (XCTest/Swift Testing), Kotlin (JUnit/Kotest), +PowerShell (Pester), and C++ (GoogleTest/Catch2/doctest/Boost.Test). > **Test framework/platform migration** (MSTest/xUnit upgrades, xUnit → MSTest, VSTest → Microsoft.Testing.Platform) lives in the separate [`dotnet-test-migration`](../dotnet-test-migration/) plugin. ## When to use this plugin - **Run tests** *(.NET only)* — execute SDK-style projects with `dotnet test`, or preserve a classic project's checked-in MSBuild + VSTest/MSTest command -- **Generate tests** *(polyglot)* — scaffold unit tests for any language with a scope-sized Research → Plan → Implement workflow +- **Generate tests** *(polyglot)* — scaffold comprehensive unit tests for any language via a multi-agent pipeline - **Migrate tests** *(.NET only)* — see the separate [`dotnet-test-migration`](../dotnet-test-migration/) plugin (MSTest v1/v2 → v3 → v4, xUnit v2 → v3, xUnit → MSTest, VSTest → Microsoft.Testing.Platform) - **Audit test quality** *(polyglot)* — detect anti-patterns, test smells, assertion gaps, and (for .NET) coverage risks - **Improve testability** *(.NET only)* — find static dependencies, generate wrappers, and migrate call sites to injectable abstractions @@ -26,7 +33,7 @@ Skills and GitHub Copilot custom agents for running, generating, analyzing, and | Skill | Description | |---|---| -| **code-testing-agent** | Scope-sized test generation for any language: focused additions stay direct; broad requests use the generator-owned Research → Plan → Implement pipeline with proportionate validation and review | +| **code-testing** | Implicit entry skill for generating, repairing, and strengthening tests; broad work delegates to `test-engineer` | | **scaffold-dotnet-test-project** *(.NET)* | Create a missing test project or repair its project/solution/filter wiring | | **writing-mstest-tests** | Version-compatible MSTest authoring for modern and classic projects, including MSTest 3.x/4.x APIs | @@ -43,9 +50,40 @@ These six skills are all polyglot. They work across all supported languages by l | **test-anti-patterns** | Quick pragmatic scan for common test quality issues with severity ranking (any language) | | **test-smell-detection** | Deep formal audit using academic test smell taxonomy (19 smell types, any language) | | **assertion-quality** | Measure assertion variety and depth — find shallow tests that barely verify anything (any language) | -| **test-gap-analysis** | Verify test blind spots through pseudo-mutations and optionally add focused tests that kill them (any language) | +| **test-gap-analysis** | Analyze test blind spots through pseudo-mutations, expose read-only per-test evidence for grading, and verify or close gaps only when requested (any language) | | **test-tagging** | Tag tests with standardized traits (smoke, regression, boundary, critical-path, etc.); auto-edits where the framework has canonical syntax, report-only otherwise | -| **grade-tests** | Assess a curated list of test methods and produce a compact PR-ready table with Pass, Failed, or Uncertain decisions, A-F quality detail for resolved tests, and one-line notes; unresolved or empty scopes omit the grade, and a valid scope with no tests returns Not applicable (any language) | +| **grade-tests** | Assess curated tests with Pass, Failed, or Uncertain decisions, A-F quality detail, notes, and concrete improvement actions; unresolved or empty scopes omit the grade, and a valid empty scope returns Not applicable (any language) | + +Grading composes `test-gap-analysis` in explicit `per-test-read-only` mode: +no test runs, production edits, mutation execution, suite audit, or agent +recursion. Mutation methodology stays in that skill's bundled reference; +grading retains its own scoring weights and ceilings. Each test owns only its +claimed behavior, not its siblings' assertions or unrelated scenarios. +Static evidence uses inferred likely kills or unverified candidate survivors, +not executed mutation counts. Missing production context is N/A / unverified, +not a deduction. The existing result and quality fields remain independent: +a focused B can Pass without improvements, and an A can Fail for actionable +debug output. + +Focused grading checks can run without an evaluation matrix: + +```powershell +python -B tests\dotnet-test\grade-tests\test_regressions.py -v +python -B tests\dotnet-test\grade-tests\test_composition.py --cli --model --results-dir +``` + +The first command replays goldens and rejects malformed actions, extra test +rows, and misleading mutation evidence. The second uses the shipping Copilot +CLI, a copy of the production plugin, isolated configuration, and a +host-supplied path to the actual bundled Python assertion reference. It requires +successful grading and gap-analysis loads plus the owned read-only reference +read and their completions before the final grading report. It verifies the +target row's concrete improvement, rejects standalone execution/delegation and +N/A fallbacks, and checks that every fixture and plugin file is unchanged. +Generation/repair dormancy cases replay golden patches and prove that the +requested assertions reject wrong costs/flags while production stays +byte-for-byte unchanged. The CLI integration needs Copilot access via +an existing token or authenticated GitHub CLI; it does not change user settings. ### Coverage & risk *(.NET only)* @@ -81,7 +119,7 @@ deliberately have no direct `tests/dotnet-test//eval.yaml`: the experiment's skilled arm loads a single skill, which the model could never invoke here, so such an eval would compare two identical arms and score judge noise. They are measured through consumer outcomes — the polyglot analysis -skills and `grade-tests` for `test-analysis-extensions`, `code-testing-agent` +skills and `grade-tests` for `test-analysis-extensions`, `code-testing` for `code-testing-extensions`, and `run-tests` and `mtp-hot-reload` for `filter-syntax`. The `run-tests` eval covers VSTest expressions, MTP argument passing, xUnit v3 native filters, and TUnit tree-node filters. @@ -100,43 +138,56 @@ plugin's skills, but not these agents or their static handoffs. ### User-facing agents -These are the entry-point agents you invoke directly: +Use this single entry-point agent for end-to-end test work: | Agent | Purpose | |---|---| -| **test-quality-auditor** | Routes focused quality requests, including curated per-test decisions, and runs multi-skill pipelines for comprehensive suite assessment | -| **testability-migration** | End-to-end testability improvement: detect → generate wrappers → migrate call sites → add deterministic tests when requested | +| **test-engineer** | Generates, repairs, runs, audits, and improves tests while coordinating the internal specialists below | > **Test framework/platform migration** is handled by the `test-migration` agent in the separate [`dotnet-test-migration`](../dotnet-test-migration/) plugin. ### Internal subagents -These agents are internal (`user-invocable: false`); you do not need to call them directly. For broad requests, the `code-testing-agent` skill invokes the named `code-testing-generator` once when available. The generator owns the pipeline and returns evidence for the caller to reuse. Research, planning, implementation, and review remain required, but run inline by default, including bounded project-wide suites. The other agents below are optional workers for substantial work that benefits from separate context, used only when available in the runtime. Focused additions stay direct, without intermediate state artifacts or agent fan-out. +These specialists are available to `test-engineer` (`user-invocable: false`); +you do not need to call them directly. The agent owns research, planning, +implementation, and review inline by default, and delegates only substantial +work that benefits from separate context. -| Agent | Caller | Purpose | +| Agent | Called by | Purpose | |---|---|---| -| **code-testing-generator** | code-testing-agent skill (broad requests) | Owns the full test generation pipeline, with phases inline by default | -| **code-testing-researcher** | code-testing-generator (optional) | Analyzes codebase structure, testing patterns, and testability | -| **code-testing-planner** | code-testing-generator (optional) | Creates phased test implementation plans from research findings | -| **code-testing-implementer** | code-testing-generator (optional) | Implements one phase from the plan, runs build-test-fix cycles | -| **code-testing-builder** | code-testing-generator or delegated implementer (optional) | Runs build/compile commands and reports results | -| **code-testing-tester** | code-testing-generator or delegated implementer (optional) | Runs test commands and reports pass/fail results | -| **code-testing-fixer** | code-testing-generator or delegated implementer (optional) | Fixes compilation errors in source or test files | -| **code-testing-linter** | code-testing-generator or delegated implementer (optional) | Runs code formatting and linting | - -> **VS Code — optional nested delegation:** The pipeline does not require phase-agent fan-out. When substantial work warrants a subagent invoking another available named agent, VS Code gates that *nested* delegation behind a setting that is **off by default**. To allow it, enable this in your VS Code settings: +| **test-quality-auditor** | test-engineer | Runs multi-skill audit pipelines for comprehensive test-suite assessment | +| **testability-migration** | test-engineer | Performs explicit .NET production-code testability refactors and adds deterministic tests | +| **code-testing-researcher** | test-engineer | Analyzes codebase structure, testing patterns, and testability | +| **code-testing-planner** | test-engineer | Creates phased test implementation plans from research findings | +| **code-testing-implementer** | test-engineer | Implements one phase from the plan, runs build-test-fix cycles | +| **code-testing-builder** | code-testing-implementer | Runs build/compile commands and reports results | +| **code-testing-tester** | code-testing-implementer | Runs test commands and reports pass/fail results | +| **code-testing-fixer** | code-testing-implementer | Fixes compilation errors in source or test files | +| **code-testing-linter** | code-testing-implementer | Runs code formatting and linting | + +> **VS Code — optional nested delegation:** The pipeline does not require +> phase-agent fan-out. When substantial work warrants a subagent invoking +> another available named agent, VS Code gates that nested delegation behind a +> setting that is **off by default**. To allow it, enable: > > ```jsonc > "chat.subagents.allowInvocationsFromSubagents": true > ``` > -> Without it, the generator or delegated implementer completes the phases inline; required validation and review are unchanged. The GitHub Copilot CLI has no such gate, but delegation is still optional: phases stay inline by default, and agents are used only for substantial separate-context work when available. +> Without it, `test-engineer` or a delegated implementer completes the phases +> inline; required validation and review are unchanged. The GitHub Copilot CLI +> has no such gate, but delegation remains optional. ## Prerequisites ### For polyglot skills and agents -The test-generation pipeline (`code-testing-generator` and friends) and the six test-analysis skills (`test-anti-patterns`, `test-smell-detection`, `assertion-quality`, `test-gap-analysis`, `test-tagging`, `grade-tests`) plus the `test-quality-auditor` agent work with any of the supported languages above. You just need a working test runtime for the language you're targeting (e.g., `python` + `pytest`, `node` + `npm test`, `mvn` / `gradle`, `go`, `bundle exec rspec`, `cargo test`, `swift test`, `pwsh` + Pester, `cmake` + your C++ test runner). The skills will detect the framework automatically. +The `test-engineer` agent, `code-testing` skill, generation workers, internal +quality auditor, and six test-analysis skills (`test-anti-patterns`, +`test-smell-detection`, `assertion-quality`, `test-gap-analysis`, +`test-tagging`, `grade-tests`) work with any supported language above. You just +need a working test runtime for the target language (for example `pytest`, +`npm test`, `mvn`, `go`, `cargo test`, Pester, or CMake plus a C++ test runner). ### For .NET-only skills and agents diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-implementer.agent.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-implementer.agent.md index 0bc2201f..e0b96d01 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-implementer.agent.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-implementer.agent.md @@ -24,8 +24,8 @@ You implement a single phase from the test plan. You are polyglot — you work w > Call `code-testing-extensions` only when the required implementation or > harness-discovery section is missing and the skill is available. -Stay in the caller's phase: never invoke the public `code-testing-agent` skill -or delegate back to `code-testing-generator`. Use supplied guidance and known +Stay in the caller's phase: never invoke the public `code-testing` skill +or delegate back to `test-engineer`. Use supplied guidance and known paths instead. Record unavailable skills and denied operations once; do not retry aliases, alternate shells, or another agent for the same restriction. Continue permitted test edits and static review when execution is blocked, @@ -99,9 +99,9 @@ These rules apply to every language and override any pattern an existing test fi #### Test depth (cross-language invariants) -Coverage alone gives false confidence — every test must *pin down behavior* so it would fail under a plausible bug. Apply the `code-testing-agent` skill's `unit-test-generation.prompt.md` → "Write Tests That Pin Down Behavior" section: mutation thinking (each assertion fails under a plausible mutation), no tautological round-trip assertions, property intersections, secondary observables when they are contractual or prove a requested interaction, and realistic (non-degenerate) fixtures. This is a depth requirement on top of the happy/edge/error-path and mocking rules above, and applies to every language. +Coverage alone gives false confidence — every test must *pin down behavior* so it would fail under a plausible bug. Apply the `code-testing` skill's `unit-test-generation.prompt.md` → "Write Tests That Pin Down Behavior" section: mutation thinking (each assertion fails under a plausible mutation), no tautological round-trip assertions, property intersections, secondary observables when they are contractual or prove a requested interaction, and realistic (non-degenerate) fixtures. This is a depth requirement on top of the happy/edge/error-path and mocking rules above, and applies to every language. -Also apply [Report-safe test names and result validation](../skills/code-testing-agent/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation) +Also apply [Report-safe test names and result validation](../skills/code-testing/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation) when naming cases and accepting test results. Preserve risky data and assertions; pass the contract to a delegated tester rather than relying on console-green. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-tester.agent.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-tester.agent.md index 4fcf6108..1b600254 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-tester.agent.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-tester.agent.md @@ -23,7 +23,7 @@ You run tests and report the results. You are polyglot — you work with any pro Run the appropriate test command and report pass/fail with actionable details. Do not modify tests, production code, dependencies, or runner configuration. -Apply [Report-safe test names and result validation](../skills/code-testing-agent/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation) +Apply [Report-safe test names and result validation](../skills/code-testing/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation) before reporting passage. Report unsafe metadata or export failures to the caller for repair; do not change test data or runner configuration yourself. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-generator.agent.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/test-engineer.agent.md similarity index 86% rename from external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-generator.agent.md rename to external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/test-engineer.agent.md index a1fe1f2a..f865b464 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/code-testing-generator.agent.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/test-engineer.agent.md @@ -1,13 +1,17 @@ --- description: >- - Required internal implementation agent for broad or comprehensive - code-testing-agent requests spanning a project, package, or multiple modules. - Orchestrates the Research-Plan-Implement pipeline after the public entry-point - skill delegates. Do not route user prompts here directly. -name: code-testing-generator -user-invocable: false -tools: ["agent", "skill", "read", "search", "edit", "execute", "Task", "Skill", "Read", "Glob", "Grep", "Edit", "Write", "Bash", "read_file", "replace", "write_file", "glob", "grep_search", "run_shell_command"] + Primary test engineering agent for generating, repairing, running, auditing, + and improving tests across supported languages. Handles focused work + directly; coordinates broad generation through specialist workers, quality + assessment through test-quality-auditor, and explicit .NET testability + refactors through testability-migration. Use for end-to-end test work. Do not + use for test framework or platform migrations; use test-migration instead. +name: test-engineer +user-invocable: true +disable-model-invocation: false agents: + - test-quality-auditor + - testability-migration - code-testing-researcher - code-testing-planner - code-testing-implementer @@ -18,24 +22,70 @@ agents: license: MIT --- -# Test Generator Agent +# Test Engineer Agent -Your active identity is `code-testing-generator`, including when the host -qualifies it as `dotnet-test:code-testing-generator`. You are not the public -entry-point caller that needs to invoke this agent. +You are the single public entry point for test engineering. You generate, +repair, execute, audit, and improve tests, delegating to internal specialists +only when that produces a better result than handling the request directly. +You are polyglot and preserve each repository's existing framework and +conventions. + +## Intent Routing + +Classify the request before acting: + +| Intent | Route | +| --- | --- | +| Add, write, or generate focused tests | Work directly using the Direct strategy below | +| Generate tests across multiple files, modules, or projects | Use the Research-Plan-Implement workflow below | +| Fix failing, flaky, or weak tests | Reproduce the narrow failure, fix its root cause, and run the smallest covering test command | +| Audit test quality without edits | Delegate to `test-quality-auditor`, then return its prioritized findings | +| Audit and improve tests | Delegate the assessment to `test-quality-auditor`, then implement and verify the agreed or explicitly requested fixes | +| Run tests without requesting changes | Use `run-tests` for .NET or the repository's native runner for other languages | +| Remove static coupling or create a missing test seam | Delegate to `testability-migration` only when the user explicitly requests a production testability refactor | +| Migrate a test framework or platform | Stop and route to the separate `test-migration` agent | + +Do not bounce the user between internal agents. Preserve the original request, +collect specialist results, and deliver one coherent outcome. When invoked by +the `code-testing` skill, continue the task directly; never invoke another +`test-engineer`. If a named internal specialist is unavailable, execute its +documented skill workflow inline rather than dropping that part of the request. + +## Repair Workflow + +For failing, flaky, or weak tests: + +1. Reproduce the smallest relevant failure before editing. +2. Classify the cause as an incorrect expectation, a production regression, a + nondeterministic test dependency, or test infrastructure/configuration. +3. Fix the root cause without weakening assertions, skipping tests, adding + arbitrary retries, or changing intended production behavior. +4. Run the narrow covering command, then the repository's normal test entry + point when the change can affect a broader scope. +5. Report the failing evidence, the correction, and the clean validation + command. + +## Quality Workflow + +For analysis-only audits, delegate to `test-quality-auditor` and preserve the +requested read-only scope. For audit-and-fix requests, use the auditor's +prioritized findings as an implementation checklist, fix the highest-impact +false-confidence and coverage gaps in scope, and rerun the affected tests. +Never treat aggregate coverage alone as proof that the requested behavior is +tested. You own the Research-Plan-Implement (RPI) pipeline for the caller's bounded test generation request. You are polyglot — you work with any programming language. -For every strategy, apply [Report-safe test names and result validation](../skills/code-testing-agent/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). +For every strategy, apply [Report-safe test names and result validation](../skills/code-testing/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). Pass that contract with the relevant guidance to delegated implementers/testers. ## Execution ownership and capability limits -- **Do not re-enter the public entry point.** You are already the generator. - Do not invoke `code-testing-agent`, delegate to `code-testing-generator`, or - ask another agent to restart the pipeline. Reuse guidance already supplied - by the caller; read a specific supporting document only when needed. +- **Do not re-enter the public entry point.** When `code-testing` invoked you, + continue the request directly. Never invoke another `test-engineer` or reload + `code-testing` to restart the pipeline. Reuse guidance already supplied by + the caller; read a specific supporting document only when needed. - **Phases are not agent calls.** Complete research, planning, implementation, and review in this context by default, including small project-wide suites. Delegate only substantial work that benefits from separate context, to a @@ -69,7 +119,7 @@ When shell execution is unavailable, review the recorded file edits and permitted file-tool output instead of running `git status` for the final working-tree review. That review does not authorize another denied command. -## Pipeline Overview +## Generation Pipeline Overview 1. **Research** — Understand the codebase structure, testing patterns, and what needs testing 2. **Plan** — Create a phased test implementation plan @@ -84,7 +134,7 @@ framework preferences. If details are incomplete, make the narrowest reasonable assumption from the working directory and repository conventions, state it, and proceed. If the user provides no details or a very basic prompt (e.g., "generate tests"), use -[unit-test-generation.prompt.md](../skills/code-testing-agent/unit-test-generation.prompt.md) +[unit-test-generation.prompt.md](../skills/code-testing/unit-test-generation.prompt.md) for default conventions, coverage goals, and test quality guidelines. Before writing code, use the available language-specific base extension or diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/test-quality-auditor.agent.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/test-quality-auditor.agent.md index 8ada2582..6b9c1bad 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/test-quality-auditor.agent.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/test-quality-auditor.agent.md @@ -1,13 +1,12 @@ --- name: test-quality-auditor description: >- - MUST USE for test-suite quality audits, from focused assertion, anti-pattern, - smell, gap, coverage, mock, or tagging reviews through broad multi-dimensional - health checks across a project/workspace. For a focused request, invoke only - the matching specialist skill; reserve the combined audit pipeline for broad - requests. Supports .NET and common non-.NET test frameworks. DO NOT USE to - write, generate, or fix tests; use the public code-testing-agent skill instead. -user-invokable: true + Internal quality specialist for the test-engineer agent. Handles focused + assertion, anti-pattern, smell, gap, coverage, mock, or tagging reviews and + broad multi-dimensional health checks. For focused requests, invoke only the + matching specialist skill; reserve the combined audit pipeline for broad + requests. Supports .NET and common non-.NET test frameworks. +user-invocable: false disable-model-invocation: false license: MIT --- @@ -34,7 +33,7 @@ broad health check: | CRAP or coverage-and-complexity risk for one named method, class, or file | `crap-score` | | Tags, traits, or test-type distribution | `test-tagging` | | Curated tests needing a PR-ready Pass / Failed / Uncertain decision | `grade-tests` | -| Generate or repair tests | `code-testing-agent`; it uses its direct workflow for focused work and delegates broad work to `code-testing-generator` | +| Generate or repair tests | Return the findings to the invoking `test-engineer`; generation and repair are outside this diagnostic specialist | For a focused request, invoke the matching skill once and stop. A request to grade a curated list is a focused decision report, not an audit dimension: route diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/testability-migration.agent.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/testability-migration.agent.md index d32b63ad..da57c655 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/testability-migration.agent.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/agents/testability-migration.agent.md @@ -1,23 +1,13 @@ --- description: >- - MUST USE for .NET testability migration requests, from static-dependency - inventories and one named dependency migration through broad end-to-end work - coordinating seam selection, call-site migration, production wiring, and - deterministic tests. Scale to the request: invoke one specialist for focused - work and the full pipeline only for multi-phase or multi-dependency work. DO - NOT USE when one bounded behavior needs both a new minimal seam and tests - (testability-obstacle), or when an existing seam only needs tests. + Internal .NET testability specialist for the test-engineer agent. Handles + static-dependency inventories and named dependency migrations through broad + seam selection, call-site migration, production wiring, and deterministic + tests. Scale to the request and use testability-obstacle when one bounded + behavior needs both a minimal new seam and tests. name: testability-migration -agents: - - code-testing-generator -handoffs: - - label: Generate Tests for Migrated Code - agent: code-testing-generator - prompt: >- - The code has been migrated to use injectable abstractions. Please - generate unit tests for the migrated classes, using test doubles for - the new wrapper interfaces. - send: false +user-invocable: false +disable-model-invocation: false license: MIT --- @@ -30,8 +20,9 @@ You are a testability migration agent for .NET codebases. Your mission is to hel Choose one of three paths: - **Migration pipeline:** **Detect → Generate → Migrate → Test** for a broad or - multi-call-site migration. After migration, the seam exists; generate tests - through `code-testing-generator`. + multi-call-site migration. After migration, the seam exists; write the + deterministic tests inline. Do not invoke `code-testing` or `test-engineer` + from this internal specialist. - **Focused migration:** for an inventory-only request, invoke `detect-static-dependencies` and stop. For one named dependency, invoke `migrate-static-to-wrapper`; stop after migration only when tests were not @@ -105,7 +96,8 @@ Use the `migrate-static-to-wrapper` skill to: ### Phase 4: Test -After Phase 3, use `code-testing-generator` to: +After Phase 3, write the requested tests inline. Do not invoke `code-testing` +or `test-engineer`; this agent is already running under the public orchestrator. 1. Reuse the migrated seam rather than introducing another abstraction. 2. Use `FakeTimeProvider`, an in-memory filesystem, or a hand-rolled fake. @@ -128,7 +120,7 @@ Use `testability-obstacle` instead of Phases 1–4 when all are true: 3. The user asks for both the minimal production refactor and deterministic tests. Do not first generate/migrate a wrapper and then invoke `testability-obstacle`; -once the seam exists, test it with `code-testing-generator`. +once the seam exists, test it directly. ## Decision Rules diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/plugin.json b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/plugin.json index 84622731..d41265fb 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/plugin.json +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/plugin.json @@ -1,10 +1,10 @@ { "name": "dotnet-test", - "version": "0.2.23", + "version": "1.0.0", "description": "Skills for running, generating, analyzing, and improving .NET tests: test execution, filtering, platform detection, coverage, testability, and MSTest workflows.", "skills": ["./skills/"], "agents": [ - "./agents/code-testing-generator.agent.md", + "./agents/test-engineer.agent.md", "./agents/code-testing-researcher.agent.md", "./agents/code-testing-planner.agent.md", "./agents/code-testing-implementer.agent.md", diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/assertion-quality/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/assertion-quality/SKILL.md index e6649488..95fbeba0 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/assertion-quality/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/assertion-quality/SKILL.md @@ -1,6 +1,6 @@ --- name: assertion-quality -description: "Analyze assertion quality, depth, variety, and false confidence in existing tests. ALWAYS USE when asked about weak, shallow, trivial, always-true, self-referential, assertion-free, presence/truthiness-only, or insufficiently diverse assertions, including MSTest, Jest, pytest, and Go. DO NOT USE for direct fixes: writing-mstest-tests owns supplied MSTest assertions; code-testing-agent owns new cases. Use test-gap-analysis when asked whether tests would catch a production change, and test-anti-patterns for general severity-ranked audits." +description: "Analyze assertion quality, depth, variety, and false confidence in existing tests. ALWAYS USE when asked about weak, shallow, trivial, always-true, self-referential, assertion-free, presence/truthiness-only, or insufficiently diverse assertions, including MSTest, Jest, pytest, and Go. DO NOT USE for direct fixes: writing-mstest-tests owns supplied MSTest assertions; code-testing owns new cases. Use test-gap-analysis when asked whether tests would catch a production change, and test-anti-patterns for general severity-ranked audits." license: MIT --- @@ -30,11 +30,11 @@ Low assertion diversity signals shallow testing. Tests may pass while bugs hide - User wants to know if test assertions are too shallow or trivial - User asks for assertion coverage metrics or diversity analysis - User suspects tests give false confidence despite passing -- The `code-testing-generator` agent (or any test-generation workflow) calls this skill as a pre-completion self-review step on freshly generated tests, before declaring the run finished +- The `test-engineer` agent (or any test-generation workflow) calls this skill as a pre-completion self-review step on freshly generated tests, before declaring the run finished ## When Not to Use -- User wants to write new tests (use `code-testing-agent` for any language, or `writing-mstest-tests` for MSTest specifically) +- User wants to write new tests (use `code-testing` for any language, or `writing-mstest-tests` for MSTest specifically) - User wants to detect anti-patterns beyond assertions (use `test-anti-patterns`) - User wants to fix or rewrite assertions (help them directly) - User asks about code coverage percentages (out of scope — this analyzes assertion quality, not line coverage) diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-agent/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-agent/SKILL.md index 31d6cd94..d746de1f 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-agent/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-agent/SKILL.md @@ -1,372 +1,19 @@ --- name: code-testing-agent description: >- - ALWAYS USE whenever asked to write, add, or generate unit tests for existing - code in xUnit, MSTest, NUnit, pytest, Vitest/Jest, Go, or another framework, - including "tests only for" one helper, function, class, or missing regression - case as well as project-wide suites. Also use for "cover this untested method", - scaffolding tests where none exist, sparse workspaces, classic packages.config - MSTest, and extending healthy suites. Focused requests use a proportional - direct workflow; broad requests use the full pipeline. DO NOT USE for only - running/diagnosing tests, coverage/audits, a test blocked on a missing - production seam (testability-obstacle), or correcting supplied MSTest - assertions, attributes, lifecycle, or configuration without designing new - cases (writing-mstest-tests). Within an active code-testing-generator - pipeline, reuse supplied guidance; do not re-enter this skill. + Legacy compatibility alias for the code-testing skill. Do not self-activate + for user requests and do not treat this name as a custom agent. When invoked + explicitly by older instructions, load code-testing and follow it instead. +user-invocable: false +disable-model-invocation: true license: MIT --- -# Code Testing Generation Skill +# Legacy Code Testing Alias -Generate comprehensive, workable unit tests for any programming language using -a bounded Research → Plan → Implement workflow. +This skill preserves explicit calls that still use the former +`code-testing-agent` name. The public custom agent is `test-engineer`; the +model-facing implicit skill is `code-testing`. -## Non-negotiable execution contract - -**Check pipeline ownership first.** If the active agent is -`code-testing-generator` (including a plugin-qualified name such as -`dotnet-test:code-testing-generator`), or the caller assigned you a phase of -that pipeline, do not delegate to another generator. Continue the assigned -work inline. This guard takes precedence over every broad-scope delegation -instruction below, even if this skill was loaded automatically. - -Classify scope **before editing**: - -- **Broad** (a project/package-wide suite, or multiple production - files/modules): create `research.md` and `plan.md` in a resolved - non-stageable `` before implementation, then `status.md` there - after the final test-quality review. When `code-testing-generator` is - available, invoke that named custom agent before implementing; do not replace - it with a generic subagent carrying the same label or implement the broad - request inline. If the state files are absent, the broad workflow is - incomplete. -- **Focused** (the user explicitly limits work to one function/class/file or one - missing method): do not create intermediate state files or fan out to multiple - agents. A sparse project-wide request remains broad even when only one source - module is present. - -For either scope, run the narrowest relevant test command to a clean exit. -Always apply [Report-safe test names and result validation](unit-test-generation.prompt.md#report-safe-test-names-and-result-validation), -including when the caller supplies conventions. Pass this contract to delegated -implementers/testers; preserve edge-case data and validate configured reports, -not just console output. -Keep the handoff proportional: for one to three focused requirements, use a -compact bullet list under a **Requirement coverage** label that names the tests -and successful command; for broader or multi-requirement work, use a -`Requirement | Evidence` table. Each requested behavior must cite an exact test -name. - -Before sending a broad-scope final response, check that the response itself -contains `| Requirement | Evidence |` and exact test names for every behavioral -row. A table in a child report or internal plan is not enough. Do not summarize -away those names into module-level bullets or an `Area | Tests` table. - -Intermediate state files are internal working data, never deliverables. Keep -`` non-stageable, never place it or its files in -version-controlled workspace content, and never modify `.gitignore` to hide -them. - -Treat completeness as a requirement matrix, not a test-count target. Give every -independently requested state, boundary, error path, or interaction its own -concrete assertion. Combine cases only when one execution genuinely proves the -whole requested combination; do not let a parameterized happy-path case stand in -for an empty state, invalid discriminator, or before/at/after boundary. -For broad requests that name several production modules or layers, give each -named module direct tests for its non-trivial public behavior. Cross-module tests -prove composition, but do not substitute for the requested module-level -coverage. Judge breadth by the behavior matrix, never by matching or exceeding a -raw test count. - -At the public entry point, delegate broad work to `code-testing-generator` -once. Research, plan, implementation, and review remain required, but they -need not be separate sub-agent calls. - -Use only capabilities available in the current runtime. Do not retry a missing -skill under aliases or use another agent to retry a policy-denied operation. -If scratch storage is denied, keep the research and plan in context, continue -permitted test edits, and report the missing state artifacts. If execution is -denied, continue permitted static review and report tests as unrun, never passed. -Neither blocker authorizes modifying production code or weakening requirements. - -For a **broad or comprehensive** request, the explicit matrix is the floor, not -the ceiling. Treat each requested module or layer as an inventory heading, not -one behavior: expand it into the bounded public operations and their distinct -validation paths, branches, boundaries, interactions, and state transitions. -After satisfying the explicit matrix, inspect each target API for observable -equivalence partitions and invariants that the prompt did not name: identity, -empty, singleton and representative interior inputs; exact boundaries plus an -immediately adjacent value; invalid partitions; and ordering, monotonicity, -rollover, capacity, truncation, or state invariants implied by the implementation. -Add one mutation-relevant case per distinct partition not already proved, using -parameterized or table-driven cases only for siblings that prove the same -behavior. A passing coverage threshold is validation, not a breadth stop -condition. Stop when remaining inputs exercise the same branch and invariant, -not merely when the explicit checklist is complete; never add cases only to -raise the count. - -## When to Use This Skill - -Use this skill when you need to: - -- Generate unit tests for an entire project or specific files -- Improve test coverage for existing codebases -- Create test files that follow project conventions -- Write tests that actually compile and pass -- Add tests for new features or untested code -- Generate or extend MSTest suites; load `writing-mstest-tests` as supporting - guidance after this entry skill has established scope and project conventions - -## When Not to Use - -- Running or executing existing tests (use the `run-tests` skill) -- Migrating between test frameworks (use migration skills) -- Answering an MSTest API/pattern or modernization question that does not ask to - generate tests (use `writing-mstest-tests`) -- Debugging failing test logic - -## How It Works - -This skill coordinates multiple specialized agents in a **Research → Plan → Implement** pipeline: - -### Pipeline Overview - -```text -┌─────────────────────────────────────────────────────────────┐ -│ TEST GENERATOR │ -│ Coordinates the full pipeline and manages state │ -└─────────────────────┬───────────────────────────────────────┘ - │ - ┌─────────────┼─────────────┐ - ▼ ▼ ▼ -┌───────────┐ ┌───────────┐ ┌───────────────┐ -│ RESEARCHER│ │ PLANNER │ │ IMPLEMENTER │ -│ │ │ │ │ │ -│ Analyzes │ │ Creates │ │ Writes tests │ -│ codebase │→ │ phased │→ │ per phase │ -│ │ │ plan │ │ │ -└───────────┘ └───────────┘ └───────┬───────┘ - │ - ┌─────────┬───────┼───────────┐ - ▼ ▼ ▼ ▼ - ┌─────────┐ ┌───────┐ ┌───────┐ ┌───────┐ - │ BUILDER │ │TESTER │ │ FIXER │ │LINTER │ - │ │ │ │ │ │ │ │ - │ Compiles│ │ Runs │ │ Fixes │ │Formats│ - │ code │ │ tests │ │ errors│ │ code │ - └─────────┘ └───────┘ └───────┘ └───────┘ -``` - -## Step-by-Step Instructions - -### Step 1: Determine the user request - -Make sure you understand what user is asking and for what scope. -When the user does not express strong requirements for test style, coverage goals, or conventions, source the guidelines from [unit-test-generation.prompt.md](unit-test-generation.prompt.md). This prompt provides best practices for discovering conventions, parameterization strategies, behavior-focused coverage, and language-specific patterns. - -### Step 2: Size the request before invoking anything - -Match the machinery to the scope. Running the full pipeline on a one-file -request costs turns and tool calls without improving the tests. - -| Scope | What it looks like | How to run it | -| --- | --- | --- | -| **Focused** | One function, class, or file; "tests for X only"; extending an existing suite with the missing cases | Skip intermediate state files and the sub-agent fan-out. Keep the requirement checklist in your head (or in the final table), read only the target and one neighbouring test for conventions, write the tests, run the narrowest test command, review your own assertions inline. | -| **Broad** | A project, package, or module set; "comprehensive suite"; a coverage threshold to clear across several files | Run the full Research → Plan → Implement pipeline in Step 3, with intermediate state files under `` and the completion contract below. | - -When in doubt, start focused and escalate only if the request turns out to span -several files. Escalating costs one extra pass; running the broad pipeline on a -focused request costs several. - -Before ending a focused request, check all three conditions together: - -1. every named behavior has a concrete assertion, including each requested - boundary or error path; -2. the narrow test command exited successfully; -3. the final handoff maps those behaviors to exact test names and cites that - successful command. - -Do not replace requirement-level evidence with a generic list of covered areas. - -### Step 3: Invoke the Test Generator (broad scope) - -Start by invoking the named `code-testing-generator` custom agent with your test -generation request. Do not use a generic/general-purpose subagent merely named -`code-testing-generator`: - -```text -You are the sole pipeline owner for this request. Do not invoke code-testing-agent or another code-testing-generator; complete the phases in your current context. Generate unit tests for [path or description of what to test], following the [unit-test-generation.prompt.md](unit-test-generation.prompt.md) guidelines. Treat the current workspace as authoritative even when it is sparse, gutted-looking, synthetic, or missing tracked files; never restore or reconstruct it, including with `git checkout`, `git restore`, `git reset`, or `git clean`. -``` - -The Test Generator owns the pipeline. After it returns, consume its recorded -quality checks, validation results, and requirement matrix instead of repeating -Steps 4 and 5 as another pipeline. Do not reload review skills or rerun unchanged -passing commands. Preserve exact test names from its evidence in the final -handoff. If evidence is missing, inspect or follow up on that specific gap -without restarting generation. A reported capability-wide denial also applies -to the caller; do not attempt another command using that capability. - -If `code-testing-generator` is unavailable, do not skip the workflow. Execute the -same Research → Plan → Implement sequence inline, resolve `` as -described below, create the intermediate state files there, and apply the same -completion contract. - -For broad scope, resolve one absolute `` before creating -intermediate state files: - -1. Prefer a host-provided session artifact or scratch directory. -2. Otherwise, in a Git worktree run - `git rev-parse --path-format=absolute --git-path testagent`; this returns a - path in worktree-specific Git metadata that cannot be staged. -3. Outside Git, create a unique directory under the operating system's - temporary directory. - -Pass the absolute directory to every pipeline agent. The path may be inside the -repository's `.git` metadata directory, but it must not be version-controlled -workspace content, appear in `git status`, or be stageable. - -### Step 4: Execute with bounded context - -For multi-file requests: - -1. Turn every explicit user requirement into a checklist before implementation. Include requested layers, collaborators to mock, boundary cases, integrations, coverage thresholds, and report artifacts. Copy multi-condition requirements verbatim — they must each map to one test that exercises the whole combination. -2. Research only the requested module or project and write the checklist plus a compact target inventory to `/research.md`. -3. Reuse manifests, symbol references, and deterministic pairing tools instead of reading every source and test file. -4. When an available `find-untested-sources` skill is useful for a substantial multi-file inventory, run it once and reuse its pairing and suggested-path output. Otherwise pair the bounded targets manually once; do not probe for an unavailable skill. -5. Plan each target file once, then implement phases sequentially. Map every checklist item to at least one concrete test or explain why it is blocked. -6. Build and test the narrow target during fix cycles. Run workspace-level - validation once at the end only for broad work, when the repository contract - requires that entry point, or when the changes can affect other projects. -7. Before reporting success, re-open the generated tests and verify every checklist item against concrete test names and assertions. Coverage alone is not evidence that a requested mock seam, boundary, state transition, or property combination was tested. -8. Read a language example from `code-testing-extensions` only when the repository has no representative tests and the base extension is insufficient. -9. For .NET, classify SDK-style vs. classic non-SDK before choosing commands or creating files. In classic projects, preserve `packages.config`, existing framework/mock versions and custom base fixtures, add every new test file to the project's explicit `` items, and use the repository's MSBuild/test-runner commands. Never modernize the project or dependency stack merely to generate tests. -10. For MSTest, inspect the pinned package version before choosing exception - assertions. MSTest 3.5.x uses `Assert.ThrowsException`; do not substitute - `[ExpectedException]`, `Assert.Throws`, or `Assert.ThrowsExactly`. - -### Completion contract - -Every scope must satisfy points 3–5 below. Points 1 and 2 are the **broad-scope** -artifacts: on a focused request the same reasoning happens inline and no -intermediate state files are written. - -Do not report completion until all of these are true: - -1. *(broad scope)* `/research.md` records the bounded target - inventory, existing test conventions, and the acceptance checklist. -2. *(broad scope)* `/plan.md` maps each checklist item to a planned - test or an explicit blocker. -3. Generated tests compile and pass with the narrowest relevant test command, - satisfying the shared report-safe naming and result-validation contract. -4. Every explicit user requirement is backed by a concrete test and assertion. - Fix missing mock seams, boundary cases, state transitions, and property - combinations even when coverage already passes. In the final summary, cite - at least one generated test name for every checklist item so completion is - auditable; if an item has no test to cite, keep implementing or report it as - blocked. For non-behavioral requirements such as scaffolding, scope limits, - commands, or coverage artifacts, cite the relevant file, command, or report - instead of forcing a test-name mapping. - A passing suite with fewer tests is not automatically weaker: judge - completeness by whether every independently requested behavior has direct, - nonredundant evidence, not by raw test volume. - For broad/comprehensive scope, also verify that every observable equivalence - partition and invariant discovered in the bounded target APIs has one - mutation-relevant case, even when the prompt did not name it. - When the request names multiple modules, verify that each module's own - non-trivial public behavior has direct test evidence in addition to any - end-to-end composition test. -5. Review the generated tests for behavior gaps and weak assertions. On a broad - scope, invoke `test-gap-analysis` and `assertion-quality` when available and - record the findings and fixes in `/status.md`. On a focused scope, - do the equivalent review inline — re-read each generated assertion against - the source — without spawning extra passes. - -The final response must provide requirement-by-requirement evidence. Use compact -bullets under a **Requirement coverage** label for one to three focused -requirements; use a `Requirement | Evidence` table for broader scopes. -Behavioral evidence cites exact generated test names. Non-behavioral evidence -cites the relevant project file, validation command, or coverage report. A -generic list of tested areas is not a substitute. - -Preserve the user's exact meaning in each evidence item; quote verbatim only -when wording distinguishes a required combination. A test that merely exercises -the same collaborators does not satisfy a requirement about their interaction, -and per-class requirements need a citation per class. - -**Cite a clean run, not an attempt.** The commands behind the final evidence must -have finished successfully: quote the final passing test summary and, when -thresholds were requested, the per-module coverage table from a run that exited -0. If the last coverage run exited non-zero, fix it and re-run before reporting; -never infer threshold clearance from a failed or partial run. - -Before reporting, inspect the final working-tree changes and confirm that -`research.md`, `plan.md`, `status.md`, and any other intermediate state files are -not among the changes intended for commit. - -## State Management - -Broad-scope runs store intermediate state files in a non-stageable -`` backed by host scratch storage, Git metadata, or OS temp. A -focused request does not create these files: - -| File | Purpose | -| ------------------------ | ---------------------------- | -| `/research.md` | Codebase analysis results | -| `/plan.md` | Phased implementation plan | -| `/status.md` | Final quality review and fixes | - -## Agent Reference - -| Agent | Purpose | -| -------------------------- | -------------------- | -| `code-testing-generator` | Coordinates pipeline | -| `code-testing-researcher` | Analyzes codebase | -| `code-testing-planner` | Creates test plan | -| `code-testing-implementer` | Writes test files | -| `code-testing-builder` | Compiles code | -| `code-testing-tester` | Runs tests | -| `code-testing-fixer` | Fixes errors | -| `code-testing-linter` | Formats code | - -## Requirements - -- Project must have a build/test system configured -- Testing framework should be installed (or installable) -- VS Code with GitHub Copilot extension - -Classic non-SDK .NET projects are supported when their existing build/test -toolchain is available. When it is not available on the current machine, the -agent can still add and register version-compatible tests, but must report -execution as blocked rather than substituting `dotnet test`. - -## Troubleshooting - -### Tests don't compile - -The `code-testing-fixer` agent will attempt to resolve compilation errors. Check -`/plan.md` for the expected test structure. Call the -`code-testing-extensions` skill and read the language-specific extension file -for error code references (e.g., `dotnet.md` for .NET). - -### Tests fail - -Most failures in generated tests are caused by **wrong expected values in assertions**, not production code bugs: - -1. Read the actual test output -2. Read the production code to understand correct behavior -3. Fix the assertion, not the production code -4. Never mark tests `[Ignore]` or `[Skip]` just to make them pass - -### Wrong testing framework detected - -Specify your preferred framework in the initial request: "Generate Jest tests for..." - -### Environment-dependent tests fail - -Tests that depend on external services, network endpoints, specific ports, or precise timing will fail in CI environments. Focus on unit tests with mocked dependencies instead. - -### Broader validation fails - -During implementation, build and test the narrow target. Run a solution or -workspace-level command only for broad work, when the repository contract uses -that entry point, or when the targeted change can affect other projects. Do not -turn a focused test request into an unconditional full non-incremental build. +1. Invoke the `code-testing` skill with the original user request. +2. Follow that skill's workflow without adding another interpretation layer. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/dotnet-examples.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/dotnet-examples.md index 8e766021..65fba2c6 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/dotnet-examples.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/dotnet-examples.md @@ -336,7 +336,7 @@ var sut = new InvoiceService(repositoryMock.Object); ## Sample Final Report -What `code-testing-generator` produces at Step 9: +What `test-engineer` produces at Step 9: ```markdown ## Test Generation Report diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/powershell.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/powershell.md index 25df6742..06cafc87 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/powershell.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/powershell.md @@ -122,7 +122,7 @@ Pester v5 runs in **two phases**: Discovery (collects test metadata) then Run (e ## Parameterized Test Display Names -Apply [Report-safe test names and result validation](../../code-testing-agent/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). +Apply [Report-safe test names and result validation](../../code-testing/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). Use an explicit safe `Name`/`Case` in `-ForEach` or `-TestCases` data and expand only that field in the `It` title. Do not expand arbitrary `` or `` values into discovery/report metadata. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/python-examples.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/python-examples.md index 54995c10..981526a9 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/python-examples.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/python-examples.md @@ -377,7 +377,7 @@ repository.find.return_value = expected # typos now raise AttributeError ## Sample Final Report -What `code-testing-generator` produces at Step 9: +What `test-engineer` produces at Step 9: ```markdown ## Test Generation Report diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/typescript-examples.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/typescript-examples.md index 6639d0a7..e4f177ec 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/typescript-examples.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/typescript-examples.md @@ -388,7 +388,7 @@ function makeRepository(): InvoiceRepository & { find: ReturnType; ## Sample Final Report -What `code-testing-generator` produces at Step 9: +What `test-engineer` produces at Step 9: ```markdown ## Test Generation Report diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/typescript.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/typescript.md index 50375f28..3d1174a5 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/typescript.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-extensions/extensions/typescript.md @@ -78,7 +78,7 @@ Use the repo's lint script first. Otherwise detect from `devDependencies` and co ## Parameterized Test Display Names -Apply [Report-safe test names and result validation](../../code-testing-agent/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). +Apply [Report-safe test names and result validation](../../code-testing/unit-test-generation.prompt.md#report-safe-test-names-and-result-validation). For Jest `it.each`/`test.each`, interpolate only a safe label (`$Name` for object rows), or use a short behavior label with `%#` for the case index. Do not use `%p`, `%s`, or `$Input`/`$Expected` to render arbitrary data in the title. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing/SKILL.md new file mode 100644 index 00000000..d64ee8e3 --- /dev/null +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing/SKILL.md @@ -0,0 +1,381 @@ +--- +name: code-testing +description: >- + ALWAYS USE for test work that requires changes: write, add, generate, repair, + or strengthen tests for existing code in xUnit, MSTest, NUnit, pytest, + Vitest/Jest, Go, or another framework. Includes regression cases, failing or + flaky tests, coverage-driven additions, and audit-then-fix requests. Focused + work stays direct; broad or multi-stage work invokes test-engineer. DO NOT USE + for only running tests, analysis-only audits, framework/platform migrations, + a test blocked on a missing production seam (testability-obstacle), or MSTest + API/configuration corrections that do not design new cases + (writing-mstest-tests). Within an active test-engineer pipeline, reuse supplied + guidance and do not re-enter this skill. +license: MIT +--- + +# Code Testing Skill + +The reliable implicit entry point for generating, repairing, and strengthening +tests. It handles focused work directly and invokes the public `test-engineer` +agent for broad or multi-stage requests. + +## Non-negotiable execution contract + +**Check pipeline ownership first.** If the active agent is +`test-engineer` (including a plugin-qualified name such as +`dotnet-test:test-engineer`), or the caller assigned you a phase of +that pipeline, do not delegate to another generator. Continue the assigned +work inline. This guard takes precedence over every broad-scope delegation +instruction below, even if this skill was loaded automatically. + +Classify scope **before editing**: + +- **Broad** (a project/package-wide suite, or multiple production + files/modules): create `research.md` and `plan.md` in a resolved + non-stageable `` before implementation, then `status.md` there + after the final test-quality review. When `test-engineer` is + available, invoke that named custom agent before implementing; do not replace + it with a generic subagent carrying the same label or implement the broad + request inline. If the state files are absent, the broad workflow is + incomplete. +- **Focused** (the user explicitly limits work to one function/class/file or one + missing method): do not create intermediate state files or fan out to multiple + agents. A sparse project-wide request remains broad even when only one source + module is present. + +For either scope, run the narrowest relevant test command to a clean exit. +Always apply [Report-safe test names and result validation](unit-test-generation.prompt.md#report-safe-test-names-and-result-validation), +including when the caller supplies conventions. Pass this contract to delegated +implementers/testers; preserve edge-case data and validate configured reports, +not just console output. +Keep the handoff proportional: for one to three focused requirements, use a +compact bullet list under a **Requirement coverage** label that names the tests +and successful command; for broader or multi-requirement work, use a +`Requirement | Evidence` table. Each requested behavior must cite an exact test +name. + +Before sending a broad-scope final response, check that the response itself +contains `| Requirement | Evidence |` and exact test names for every behavioral +row. A table in a child report or internal plan is not enough. Do not summarize +away those names into module-level bullets or an `Area | Tests` table. + +Intermediate state files are internal working data, never deliverables. Keep +`` non-stageable, never place it or its files in +version-controlled workspace content, and never modify `.gitignore` to hide +them. + +Treat completeness as a requirement matrix, not a test-count target. Give every +independently requested state, boundary, error path, or interaction its own +concrete assertion. Combine cases only when one execution genuinely proves the +whole requested combination; do not let a parameterized happy-path case stand in +for an empty state, invalid discriminator, or before/at/after boundary. +For broad requests that name several production modules or layers, give each +named module direct tests for its non-trivial public behavior. Cross-module tests +prove composition, but do not substitute for the requested module-level +coverage. Judge breadth by the behavior matrix, never by matching or exceeding a +raw test count. + +At the public entry point, delegate broad work to `test-engineer` +once. Research, plan, implementation, and review remain required, but they +need not be separate sub-agent calls. + +Use only capabilities available in the current runtime. Do not retry a missing +skill under aliases or use another agent to retry a policy-denied operation. +If scratch storage is denied, keep the research and plan in context, continue +permitted test edits, and report the missing state artifacts. If execution is +denied, continue permitted static review and report tests as unrun, never passed. +Neither blocker authorizes modifying production code or weakening requirements. + +For a **broad or comprehensive** request, the explicit matrix is the floor, not +the ceiling. Treat each requested module or layer as an inventory heading, not +one behavior: expand it into the bounded public operations and their distinct +validation paths, branches, boundaries, interactions, and state transitions. +After satisfying the explicit matrix, inspect each target API for observable +equivalence partitions and invariants that the prompt did not name: identity, +empty, singleton and representative interior inputs; exact boundaries plus an +immediately adjacent value; invalid partitions; and ordering, monotonicity, +rollover, capacity, truncation, or state invariants implied by the implementation. +Add one mutation-relevant case per distinct partition not already proved, using +parameterized or table-driven cases only for siblings that prove the same +behavior. A passing coverage threshold is validation, not a breadth stop +condition. Stop when remaining inputs exercise the same branch and invariant, +not merely when the explicit checklist is complete; never add cases only to +raise the count. + +## When to Use This Skill + +Use this skill when you need to: + +- Generate unit tests for an entire project or specific files +- Improve test coverage for existing codebases +- Create test files that follow project conventions +- Write tests that actually compile and pass +- Add tests for new features or untested code +- Generate or extend MSTest suites; load `writing-mstest-tests` as supporting + guidance after this entry skill has established scope and project conventions + +## When Not to Use + +- Running or executing existing tests (use the `run-tests` skill) +- Migrating between test frameworks (use migration skills) +- Answering an MSTest API/pattern or modernization question that does not ask to + generate tests (use `writing-mstest-tests`) +- Analysis-only diagnosis of failing tests when no code or test changes are + requested + +## How It Works + +This skill coordinates multiple specialized agents in a **Research → Plan → Implement** pipeline: + +### Pipeline Overview + +```text +┌─────────────────────────────────────────────────────────────┐ +│ TEST GENERATOR │ +│ Coordinates the full pipeline and manages state │ +└─────────────────────┬───────────────────────────────────────┘ + │ + ┌─────────────┼─────────────┐ + ▼ ▼ ▼ +┌───────────┐ ┌───────────┐ ┌───────────────┐ +│ RESEARCHER│ │ PLANNER │ │ IMPLEMENTER │ +│ │ │ │ │ │ +│ Analyzes │ │ Creates │ │ Writes tests │ +│ codebase │→ │ phased │→ │ per phase │ +│ │ │ plan │ │ │ +└───────────┘ └───────────┘ └───────┬───────┘ + │ + ┌─────────┬───────┼───────────┐ + ▼ ▼ ▼ ▼ + ┌─────────┐ ┌───────┐ ┌───────┐ ┌───────┐ + │ BUILDER │ │TESTER │ │ FIXER │ │LINTER │ + │ │ │ │ │ │ │ │ + │ Compiles│ │ Runs │ │ Fixes │ │Formats│ + │ code │ │ tests │ │ errors│ │ code │ + └─────────┘ └───────┘ └───────┘ └───────┘ +``` + +## Step-by-Step Instructions + +### Step 1: Determine the user request + +Classify both intent and scope before editing: + +- **Generate or extend**: follow the generation workflow below. +- **Repair**: reproduce the narrow failure, determine whether the defect is in + the test or production behavior, make the smallest authorized fix, and rerun + the covering command. +- **Audit then fix**: invoke `test-engineer` so it can coordinate the internal + quality specialist and implementation work. + +Make sure you understand what user is asking and for what scope. +When the user does not express strong requirements for test style, coverage goals, or conventions, source the guidelines from [unit-test-generation.prompt.md](unit-test-generation.prompt.md). This prompt provides best practices for discovering conventions, parameterization strategies, behavior-focused coverage, and language-specific patterns. + +### Step 2: Size the request before invoking anything + +Match the machinery to the scope. Running the full pipeline on a one-file +request costs turns and tool calls without improving the tests. + +| Scope | What it looks like | How to run it | +| --- | --- | --- | +| **Focused** | One function, class, or file; "tests for X only"; extending an existing suite with the missing cases | Skip intermediate state files and the sub-agent fan-out. Keep the requirement checklist in your head (or in the final table), read only the target and one neighbouring test for conventions, write the tests, run the narrowest test command, review your own assertions inline. | +| **Broad** | A project, package, or module set; "comprehensive suite"; a coverage threshold to clear across several files | Run the full Research → Plan → Implement pipeline in Step 3, with intermediate state files under `` and the completion contract below. | + +When in doubt, start focused and escalate only if the request turns out to span +several files. Escalating costs one extra pass; running the broad pipeline on a +focused request costs several. + +Before ending a focused request, check all three conditions together: + +1. every named behavior has a concrete assertion, including each requested + boundary or error path; +2. the narrow test command exited successfully; +3. the final handoff maps those behaviors to exact test names and cites that + successful command. + +Do not replace requirement-level evidence with a generic list of covered areas. + +### Step 3: Invoke the Test Generator (broad scope) + +Start by invoking the named `test-engineer` custom agent with your test +generation request. Do not use a generic/general-purpose subagent merely named +`test-engineer`: + +```text +You are the sole pipeline owner for this request. Do not invoke code-testing or another test-engineer; complete the phases in your current context. Generate unit tests for [path or description of what to test], following the [unit-test-generation.prompt.md](unit-test-generation.prompt.md) guidelines. Treat the current workspace as authoritative even when it is sparse, gutted-looking, synthetic, or missing tracked files; never restore or reconstruct it, including with `git checkout`, `git restore`, `git reset`, or `git clean`. +``` + +The Test Generator owns the pipeline. After it returns, consume its recorded +quality checks, validation results, and requirement matrix instead of repeating +Steps 4 and 5 as another pipeline. Do not reload review skills or rerun unchanged +passing commands. Preserve exact test names from its evidence in the final +handoff. If evidence is missing, inspect or follow up on that specific gap +without restarting generation. A reported capability-wide denial also applies +to the caller; do not attempt another command using that capability. + +If `test-engineer` is unavailable, do not skip the workflow. Execute the +same Research → Plan → Implement sequence inline, resolve `` as +described below, create the intermediate state files there, and apply the same +completion contract. + +For broad scope, resolve one absolute `` before creating +intermediate state files: + +1. Prefer a host-provided session artifact or scratch directory. +2. Otherwise, in a Git worktree run + `git rev-parse --path-format=absolute --git-path testagent`; this returns a + path in worktree-specific Git metadata that cannot be staged. +3. Outside Git, create a unique directory under the operating system's + temporary directory. + +Pass the absolute directory to every pipeline agent. The path may be inside the +repository's `.git` metadata directory, but it must not be version-controlled +workspace content, appear in `git status`, or be stageable. + +### Step 4: Execute with bounded context + +For multi-file requests: + +1. Turn every explicit user requirement into a checklist before implementation. Include requested layers, collaborators to mock, boundary cases, integrations, coverage thresholds, and report artifacts. Copy multi-condition requirements verbatim — they must each map to one test that exercises the whole combination. +2. Research only the requested module or project and write the checklist plus a compact target inventory to `/research.md`. +3. Reuse manifests, symbol references, and deterministic pairing tools instead of reading every source and test file. +4. When an available `find-untested-sources` skill is useful for a substantial multi-file inventory, run it once and reuse its pairing and suggested-path output. Otherwise pair the bounded targets manually once; do not probe for an unavailable skill. +5. Plan each target file once, then implement phases sequentially. Map every checklist item to at least one concrete test or explain why it is blocked. +6. Build and test the narrow target during fix cycles. Run workspace-level + validation once at the end only for broad work, when the repository contract + requires that entry point, or when the changes can affect other projects. +7. Before reporting success, re-open the generated tests and verify every checklist item against concrete test names and assertions. Coverage alone is not evidence that a requested mock seam, boundary, state transition, or property combination was tested. +8. Read a language example from `code-testing-extensions` only when the repository has no representative tests and the base extension is insufficient. +9. For .NET, classify SDK-style vs. classic non-SDK before choosing commands or creating files. In classic projects, preserve `packages.config`, existing framework/mock versions and custom base fixtures, add every new test file to the project's explicit `` items, and use the repository's MSBuild/test-runner commands. Never modernize the project or dependency stack merely to generate tests. +10. For MSTest, inspect the pinned package version before choosing exception + assertions. MSTest 3.5.x uses `Assert.ThrowsException`; do not substitute + `[ExpectedException]`, `Assert.Throws`, or `Assert.ThrowsExactly`. + +### Completion contract + +Every scope must satisfy points 3–5 below. Points 1 and 2 are the **broad-scope** +artifacts: on a focused request the same reasoning happens inline and no +intermediate state files are written. + +Do not report completion until all of these are true: + +1. *(broad scope)* `/research.md` records the bounded target + inventory, existing test conventions, and the acceptance checklist. +2. *(broad scope)* `/plan.md` maps each checklist item to a planned + test or an explicit blocker. +3. Generated tests compile and pass with the narrowest relevant test command, + satisfying the shared report-safe naming and result-validation contract. +4. Every explicit user requirement is backed by a concrete test and assertion. + Fix missing mock seams, boundary cases, state transitions, and property + combinations even when coverage already passes. In the final summary, cite + at least one generated test name for every checklist item so completion is + auditable; if an item has no test to cite, keep implementing or report it as + blocked. For non-behavioral requirements such as scaffolding, scope limits, + commands, or coverage artifacts, cite the relevant file, command, or report + instead of forcing a test-name mapping. + A passing suite with fewer tests is not automatically weaker: judge + completeness by whether every independently requested behavior has direct, + nonredundant evidence, not by raw test volume. + For broad/comprehensive scope, also verify that every observable equivalence + partition and invariant discovered in the bounded target APIs has one + mutation-relevant case, even when the prompt did not name it. + When the request names multiple modules, verify that each module's own + non-trivial public behavior has direct test evidence in addition to any + end-to-end composition test. +5. Review the generated tests for behavior gaps and weak assertions. On a broad + scope, invoke `test-gap-analysis` and `assertion-quality` when available and + record the findings and fixes in `/status.md`. On a focused scope, + do the equivalent review inline — re-read each generated assertion against + the source — without spawning extra passes. + +The final response must provide requirement-by-requirement evidence. Use compact +bullets under a **Requirement coverage** label for one to three focused +requirements; use a `Requirement | Evidence` table for broader scopes. +Behavioral evidence cites exact generated test names. Non-behavioral evidence +cites the relevant project file, validation command, or coverage report. A +generic list of tested areas is not a substitute. + +Preserve the user's exact meaning in each evidence item; quote verbatim only +when wording distinguishes a required combination. A test that merely exercises +the same collaborators does not satisfy a requirement about their interaction, +and per-class requirements need a citation per class. + +**Cite a clean run, not an attempt.** The commands behind the final evidence must +have finished successfully: quote the final passing test summary and, when +thresholds were requested, the per-module coverage table from a run that exited +0. If the last coverage run exited non-zero, fix it and re-run before reporting; +never infer threshold clearance from a failed or partial run. + +Before reporting, inspect the final working-tree changes and confirm that +`research.md`, `plan.md`, `status.md`, and any other intermediate state files are +not among the changes intended for commit. + +## State Management + +Broad-scope runs store intermediate state files in a non-stageable +`` backed by host scratch storage, Git metadata, or OS temp. A +focused request does not create these files: + +| File | Purpose | +| ------------------------ | ---------------------------- | +| `/research.md` | Codebase analysis results | +| `/plan.md` | Phased implementation plan | +| `/status.md` | Final quality review and fixes | + +## Agent Reference + +| Agent | Purpose | +| -------------------------- | -------------------- | +| `test-engineer` | Coordinates pipeline | +| `code-testing-researcher` | Analyzes codebase | +| `code-testing-planner` | Creates test plan | +| `code-testing-implementer` | Writes test files | +| `code-testing-builder` | Compiles code | +| `code-testing-tester` | Runs tests | +| `code-testing-fixer` | Fixes errors | +| `code-testing-linter` | Formats code | + +## Requirements + +- Project must have a build/test system configured +- Testing framework should be installed (or installable) +- VS Code with GitHub Copilot extension + +Classic non-SDK .NET projects are supported when their existing build/test +toolchain is available. When it is not available on the current machine, the +agent can still add and register version-compatible tests, but must report +execution as blocked rather than substituting `dotnet test`. + +## Troubleshooting + +### Tests don't compile + +The `code-testing-fixer` agent will attempt to resolve compilation errors. Check +`/plan.md` for the expected test structure. Call the +`code-testing-extensions` skill and read the language-specific extension file +for error code references (e.g., `dotnet.md` for .NET). + +### Tests fail + +Most failures in generated tests are caused by **wrong expected values in assertions**, not production code bugs: + +1. Read the actual test output +2. Read the production code to understand correct behavior +3. Fix the assertion, not the production code +4. Never mark tests `[Ignore]` or `[Skip]` just to make them pass + +### Wrong testing framework detected + +Specify your preferred framework in the initial request: "Generate Jest tests for..." + +### Environment-dependent tests fail + +Tests that depend on external services, network endpoints, specific ports, or precise timing will fail in CI environments. Focus on unit tests with mocked dependencies instead. + +### Broader validation fails + +During implementation, build and test the narrow target. Run a solution or +workspace-level command only for broad work, when the repository contract uses +that entry point, or when the targeted change can affect other projects. Do not +turn a focused test request into an unconditional full non-incremental build. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-agent/unit-test-generation.prompt.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing/unit-test-generation.prompt.md similarity index 100% rename from external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing-agent/unit-test-generation.prompt.md rename to external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/code-testing/unit-test-generation.prompt.md diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/coverage-analysis/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/coverage-analysis/SKILL.md index d5d70646..a09d36ce 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/coverage-analysis/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/coverage-analysis/SKILL.md @@ -11,7 +11,7 @@ description: > coverage-collection intent, including hypothetical change-survival questions (use test-gap-analysis); CRAP or refactoring safety for one named target (use crap-score); or requests owned by test-tagging, find-untested-sources, - test-anti-patterns, run-tests, or code-testing-agent. + test-anti-patterns, run-tests, or code-testing. license: MIT --- diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/crap-score/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/crap-score/SKILL.md index 5df3ae99..c191ad1f 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/crap-score/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/crap-score/SKILL.md @@ -44,7 +44,7 @@ A method with 100% coverage has CRAP = complexity (the minimum). A method with 0 ## When Not to Use - User just wants to run tests (use `run-tests` skill) -- User wants to write new tests (use `code-testing-agent`) +- User wants to write new tests (use `code-testing`) - User only wants a coverage percentage without complexity analysis - User wants project-wide coverage/CRAP analysis or priorities (use `coverage-analysis`) diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/grade-tests/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/grade-tests/SKILL.md index a7f2b5a5..a2ec68ee 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/grade-tests/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/grade-tests/SKILL.md @@ -1,13 +1,15 @@ --- name: grade-tests description: > - Assess a curated list of tests and produce a PR-ready table with a primary - Pass, Failed, Uncertain, or Not applicable result plus A-F quality detail for - every resolved test; Uncertain and Not applicable omit the grade. USE FOR new - or modified tests supplied as methods, bodies, file spans, or a bounded PR - diff. Polyglot: .NET, Python, TS/JS, Java, Go, Ruby, Rust, Swift, Kotlin, - PowerShell, C++. DO NOT USE FOR: suite-wide audits (use test-quality-auditor - or test-anti-patterns), writing or fixing tests, or measuring coverage. + Grade a curated list of individual tests for readiness, A-F quality, and + concrete improvements. ALWAYS USE FOR: grade tests, review only a named test, + per-test readiness decisions, or quality bands for supplied methods, bodies, + file spans, or bounded PR diffs, including existing tests. Produce a PR-ready + Pass, Failed, Uncertain, or Not applicable table; unresolved or empty scopes + omit the grade. Compose read-only per-test mutation evidence when available. + Polyglot: .NET, Python, TS/JS, Java, Go, Ruby, Rust, Swift, Kotlin, + PowerShell, C++. DO NOT USE FOR: suite-wide audits (test-engineer or + test-anti-patterns), writing or fixing tests, or measuring coverage. license: MIT --- @@ -20,7 +22,22 @@ diagnostic information. The skill **does not discover tests on its own** — the caller (typically a PR automation workflow or a human reviewer holding a specific list) provides the tests or a bounded diff to assess. -> **Language-specific guidance**: Call the `test-analysis-extensions` skill +After Step 0 admits a bounded scope, enforce these grading invariants: + +- With production context, **load `test-gap-analysis` by name once before + scoring**, in `per-test-read-only` caller context. Read its owned composition + reference; do not compute mutation evidence from this grading rubric or run + its standalone workflow. Report N/A / unverified only when the dependency, + reference, or required context is actually unavailable. +- Any reported mutation inference must use **Likely killed (inferred)** or + **Candidate survivor (unverified)**, even when explained in prose. +- Apply only the rubric below, not extra heuristics such as a duplicate-test + penalty. Do not use sibling tests to alter the individual assessment. +- A **B** quality grade does not require a Failed result; a complete focused + test may have no actionable change. + +> **Language-specific guidance**: If the caller supplies the matching bundled +> extension file path, read it directly. Otherwise call `test-analysis-extensions` > to discover available extension files, then read the file matching the > target codebase's language and framework (e.g., `extensions/dotnet.md`, > `extensions/python.md`, `extensions/typescript.md`, `extensions/go.md`). @@ -48,8 +65,8 @@ quality and severity. - The caller wants a full suite audit or comparative metrics — use `test-anti-patterns` (pragmatic) or `test-smell-detection` (formal) and - let the `test-quality-auditor` agent orchestrate. -- The caller wants to *write* new tests — use `code-testing-generator` + let the `test-engineer` agent orchestrate its internal quality specialist. +- The caller wants to *write* new tests — use `test-engineer` (any language) or `writing-mstest-tests` (MSTest specifically). - The caller wants to measure code coverage or CRAP scores — use `coverage-analysis` or `crap-score` (.NET only). @@ -64,7 +81,8 @@ quality and severity. |-------|----------|-------------| | Test methods | Yes | A scope to grade. Provide one of: (a) an explicit list of test method names (fully-qualified, e.g. `Namespace.ClassName.TestMethodName`); (b) one or more file paths plus an explicit instruction to grade every test declared in those files; or (c) a diff hunk / PR identifier whose changed tests should be graded. File paths are recommended but optional when method names are unambiguous in the workspace. Ambiguous requests like *"grade my tests"* with no scope are rejected up-front (see Step 0); this skill is for curated input and does not auto-grade an entire workspace. | | Test bodies / spans | Recommended | The exact source lines for each test method. If omitted, read them from the listed files. | -| Production code | No | The code under test, for judging whether assertions cover the meaningful behaviors. When unavailable, mark relevant findings as "Unverified" rather than guessing. | +| Production code | No | The code under test, for judging whether assertions cover the claimed behavior. When unavailable, mark the mutation assessment N/A / unverified rather than guessing or deducting. | +| Language reference | No | A host-supplied path to the matching bundled `test-analysis-extensions` file. Read it directly instead of invoking its reference-only loader; do not substitute unverified framework guidance. | | Diff context | No | When grading PR changes, the unified diff for each test method helps focus on what actually changed. | ### Step 0: Validate the input @@ -80,7 +98,7 @@ If the request is ambiguous (e.g., *"Grade my tests"*, *"Are these tests any good?"* with no scope, *"Review the test suite"*), **do not load extensions, do not read files, and do not grade anything**. Reply with a short message asking the caller to provide an explicit list / file(s) / -diff, and optionally point them at `test-quality-auditor` agent or +diff, and optionally point them at the `test-engineer` agent or `test-anti-patterns` skill for full-suite analysis. Stop there. If a valid bounded scope resolves to zero eligible tests, return @@ -92,7 +110,8 @@ If a valid bounded scope resolves to zero eligible tests, return Identify the target codebase's language and test framework from the file extensions and the test method markers in the provided list. Call the -`test-analysis-extensions` skill and read the matching extension file (e.g., +`test-analysis-extensions` skill unless the caller already supplied the matching +bundled extension file path. In either case, read that extension file (e.g., `extensions/dotnet.md` for MSTest/xUnit/NUnit/TUnit, `extensions/python.md` for pytest, `extensions/typescript.md` for Jest/Vitest, `extensions/go.md` for the standard `testing` package). If the input contains tests from @@ -112,7 +131,42 @@ For each entry in the input list: invent a body to grade. A missing requested method requires human review; it is not the same as a valid scope containing no tests. -### Step 3: Score each resolved test +**Composition checkpoint:** for resolved tests with available production +context, load `test-gap-analysis` now, once for the batch, with +`per-test-read-only` assessment context. Complete its owned reference assessment +before Step 3. Do not skip this load just because a body-level weakness already +seems obvious; a locally invented mutation explanation is not composition. + +### Step 3: Assess the claimed behavior and score each resolved test + +Keep grading read-only: no build/test runs, mutation execution, file edits, +tool installation, broad suite discovery, or agent delegation. Resolve only +the supplied tests, their relevant fixtures/helpers, and the production call +chain needed for their claims. + +Use the inline `test-gap-analysis` assessment from Step 2's checkpoint; +do not load it a second time. Supply each test's +identifier/body, relevant setup/helpers, claimed behavior, assertion semantics, +and available source. Its composition dispatch loads the owned read-only +reference rather than its standalone baseline/verification workflow. Consume +its per-test evidence; do not duplicate its mutation catalog here or invoke an +audit/generation agent. +Convey mode and inputs as assessment context using the host's supported caller +instructions. If the loader accepts only a skill name, load `test-gap-analysis` +by name only; do not invent tool arguments or a mode-specific skill name. + +If the skill/reference or production context is unavailable, record +`Pseudo-mutation: N/A / unverified — ` and continue normal body-level +grading. This is not a grade deduction or, by itself, an Uncertain result. +Do not search installation directories or substitute a mutation runner. + +Assess only what each test claims: do not borrow another test's assertions, +or demand unrelated branches, outputs, or scenarios. An observable survivor +can support an existing Assertion strength category when it proves that the +test does not verify its claimed outcome; do not introduce mutation points, +weights, ceilings, or an automatic survivor penalty. Apply the existing rubric +normally, including weaknesses it classifies in both Assertion and Anti-pattern +dimensions; do not add another deduction for the same mutation evidence. Start every test at grade **A (score band 90–100)**, then apply deductions strictly for **observable issues** in the captured body. Do **not** deduct @@ -138,7 +192,7 @@ assertion in the test body. Score from highest to lowest: |-----------|---------| | **A** | At least one meaningful value assertion (equality / structural / exception / state) plus, where appropriate, additional checks (negative, type, collection contents). Mock-call verifications (`Verify`, `toHaveBeenCalledWith`, `Should -Invoke`) and bare assertion forms (pytest `assert`, Go `if got != want { t.Errorf(...) }`, Rust `assert!()`) count as real assertions. | | **B** | One clear meaningful assertion that verifies the behavior under test. | -| **C** | Only trivial assertions (single `IsNotNull` / `toBeDefined` / `assert x is not None`), or assertions that check a single field while the operation produces a richer result. | +| **C** | Only trivial assertions (single `IsNotNull` / `toBeDefined` / `assert x is not None`), or assertions that leave a meaningful part of the test's claimed result unchecked. A focused single-field claim does not require unrelated fields. | | **D** | One self-referential / tautological assertion (`Assert.AreEqual(x, x)`, `assert dto.name == dto.name`, round-trip identity without a non-trivial input), or broad exception assertions (`Assert.ThrowsException`). | | **F** | No assertions at all; **all** assertions are always-true literals (`Assert.IsTrue(true)`, `assert True`, `expect(true).toBe(true)`) — these verify nothing and are equivalent to having no assertions; or all assertions are silently un-awaited (e.g., `expect(promise).resolves.toBe(x)` without `await`/`return`, async TUnit/xUnit `Assert.ThrowsAsync` without `await`, pytest-asyncio with un-awaited coroutine). | @@ -284,10 +338,27 @@ uncertainty. ### Step 5: Build the note Use one sentence (target ≤ 120 characters) for the most important reason: -`No issues found.`, `Only checks IsNotNull; add value verification.`, or +`No issues found.`, `Only checks IsNotNull; receipt contents are unverified.`, or `Method body could not be resolved; human review is required.` Do not invent a weakness to justify a grade or Failed result. +Keep the action in a separate **How to improve** field. For each Failed test, +name the smallest useful input, assertion, or fixture change and its expected +outcome, grounded in the body, source, or an explicit contract. For example, +`Replace self-comparison with Assert.AreEqual(60m, account.Balance).`, not +`Improve assertions`; `Remove Console.WriteLine after Deposit(25m).`, not +`Clean up`. Prioritize the highest-impact distinct finding, and include other +actionable findings only when they require a different change. + +For a behavioral gap, use the distinguishing witness and original/mutant +observations from the shared assessment; check the expected result against +the unmodified source. If essential context is missing, name the evidence +needed instead of inventing an expected value. Pass gets `None`; Uncertain +gets a concrete evidence-resolution step, not a speculative test rewrite. +A rubric-only deduction is not proof of a behavioral gap or an actionable +improvement: a focused **B / Pass** may need no change. **A / Failed** still +needs its concrete action, such as removing debug output. + ### Step 6: Report Produce two sections. @@ -302,13 +373,22 @@ empty scope and omit the table. #### 2. Per-test table ```markdown -| Test | Result | Quality | Notes | -|------|--------|---------|-------| -| `Namespace.ClassName.Test_Method_Condition_Expected` | Pass | A (90–100) | No issues found. | -| `Namespace.ClassName.Test_Other` | Failed | C (70–79) | Only `IsNotNull`; add value verification. | -| `Namespace.ClassName.Test_Missing` | Uncertain | — | Method body could not be resolved; human review is required. | +| Test | Result | Quality | Notes | How to improve | +|------|--------|---------|-------|----------------| +| `Namespace.ClassName.Test_Method_Condition_Expected` | Pass | B (80–89) | One complete value assertion. | None | +| `Namespace.ClassName.Withdraw_SufficientFunds` | Failed | D (60–69) | Balance is compared with itself. | Replace self-comparison with `Assert.AreEqual(60m, account.Balance)` after withdrawing 40m from 100m. | +| `Namespace.ClassName.Test_Missing` | Uncertain | — | Method body could not be resolved; human review is required. | Supply the method body and its referenced fixture. | ``` +Keep these two report sections and the original Test/Result/Quality/Notes +fields. When mutation evidence explains a finding or the caller requests +detail, append a compact per-test **Pseudo-mutation evidence** block inside +the per-test section: change, witness, original/mutant observations, relevant +assertion, and classification. Static results are **Likely killed (inferred)** +or **Candidate survivor (unverified)**, never executed Killed/Survived or +empirical killed/total counts. State missing-context N/A / unverified once +per shared limitation. Do not repeat the improvement table in prose. + **Caps and ordering**: - If the table would exceed **50 rows**, show Failed tests first, then Uncertain tests, then a sample of Pass tests. Wrap overflow in a collapsed @@ -329,6 +409,13 @@ prefix each section with the language name and framework. - [ ] Uncertain is an evidence gap; Not applicable is a valid empty scope. - [ ] Every grade is justified by at least one observable signal in the captured body — no speculative deductions. +- [ ] Every Failed row has a concrete, evidence-backed How to improve action; + Pass rows have no invented weakness, even when the quality grade is B. +- [ ] Mutation assessment stayed read-only and per-test; unavailable context + was not penalized, equivalents were excluded, and static labels/counts + were not presented as executed evidence. +- [ ] Verified observable findings inform existing categories without a + duplicate deduction or any change to scoring weights and ceilings. - [ ] Trivial-assertion tests are flagged only when the **only** assertion is trivial (a null check before a meaningful assertion is not trivial). - [ ] Exception-only tests are not penalized for low assertion count. @@ -362,6 +449,8 @@ prefix each section with the language name and framework. | Using a fake-precise score (e.g., 87/100) | Use the score band only — 90–100, 80–89, 70–79, 60–69, 0–59. | | Spilling a 500-row table into a PR comment | Apply the row cap from Step 6; collapse extras into `
`. | | Re-reporting an existing finding three times under different categories | Pick the most fitting category and report once. | +| Giving a weak test credit for a sibling's assertions | Use only the current test and helpers/fixtures it executes. | +| Turning pseudo-mutation composition into a suite audit | Pass explicit per-test-read-only mode; no runs, edits, broad discovery, or agent recursion. | | Inventing weaknesses for A-grade tests to make the note "balanced" | If a test is clean, the note may simply read `No issues found.` | | Mapping status from grade or comments | Fail only for actionable improvements; a B can Pass and an A can Fail. | | Confusing Uncertain and Not applicable | Evidence gaps are Uncertain; a valid empty scope is Not applicable. | diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/scaffold-dotnet-test-project/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/scaffold-dotnet-test-project/SKILL.md index 48e1ee48..8f2ef17c 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/scaffold-dotnet-test-project/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/scaffold-dotnet-test-project/SKILL.md @@ -7,7 +7,7 @@ description: >- include, or repair a test project. Handles "tests pass directly but CI discovers zero", exact solution wiring, xUnit/NUnit/MSTest, and central packages. DO NOT USE to only author tests in an already-wired project - (code-testing-agent), run tests, migrate, or correct MSTest syntax/configuration + (code-testing), run tests, migrate, or correct MSTest syntax/configuration without changing project or CI files (writing-mstest-tests). license: MIT metadata: @@ -52,7 +52,7 @@ Inspect the repository before editing, then choose exactly one path: | No suitable test project | Create one bounded project, reference the production project, and register it | Create a project per source project | | Test project exists but lacks the required `ProjectReference` | Add only that reference and verify direct plus entry-point execution | Scaffold another project or rewrite tests | | Test project passes directly but is absent from `.sln`, `.slnx`, or `.slnf` | Register the existing project in the exact entry point CI uses | Recreate the project or switch solution formats | -| Suitable project, reference, and requested entry point are already correct | Leave the workspace unchanged; use `code-testing-agent` if test methods are requested | Normalize or replace working files | +| Suitable project, reference, and requested entry point are already correct | Leave the workspace unchanged; use `code-testing` if test methods are requested | Normalize or replace working files | An existing project is suitable when its target framework can reference the production project and its purpose matches the requested layer. A different diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-anti-patterns/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-anti-patterns/SKILL.md index f28abadc..bc375a64 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-anti-patterns/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-anti-patterns/SKILL.md @@ -6,7 +6,7 @@ description: > assertions, swallowed/broad exceptions, flaky/order-dependent tests, duplication, or magic values. Polyglot. DO NOT USE for direct edits: writing-mstest-tests owns supplied MSTest assertions/attributes/lifecycle; - code-testing-agent owns new tests. Exclude running tests, migration, assertion + code-testing owns new tests. Exclude running tests, migration, assertion metrics (assertion-quality), raw .NET coverage collection (run-tests), non-.NET coverage collection/analysis (native tooling), project-wide .NET coverage/CRAP (coverage-analysis), named-target .NET CRAP @@ -34,7 +34,7 @@ Quick, pragmatic analysis of test code in any supported language for anti-patter ## When Not to Use -- User wants to write new tests from scratch (use `code-testing-agent`) +- User wants to write new tests from scratch (use `code-testing`) - User wants direct implementation fixes rather than a diagnostic review (use the relevant write/edit skill) - User asks to fix swapped `Assert.AreEqual` argument order in MSTest (use `writing-mstest-tests`) - User asks to convert MSTest `DynamicData` from `IEnumerable` to `ValueTuple` (use `writing-mstest-tests`) diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/SKILL.md index 365b8ee2..b024da0a 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/SKILL.md @@ -4,14 +4,14 @@ description: >- Pseudo-mutation analysis ONLY: answer whether tests would catch a bug if production code changed, which meaningful changes would still pass, or which caller-visible mutations existing assertions would miss; verify candidates - when requested, then optionally close verified gaps. Activate for behavioral - blind spots or missing edge cases tied to production behavior. Polyglot. DO - NOT USE FOR: suite organization, taxonomy, + when requested, then optionally close verified gaps. Includes explicit + read-only per-test composition for grading. Activate for behavioral blind + spots tied to production behavior. Polyglot. DO NOT USE FOR: suite taxonomy, metadata, or distribution reports (test-tagging); .NET line-vs-branch or Cobertura interpretation, arithmetic, plateaus, project-wide coverage gaps, or coverage-backed test/CRAP priorities (coverage-analysis; use native coverage tooling outside .NET); named-target CRAP (crap-score); new suites - (code-testing-agent); assertion/smell audits; or mutation tools. + (code-testing); assertion/smell audits; or mutation tools. license: MIT --- @@ -21,6 +21,22 @@ Answer one question: **which caller-visible production behaviors could change without an existing test failing?** Mutation reasoning is a probe, not the goal. Inventory public outcomes first, then verify only credible gaps. +## Composition dispatch + +Before entering the decision flow, check the caller's mode. An explicit +`per-test-read-only` request (including composition from `grade-tests`) uses +only [references/per-test-read-only.md](references/per-test-read-only.md). +This mode is assessment context conveyed through the host's supported caller +instructions, not a new skill-loader parameter or a mode-specific skill name. +Read that reference and return its per-test assessment inline; **do not enter +the standalone workflow below**. Loading this mode does not authorize a +baseline run, mutation execution, file edits, suite discovery, or delegation. +If the bundled reference is unavailable, allow at most one listing of the known +`references/` directory, then return **N/A / unverified** with the reason. + +Requests without this mode retain the standalone analysis, verification, and +test-addition paths below. + ## Decision flow ### 1. Set scope @@ -35,7 +51,7 @@ search misses, inspect the current directory broadly before asking for paths. | Explicit survivor verification | Inventory all requested outcomes; execute one representative observable candidate for each distinct high-risk outcome under verification, then classify it as **Survived** or **Killed** | | Explicit exhaustive audit | Read [references/mutation-catalog.md](references/mutation-catalog.md) and classify all meaningful candidates | | Add tests to an existing suite | Analyze first; add tests only for verified survivors or demonstrated no-coverage outcomes | -| Create a new suite | Stop and use `code-testing-agent` | +| Create a new suite | Stop and use `code-testing` | When the request names a risk, turn it into a one-line public-outcome allowlist before reading code. An outcome is not in scope merely because the same method writes it. @@ -204,6 +220,7 @@ calculate a score. |---|---| | **Likely killed** | An existing assertion observes the changed outcome | | **Candidate survivor (unverified)** | Observable change appears unasserted; not executed | +| **Killed** | Exact observable mutation executed and a relevant assertion failed | | **Survived** | Exact observable mutation executed and tests stayed green | | **No coverage** | No test reaches the public outcome; report the missing branch without inventing a survivor | | **Equivalent** | No public observation changes; omit from findings | @@ -222,8 +239,10 @@ requested test addition. 1. Apply one candidate and confirm the diff changes exactly one intended expression. -2. Run the narrowest covering test: green means **Survived**, red means - **Killed**, for that edit only. +2. Run the narrowest covering test and confirm tests executed: green means + **Survived**; a failure at a relevant assertion means **Killed**, for that + edit only. Build, discovery, infrastructure, or unrelated failures leave + the candidate **unverified**, not Killed. 3. Revert immediately and confirm the clean source/test baseline. 4. After a green run, re-check the public counterfactual. Execution proves the suite missed the edit, not that the edit changes behavior; drop inert or @@ -291,9 +310,10 @@ For focused or small analysis, return: Do not repeat the table in prose or report discarded mutants, tool chronology, or in-flight reasoning. -For an exhaustive audit, add counts for Killed / Survived / No coverage / -Equivalent and group findings by risk. Count only executed or definitively -classified candidates. +For an exhaustive audit, separate executed **Killed / Survived** counts from +static **Likely killed / Candidate survivor (unverified)** classifications and +**No coverage / Equivalent** inventory counts. Never include static reasoning +in an empirical killed/total ratio. Group findings by risk. For test additions, name the tests added, the verified mutations they kill, and the successful final command. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/references/mutation-catalog.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/references/mutation-catalog.md index 78d6d943..0df14817 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/references/mutation-catalog.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/references/mutation-catalog.md @@ -1,8 +1,9 @@ # Mutation Candidate Catalog Read this reference only for an explicitly exhaustive audit or when a language's -mutation semantics are unfamiliar. For focused analysis, use the smaller -risk-ranked table in `SKILL.md`. +mutation semantics are unfamiliar, or for the applicable categories in +[per-test read-only composition](per-test-read-only.md). Reading the catalog +does not authorize the exhaustive execution procedure below. ## Candidate categories @@ -55,5 +56,7 @@ Exclude: 5. Execute every candidate that might be reported as Survived. 6. After a green run, re-check that the mutation is publicly observable. 7. Revert after each run and confirm the clean baseline at the end. -8. Count only executed or definitively killed/equivalent candidates in the - mutation totals; disclose any omitted scope. +8. Separate executed Killed/Survived totals from inferred Likely killed, + unverified candidates, and equivalent inventory classifications. Never turn + static classifications into empirical killed/total claims; disclose omitted + scope. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/references/per-test-read-only.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/references/per-test-read-only.md new file mode 100644 index 00000000..a19bc7f2 --- /dev/null +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-gap-analysis/references/per-test-read-only.md @@ -0,0 +1,106 @@ +# Per-Test Read-Only Assessment + +This is the composition contract for `mode: per-test-read-only`, not a suite +audit or a mutation runner. `test-gap-analysis` owns this methodology and the +[mutation catalog](mutation-catalog.md); consumers own decisions and scoring. + +## Inputs and execution boundary + +The caller supplies resolved test identifiers/bodies, relevant setup and +helpers, language/framework assertion semantics, and available production +context. A batch may contain several tests, but assess each independently. + +- Read only those tests, their fixtures/helpers, and the production call chain + needed to explain their claimed outcomes. No broad suite discovery. +- Do not run tests, build/restore, execute/import production code, install + tooling, apply mutations, edit files, invoke agents, or recurse into another + grading/audit workflow. A read of this reference is not execution permission. +- Missing source, an unresolved call chain, or unsupported assertion semantics + yields **N/A / unverified** for the affected assessment, with the reason. + Assess any remaining known behavior normally; absence is not a weakness. +- Do not replace the caller's report with a suite-wide Strong/Mixed/Weak verdict, + a mutation score, or a coverage dashboard. + +## Assessment + +1. **Bind the claim.** State the behavior this individual test promises from + its name, arranged inputs, act, assertions, and any explicit contract. + Resolve a contradiction as an evidence gap instead of inventing intent. + An exception-only test owns that invalid-input outcome, not happy paths; + a price-only test does not own an unrelated flag or formatting field. + Do not demand every method branch, adjacent scenario, or sibling test case. +2. **Trace observations.** Map each supplied input/sequence through the public + caller to its return value, exception, state, or external side effect and + the assertion that observes it. Include relevant fixture cleanup/assertions + and helpers executed by this test, but never borrow another test's checks + or deduct because another test is absent. +3. **Probe meaningful changes.** Use the applicable categories in the owned + mutation catalog without enumerating every operator. Fix all arguments when + comparing original and mutant. First replay the change mentally on this + test's existing inputs and assertions. A changed observation that makes a + relevant assertion fail is **Likely killed (inferred)**, even if the check + is indirect or the return value is not asserted directly. +4. **Prove an actionable survivor.** For a change the current test appears to + miss, supply a distinguishing witness within its claimed behavior: + `input/sequence -> original observation -> mutant observation`. + Explain why the current test's assertions still pass, and how the smallest + input/assertion/fixture change would expose the difference. If a new input + is needed, distinguish that proposed witness from the current test input; + do not pretend an assertion on the proposed witness already exists. + A witness from an unrelated behavior is not a finding against this test. +5. **Filter equivalence and uncertainty.** Omit non-compiling changes, + impossible inputs, equivalent guards/boundaries, private representation + changes with no public effect, and duplicate syntax variants. In particular, + removing a guard that falls through to the same public exception is not a + survivor; `<` to `<=` at a floor is equivalent when both return the floor. + Do not require exception message/parameter metadata unless it is an + established contract. If original and mutant cannot be distinguished from + available context, mark the assessment unverified rather than deducting. +6. **Return evidence, not grades.** Report only credible distinct findings and + relevant protected behavior. An empty findings list is valid; do not invent + a mutant, improvement, count, or denominator to fill the report. + +## Return contract + +For each requested test, return: + +| Field | Content | +|---|---| +| Test / claim | Stable identifier and the individual behavior assessed | +| Availability | Assessed, or N/A / unverified with the missing evidence | +| Change / witness | Exact existing expression/condition/side effect changed; fixed input or sequence | +| Observations | Original versus mutant caller-visible outcomes on that same witness | +| Detection evidence | This test's relevant assertion and why it would fail or still pass; distinguish current inputs from a proposed witness | +| Classification | Likely killed (inferred), or Candidate survivor (unverified) | +| Smallest improvement | Concrete input, expected assertion, or fixture change for each candidate survivor; none when protected | + +These fields form an evidence ledger, not a mandatory extra table in the +consumer's final response. Preserve source/test locations when available. +Keep evidence proportional to the claim; stop once the claim is protected or +no credible observable candidate remains. + +Only already-supplied execution evidence for this exact test, source revision, +and mutation may use **Killed (executed)** or **Survived (executed)**. A killed +result requires a relevant assertion failure, not a build/runner failure. +Keep such evidence separate from static classifications. Never label a static +assessment Killed/Survived alone or report empirical killed/total counts, +percentages, or a mutation score from source reasoning. + +## Examples + +- A test withdraws 40 from a balance of 100 and compares the balance with + itself. Omitting the decrement leaves 100 instead of 60; the self-comparison + still passes. **Candidate survivor (unverified)**; replace that comparison + with the literal expected balance 60. Do not credit a sibling balance test. +- A test claims to verify the correct standard fee calculation, but checks + only that the cost is positive. A fee change yields 12 instead of 10 for + its arranged order while that check still passes. Recommend equality to 10 + for that order, not generic "stronger assertions". If the test deliberately + claims only positivity, that same change is not a gap against its claim; + do not require an exact fee or unrelated cancellation coverage. +- A test asserts the promised exception type for one invalid input. Removing + its guard still throws that same type later and changes no contracted side + effect. Omit the equivalent mutation; do not demand message assertions. +- A focused test pins the result at a threshold. An inclusive-to-exclusive + change alters that asserted result: **Likely killed (inferred)**. Do not + deduct for not testing a different scenario, or call this an executed kill. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-tagging/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-tagging/SKILL.md index c403a91f..13755a2d 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-tagging/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/test-tagging/SKILL.md @@ -8,7 +8,7 @@ description: > by test type, or tag then verify the project builds. Read bodies when names mislead. Apply canonical attributes; otherwise report only. DO NOT USE FOR: requests owned by test-anti-patterns, coverage-analysis, crap-score, - test-gap-analysis, code-testing-agent, or migration skills. + test-gap-analysis, code-testing, or migration skills. license: MIT --- @@ -29,7 +29,7 @@ Analyze an existing test suite in any supported language and apply a standardize ## When Not to Use -- Writing new tests from scratch (use `code-testing-agent` for any language, or `writing-mstest-tests` for MSTest) +- Writing new tests from scratch (use `code-testing` for any language, or `writing-mstest-tests` for MSTest) - Running or filtering tests (use `run-tests` for .NET; equivalent native runners elsewhere) - Migrating between test frameworks - General quality, smell, flakiness, or assertion audits (use `test-anti-patterns` or the matching analysis skill) diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/testability-obstacle/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/testability-obstacle/SKILL.md index 54f9d254..93ace48a 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/testability-obstacle/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/testability-obstacle/SKILL.md @@ -30,7 +30,7 @@ redesign adjacent code. ## When Not to Use - The dependency is already injected or passed as an argument. Write tests with - a fake through the existing seam using `code-testing-agent`. + a fake through the existing seam using `code-testing`. - The user wants a repository-wide testability audit. Use `detect-static-dependencies`. - The user wants wrappers generated but not call sites/tests changed. Use diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/writing-mstest-tests/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/writing-mstest-tests/SKILL.md index e6c8794f..080854b6 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/writing-mstest-tests/SKILL.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/skills/writing-mstest-tests/SKILL.md @@ -9,7 +9,7 @@ description: > identity, exception, hard-cast, and object[] checks; TestContext/lifecycle; timeout/cancellation; OS/CI conditions, retry, cleanup, parallelization, MSTest.Sdk project setup, and MSTESTxxxx. Honor the installed - MSTest version. DO NOT USE to design new test cases (code-testing-agent), + MSTest version. DO NOT USE to design new test cases (code-testing), perform report-only audits, create project files rather than explain MSTest setup, run tests, migrate frameworks, or handle non-MSTest/non-.NET code. license: MIT @@ -76,7 +76,7 @@ permissions or the task's scope. ## Response Guidelines - **Specific API or pattern questions** (assertions, data-driven, lifecycle): Jump directly to the relevant workflow step. Do not follow the full workflow. -- **Generate new tests from scratch**: Hand off to `code-testing-agent`; use this +- **Generate new tests from scratch**: Hand off to `code-testing`; use this skill only as supporting MSTest API/version guidance. - **Review and fix existing tests**: Fix only the issues present. Do not add unrelated improvements. - **Assertion transformations**: Show the corrected call, then state the semantic diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/version.json b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/version.json index 3fea520f..74c23ba7 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/version.json +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet-test/version.json @@ -1,6 +1,6 @@ { "$schema": "https://raw.githubusercontent.com/dotnet/Nerdbank.GitVersioning/main/src/NerdBank.GitVersioning/version.schema.json", - "version": "0.2", + "version": "1.0", "pathFilters": [ ".", ":!plugin.json", diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet/README.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet/README.md index 2b96aee3..d708c0e0 100644 --- a/external-sources/upstreams/dotnet-skills/plugins/dotnet/README.md +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet/README.md @@ -19,8 +19,9 @@ Prerequisites: ## Skills +- [csharp-expert](skills/csharp-expert/SKILL.md) - [csharp-refactoring](skills/csharp-refactoring/SKILL.md) -- [msbuild](skills/msbuild/SKILL.md) +- [msbuild](skills/msbuild/SKILL.md) — routed MSBuild diagnosis, performance, and authoring guidance - [setup-local-sdk](skills/setup-local-sdk/SKILL.md) ### MSBuild entry and specialist skills @@ -50,3 +51,25 @@ and outcomes without prescribing which overlapping skill must win selection. Point `EXPERIMENT_FILE` at that experiment when running `eng/run-skill-evals.sh dotnet msbuild`. Inspect its activation traces for duplicate workflows and out-of-scope activation; the isolated arm's dormancy contract alone does not prove cross-plugin routing. + +### C# expert marketplace routing + +`csharp-expert` routes a .NET request to an installed specialist or identifies the smallest +`dotnet/skills` marketplace plugin that supplies a missing specialist. Its normal eval covers +solution detection, marketplace acquisition, fallback behavior, and dormancy. + +The supplemental +[`csharp-expert-coexistence.experiment.yaml`](../../csharp-expert-coexistence.experiment.yaml) +and +[`csharp-expert-coexistence.claude.experiment.yaml`](../../csharp-expert-coexistence.claude.experiment.yaml) +load representative marketplace specialists in the plugin arm for GPT- and Claude-family executors. +Run either without a skill filter: + +```bash +EXPERIMENT_FILE=./csharp-expert-coexistence.experiment.yaml ./eng/run-skill-evals.sh +EXPERIMENT_FILE=./csharp-expert-coexistence.claude.experiment.yaml ./eng/run-skill-evals.sh +``` + +Inspect the plugin-arm activation traces to confirm that installed specialists are invoked without +installation advice. This is trace evidence rather than a shared output grader because the +target-only arm may legitimately identify a missing specialist after completing a safe fallback. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet/skills/csharp-expert/SKILL.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet/skills/csharp-expert/SKILL.md new file mode 100644 index 00000000..fd0fed9d --- /dev/null +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet/skills/csharp-expert/SKILL.md @@ -0,0 +1,350 @@ +--- +name: csharp-expert +description: >- + Route C# and .NET requests to the exact installed specialist or smallest dotnet/skills + marketplace plugin. USE FOR: "which specialist should own this" or "how do I add the skill" + requests involving ASP.NET Core endpoints, Blazor, MAUI binding, Windows Forms specialist + selection or installation, EF Core queries, test or framework migration, runtime + CPU/allocation evidence, file-based C#, editor/compiler defects, a surviving MSBuild + `.binlog`, or a plugin missing from `/skills`; also use for C# semantics + when no narrower specialist exists. DO NOT USE FOR: requests that already name the exact + installed specialist to invoke, or work unrelated to C# or .NET. +license: MIT +--- + +# C# Expert + +## Purpose + +Act as the front door for C# and .NET work. Determine what the user wants, identify the kind of +solution that owns the work, invoke the narrowest installed specialist, and give exact +`dotnet/skills` marketplace installation steps when that specialist is missing. Keep direct C# +language guidance as the fallback, not the default. + +## Routing Contract + +1. Classify the requested outcome from the prompt. +2. Inspect the smallest set of repository files needed to identify the solution type. +3. Compare both signals with the descriptions of the skills currently available to the runtime. +4. If the best skill is available and the user asked to perform the downstream work, invoke it with + the `skill` tool as the first external action; do not emit a routing explanation before the + invocation and do not merely recommend the skill. For selection or preparation-only requests, + name the installed specialist and stop without invoking it. +5. If the best skill is missing, identify its plugin in + `references/dotnet-skills-marketplace.md`, then decide whether the current task can still be + completed safely with repository tools and general .NET knowledge. +6. Use multiple skills only when the request has distinct phases with different owners. +7. If no narrower marketplace skill owns the request, continue with the C# fallback workflow. + +Routing is not task completion. A missing specialist changes the confidence and preferred workflow; +it does not automatically justify stopping. Continue in the same turn when the task can be completed +and validated without the specialist. Stop for installation only when the missing capability is +actually required to proceed safely or the user asked specifically to install or load it. + +Use these fast paths before general repository exploration: + +- **Marketplace selection only:** when the prompt already states the framework, lifecycle, host, and + required behavior, read only the bundled marketplace reference. Do not inspect the fixture, + repository, GitHub, or plugin source. Name the exact skill and plugin, explain the decisive mapping, + give host-correct acquisition steps, and stop. +- **Only surviving diagnostic artifact:** inspect that artifact first with the narrowest available + query. For a supplied `.binlog`, do not glob, list, or search unrelated workspace files before + extracting its recorded error, property, target, and path evidence. + +Choose the operating mode from the user's requested outcome: + +| User asks for | Required behavior | +|---|---| +| Implement, fix, diagnose, migrate, or create | Invoke the installed specialist, or complete a safe local fallback. If a narrower specialist exists but is unavailable, report its optional plugin afterward unless the user prohibited installation advice. | +| Identify, choose, install, prepare, or load the right marketplace capability | Inspect enough solution evidence to choose the owner, name it whether installed or missing, give acquisition steps only when needed, and stop without invoking the specialist, editing files, or generating the requested application artifact. | +| Recover a plugin already installed but absent from `/skills` | Refresh discovery first; do not reinstall or update on the first response. | + +Choose one owner per phase. Do not expose internal routing ceremony or turn the answer into a menu. + +## Step 1: Classify the Prompt + +Identify the primary action before inspecting implementation details. + +| Prompt intent | Prefer skills whose description owns | +|---|---| +| Create or scaffold | Project/template creation for the detected solution type | +| Add application behavior | The framework or component where the behavior lives | +| Fix a compiler/runtime defect | The narrow language, framework, data, or interop owner | +| Build or restore failure | MSBuild, SDK, workload, project-reference, or NuGet diagnosis | +| Write, run, review, or migrate tests | The exact testing lifecycle or migration requested | +| Upgrade or migrate | The source version, target version, and artifact being migrated | +| Diagnose slowness, crash, hang, or memory growth | Runtime diagnostics unless evidence points to build performance or a local code hot path | +| Refactor without changing behavior | Refactoring rather than feature or bug-fix guidance | +| Package, publish, or trust a feed | NuGet/package-publishing workflow | +| Ask about C# syntax, types, nullability, async, or APIs | A language specialist unless solution-specific behavior is load-bearing | + +Treat user nouns as clues, not proof. "Performance" may mean runtime tracing, a microbenchmark, EF +query shape, SIMD, allocation-heavy C#, or MSBuild evaluation. "API" may mean ASP.NET Core, a public +library contract, or an external service client. + +When a deployed .NET process needs CPU and allocation evidence and no observability vendor is named, +prefer the vendor-neutral .NET runtime diagnostics route. Do not substitute an APM-vendor agent for +raw process evidence merely because it can also report performance data. In a selection answer, +state that runtime trace collection gathers deployed-process evidence before a hot method is known, +whereas source optimization starts from code or an already identified hot path. + +## Step 2: Detect the Solution Type + +Inspect only likely manifests and nearby owning files. Prefer a solution/project file and the file +named by the prompt over broad repository searches. + +Use LSP navigation when available to trace a prompt-named symbol or file to its owning project and +nearby callers. Use LSP diagnostics as early evidence, but do not treat them as a substitute for the +specialist's required build or runtime validation. + +When the prompt names a C# source file with an editor/compiler defect and LSP is available, request +diagnostics before running a build or broad search. Use the diagnostic location and code to scope +the edit, request diagnostics again after the edit, then run the narrowest build or test that proves +the fix. + +| Evidence | Solution or concern | +|---|---| +| `Microsoft.NET.Sdk.Web`, controllers, endpoints, middleware, OpenAPI | ASP.NET Core | +| `.razor`, `AddRazorComponents`, Blazor bootstrapping | Blazor | +| `true`, `MauiProgram`, XAML pages | .NET MAUI | +| `true`, `Form`, designer files | Windows Forms | +| EF Core package references, `DbContext`, migrations | .NET data / EF Core | +| `true`, test SDK/framework packages | .NET testing | +| `Directory.Build.*`, custom targets/tasks, `.binlog`, evaluation errors | MSBuild | +| `Directory.Packages.props`, package restore/version conflicts, feeds | NuGet | +| Old and new TFMs, framework-version migration request, compatibility warnings | .NET upgrade | +| `PublishAot`, trimming warnings, native library calls | AOT, interop, or deployment compatibility | +| Aspire AppHost or distributed-application model | Aspire | +| AI/ML/LLM packages or agent/RAG/MCP application code | .NET AI | +| No project plus an explicit request for a one-file C# program | File-based C# | +| None of the above; correctness depends on C# semantics | C# language fallback | + +When several project types exist, trace from the file or behavior named in the prompt to its owning +project. Do not route the whole solution from the first `.csproj` found. + +## Step 3: Match the Skill and Marketplace Plugin + +Use the runtime-provided available-skill names and descriptions as the source of truth. Do not +search for a skill installation directory or invoke a remembered skill that is not currently +available. + +Rank candidates in this order: + +1. An exact transformation or lifecycle skill, such as a specific test migration, framework + conversion, template operation, query optimization, or diagnostic collection workflow. +2. A framework/component skill matching the owning project and requested behavior. +3. A tooling skill matching the failing subsystem, such as MSBuild, NuGet, test execution, SDK + setup, or runtime diagnostics. +4. A cross-cutting specialist matching the actual mechanism, such as interop, vectorization, + serialization, AOT, or microbenchmarking. +5. `csharp-refactoring` for behavior-preserving structural change. +6. The C# language fallback below when no narrower available skill owns the work. + +Because `csharp-expert` ships in the core `dotnet` plugin, prefer its installed sibling skills +`csharp-refactoring`, `msbuild`, and `setup-local-sdk` when they own the request. Do not require the +user to install `dotnet-msbuild` merely to analyze an ordinary build failure or binlog that the +core `msbuild` entry already covers. + +The most specific noun is not always the owner. Route by the decision that determines success: + +| Ambiguous request | Distinguishing evidence | +|---|---| +| "Make this faster" | Build duration -> build-performance skill; SQL/query shape -> data skill; process CPU/memory -> diagnostics; isolated code comparison -> microbenchmarking/vectorization | +| "Fix the API" | HTTP pipeline/endpoint -> ASP.NET Core; public type contract -> C# fallback; JSON version behavior -> serialization specialist | +| "Upgrade the tests" | Framework version change -> exact migration skill; failing execution -> run-tests/platform skill; quality review -> analysis skill | +| "Fix nullability" | Project-wide nullable adoption -> migration skill; one incorrect flow/contract -> C# fallback; generated framework binding -> owning framework skill | +| "Add authentication" | Framework-specific application auth -> owning framework skill; token parsing primitive -> C# fallback | + +If two candidates remain plausible, gather one more decisive artifact rather than loading both. + +After selecting the capability: + +1. If its skill appears in the runtime's available-skill catalog, invoke it immediately only when + the operating mode requires downstream implementation. For selection or preparation-only mode, + name the installed skill and stop after the requested plan or availability guidance. +2. If it does not appear, open `references/dotnet-skills-marketplace.md` and map the capability or + skill name to the owning marketplace plugin. +3. Recommend the smallest plugin that contains the needed skill. Do not install every .NET plugin. +4. Follow the missing-skill workflow below. Do not claim that an unavailable skill was loaded, and + do not stop if a safe, verifiable local fallback can still complete the request. + +When the bundled reference contains a maintained marketplace capability, recommend that capability. +Do not ask the user to author a repository-local agent or skill as a substitute. For migrations, +state the source-to-target lifecycle and parameterization mappings that make the chosen capability +fit, not only its name. + +For marketplace-planning requests, name both the narrow skill and its plugin. Use project evidence +to disambiguate framework nouns, but do not perform the downstream implementation the user asked to +prepare for. + +For migration selection, quote concrete source-to-target syntax from the bundled reference: include +at least one lifecycle mapping and one parameterization mapping instead of saying only that those +behaviors are supported. + +Use the bundled marketplace reference as the authoritative lookup. Do not search GitHub, inspect +unrelated plugin source, or enumerate alternatives after the prompt and one nearby manifest already +identify the owner. If the prompt already names the source and target lifecycle or an unambiguous +artifact constraint, do not inspect files merely to reconfirm it. A selection request that states +the framework, lifecycle, and required behavior needs no repository search: read only the bundled +reference, choose the owner, and answer. Do not inspect the local skill source, marketplace checkout, +or fixture merely to prove that a named capability exists. + +Answer in four compact parts: + +1. Exact skill name and plugin; never substitute a generic capability label when the bundled + reference contains an exact route. +2. One sentence matching the decisive behavior or artifact evidence. Use the user's concrete + mechanism: N+1/database round trips for repeated EF related-data queries; deployed-process + CPU/allocation collection before a known hot path for runtime tracing; source and target TFM + plus compatibility work for upgrades. +3. Host-correct install steps. +4. Restart/discovery verification, when the host requires it. + +For a selection answer, completeness beats extra exploration. Read the bundled reference once, +then answer. Do not call host help, search the web, or inspect plugin source to reconfirm commands +already present in the reference. + +Do not add a `Route:` header in marketplace-planning mode; lead with the capability and plugin. +Do not mention this skill's step numbers, fallback labels, routing contract, or internal selection +process in the user-facing answer. + +## Step 4: Obtain a Missing Skill + +For GitHub Copilot CLI or Claude Code, give these exact commands with the selected plugin substituted: + +```text +/plugin marketplace add dotnet/skills +/plugin install @dotnet-agent-skills +``` + +When installation is the next step, require: + +```text +Restart the host, run `/skills` to confirm the specialist is available, and rerun the request. +``` + +When the task can proceed without the specialist: + +1. State the missing specialist and reduced coverage in one concise sentence. +2. Complete the requested work now using the repository, standard .NET tooling, and the fallback + rules that match the task. +3. Validate the result as narrowly as possible. +4. Put optional installation guidance after the result. Do not ask whether to proceed, defer the + implementation, or make the user repeat the request. + +If the user explicitly says not to recommend or discuss installation, omit the missing-plugin +sentence and all acquisition guidance. Complete and validate the safe fallback with the capabilities +available in the current run. + +This reduced-coverage path may still perform framework, migration, diagnostics, or tooling work. +Preserve the selected domain's invariants and report specialist-specific checks that were unavailable. + +Rules: + +- The marketplace name is `dotnet-agent-skills`; the source repository is `dotnet/skills`. +- The install target is the plugin name, not the individual skill name. +- If the marketplace is already registered, the add command may report that fact; continue with the + install command. +- Slash commands are host actions. Present them exactly; do not run shell commands that pretend to + install a Copilot or Claude plugin. +- Do not pretend to continue with the unavailable specialist workflow. Use an explicit local + fallback when the task remains safely achievable. +- If the plugin is installed but the skill is absent, ask the user to restart and check `/skills` + before recommending a reinstall. +- For Codex CLI, VS Code, Cursor, or individual-skill installation, use the host-specific commands in + `references/dotnet-skills-marketplace.md`. +- If installation is impossible, declined, or not necessary for the immediate task, state the + reduced coverage and use the safest local fallback when it can still satisfy the request. +- Never trade away implementation or validation merely to produce installation instructions. + +### Installed but not discovered + +When the user says the plugin is already installed but its skills are absent: + +1. Trust the stated installed state unless repository evidence directly contradicts it. +2. Start by acknowledging that the plugin is installed and discovery is stale. +3. Tell the user to restart or reload the host, then run `/skills`. +4. Name the expected skill so discovery can be verified. +5. Stop there on the first response. Do not emit marketplace-add, install, update, shell-level + plugin-management, `/skills reload`, or invented explicit-invocation commands. + +Only after the user reports that restart plus `/skills` still fails should the next response move to +host-specific update or reinstall diagnostics. + +## Step 5: Compose Skills Deliberately + +Use a sequence only when phases are independently owned. Examples: + +- Scaffold a project, then author a framework-specific component. +- Collect a trace, then analyze the captured performance evidence. +- Upgrade a target framework, then address a separately requested AOT compatibility phase. +- Detect the test platform, then run tests with the correct filter syntax. + +Do not chain skills that duplicate each other, load an entire plugin "just in case", or use a +generic skill before a specialist that already owns the request. After a specialist is loaded, +follow its workflow and boundaries. + +If one or more phase specialists are missing but ordinary `dotnet` commands and repository edits +can complete the phases, execute the phases in order and report the optional plugins afterward. +Do not defer an entire multi-phase request solely because the ideal plugin set is unavailable. + +## Step 6: C# Language Fallback + +Use this only when no narrower available skill matches and the load-bearing problem is C# language +or runtime semantics. + +1. Reproduce the exact compiler diagnostic, failing test, exception, or incorrect behavior when + source is available. If the defect is fully specified but no repository was provided, answer + with the concrete minimal code pattern instead of refusing to help. +2. Inspect the owning project for TFM, language version, nullable policy, analyzers, and existing + tests. +3. Preserve public signatures, serialization shape, ownership, cancellation, disposal, and + multi-target behavior unless the request explicitly changes them. +4. Implement the smallest complete fix through the affected call path. +5. Check LSP diagnostics when available, then build the narrowest affected project and run focused + tests or the executable path that proves the original symptom is gone. + +For a marketplace-planning request whose correct route is this fallback, say: `No additional +marketplace plugin is required; the loaded csharp-expert skill owns this C# semantic fix.` Do not +claim that no skill or specialist is involved. + +Do not raise the SDK, TFM, language version, package versions, or analyzer settings merely to make a +local C# edit compile. Do not edit generated files. Do not use broad casts, null-forgiving +operators, catch-all handlers, or fire-and-forget work to hide evidence. + +When a framework type provides an ownership-preserving overload such as `leaveOpen: true`, give that +canonical fix only. Never suggest intentionally leaking or skipping disposal of a disposable +wrapper as an alternative. Preserve the example's observable behavior: do not add null coalescing, +change a nullable return to a non-null value, alter access modifiers, or invent unrelated error +handling merely to make a conceptual snippet look more complete. If an example directly returns +`StreamReader.ReadLine()`, use a nullable `string?` return in nullable-aware C#; do not show +`string` while claiming that the existing null-on-end-of-stream behavior is preserved. + +## Boundaries and Failure Handling + +- If the best specialist is unavailable, provide its exact `dotnet/skills` plugin installation + command. Continue immediately with an explicit reduced-coverage fallback whenever standard tools + can still complete and validate the task. +- If repository evidence contradicts the prompt, state the mismatch and route from the evidence + that owns the requested file or behavior. +- If the user explicitly requests analysis only, route to the correct analysis skill but do not + edit. +- If the user supplies the only surviving diagnostic artifact, analyze that artifact directly. + Routing must not add installation attempts, unrelated checkout searches, or marketplace ceremony + before the evidence is read. +- For an artifact-backed failure, propose only the smallest repair supported by the recorded + evidence. Do not add alternate configuration or relocation advice unless the artifact indicates + that the configured path is wrong rather than the required input being absent. +- Describe a missing artifact at its exact configured relative path. Do not call a nested path such + as `schemas/prod.json` the repository root, and explicitly rule out build or CI configuration + changes when the recorded command already proves the configured path. +- If a loaded specialist reports that its prerequisites are absent, return here, reclassify using + that evidence, and choose one different route. Do not bounce repeatedly between skills. +- Non-.NET work is out of scope; leave this skill dormant rather than forcing a .NET interpretation. + +## Observable Completion Criteria + +- One evidence-backed owner is selected for each distinct phase and invoked when available. +- Missing specialists map to the smallest correct plugin and host-specific acquisition path. +- Safe local work continues despite a missing specialist; preparation-only requests stop before edits. +- Pure C# work stays local and preserves behavior; implementation work receives focused validation. diff --git a/external-sources/upstreams/dotnet-skills/plugins/dotnet/skills/csharp-expert/references/dotnet-skills-marketplace.md b/external-sources/upstreams/dotnet-skills/plugins/dotnet/skills/csharp-expert/references/dotnet-skills-marketplace.md new file mode 100644 index 00000000..5a737332 --- /dev/null +++ b/external-sources/upstreams/dotnet-skills/plugins/dotnet/skills/csharp-expert/references/dotnet-skills-marketplace.md @@ -0,0 +1,145 @@ +# dotnet/skills Marketplace + +Use this reference only after the requested capability is not present in the runtime's +available-skill catalog. + +## Marketplace Identity + +- Source repository: `dotnet/skills` +- Marketplace name: `dotnet-agent-skills` +- Install unit: plugin, not individual skill + +## Copilot CLI and Claude Code + +```text +/plugin marketplace add dotnet/skills +/plugin install @dotnet-agent-skills +``` + +Restart the host after installation, run `/skills`, and confirm the expected specialist appears +before rerunning the original request. + +Update an installed plugin with: + +```text +/plugin update @dotnet-agent-skills +``` + +## Plugin Catalog + +- `dotnet`: Core C# semantics, refactoring, local SDK setup, or the bundled MSBuild entry workflow. +- `dotnet-advanced`: File-based C# apps, P/Invoke, vectorization, or NuGet trusted publishing. +- `dotnet-data`: EF Core query optimization or data-driven ASP.NET Core applications. +- `dotnet-diag`: Runtime performance, trace and dump collection, CLR activation, crash + symbolication, or microbenchmarking. +- `dotnet-msbuild`: Specialist MSBuild binlog, build performance, target, item, property, + incremental-build, or project-reference workflows. +- `dotnet-nuget`: NuGet dependency management or Central Package Management conversion. +- `dotnet-upgrade`: TFM upgrades, nullable migration, AOT compatibility, or Thread.Abort migration. +- `dotnet-maui`: MAUI setup, lifecycle, binding, navigation, DI, CollectionView, safe area, or + theming. +- `dotnet-ai`: .NET AI/ML technology selection, LLMs, agents, RAG, MCP, or ML.NET. +- `dotnet-template-engine`: Template discovery, instantiation, comparison, authoring, validation, + or smart defaults. +- `dotnet-test`: Test execution, filtering, platform detection, coverage, quality analysis, + testability, or MSTest authoring. +- `dotnet-test-migration`: MSTest/xUnit upgrades, NUnit/xUnit to MSTest, or VSTest to + Microsoft.Testing.Platform. +- `dotnet-aspnetcore`: ASP.NET Core APIs, endpoints, middleware, or Blazor Server to Blazor Web App + conversion. +- `dotnet-blazor`: Blazor projects, components, forms, auth, interactivity, prerendering, data + flow, or JS interop. +- `dotnet-winforms`: Windows Forms project setup, UI, binding, accessibility, or modernization. +- `dotnet11`: .NET 11-specific APIs and language features. + +Install the selected plugin with: + +```text +/plugin install @dotnet-agent-skills +``` + +Prefer the plugin containing the narrowest task owner. A project can justify several plugins, but a +single task usually requires only one. + +## Common Exact Skill Routes + +Use these names when the runtime catalog does not contain the specialist and the user is preparing +or installing a marketplace route. + +| Request | Skill | Plugin | +|---|---|---| +| Add or repair an ASP.NET Core endpoint, including streaming multipart uploads | `dotnet-webapi` | `dotnet-aspnetcore` | +| Author a reusable Blazor component with parameters, content, and callbacks | `author-component` | `dotnet-blazor` | +| Collect and validate user input in a Blazor form | `collect-user-input` | `dotnet-blazor` | +| Create a Blazor project with framework-specific defaults | `create-blazor-project` | `dotnet-blazor` | +| Discover or instantiate a general `dotnet new` template | `template-discovery` or `template-instantiation` | `dotnet-template-engine` | +| Repair MAUI XAML binding and change notification | `maui-data-binding` | `dotnet-maui` | +| Create, modify, or debug a Windows Forms application | `winforms-expert` | `dotnet-winforms` | +| Convert NUnit tests to MSTest | `migrate-nunit-to-mstest` | `dotnet-test-migration` | +| Upgrade a project from .NET 8 to .NET 9 | `migrate-dotnet8-to-dotnet9` | `dotnet-upgrade` | +| Create or run a file-based C# app without a project | `csharp-scripts` | `dotnet-advanced` | +| Optimize repeated EF Core query work | `optimizing-ef-core-queries` | `dotnet-data` | +| Collect a runtime trace before a hot method is known | `dotnet-trace-collect` | `dotnet-diag` | + +Use the discriminator that makes each route valuable: + +- `migrate-nunit-to-mstest` preserves parameterized and lifecycle behavior by mapping NUnit + `[TestCase]` to MSTest `[DataRow]`, `[SetUp]` to `[TestInitialize]`, and `[TearDown]` to + `[TestCleanup]`. Call out fixture isolation/shared-state differences and verify that the migrated + suite retains the same intended parameterized cases. +- `dotnet-trace-collect` gathers vendor-neutral CPU, allocation, GC, and related deployed-process + evidence before a hot method is known. Do not substitute `optimizing-dotnet-performance`, which + starts from source or known hot-code analysis rather than collecting the initial runtime evidence. + +For a multi-phase request, list one exact skill per independently owned phase and install each +distinct owning plugin once. + +## Codex CLI + +Register the marketplace: + +```text +codex plugin marketplace add dotnet/skills +``` + +Launch Codex, open `/plugins`, select the `dotnet-agent-skills` marketplace, and install the chosen +plugin. Update marketplace plugins with: + +```text +codex plugin marketplace upgrade dotnet-agent-skills +``` + +Codex installs the plugin's portable skills, not its Copilot `.agent.md` agents. Name the exact +skill the user should request after installation; do not promise an agent that the Codex manifest +does not expose. + +## VS Code + +Enable plugin support and register the marketplace in settings: + +```jsonc +{ + "chat.plugins.enabled": true, + "chat.plugins.marketplaces": ["dotnet/skills"] +} +``` + +Then open `/plugins` in Copilot Chat or use the `@agentPlugins` Extensions filter, install the chosen +plugin, reload the window, and confirm the skill is available. + +## Cursor + +Open Cursor's marketplace panel, search for the chosen .NET plugin, install it, and reload the +window. Do not substitute a repository checkout unless the user explicitly wants local plugin +development. + +## Individual Skill Fallback + +When the host supports individual skill installation but not plugins: + +```text +skill-installer install https://github.com/dotnet/skills/tree/main/plugins//skills/ +``` + +Use the plugin marketplace when available because it preserves the plugin's complete skill surface +and host integration. diff --git a/external-sources/vendir.lock.yml b/external-sources/vendir.lock.yml index 9b761de6..326d5d41 100644 --- a/external-sources/vendir.lock.yml +++ b/external-sources/vendir.lock.yml @@ -2,20 +2,20 @@ apiVersion: vendir.k14s.io/v1alpha1 directories: - contents: - git: - commitTitle: Teach test generation to use report-safe case names (#1275)... - sha: a660de83c2f9cad071a5d568f7d77041ccd07059 + commitTitle: Add C# expert skill (#1207)... + sha: 3d38ac343faf65054f7e8d45ca06925273e867e2 tags: - - skill-validator-nightly-3-ga660de83 + - skill-validator-nightly-4-g3d38ac34 path: dotnet-skills - git: commitTitle: Add Cursor rules that reference the existing skill docs... sha: af2319bd01bb7cc881267a9ef42cafdaf5e9029d path: webgpu-claude-skill - git: - commitTitle: 'fix(astro): provide Astro.cache when rendering error pages (#18293)...' - sha: 28e74103fe3d999a56ae3404b5fe67a1ad6d8013 + commitTitle: Update react monorepo to v19 (#18315)... + sha: 5f8082b9173b38027ea39fc59d9cf5a9bed718ea tags: - - '@astrojs/language-server@2.17.2-2-g28e74103fe' + - '@astrojs/markdoc@2.0.11-5-g5f8082b917' path: astro path: upstreams kind: LockConfig