diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index d397977c..6dc565a0 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -50,7 +50,11 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 5 steps: + # Full history: AI integration tests skip themselves unless the AI + # surface changed vs origin/master (no merge-base in a shallow clone). - uses: actions/checkout@v7 + with: + fetch-depth: 0 - uses: ./.github/actions/setup-ruby-and-dependencies with: @@ -71,6 +75,10 @@ jobs: steps: - name: Checkout code uses: actions/checkout@v7 + # Full history: AI integration tests skip themselves unless the AI + # surface changed vs origin/master (no merge-base in a shallow clone). + with: + fetch-depth: 0 - uses: ./.github/actions/setup-ruby-and-dependencies with: @@ -107,10 +115,9 @@ jobs: contains(github.event.pull_request.labels.*.name, 'full-ci') needs: [ functional-test ] runs-on: ubuntu-latest - # Must fit `max_attempts * timeout_minutes` below, plus ~1 min of setup, - # or the last attempt gets killed mid-run and the cell reports `cancelled` - # -- a dead gate. JRuby: 1 + 20 + 20 = 41. MRI: 1 + 4 + 4 = 9. - timeout-minutes: ${{ contains(matrix.ruby-version, 'jruby') && 41 || 9 }} + # ~1 min of setup plus one 4-min attempt; re-measure if the suite grows + # into it -- a killed attempt reports `cancelled`, a dead gate. + timeout-minutes: 6 continue-on-error: ${{ matrix.experimental }} strategy: matrix: @@ -131,17 +138,47 @@ jobs: - ruby-version: "4.0" gemfile: edge_gems.rb experimental: true - # One cell: JRuby's differences (no Kernel#fork, thread semantics, - # the FFI/vips path) are JVM-level, not Rails-level. - - ruby-version: jruby-10.1 - gemfile: rails81_gems.rb - experimental: false env: BUNDLE_GEMFILE: gemfiles/${{ matrix.gemfile }} # JRuby is the only place ruby-vips' FFI path runs on a non-MRI engine. # Test Drivers covers both backends on CRuby. - SCREENSHOT_DRIVER: ${{ contains(matrix.ruby-version, 'jruby') && 'vips' || 'chunky_png' }} + SCREENSHOT_DRIVER: chunky_png + + steps: + - uses: actions/checkout@v7 + + - uses: ./.github/actions/setup-ruby-and-dependencies + with: + ruby-version: ${{ matrix.ruby-version }} + ruby-cache-version: ${{ matrix.ruby-version }}-${{ matrix.gemfile }}-1 + cache-apt-packages: true + + # Measured, not chosen: slowest MRI attempt is ~146s. The suite has + # timed out before when it grew into this -- re-measure then. + - run: bin/rake test + timeout-minutes: 4 + + matrix-jruby: + name: Test JRuby + # The single most expensive cell (~12 min x up to 2 attempts): JRuby's + # differences (no Kernel#fork, thread semantics, FFI/vips) are JVM-level, + # so the weekly drift check covers them. Not on every master push. + if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' + needs: [ functional-test ] + runs-on: ubuntu-latest + # 1 min setup + 20 + 20: the retry below exists only for the + # intermittent JRuby teardown hang (#244). + timeout-minutes: 41 + strategy: + matrix: + ruby-version: [ jruby-10.1 ] + gemfile: [ rails81_gems.rb ] + + env: + BUNDLE_GEMFILE: gemfiles/${{ matrix.gemfile }} + # The one place ruby-vips' FFI path runs on a non-MRI engine. + SCREENSHOT_DRIVER: vips steps: - uses: actions/checkout@v7 @@ -152,15 +189,10 @@ jobs: ruby-cache-version: ${{ matrix.ruby-version }}-${{ matrix.gemfile }}-1 cache-apt-packages: true - - name: Run tests (with 1 retry) + - name: Run tests (with 1 retry for the #244 teardown hang) uses: nick-fields/retry@v4 with: - # Measured, not chosen: slowest attempt is ~880s JRuby, ~146s MRI. - # Both have timed out before when the suite grew into them -- when - # that happens, re-measure and raise this AND timeout-minutes above. - timeout_minutes: ${{ contains(matrix.ruby-version, 'jruby') && 20 || 4 }} - # Exists only for the intermittent JRuby teardown hang (#244). - # Once that is fixed, drop to 1 and halve the JRuby bill. + timeout_minutes: 20 max_attempts: 2 command: bin/rake test @@ -213,8 +245,9 @@ jobs: # gemfile, so a rule listing them stops requiring anything the moment the # matrix changes shape. `skipped` passes -- the matrix is off on PRs. if: always() - needs: [ test-minimal-setup, functional-test, matrix, matrix-screenshot-driver ] + needs: [ test-minimal-setup, functional-test, matrix, matrix-jruby, matrix-screenshot-driver ] runs-on: ubuntu-latest + timeout-minutes: 2 steps: - if: contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled') run: exit 1 diff --git a/CHANGELOG.md b/CHANGELOG.md index 9cf283f0..531e8003 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -18,6 +18,19 @@ green while comparing nothing**, and all four are fixed here — that is the rea The sections below the divider are the prerelease notes (alpha1 → beta3) and are kept as history. This entry is the one to read if you are coming from **1.15.1**. +### Added + +- **AI triage (optional, offline-first, advisory by default).** `SnapDiff::Reporters::AISimple` + classifies every failed comparison as `flaky` / `intentional` / `real_bug`, + logs one line per diff, quotes the verdict in the failure message, and + annotates the HTML report with a verdict badge and summary — all via one + shared in-memory store (`SnapDiff::AI`), no files written. With + `fail_on: %w[real_bug]` the verdict gates pass/fail: + flaky/intentional diffs stay green, `unknown` always fails. Default backend + is offline CLIP via the `informers` gem; custom backends (local VLM, + decision APIs) register by name via `SnapDiff::AI.register`. Zero new hard + dependencies. See `docs/ai.md`. + ### Upgrading from 1.15.1: change the version, run your suite ```ruby diff --git a/README.md b/README.md index fbf6c856..c6889602 100644 --- a/README.md +++ b/README.md @@ -237,6 +237,28 @@ After tests run, open `doc/screenshots/snap_diff_report.html`: See [Web UI & Custom Reporters](docs/reporters.md) for full feature details and [CI Integration](docs/ci-integration.md) for GitHub Actions setup. +## AI-Assisted Triage (optional) + +AI tells you *which* failures to look at first. One line enables offline +CLIP triage — every failed comparison is classified as `flaky` / +`intentional` / `real_bug`, logged per diff, quoted in the failure message, +and badged in the HTML report (all via one shared in-memory store — no +files). Advisory by +default; pass `fail_on: %w[real_bug]` to keep the suite green on flaky and +intentional diffs and fail only on what AI calls a real bug: + +```ruby +# Gemfile +gem "informers" # offline CLIP via ONNX, no API key + +# test/test_helper.rb +require "snap_diff/reporters/ai_simple" +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new) +``` + +Plug in your own backend (a local VLM, a decision API) with +`SnapDiff::AI.register(:name) { ... }` — see [AI triage](docs/ai.md). + ## Compare Any Two Images Works without a browser — PDFs, generated images, CI artifacts: @@ -328,6 +350,7 @@ instead of failing quietly — useful when `snap_diff_report.html` is missing en - [Image Processing Drivers](docs/drivers.md) — VIPS, ChunkyPNG, perceptual threshold - [Screenshot Organization](docs/organization.md) — groups, sections, cropping, multi-browser - [Web UI & Custom Reporters](docs/reporters.md) — interactive report, custom reporters +- [AI-Assisted Triage](docs/ai.md) — optional flaky/intentional/real_bug classification, pluggable backends ## Development diff --git a/docs/ai.md b/docs/ai.md new file mode 100644 index 00000000..a468877d --- /dev/null +++ b/docs/ai.md @@ -0,0 +1,234 @@ +# AI-assisted diff triage + +> Optional and offline-first. Advisory by default — the pixel comparison +> stays the verdict and AI classifies failures so you know which reds to +> look at first. Opt into `fail_on:` to let the AI verdict gate failures. + +`SnapDiff::Reporters::AISimple` analyzes every comparison that **already +failed** and labels it: + +| Verdict | Meaning | Typical cause | +| --- | --- | --- | +| `flaky` | Pixels differ, semantics identical | Anti-aliasing, timestamps, avatars | +| `intentional` | Semantics shifted within a familiar shape | Redesign, copy change | +| `real_bug` | Semantics moved substantially | Clipped text, overlap, missing element | +| `unknown` | Backend could not score | Missing/unreadable image | + +One line per diff in the test output, and — automatically, via the +shared store — a verdict badge plus one-line summary per failure in +`snap_diff_report.html`. No files, no duplicate storage. + +## How the HTML report gets AI annotations + +Both reporters read the same shared store: `AISimple` writes each result +into `SnapDiff::AI` (keyed by screenshot name), and the HTML reporter +attaches `SnapDiff::AI[name]` to the failure entry — behind a `defined?` +guard, so `html.rb` never requires the AI module and nothing changes +when AI triage isn't loaded. No wiring between the two reporters: + +```ruby +require "snap_diff/reporters/html" # already auto-registers +require "snap_diff/reporters/ai_simple" +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new) +``` + +The report then shows, per failure: an `AI: real bug` / `AI: flaky` badge +in the sidebar and top strip, plus `similarity · confidence · summary` +when the backend provides them. Under fork-parallel, results merge in the +parent before the report renders. + +``` +[snap_diff:ai] homepage: FLAKY (clip, similarity=0.9912) + pixels differ but semantics match -- candidate for skip_area or a tolerance bump +[snap_diff:ai] checkout: REAL_BUG (clip, similarity=0.7312) +[snap_diff:ai] 2 diff(s) analyzed: 1 real_bug, 1 flaky +``` + +## Architecture + +``` +lib/snap_diff/ai.rb # backend registry, verdict thresholds, shared result store +lib/snap_diff/ai/backends/clip.rb # built-in offline backend (informers) +lib/snap_diff/reporters/ai_simple.rb # record/finalize/summary; writes the store +lib/snap_diff/reporters/html.rb # annotates failures from the store (defined? guard) +lib/snap_diff/screenshot_assertion.rb# validate: fail-gate + AI line in the failure message +``` + +A backend is any object with `#call(name:, base:, current:, meta:) -> Hash`. +Extend without editing existing files: + +```ruby +SnapDiff::AI.register(:jev) { JevBackend.new } # optional requires go in the block +``` + +The returned hash may carry `:verdict` (used as-is) or `:similarity` +(classified by the shared thresholds); any other keys (`:confidence`, +`:summary`, `:auto_accept`) pass through to the report untouched. + +## Setup + +Default backend: CLIP via [`informers`](https://github.com/ankane/informers) +— ONNX Runtime, fully offline, ~90 MB quantized model, ~40 ms/image. + +```ruby +# Gemfile +gem "informers" # pulls onnxruntime itself + +# test/test_helper.rb +require "snap_diff/reporters/ai_simple" +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new) +``` + +Without `informers`, one warning at registration, then silence — never a +failed build. + +**Prefetch the model.** The ~90 MB download happens on the first analyzed +diff, so a fully offline run needs a prefilled cache. Warm it at suite +setup (or as a cached CI step): + +```ruby +SnapDiff::AI::Clip.new.prefetch! +``` + +Custom thresholds: + +```ruby +SnapDiff::Reporters::AISimple.new(flaky: 0.99, intentional: 0.85) +``` + +## Gating failures on AI verdicts + +Advisory mode never touches pass/fail. Add `fail_on:` and the verdict +decides: diffs classified as anything not listed are **suppressed** — the +test stays green, the result is still logged, stored, and shown in the +HTML report. `unknown` always fails: AI can downgrade a diff, never vouch +for one it could not classify. + +```ruby +# Fail only on what AI calls a real bug; flaky/intentional stay green. +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(fail_on: %w[real_bug])) +``` + +Failures that survive the gate quote the verdict inline, no clicks needed: + +``` +Screenshot does not match for 'checkout': the change spans ... + AI triage: REAL_BUG (clip, similarity=0.7312) -- CTA clipped +``` + +Suppressed diffs log one line instead: + +``` +[snap_diff:ai] homepage: failure suppressed -- FLAKY (clip, similarity=0.9912) +``` + +## Backend recipes + +### Typed decisions (TypeSafe Jev) + +Jev answers typed questions (Choice / Score / Noul) over a state — text and +structured data, not images — so we send the diff *metrics*, not the pixels. +One call returns a verdict CI can branch on. Requires the community +[`typesafe-sdk`](https://rubygems.org/gems/typesafe-sdk) gem and +`TYPESAFE_API_KEY`: + +```ruby +class JevBackend + def name = "jev" + + def initialize + require "typesafe/sdk" + @client = Typesafe::SDK::Client.new(api_key: ENV.fetch("TYPESAFE_API_KEY")) + end + + def call(name:, base:, current:, meta: {}) + resp = @client.system_one( + state: {screenshot: name, changed_area_px: meta[:area_size], changed_region: meta[:region]}, + questions: { + verdict: Typesafe::SDK::Choice.new( + instructions: "Classify this visual diff", + criteria: { + flaky: "timestamp, anti-aliasing, or avatar noise", + intentional: "deliberate redesign or copy change", + real_bug: "clipped, overlapping, or missing UI" + } + ), + auto_accept: Typesafe::SDK::Noul.new(instructions: "Safe to accept the new rendering as the baseline") + } + ) + answers = resp.answers + {verdict: answers["verdict"].choice, + confidence: answers["verdict"].confidence.round(2), + auto_accept: answers["auto_accept"].noul.round(3)} + end +end + +SnapDiff::AI.register(:jev) { JevBackend.new } +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(backend: :jev)) +``` + +### Local VLM explanations (Qwen2.5-VL via Ollama) + +```ruby +class QwenBackend + def name = "qwen" + + def initialize + require "ollama-ai" + require "base64" + @ollama = Ollama.new + end + + def call(name:, base:, current:, meta: {}) + res = @ollama.chat(model: "qwen2.5vl:3b", messages: [{ + role: "user", + content: "You are visual QA. Compare baseline vs current. " \ + 'Return JSON {"summary": one sentence, "verdict": "real_bug|intentional|flaky"}.', + images: [base, current].map { |p| Base64.strict_encode64(File.binread(p)) } + }]) + content = res.dig("message", "content").to_s + JSON.parse(content, symbolize_names: true).slice(:verdict, :summary) + rescue JSON::ParserError + {summary: content} + end +end +``` + +### Hybrid: CLIP first, Qwen only where it pays + +```ruby +class ClipThenQwen + def name = "clip+qwen" + + def initialize + @clip = SnapDiff::AI::Clip.new + @qwen = QwenBackend.new + end + + def call(name:, base:, current:, meta: {}) + result = @clip.call(name: name, base: base, current: current, meta: meta) + sim = result[:similarity] + return result if sim.nil? || sim >= 0.985 || sim < 0.90 + + @qwen.call(name: name, base: base, current: current, meta: meta).merge(result) + end +end +``` + +## CI gating + +Use the built-in gate — no JSON parsing, no extra script: + +```ruby +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(fail_on: %w[real_bug])) +``` + +The suite then fails only on verdicts you list; see +[Gating failures on AI verdicts](#gating-failures-on-ai-verdicts). + +## Guarantees + +- **Advisory by default**: pass/fail changes only with an explicit `fail_on:` opt-in. +- **Thread-safe**: results behind one mutex; CLIP inference serialized. +- **Fork-parallel**: results are plain JSON-able hashes; workers merge via `dump_state`/`merge_state!`. +- **No hard deps**: `informers`, `typesafe`, `ollama-ai` all optional; gemspec unchanged. diff --git a/docs/reporters.md b/docs/reporters.md index 9154ce76..6b8b0cdb 100644 --- a/docs/reporters.md +++ b/docs/reporters.md @@ -127,4 +127,12 @@ still run. Full details in [Custom reporters](snapdiff.md#custom-reporters). +## AI triage reporter + +`SnapDiff::Reporters::AISimple` is an optional reporter: it classifies each +failed comparison as `flaky` / `intentional` / `real_bug` (offline CLIP by +default, pluggable backends) and its verdicts show up in this HTML report as +badges. Advisory by default — it never changes pass/fail unless you opt in +with `fail_on: %w[real_bug]`. Setup and recipes: [AI triage](ai.md). + [← Back to README](../README.md) diff --git a/lib/snap_diff/ai.rb b/lib/snap_diff/ai.rb new file mode 100644 index 00000000..4eb15698 --- /dev/null +++ b/lib/snap_diff/ai.rb @@ -0,0 +1,115 @@ +# frozen_string_literal: true + +# Optional AI triage. A backend is any object responding to +# #call(name:, base:, current:, meta:) -> Hash; register one by name or +# pass an instance. Builders run lazily, so optional gems load only when +# their backend is used. Advisory by default: pixel diff stays the verdict +# unless a reporter is configured with fail_on: (see AISimple). +# +# Results live in a process-wide store so ANY consumer can read them -- +# the AISimple reporter writes, the HTML reporter annotates from it, the +# fail-gate consults it. Keyed by screenshot name; later writes win. +module SnapDiff + module AI + # The only verdicts a backend may return; anything else maps to + # "unknown", which the fail-gate never suppresses. + VERDICTS = %w[real_bug intentional flaky unknown].freeze + + @backends = {} + @results = {} + @mutex = Mutex.new + + class << self + # Optional fail-gate, set by AISimple when configured with fail_on:. + # ScreenshotAssertion#validate consults it on a pixel diff: verdicts + # the gate accepts suppress the failure, the rest still fail. + attr_accessor :gate + + def gated_result(name, difference) = gate&.gated_result(name, difference) + + # No gate -> everything fails, exactly as without AI. + def fails?(verdict) = gate ? gate.fails?(verdict) : true + + def register(name, &build) + @mutex.synchronize { @backends[name.to_sym] = build } + end + + def names + @mutex.synchronize { @backends.keys } + end + + # nil/:default -> :clip; a Symbol -> registered builder; anything + # else must be a backend instance responding to #call. + def resolve(input = nil) + return build((input.nil? || input == :default) ? :clip : input) if input.nil? || input.is_a?(Symbol) + + unless input.respond_to?(:call) + raise ArgumentError, "AI backend must respond to #call(name:, base:, current:, meta:), got #{input.inspect}" + end + input + end + + def build(name) + builder = @mutex.synchronize { @backends[name.to_sym] } or + raise ArgumentError, "unknown AI backend #{name.inspect} (registered: #{names.map(&:inspect).join(", ")})" + builder.call + end + + # One-line rendering of a result, shared by the AISimple log and the + # assertion failure message: "REAL_BUG (clip, similarity=0.73) -- CTA clipped". + def format(result) + line = "#{result[:verdict].upcase} (#{result[:backend]}" + line += ", similarity=#{result[:similarity]}" if result[:similarity] + line += ", confidence=#{result[:confidence]}" if result[:confidence] + line += ")" + line += " -- #{result[:summary]}" if result[:summary] + line + end + + # similarity -> verdict. The one place thresholds live. + # Anything but a finite number is "unknown" -- NaN/Infinity would + # otherwise classify as flaky and let the gate suppress a real diff. + def verdict(similarity, flaky: 0.985, intentional: 0.90) + return "unknown" unless similarity.is_a?(Numeric) && similarity.finite? + + if similarity >= flaky + "flaky" + elsif similarity >= intentional + "intentional" + else + "real_bug" + end + end + + # --- shared result store ------------------------------------------ + + def record_result(result) + @mutex.synchronize { @results[result[:name]] = result } + end + + def [](name) + @mutex.synchronize { @results[name] } + end + + def results + @mutex.synchronize { @results.values } + end + + # Test isolation hook, same role as Reporting.reset_run_totals!. + def clear_results! + @mutex.synchronize { @results.clear } + end + + # Fork-parallel: plain hashes round-trip via JSON fragments. + def dump_state + {"results" => results} + end + + def merge_state!(state) + Array(state.is_a?(Hash) && state["results"]).each { |r| record_result(r.transform_keys(&:to_sym)) } + end + end + end +end + +require "snap_diff/ai/backends/clip" diff --git a/lib/snap_diff/ai/backends/clip.rb b/lib/snap_diff/ai/backends/clip.rb new file mode 100644 index 00000000..ede99bc2 --- /dev/null +++ b/lib/snap_diff/ai/backends/clip.rb @@ -0,0 +1,48 @@ +# frozen_string_literal: true + +module SnapDiff + module AI + # Offline default: CLIP embeddings via the +informers+ gem (ONNX, + # ~90 MB quantized, ~40 ms/image, no network). Returns only + # :similarity; verdicts come from AI.verdict. + class Clip + def name = "clip" + + def initialize(model: "Xenova/clip-vit-base-patch32", pipeline: nil) + require "informers" unless pipeline + @model = model + @pipeline = pipeline + @mutex = Mutex.new # ONNX session thread safety is not guaranteed + end + + def call(name:, base:, current:, meta: {}) + {similarity: cosine(base, current)&.round(4)} + end + + # Downloads and loads the model NOW. Run at suite setup (or a CI + # cache step) so the first diff is analyzed offline: without a + # prefetched model, the first diff pulls ~90 MB over the network. + def prefetch! + pipeline + end + + private + + def pipeline + # Lazy: the one-time model download lands on the first diff, + # not at reporter registration. Memoized per model. + @pipeline ||= Informers.pipeline("image-feature-extraction", @model, quantized: true) + end + + def cosine(path_a, path_b) + return unless path_a && path_b && File.exist?(path_a) && File.exist?(path_b) + + a, b = @mutex.synchronize { [pipeline.call(path_a).first, pipeline.call(path_b).first] } + norm = Math.sqrt(a.sum { |x| x * x }) * Math.sqrt(b.sum { |x| x * x }) + a.zip(b).sum { |x, y| x * y } / norm if norm.positive? + end + end + + register(:clip) { Clip.new } + end +end diff --git a/lib/snap_diff/reporters/ai_simple.rb b/lib/snap_diff/reporters/ai_simple.rb new file mode 100644 index 00000000..a7895305 --- /dev/null +++ b/lib/snap_diff/reporters/ai_simple.rb @@ -0,0 +1,134 @@ +# frozen_string_literal: true + +require "snap_diff/ai" + +module SnapDiff + module Reporters + # Advisory AI triage: classifies every FAILED comparison as + # flaky/intentional/real_bug and logs one line per diff. By default + # never changes pass/fail; with fail_on: the verdict gates it instead. + # + # SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new) # offline CLIP + # SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(backend: :jev)) + # SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(backend: my_backend)) + # SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(fail_on: %w[real_bug])) + # + # Results live only in the shared SnapDiff::AI store -- the HTML + # reporter annotates from it, gate reads it, no files, no wiring. + # Extension: SnapDiff::AI.register(:name) { backend } -- no edits here. + class AISimple + # fail_on: verdicts that still fail the test; every other verdict + # suppresses the pixel diff. "unknown" ALWAYS fails -- AI can + # downgrade a diff, never vouch for one it could not classify. + # nil thresholds defer to AI.verdict's defaults -- one source of truth. + def initialize(backend: nil, flaky: nil, intentional: nil, fail_on: nil) + @thresholds = {flaky: flaky, intentional: intentional}.compact + @backend = resolve(backend) + @fail_on = Array(fail_on).map(&:to_s) if fail_on + # Memoized per COMPARISON AND INPUTS: the gate at validate-time + # and the reporter pass at teardown see the same difference with + # the same name and metrics; a reassigned compare, a mutated + # result, or a later test asserting the same name all change the + # key and re-analyze -- a stale "flaky" must never suppress a + # fresh regression. + @memo = {} + @memo_mutex = Mutex.new + AI.gate = self if @fail_on + end + + def record(assertions) + return unless @backend + + assertions.each do |a| + difference = a.compare&.difference + next unless difference&.different? + + analyze_once(a.name, difference) + end + end + + # Fail-gate entry point, called by ScreenshotAssertion#validate on a + # pixel diff. Analysis runs there (before the error message is + # built) and is memoized, so #record never re-analyzes. + def gated_result(name, difference) + return unless @fail_on && @backend + + analyze_once(name, difference) + end + + def fails?(verdict) = verdict == "unknown" || @fail_on.include?(verdict) + + # Results are already in the shared store -- nothing to write out. + def finalize = nil + + def summary + results = AI.results + return if results.empty? + + counts = results.group_by { |r| r[:verdict] }.transform_values(&:size) + breakdown = AI::VERDICTS.filter_map { |v| "#{counts[v]} #{v}" if counts[v] } + "[snap_diff:ai] #{results.size} diff(s) analyzed: #{breakdown.join(", ")}" + end + + # Fork-parallel (Rails parallelize): the shared store round-trips. + def dump_state = AI.dump_state + def merge_state!(state) = AI.merge_state!(state) + + private + + # Unknown names and invalid objects are config errors and raise; + # a factory that fails to build (missing gem, init error) degrades + # to a warning and disabled triage. + def resolve(backend) + AI.resolve(backend) + rescue ArgumentError + raise + rescue LoadError => e + warn "[snap_diff:ai] backend unavailable (#{e.message}) -- AI triage disabled." + nil + rescue => e + warn "[snap_diff:ai] backend factory failed (#{e.class}: #{e.message}) -- AI triage disabled." + nil + end + + def analyze_once(name, difference) + key = [difference.object_id, name, difference.to_h] + result = @memo_mutex.synchronize do + @memo.fetch(key) { @memo[key] = analyze(name, difference) } + end + AI.record_result(result) if result + result + end + + def analyze(name, difference) + raw = @backend.call( + name: name, + base: difference.original_image_path&.to_s, + current: difference.new_image_path&.to_s, + meta: difference.to_h + ).transform_keys(&:to_sym) + + # A backend's own :verdict outranks the shared thresholds, but + # only known verdicts are honored -- anything else becomes + # "unknown", which the fail-gate never suppresses. + raw[:verdict] = AI.verdict(raw[:similarity], **@thresholds) if raw[:verdict].nil? + raw[:verdict] = "unknown" unless AI::VERDICTS.include?(raw[:verdict]) + result = {name: name, backend: backend_name}.merge(raw.except(:name, :backend)) + log(result) + result + rescue => e + warn "[snap_diff:ai] Backend failed for #{name.inspect} (#{e.class}: #{e.message})" + nil + end + + def backend_name + @backend.respond_to?(:name) ? @backend.name : "custom" + end + + def log(r) + $stdout.puts "[snap_diff:ai] #{r[:name]}: #{AI.format(r)}" + $stdout.puts " pixels differ but semantics match -- candidate for skip_area or a tolerance bump" if r[:verdict] == "flaky" + end + end + end +end diff --git a/lib/snap_diff/reporters/html.rb b/lib/snap_diff/reporters/html.rb index 9f79a2f4..b0429e81 100644 --- a/lib/snap_diff/reporters/html.rb +++ b/lib/snap_diff/reporters/html.rb @@ -97,6 +97,7 @@ def summary end def render + attach_ai_annotations ERB.new(File.read(self.class.template_path)).result(binding) end @@ -123,7 +124,23 @@ def failure_entry_for(name, compare) diff_level: difference.ratio && (difference.ratio * 100).round(2), area_size: difference.region_area_size, max_color_distance: difference.meta[:max_color_distance]&.round(1) - } + }.compact + end + + # Advisory AI triage annotations, attached at RENDER time: HTML + # records before the AI reporter (auto-registration runs first) and + # fork-parallel merges land after record, so only the final render + # can see every verdict. HTML never requires the AI module -- the + # annotation appears iff the user opted into AI triage. + def attach_ai_annotations + return unless defined?(SnapDiff::AI) + + failures.each do |entry| + # || would create a nil :ai key on misses; keep the entry clean. + if (annotation = SnapDiff::AI[entry[:name]]) + entry[:ai] ||= annotation + end + end end def resolve_image(path) diff --git a/lib/snap_diff/reporters/templates/report.html.erb b/lib/snap_diff/reporters/templates/report.html.erb index 8c756e5a..e4ad5993 100644 --- a/lib/snap_diff/reporters/templates/report.html.erb +++ b/lib/snap_diff/reporters/templates/report.html.erb @@ -81,6 +81,16 @@ .thumb-badge-fail { background: var(--red-dim); color: var(--red); } .thumb-badge-pass { background: var(--green-dim); color: var(--green); } + /* ── AI triage ─────────────────────────────────────────────────── */ + .ai-badge { font-size: .625rem; font-family: var(--font-mono); font-weight: 600; padding: .1rem .375rem; border-radius: 10px; white-space: nowrap; flex-shrink: 0; } + .ai-real_bug { background: var(--red-dim); color: var(--red); } + .ai-intentional { background: rgba(251,146,60,.15); color: var(--orange); } + .ai-flaky { background: var(--green-dim); color: var(--green); } + .ai-unknown { background: var(--bg4); color: var(--text2); } + #ai-bar { display: none; align-items: center; gap: .5rem; padding: .3rem .75rem; background: var(--bg3); border-bottom: 1px solid var(--border); font-size: .6875rem; color: var(--text2); flex-shrink: 0; } + #ai-bar.visible { display: flex; } + #ai-text { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; min-width: 0; } + /* ── Main ────────────────────────────────────────────────────────── */ #main { flex: 1; overflow: hidden; display: flex; flex-direction: column; background: var(--bg); } @@ -202,6 +212,8 @@ +
+
@@ -239,6 +251,10 @@ /* ── Sidebar ──────────────────────────────────────────────────────── */ function esc(s) { return String(s).replace(/&/g,'&').replace(//g,'>').replace(/"/g,'"'); } + var AI_VERDICTS = ['real_bug', 'intentional', 'flaky', 'unknown']; + /* Class suffix only for known verdicts: a backend's verdict is model + output, and model output is not trusted enough for a class attribute. */ + function aiClass(v) { return AI_VERDICTS.indexOf(v) !== -1 ? ' ai-' + v : ''; } function buildSidebar() { sidebarList.innerHTML = ''; @@ -258,6 +274,7 @@ '
' + ''; btn.addEventListener('click', function() { selectItem(+this.dataset.idx); }); @@ -294,6 +311,22 @@ } topBadge.className = 'diff-badge ' + (hasDiff ? 'diff-badge-fail' : 'diff-badge-pass'); + /* AI triage strip: shown only when an advisory verdict exists */ + var aiBar = $('ai-bar'); + if (item.ai && item.ai.verdict) { + var aiV = $('ai-verdict'); + aiV.textContent = 'AI: ' + item.ai.verdict.replace('_', ' '); + aiV.className = 'ai-badge' + aiClass(item.ai.verdict); + var bits = []; + if (item.ai.similarity != null) bits.push('similarity ' + item.ai.similarity); + if (item.ai.confidence != null) bits.push('confidence ' + item.ai.confidence); + if (item.ai.summary) bits.push(item.ai.summary); + $('ai-text').textContent = bits.join(' · '); + aiBar.className = 'visible'; + } else { + aiBar.className = ''; + } + /* Resolve image sources based on view + annotated toggle */ var ann = state.annotated; var baseImg = ann ? (item.base_diff || item.original || '') : (item.original || ''); diff --git a/lib/snap_diff/screenshot_assertion.rb b/lib/snap_diff/screenshot_assertion.rb index 2434fb7d..1b8cc21c 100644 --- a/lib/snap_diff/screenshot_assertion.rb +++ b/lib/snap_diff/screenshot_assertion.rb @@ -64,7 +64,19 @@ def validate return unless compare if compare.different? - "Screenshot does not match for '#{name}': #{compare.error_message}\n#{caller.join("\n")}" + # Optional AI gate (SnapDiff::Reporters::AISimple with fail_on:): + # the verdict is computed here, before the message is built, so + # the failure text can quote it and accepted verdicts can skip it. + ai = SnapDiff::AI.gated_result(name, compare.difference) if defined?(SnapDiff::AI) && SnapDiff::AI.gate + if ai && !SnapDiff::AI.fails?(ai[:verdict]) + $stdout.puts "[snap_diff:ai] #{name}: failure suppressed -- #{SnapDiff::AI.format(ai)}" + return nil + end + + message = "Screenshot does not match for '#{name}': #{compare.error_message}\n#{caller.join("\n")}" + ai ||= SnapDiff::AI[name] if defined?(SnapDiff::AI) + message += "\n AI triage: #{SnapDiff::AI.format(ai)}" if ai + message else archive_baseline! nil diff --git a/test/fixtures/ai_triage_case.rb b/test/fixtures/ai_triage_case.rb new file mode 100644 index 00000000..93a0afd3 --- /dev/null +++ b/test/fixtures/ai_triage_case.rb @@ -0,0 +1,57 @@ +# frozen_string_literal: true + +# A USER'S test file with AI triage enabled, run in its own process by +# test/integration/ai_triage_test.rb: real registry, real reporters, real +# git baselines, real Minitest. The only stubs are capture (a file copy) +# and the AI backend (a name-keyed fake -- no model download, no network). +# +# Env in: +# SNAP_ROOT -- a git repo whose `screenshots/` holds the COMMITTED baselines +# SNAP_IMAGES -- directory holding the fixture PNGs a.png / b.png +# SNAP_CASES -- comma-separated subset of verified,flaky,buggy (may be empty) +# SNAP_FAIL_ON -- when "1", the reporter gates with fail_on: %w[real_bug] +require "minitest/autorun" +require "snap_diff/integrations/minitest" +require "snap_diff/reporters/html" +require "snap_diff/reporters/ai_simple" +require "fileutils" +require "pathname" + +# No browser: rack-test is enough for the capture stub below. +Capybara.app = ->(_env) { [200, {"content-type" => "text/plain"}, ["ok"]] } + +IMAGES = Pathname(ENV.fetch("SNAP_IMAGES")) +CASES = ENV.fetch("SNAP_CASES", "").split(",") + +# verified: capture equals the baseline. flaky/buggy: capture differs. +class FileCopyScreenshoter < SnapDiff::Screenshoter + CAPTURES = {"verified" => "a.png", "flaky" => "b.png", "buggy" => "b.png"}.freeze + + def take_screenshot(screenshot_path) + name = File.basename(screenshot_path.to_s).sub(/\.attempt_\d+/, "") + FileUtils.mkdir_p(File.dirname(screenshot_path)) + FileUtils.cp(IMAGES / CAPTURES.fetch(File.basename(name, ".png")), screenshot_path) + end +end + +# The fake backend: similarity keyed by screenshot name -- flaky lands in +# the flaky band, buggy in the real_bug band. +FAKE_BACKEND = ->(name:, base:, current:, meta:) { + {similarity: {"flaky" => 0.999, "buggy" => 0.42}.fetch(name, 0.5)} +} + +SnapDiff.config.root = ENV.fetch("SNAP_ROOT") +SnapDiff.config.save_path = "screenshots" +SnapDiff.config.screenshoter = FileCopyScreenshoter +SnapDiff.config.fail_if_new = false + +options = (ENV["SNAP_FAIL_ON"] == "1") ? {fail_on: %w[real_bug]} : {} +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(backend: FAKE_BACKEND, **options)) + +class AiTriageCase < Minitest::Test + include SnapDiff::Minitest::Assertions + + CASES.each do |name| + define_method(:"test_#{name}") { assert_matches_screenshot(name) } + end +end diff --git a/test/integration/ai_triage_test.rb b/test/integration/ai_triage_test.rb new file mode 100644 index 00000000..6cd9d0b1 --- /dev/null +++ b/test/integration/ai_triage_test.rb @@ -0,0 +1,95 @@ +# frozen_string_literal: true + +require "test_helper" +require "open3" +require "tmpdir" + +# AI triage end-to-end: REAL runs of a user's test file +# (test/fixtures/ai_triage_case.rb) in fresh processes -- real registry, +# real HTML reporter, real git baselines -- with a name-keyed fake backend, +# so no model download or network is involved. In-process unit tests would +# not catch a reporter that never gets registered, a gate that runs too +# late, or a report rendered before the store is filled; the finished +# process's output and files do. +class AiTriageTest < ActiveSupport::TestCase + # Each case boots a full subprocess (git init, fresh ruby), the priciest + # tests in the suite, so they run only when the AI surface itself changed: + # the AI lib, its reporter, or these tests/fixtures. Everything else is + # already covered by the fast in-process unit tests, which always run. + # Force with RUN_AI_TESTS=1. Fails OPEN: when git can't tell (shallow + # checkout, no origin/master) or we're on master, the tests run. + AI_SURFACE = %r{\A(?: + lib/snap_diff/(?:ai\.rb|reporters/ai_simple\.rb) + | test/(?:unit/reporters/ai_simple_test\.rb|integration/ai_triage_test\.rb|fixtures/ai_triage_case\.rb) + )\z}x + + def self.ai_surface_changed? + return true if ENV["RUN_AI_TESTS"] == "1" + branch, = Open3.capture2e("git", "rev-parse", "--abbrev-ref", "HEAD") + return true if branch.strip == "master" + merge_base, = Open3.capture2e("git", "merge-base", "HEAD", "origin/master") + return true if merge_base.strip.empty? + changed, = Open3.capture2e("git", "diff", "--name-only", merge_base.strip, "HEAD") + changed.split("\n").any? { |path| path.match?(AI_SURFACE) } + end + + setup do + skip "AI surface unchanged (RUN_AI_TESTS=1 to force)" unless AiTriageTest.ai_surface_changed? + end + + test "advisory mode logs and badges verdicts, and the pixel diff still fails" do + out, status, report = run_case("verified,flaky,buggy") + + refute status.success?, out + assert_includes out, "[snap_diff:ai] flaky: FLAKY (custom, similarity=0.999)" + assert_includes out, "[snap_diff:ai] buggy: REAL_BUG (custom, similarity=0.42)" + assert_includes out, "[snap_diff:ai] 2 diff(s) analyzed: 1 real_bug, 1 flaky" + assert_includes report, '"verdict":"flaky"' + assert_includes report, '"verdict":"real_bug"' + end + + test "gate mode suppresses flaky diffs and keeps the suite green" do + out, status, = run_case("flaky", fail_on: true) + + assert status.success?, out + assert_includes out, "[snap_diff:ai] flaky: failure suppressed -- FLAKY (custom, similarity=0.999)" + end + + test "gate mode still fails real bugs and quotes the verdict in the failure message" do + out, status, = run_case("flaky,buggy", fail_on: true) + + refute status.success?, out + assert_includes out, "[snap_diff:ai] flaky: failure suppressed -- FLAKY (custom, similarity=0.999)" + assert_includes out, "Screenshot does not match for 'buggy'" + assert_includes out, "AI triage: REAL_BUG (custom, similarity=0.42)" + end + + private + + # Mirrors summary_line_test.rb: a throwaway git repo with COMMITTED + # baselines, then the user's test file against it in a fresh process. + # Returns [output, exit status, rendered HTML report (nil when absent)]. + def run_case(cases, fail_on: false) + Dir.mktmpdir do |dir| + # macOS hands out /var/... symlinks; git reports the physical path, and + # baseline lookup is a relative_path_from between the two. + repo = File.realpath(dir) + FileUtils.mkdir_p("#{repo}/screenshots") + %w[verified flaky buggy].each do |name| + FileUtils.cp(fixture_image_path_from("a"), "#{repo}/screenshots/#{name}.png") + end + git = ["git", "-C", repo, "-c", "user.email=t@example.com", "-c", "user.name=t"] + Open3.capture2e(*git, "init", "-q") + Open3.capture2e(*git, "add", "screenshots") + Open3.capture2e(*git, "commit", "-qm", "baselines") + + out, status = Open3.capture2e( + {"SNAP_ROOT" => repo, "SNAP_IMAGES" => TEST_IMAGES_DIR.to_s, "SNAP_CASES" => cases, "CI" => nil, + "SNAP_FAIL_ON" => (fail_on ? "1" : nil)}, + RbConfig.ruby, "-Ilib", "-Itest", file_fixture("ai_triage_case.rb").to_s + ) + report_path = "#{repo}/screenshots/snap_diff_report.html" + [out, status, (File.read(report_path) if File.exist?(report_path))] + end + end +end diff --git a/test/unit/reporters/ai_simple_test.rb b/test/unit/reporters/ai_simple_test.rb new file mode 100644 index 00000000..21d0a050 --- /dev/null +++ b/test/unit/reporters/ai_simple_test.rb @@ -0,0 +1,366 @@ +# frozen_string_literal: true + +require "test_helper" +require "tmpdir" + +require "snap_diff/reporters/ai_simple" +require "snap_diff/reporters/html" +require "snap_diff/screenshot_assertion" + +class AISimpleReporterTest < Minitest::Test + # Stub the exact surface the reporter touches; a real comparison would + # need vips, fixtures and a checked-out baseline for no extra coverage. + StubDifference = Struct.new(:different, keyword_init: true) do + def different? = different + def original_image_path = Pathname("/nonexistent/base.png") + def new_image_path = Pathname("/nonexistent/current.png") + def to_h = {area_size: 42, region: [0, 0, 10, 10]} + end + StubCompare = Struct.new(:difference) + StubAssertion = Struct.new(:name, :compare) + + # Minimal ScreenshotAssertion surface for gate tests. + GateCompare = Struct.new(:difference) do + def different? = difference.different? + def error_message = "diff details" + end + + def setup + SnapDiff::AI.clear_results! + end + + def teardown + SnapDiff::AI.gate = nil + end + + def gate_assertion(name, different: true) + SnapDiff::ScreenshotAssertion.new(name).tap do |a| + a.compare = GateCompare.new(StubDifference.new(different: different)) + a.caller = ["test.rb:1"] + end + end + + def build_assertion(name, different: true) + StubAssertion.new(name, StubCompare.new(StubDifference.new(different: different))) + end + + def similarity_backend(value) + ->(name:, base:, current:, meta:) { {similarity: value} } + end + + def build_reporter(backend) + SnapDiff::Reporters::AISimple.new(backend: backend) + end + + def test_records_only_different_assertions + build_reporter(similarity_backend(0.5)).record([ + build_assertion("changed"), + build_assertion("same", different: false), + StubAssertion.new("pending", StubCompare.new(nil)) + ]) + + assert_equal ["changed"], SnapDiff::AI.results.map { |r| r[:name] } + assert_equal "real_bug", SnapDiff::AI.results.first[:verdict] + end + + def test_backend_verdict_wins_over_thresholds + backend = ->(name:, base:, current:, meta:) { {similarity: 0.99, verdict: "real_bug", confidence: 0.91} } + build_reporter(backend).record([build_assertion("checkout")]) + + result = SnapDiff::AI["checkout"] + assert_equal "real_bug", result[:verdict] + assert_in_delta 0.91, result[:confidence] + end + + def test_silent_when_nothing_analyzed + reporter = build_reporter(similarity_backend(0.99)) + reporter.finalize + + assert_nil reporter.summary + end + + def test_summary_breaks_down_verdicts + reporter = build_reporter(similarity_backend(0.5)) + reporter.record([build_assertion("a"), build_assertion("b")]) + + assert_equal "[snap_diff:ai] 2 diff(s) analyzed: 2 real_bug", reporter.summary + end + + def test_dump_and_merge_state_round_trip_for_fork_parallel + build_reporter(similarity_backend(0.99)).record([build_assertion("homepage")]) + fragment = JSON.parse(JSON.generate(SnapDiff::AI.dump_state)) + SnapDiff::AI.clear_results! + + build_reporter(similarity_backend(0.5)).merge_state!(fragment) + + assert_equal "flaky", SnapDiff::AI["homepage"][:verdict] + end + + def test_backend_failure_skips_the_assertion_instead_of_raising + reporter = build_reporter(->(name:, base:, current:, meta:) { raise "model exploded" }) + + _out, err = capture_io { reporter.record([build_assertion("boom")]) } + + assert_includes err, "model exploded" + assert_empty SnapDiff::AI.results + end + + def test_rejects_a_backend_that_does_not_respond_to_call + assert_raises(ArgumentError) { build_reporter(Object.new) } + end + + def test_unknown_backend_symbol_names_registered_alternatives + error = assert_raises(ArgumentError) { build_reporter(:does_not_exist) } + + assert_includes error.message, ":clip" + end + + def test_registered_backend_resolves_by_name + SnapDiff::AI.register(:test_stub) { similarity_backend(0.99) } + build_reporter(:test_stub).record([build_assertion("homepage")]) + + assert_equal "flaky", SnapDiff::AI["homepage"][:verdict] + ensure + SnapDiff::AI.instance_variable_get(:@backends).delete(:test_stub) + end + + def test_factory_failure_degrades_to_disabled_but_config_errors_raise + SnapDiff::AI.register(:exploding) { raise "cannot reach the model server" } + + reporter = nil + _out, err = capture_io { reporter = build_reporter(:exploding) } + + assert_includes err, "triage disabled" + reporter.record([build_assertion("homepage")]) + assert_empty SnapDiff::AI.results + ensure + SnapDiff::AI.instance_variable_get(:@backends).delete(:exploding) + end + + def test_clip_backend_degrades_to_disabled_without_informers + begin + require "informers" + skip "informers is installed in this environment" + rescue LoadError + # expected: exercising the absence path + end + + reporter = nil + _out, err = capture_io { reporter = build_reporter(:clip) } + + assert_includes err, "triage disabled" + reporter.record([build_assertion("homepage")]) + assert_empty SnapDiff::AI.results + end + + def test_custom_thresholds + SnapDiff::Reporters::AISimple.new( + backend: similarity_backend(0.95), flaky: 0.90, intentional: 0.80 + ).record([build_assertion("homepage")]) + + assert_equal "flaky", SnapDiff::AI["homepage"][:verdict] + end + + def test_gate_suppresses_verdicts_not_in_fail_on + SnapDiff::Reporters::AISimple.new(backend: similarity_backend(0.99), fail_on: %w[real_bug]) + + assert_nil gate_assertion("homepage").validate + assert_equal "flaky", SnapDiff::AI["homepage"][:verdict] + end + + def test_gate_still_fails_real_bugs_and_quotes_ai_in_the_message + SnapDiff::Reporters::AISimple.new(backend: similarity_backend(0.50), fail_on: %w[real_bug]) + + message = gate_assertion("checkout").validate + + assert_includes message, "Screenshot does not match for 'checkout'" + assert_includes message, "AI triage: REAL_BUG (custom, similarity=0.5)" + end + + def test_gate_always_fails_unknown_verdicts + SnapDiff::Reporters::AISimple.new(backend: similarity_backend(nil), fail_on: %w[real_bug]) + + message = gate_assertion("checkout").validate + + assert_includes message, "AI triage: UNKNOWN" + end + + def test_gate_analysis_is_memoized_for_the_reporter_pass + calls = 0 + counting_backend = ->(name:, base:, current:, meta:) { + calls += 1 + {similarity: 0.99} + } + reporter = SnapDiff::Reporters::AISimple.new(backend: counting_backend, fail_on: %w[real_bug]) + + assertion = gate_assertion("homepage") + assertion.validate + reporter.record([assertion]) + + assert_equal 1, calls + end + + def test_no_gate_means_pure_advisory + SnapDiff::Reporters::AISimple.new(backend: similarity_backend(0.99)) + + message = gate_assertion("homepage").validate + + assert_includes message, "Screenshot does not match" + refute_includes message, "AI triage:" + end + + def test_gate_reanalyzes_when_the_same_name_is_compared_again + similarities = [0.99, 0.50] + backend = ->(name:, base:, current:, meta:) { {similarity: similarities.shift} } + SnapDiff::Reporters::AISimple.new(backend: backend, fail_on: %w[real_bug]) + + # First comparison classifies flaky and is suppressed... + assert_nil gate_assertion("homepage").validate + + # ...but a FRESH comparison under the same name must be re-analyzed: + # a stale "flaky" may never suppress a new regression. + message = gate_assertion("homepage").validate + assert_includes message, "AI triage: REAL_BUG" + assert_equal "real_bug", SnapDiff::AI["homepage"][:verdict] + end + + def test_unrecognized_backend_verdict_becomes_unknown + backend = ->(name:, base:, current:, meta:) { {verdict: "uncertain"} } + build_reporter(backend).record([build_assertion("checkout")]) + + assert_equal "unknown", SnapDiff::AI["checkout"][:verdict] + end + + def test_gate_fails_on_unrecognized_verdicts + backend = ->(name:, base:, current:, meta:) { {verdict: "uncertain"} } + SnapDiff::Reporters::AISimple.new(backend: backend, fail_on: %w[real_bug]) + + assert_includes gate_assertion("checkout").validate, "AI triage: UNKNOWN" + end + + def test_memo_is_scoped_to_inputs_not_just_the_difference_object + calls = 0 + backend = ->(name:, base:, current:, meta:) { + calls += 1 + {similarity: 0.99} + } + SnapDiff::Reporters::AISimple.new(backend: backend, fail_on: %w[real_bug]) + + # Two assertion names SHARING one difference object must both analyze. + shared = StubDifference.new(different: true) + %w[one two].each do |name| + SnapDiff::ScreenshotAssertion.new(name).tap do |a| + a.compare = GateCompare.new(shared) + a.caller = [] + end.validate + end + + assert_equal 2, calls + end +end + +class AIVerdictTest < Minitest::Test + def test_bands + assert_equal "flaky", SnapDiff::AI.verdict(0.985) + assert_equal "intentional", SnapDiff::AI.verdict(0.90) + assert_equal "real_bug", SnapDiff::AI.verdict(0.50) + assert_equal "unknown", SnapDiff::AI.verdict(nil) + end + + def test_non_finite_and_non_numeric_similarities_are_unknown + assert_equal "unknown", SnapDiff::AI.verdict(Float::INFINITY) + assert_equal "unknown", SnapDiff::AI.verdict(Float::NAN) + assert_equal "unknown", SnapDiff::AI.verdict("high") + end +end + +# The mix: AISimple writes the shared store, HTML annotates from it. +class AIHtmlReporterMixTest < Minitest::Test + HtmlDifference = Struct.new(:ratio, keyword_init: true) do + def different? = true + def region_area_size = 42 + def meta = {max_color_distance: 3.21} + end + class HtmlReporterStub + def annotated_base_image_path = nil + def annotated_image_path = nil + def heatmap_diff_path = nil + end + HtmlCompare = Struct.new(:difference) do + def base_image_path = Pathname("/nonexistent/base.png") + def image_path = Pathname("/nonexistent/current.png") + def reporter = HtmlReporterStub.new + end + HtmlAssertion = Struct.new(:name, :compare) + + def setup + SnapDiff::AI.clear_results! + end + + def html_reporter(dir) + SnapDiff::Reporters::HTML.new(output_path: File.join(dir, "report.html")) + end + + def failed_assertion(name) + HtmlAssertion.new(name, HtmlCompare.new(HtmlDifference.new(ratio: 0.02))) + end + + def test_failures_carry_ai_annotation_after_render + SnapDiff::AI.record_result( + name: "checkout", verdict: "real_bug", backend: "clip", + similarity: 0.7312, summary: "CTA clipped" + ) + + Dir.mktmpdir do |dir| + reporter = html_reporter(dir) + reporter.record([failed_assertion("checkout"), failed_assertion("plain")]) + reporter.finalize + + checkout, plain = reporter.failures + assert_equal "real_bug", checkout[:ai][:verdict] + assert_equal "CTA clipped", checkout[:ai][:summary] + refute plain.key?(:ai) + end + end + + def test_ai_result_recorded_after_html_record_still_renders + # HTML auto-registers before AISimple, so its record runs first; the + # annotation must attach at render regardless of reporter order. + Dir.mktmpdir do |dir| + reporter = html_reporter(dir) + reporter.record([failed_assertion("checkout")]) + SnapDiff::AI.record_result(name: "checkout", verdict: "flaky", backend: "clip", similarity: 0.9912) + reporter.finalize + + assert_equal "flaky", reporter.failures.first[:ai][:verdict] + end + end + + def test_rendered_report_includes_ai_bar_markup + SnapDiff::AI.record_result(name: "checkout", verdict: "flaky", backend: "clip", similarity: 0.9912) + + Dir.mktmpdir do |dir| + reporter = html_reporter(dir) + reporter.record([failed_assertion("checkout")]) + reporter.finalize + + html = File.read(File.join(dir, "report.html")) + assert_includes html, "ai-bar" + assert_includes html, '"verdict":"flaky"' + end + end + + def test_rendered_report_without_ai_stays_clean + # AI not enabled (store empty): entries must not gain an :ai key, and + # the serialized DATA must contain no ai annotations at all. + Dir.mktmpdir do |dir| + reporter = html_reporter(dir) + reporter.record([failed_assertion("checkout"), failed_assertion("plain")]) + reporter.finalize + + reporter.failures.each { |entry| refute entry.key?(:ai) } + html = File.read(File.join(dir, "report.html")) + refute_includes html, '"ai":' + end + end +end