diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index d397977c..6dc565a0 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -50,7 +50,11 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 5 steps: + # Full history: AI integration tests skip themselves unless the AI + # surface changed vs origin/master (no merge-base in a shallow clone). - uses: actions/checkout@v7 + with: + fetch-depth: 0 - uses: ./.github/actions/setup-ruby-and-dependencies with: @@ -71,6 +75,10 @@ jobs: steps: - name: Checkout code uses: actions/checkout@v7 + # Full history: AI integration tests skip themselves unless the AI + # surface changed vs origin/master (no merge-base in a shallow clone). + with: + fetch-depth: 0 - uses: ./.github/actions/setup-ruby-and-dependencies with: @@ -107,10 +115,9 @@ jobs: contains(github.event.pull_request.labels.*.name, 'full-ci') needs: [ functional-test ] runs-on: ubuntu-latest - # Must fit `max_attempts * timeout_minutes` below, plus ~1 min of setup, - # or the last attempt gets killed mid-run and the cell reports `cancelled` - # -- a dead gate. JRuby: 1 + 20 + 20 = 41. MRI: 1 + 4 + 4 = 9. - timeout-minutes: ${{ contains(matrix.ruby-version, 'jruby') && 41 || 9 }} + # ~1 min of setup plus one 4-min attempt; re-measure if the suite grows + # into it -- a killed attempt reports `cancelled`, a dead gate. + timeout-minutes: 6 continue-on-error: ${{ matrix.experimental }} strategy: matrix: @@ -131,17 +138,47 @@ jobs: - ruby-version: "4.0" gemfile: edge_gems.rb experimental: true - # One cell: JRuby's differences (no Kernel#fork, thread semantics, - # the FFI/vips path) are JVM-level, not Rails-level. - - ruby-version: jruby-10.1 - gemfile: rails81_gems.rb - experimental: false env: BUNDLE_GEMFILE: gemfiles/${{ matrix.gemfile }} # JRuby is the only place ruby-vips' FFI path runs on a non-MRI engine. # Test Drivers covers both backends on CRuby. - SCREENSHOT_DRIVER: ${{ contains(matrix.ruby-version, 'jruby') && 'vips' || 'chunky_png' }} + SCREENSHOT_DRIVER: chunky_png + + steps: + - uses: actions/checkout@v7 + + - uses: ./.github/actions/setup-ruby-and-dependencies + with: + ruby-version: ${{ matrix.ruby-version }} + ruby-cache-version: ${{ matrix.ruby-version }}-${{ matrix.gemfile }}-1 + cache-apt-packages: true + + # Measured, not chosen: slowest MRI attempt is ~146s. The suite has + # timed out before when it grew into this -- re-measure then. + - run: bin/rake test + timeout-minutes: 4 + + matrix-jruby: + name: Test JRuby + # The single most expensive cell (~12 min x up to 2 attempts): JRuby's + # differences (no Kernel#fork, thread semantics, FFI/vips) are JVM-level, + # so the weekly drift check covers them. Not on every master push. + if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' + needs: [ functional-test ] + runs-on: ubuntu-latest + # 1 min setup + 20 + 20: the retry below exists only for the + # intermittent JRuby teardown hang (#244). + timeout-minutes: 41 + strategy: + matrix: + ruby-version: [ jruby-10.1 ] + gemfile: [ rails81_gems.rb ] + + env: + BUNDLE_GEMFILE: gemfiles/${{ matrix.gemfile }} + # The one place ruby-vips' FFI path runs on a non-MRI engine. + SCREENSHOT_DRIVER: vips steps: - uses: actions/checkout@v7 @@ -152,15 +189,10 @@ jobs: ruby-cache-version: ${{ matrix.ruby-version }}-${{ matrix.gemfile }}-1 cache-apt-packages: true - - name: Run tests (with 1 retry) + - name: Run tests (with 1 retry for the #244 teardown hang) uses: nick-fields/retry@v4 with: - # Measured, not chosen: slowest attempt is ~880s JRuby, ~146s MRI. - # Both have timed out before when the suite grew into them -- when - # that happens, re-measure and raise this AND timeout-minutes above. - timeout_minutes: ${{ contains(matrix.ruby-version, 'jruby') && 20 || 4 }} - # Exists only for the intermittent JRuby teardown hang (#244). - # Once that is fixed, drop to 1 and halve the JRuby bill. + timeout_minutes: 20 max_attempts: 2 command: bin/rake test @@ -213,8 +245,9 @@ jobs: # gemfile, so a rule listing them stops requiring anything the moment the # matrix changes shape. `skipped` passes -- the matrix is off on PRs. if: always() - needs: [ test-minimal-setup, functional-test, matrix, matrix-screenshot-driver ] + needs: [ test-minimal-setup, functional-test, matrix, matrix-jruby, matrix-screenshot-driver ] runs-on: ubuntu-latest + timeout-minutes: 2 steps: - if: contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled') run: exit 1 diff --git a/CHANGELOG.md b/CHANGELOG.md index 9cf283f0..531e8003 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -18,6 +18,19 @@ green while comparing nothing**, and all four are fixed here — that is the rea The sections below the divider are the prerelease notes (alpha1 → beta3) and are kept as history. This entry is the one to read if you are coming from **1.15.1**. +### Added + +- **AI triage (optional, offline-first, advisory by default).** `SnapDiff::Reporters::AISimple` + classifies every failed comparison as `flaky` / `intentional` / `real_bug`, + logs one line per diff, quotes the verdict in the failure message, and + annotates the HTML report with a verdict badge and summary — all via one + shared in-memory store (`SnapDiff::AI`), no files written. With + `fail_on: %w[real_bug]` the verdict gates pass/fail: + flaky/intentional diffs stay green, `unknown` always fails. Default backend + is offline CLIP via the `informers` gem; custom backends (local VLM, + decision APIs) register by name via `SnapDiff::AI.register`. Zero new hard + dependencies. See `docs/ai.md`. + ### Upgrading from 1.15.1: change the version, run your suite ```ruby diff --git a/README.md b/README.md index fbf6c856..c6889602 100644 --- a/README.md +++ b/README.md @@ -237,6 +237,28 @@ After tests run, open `doc/screenshots/snap_diff_report.html`: See [Web UI & Custom Reporters](docs/reporters.md) for full feature details and [CI Integration](docs/ci-integration.md) for GitHub Actions setup. +## AI-Assisted Triage (optional) + +AI tells you *which* failures to look at first. One line enables offline +CLIP triage — every failed comparison is classified as `flaky` / +`intentional` / `real_bug`, logged per diff, quoted in the failure message, +and badged in the HTML report (all via one shared in-memory store — no +files). Advisory by +default; pass `fail_on: %w[real_bug]` to keep the suite green on flaky and +intentional diffs and fail only on what AI calls a real bug: + +```ruby +# Gemfile +gem "informers" # offline CLIP via ONNX, no API key + +# test/test_helper.rb +require "snap_diff/reporters/ai_simple" +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new) +``` + +Plug in your own backend (a local VLM, a decision API) with +`SnapDiff::AI.register(:name) { ... }` — see [AI triage](docs/ai.md). + ## Compare Any Two Images Works without a browser — PDFs, generated images, CI artifacts: @@ -328,6 +350,7 @@ instead of failing quietly — useful when `snap_diff_report.html` is missing en - [Image Processing Drivers](docs/drivers.md) — VIPS, ChunkyPNG, perceptual threshold - [Screenshot Organization](docs/organization.md) — groups, sections, cropping, multi-browser - [Web UI & Custom Reporters](docs/reporters.md) — interactive report, custom reporters +- [AI-Assisted Triage](docs/ai.md) — optional flaky/intentional/real_bug classification, pluggable backends ## Development diff --git a/docs/ai.md b/docs/ai.md new file mode 100644 index 00000000..a468877d --- /dev/null +++ b/docs/ai.md @@ -0,0 +1,234 @@ +# AI-assisted diff triage + +> Optional and offline-first. Advisory by default — the pixel comparison +> stays the verdict and AI classifies failures so you know which reds to +> look at first. Opt into `fail_on:` to let the AI verdict gate failures. + +`SnapDiff::Reporters::AISimple` analyzes every comparison that **already +failed** and labels it: + +| Verdict | Meaning | Typical cause | +| --- | --- | --- | +| `flaky` | Pixels differ, semantics identical | Anti-aliasing, timestamps, avatars | +| `intentional` | Semantics shifted within a familiar shape | Redesign, copy change | +| `real_bug` | Semantics moved substantially | Clipped text, overlap, missing element | +| `unknown` | Backend could not score | Missing/unreadable image | + +One line per diff in the test output, and — automatically, via the +shared store — a verdict badge plus one-line summary per failure in +`snap_diff_report.html`. No files, no duplicate storage. + +## How the HTML report gets AI annotations + +Both reporters read the same shared store: `AISimple` writes each result +into `SnapDiff::AI` (keyed by screenshot name), and the HTML reporter +attaches `SnapDiff::AI[name]` to the failure entry — behind a `defined?` +guard, so `html.rb` never requires the AI module and nothing changes +when AI triage isn't loaded. No wiring between the two reporters: + +```ruby +require "snap_diff/reporters/html" # already auto-registers +require "snap_diff/reporters/ai_simple" +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new) +``` + +The report then shows, per failure: an `AI: real bug` / `AI: flaky` badge +in the sidebar and top strip, plus `similarity · confidence · summary` +when the backend provides them. Under fork-parallel, results merge in the +parent before the report renders. + +``` +[snap_diff:ai] homepage: FLAKY (clip, similarity=0.9912) + pixels differ but semantics match -- candidate for skip_area or a tolerance bump +[snap_diff:ai] checkout: REAL_BUG (clip, similarity=0.7312) +[snap_diff:ai] 2 diff(s) analyzed: 1 real_bug, 1 flaky +``` + +## Architecture + +``` +lib/snap_diff/ai.rb # backend registry, verdict thresholds, shared result store +lib/snap_diff/ai/backends/clip.rb # built-in offline backend (informers) +lib/snap_diff/reporters/ai_simple.rb # record/finalize/summary; writes the store +lib/snap_diff/reporters/html.rb # annotates failures from the store (defined? guard) +lib/snap_diff/screenshot_assertion.rb# validate: fail-gate + AI line in the failure message +``` + +A backend is any object with `#call(name:, base:, current:, meta:) -> Hash`. +Extend without editing existing files: + +```ruby +SnapDiff::AI.register(:jev) { JevBackend.new } # optional requires go in the block +``` + +The returned hash may carry `:verdict` (used as-is) or `:similarity` +(classified by the shared thresholds); any other keys (`:confidence`, +`:summary`, `:auto_accept`) pass through to the report untouched. + +## Setup + +Default backend: CLIP via [`informers`](https://github.com/ankane/informers) +— ONNX Runtime, fully offline, ~90 MB quantized model, ~40 ms/image. + +```ruby +# Gemfile +gem "informers" # pulls onnxruntime itself + +# test/test_helper.rb +require "snap_diff/reporters/ai_simple" +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new) +``` + +Without `informers`, one warning at registration, then silence — never a +failed build. + +**Prefetch the model.** The ~90 MB download happens on the first analyzed +diff, so a fully offline run needs a prefilled cache. Warm it at suite +setup (or as a cached CI step): + +```ruby +SnapDiff::AI::Clip.new.prefetch! +``` + +Custom thresholds: + +```ruby +SnapDiff::Reporters::AISimple.new(flaky: 0.99, intentional: 0.85) +``` + +## Gating failures on AI verdicts + +Advisory mode never touches pass/fail. Add `fail_on:` and the verdict +decides: diffs classified as anything not listed are **suppressed** — the +test stays green, the result is still logged, stored, and shown in the +HTML report. `unknown` always fails: AI can downgrade a diff, never vouch +for one it could not classify. + +```ruby +# Fail only on what AI calls a real bug; flaky/intentional stay green. +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(fail_on: %w[real_bug])) +``` + +Failures that survive the gate quote the verdict inline, no clicks needed: + +``` +Screenshot does not match for 'checkout': the change spans ... + AI triage: REAL_BUG (clip, similarity=0.7312) -- CTA clipped +``` + +Suppressed diffs log one line instead: + +``` +[snap_diff:ai] homepage: failure suppressed -- FLAKY (clip, similarity=0.9912) +``` + +## Backend recipes + +### Typed decisions (TypeSafe Jev) + +Jev answers typed questions (Choice / Score / Noul) over a state — text and +structured data, not images — so we send the diff *metrics*, not the pixels. +One call returns a verdict CI can branch on. Requires the community +[`typesafe-sdk`](https://rubygems.org/gems/typesafe-sdk) gem and +`TYPESAFE_API_KEY`: + +```ruby +class JevBackend + def name = "jev" + + def initialize + require "typesafe/sdk" + @client = Typesafe::SDK::Client.new(api_key: ENV.fetch("TYPESAFE_API_KEY")) + end + + def call(name:, base:, current:, meta: {}) + resp = @client.system_one( + state: {screenshot: name, changed_area_px: meta[:area_size], changed_region: meta[:region]}, + questions: { + verdict: Typesafe::SDK::Choice.new( + instructions: "Classify this visual diff", + criteria: { + flaky: "timestamp, anti-aliasing, or avatar noise", + intentional: "deliberate redesign or copy change", + real_bug: "clipped, overlapping, or missing UI" + } + ), + auto_accept: Typesafe::SDK::Noul.new(instructions: "Safe to accept the new rendering as the baseline") + } + ) + answers = resp.answers + {verdict: answers["verdict"].choice, + confidence: answers["verdict"].confidence.round(2), + auto_accept: answers["auto_accept"].noul.round(3)} + end +end + +SnapDiff::AI.register(:jev) { JevBackend.new } +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(backend: :jev)) +``` + +### Local VLM explanations (Qwen2.5-VL via Ollama) + +```ruby +class QwenBackend + def name = "qwen" + + def initialize + require "ollama-ai" + require "base64" + @ollama = Ollama.new + end + + def call(name:, base:, current:, meta: {}) + res = @ollama.chat(model: "qwen2.5vl:3b", messages: [{ + role: "user", + content: "You are visual QA. Compare baseline vs current. " \ + 'Return JSON {"summary": one sentence, "verdict": "real_bug|intentional|flaky"}.', + images: [base, current].map { |p| Base64.strict_encode64(File.binread(p)) } + }]) + content = res.dig("message", "content").to_s + JSON.parse(content, symbolize_names: true).slice(:verdict, :summary) + rescue JSON::ParserError + {summary: content} + end +end +``` + +### Hybrid: CLIP first, Qwen only where it pays + +```ruby +class ClipThenQwen + def name = "clip+qwen" + + def initialize + @clip = SnapDiff::AI::Clip.new + @qwen = QwenBackend.new + end + + def call(name:, base:, current:, meta: {}) + result = @clip.call(name: name, base: base, current: current, meta: meta) + sim = result[:similarity] + return result if sim.nil? || sim >= 0.985 || sim < 0.90 + + @qwen.call(name: name, base: base, current: current, meta: meta).merge(result) + end +end +``` + +## CI gating + +Use the built-in gate — no JSON parsing, no extra script: + +```ruby +SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(fail_on: %w[real_bug])) +``` + +The suite then fails only on verdicts you list; see +[Gating failures on AI verdicts](#gating-failures-on-ai-verdicts). + +## Guarantees + +- **Advisory by default**: pass/fail changes only with an explicit `fail_on:` opt-in. +- **Thread-safe**: results behind one mutex; CLIP inference serialized. +- **Fork-parallel**: results are plain JSON-able hashes; workers merge via `dump_state`/`merge_state!`. +- **No hard deps**: `informers`, `typesafe`, `ollama-ai` all optional; gemspec unchanged. diff --git a/docs/reporters.md b/docs/reporters.md index 9154ce76..6b8b0cdb 100644 --- a/docs/reporters.md +++ b/docs/reporters.md @@ -127,4 +127,12 @@ still run. Full details in [Custom reporters](snapdiff.md#custom-reporters). +## AI triage reporter + +`SnapDiff::Reporters::AISimple` is an optional reporter: it classifies each +failed comparison as `flaky` / `intentional` / `real_bug` (offline CLIP by +default, pluggable backends) and its verdicts show up in this HTML report as +badges. Advisory by default — it never changes pass/fail unless you opt in +with `fail_on: %w[real_bug]`. Setup and recipes: [AI triage](ai.md). + [← Back to README](../README.md) diff --git a/lib/snap_diff/ai.rb b/lib/snap_diff/ai.rb new file mode 100644 index 00000000..4eb15698 --- /dev/null +++ b/lib/snap_diff/ai.rb @@ -0,0 +1,115 @@ +# frozen_string_literal: true + +# Optional AI triage. A backend is any object responding to +# #call(name:, base:, current:, meta:) -> Hash; register one by name or +# pass an instance. Builders run lazily, so optional gems load only when +# their backend is used. Advisory by default: pixel diff stays the verdict +# unless a reporter is configured with fail_on: (see AISimple). +# +# Results live in a process-wide store so ANY consumer can read them -- +# the AISimple reporter writes, the HTML reporter annotates from it, the +# fail-gate consults it. Keyed by screenshot name; later writes win. +module SnapDiff + module AI + # The only verdicts a backend may return; anything else maps to + # "unknown", which the fail-gate never suppresses. + VERDICTS = %w[real_bug intentional flaky unknown].freeze + + @backends = {} + @results = {} + @mutex = Mutex.new + + class << self + # Optional fail-gate, set by AISimple when configured with fail_on:. + # ScreenshotAssertion#validate consults it on a pixel diff: verdicts + # the gate accepts suppress the failure, the rest still fail. + attr_accessor :gate + + def gated_result(name, difference) = gate&.gated_result(name, difference) + + # No gate -> everything fails, exactly as without AI. + def fails?(verdict) = gate ? gate.fails?(verdict) : true + + def register(name, &build) + @mutex.synchronize { @backends[name.to_sym] = build } + end + + def names + @mutex.synchronize { @backends.keys } + end + + # nil/:default -> :clip; a Symbol -> registered builder; anything + # else must be a backend instance responding to #call. + def resolve(input = nil) + return build((input.nil? || input == :default) ? :clip : input) if input.nil? || input.is_a?(Symbol) + + unless input.respond_to?(:call) + raise ArgumentError, "AI backend must respond to #call(name:, base:, current:, meta:), got #{input.inspect}" + end + input + end + + def build(name) + builder = @mutex.synchronize { @backends[name.to_sym] } or + raise ArgumentError, "unknown AI backend #{name.inspect} (registered: #{names.map(&:inspect).join(", ")})" + builder.call + end + + # One-line rendering of a result, shared by the AISimple log and the + # assertion failure message: "REAL_BUG (clip, similarity=0.73) -- CTA clipped". + def format(result) + line = "#{result[:verdict].upcase} (#{result[:backend]}" + line += ", similarity=#{result[:similarity]}" if result[:similarity] + line += ", confidence=#{result[:confidence]}" if result[:confidence] + line += ")" + line += " -- #{result[:summary]}" if result[:summary] + line + end + + # similarity -> verdict. The one place thresholds live. + # Anything but a finite number is "unknown" -- NaN/Infinity would + # otherwise classify as flaky and let the gate suppress a real diff. + def verdict(similarity, flaky: 0.985, intentional: 0.90) + return "unknown" unless similarity.is_a?(Numeric) && similarity.finite? + + if similarity >= flaky + "flaky" + elsif similarity >= intentional + "intentional" + else + "real_bug" + end + end + + # --- shared result store ------------------------------------------ + + def record_result(result) + @mutex.synchronize { @results[result[:name]] = result } + end + + def [](name) + @mutex.synchronize { @results[name] } + end + + def results + @mutex.synchronize { @results.values } + end + + # Test isolation hook, same role as Reporting.reset_run_totals!. + def clear_results! + @mutex.synchronize { @results.clear } + end + + # Fork-parallel: plain hashes round-trip via JSON fragments. + def dump_state + {"results" => results} + end + + def merge_state!(state) + Array(state.is_a?(Hash) && state["results"]).each { |r| record_result(r.transform_keys(&:to_sym)) } + end + end + end +end + +require "snap_diff/ai/backends/clip" diff --git a/lib/snap_diff/ai/backends/clip.rb b/lib/snap_diff/ai/backends/clip.rb new file mode 100644 index 00000000..ede99bc2 --- /dev/null +++ b/lib/snap_diff/ai/backends/clip.rb @@ -0,0 +1,48 @@ +# frozen_string_literal: true + +module SnapDiff + module AI + # Offline default: CLIP embeddings via the +informers+ gem (ONNX, + # ~90 MB quantized, ~40 ms/image, no network). Returns only + # :similarity; verdicts come from AI.verdict. + class Clip + def name = "clip" + + def initialize(model: "Xenova/clip-vit-base-patch32", pipeline: nil) + require "informers" unless pipeline + @model = model + @pipeline = pipeline + @mutex = Mutex.new # ONNX session thread safety is not guaranteed + end + + def call(name:, base:, current:, meta: {}) + {similarity: cosine(base, current)&.round(4)} + end + + # Downloads and loads the model NOW. Run at suite setup (or a CI + # cache step) so the first diff is analyzed offline: without a + # prefetched model, the first diff pulls ~90 MB over the network. + def prefetch! + pipeline + end + + private + + def pipeline + # Lazy: the one-time model download lands on the first diff, + # not at reporter registration. Memoized per model. + @pipeline ||= Informers.pipeline("image-feature-extraction", @model, quantized: true) + end + + def cosine(path_a, path_b) + return unless path_a && path_b && File.exist?(path_a) && File.exist?(path_b) + + a, b = @mutex.synchronize { [pipeline.call(path_a).first, pipeline.call(path_b).first] } + norm = Math.sqrt(a.sum { |x| x * x }) * Math.sqrt(b.sum { |x| x * x }) + a.zip(b).sum { |x, y| x * y } / norm if norm.positive? + end + end + + register(:clip) { Clip.new } + end +end diff --git a/lib/snap_diff/reporters/ai_simple.rb b/lib/snap_diff/reporters/ai_simple.rb new file mode 100644 index 00000000..a7895305 --- /dev/null +++ b/lib/snap_diff/reporters/ai_simple.rb @@ -0,0 +1,134 @@ +# frozen_string_literal: true + +require "snap_diff/ai" + +module SnapDiff + module Reporters + # Advisory AI triage: classifies every FAILED comparison as + # flaky/intentional/real_bug and logs one line per diff. By default + # never changes pass/fail; with fail_on: the verdict gates it instead. + # + # SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new) # offline CLIP + # SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(backend: :jev)) + # SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(backend: my_backend)) + # SnapDiff::Reporting.register(SnapDiff::Reporters::AISimple.new(fail_on: %w[real_bug])) + # + # Results live only in the shared SnapDiff::AI store -- the HTML + # reporter annotates from it, gate reads it, no files, no wiring. + # Extension: SnapDiff::AI.register(:name) { backend } -- no edits here. + class AISimple + # fail_on: verdicts that still fail the test; every other verdict + # suppresses the pixel diff. "unknown" ALWAYS fails -- AI can + # downgrade a diff, never vouch for one it could not classify. + # nil thresholds defer to AI.verdict's defaults -- one source of truth. + def initialize(backend: nil, flaky: nil, intentional: nil, fail_on: nil) + @thresholds = {flaky: flaky, intentional: intentional}.compact + @backend = resolve(backend) + @fail_on = Array(fail_on).map(&:to_s) if fail_on + # Memoized per COMPARISON AND INPUTS: the gate at validate-time + # and the reporter pass at teardown see the same difference with + # the same name and metrics; a reassigned compare, a mutated + # result, or a later test asserting the same name all change the + # key and re-analyze -- a stale "flaky" must never suppress a + # fresh regression. + @memo = {} + @memo_mutex = Mutex.new + AI.gate = self if @fail_on + end + + def record(assertions) + return unless @backend + + assertions.each do |a| + difference = a.compare&.difference + next unless difference&.different? + + analyze_once(a.name, difference) + end + end + + # Fail-gate entry point, called by ScreenshotAssertion#validate on a + # pixel diff. Analysis runs there (before the error message is + # built) and is memoized, so #record never re-analyzes. + def gated_result(name, difference) + return unless @fail_on && @backend + + analyze_once(name, difference) + end + + def fails?(verdict) = verdict == "unknown" || @fail_on.include?(verdict) + + # Results are already in the shared store -- nothing to write out. + def finalize = nil + + def summary + results = AI.results + return if results.empty? + + counts = results.group_by { |r| r[:verdict] }.transform_values(&:size) + breakdown = AI::VERDICTS.filter_map { |v| "#{counts[v]} #{v}" if counts[v] } + "[snap_diff:ai] #{results.size} diff(s) analyzed: #{breakdown.join(", ")}" + end + + # Fork-parallel (Rails parallelize): the shared store round-trips. + def dump_state = AI.dump_state + def merge_state!(state) = AI.merge_state!(state) + + private + + # Unknown names and invalid objects are config errors and raise; + # a factory that fails to build (missing gem, init error) degrades + # to a warning and disabled triage. + def resolve(backend) + AI.resolve(backend) + rescue ArgumentError + raise + rescue LoadError => e + warn "[snap_diff:ai] backend unavailable (#{e.message}) -- AI triage disabled." + nil + rescue => e + warn "[snap_diff:ai] backend factory failed (#{e.class}: #{e.message}) -- AI triage disabled." + nil + end + + def analyze_once(name, difference) + key = [difference.object_id, name, difference.to_h] + result = @memo_mutex.synchronize do + @memo.fetch(key) { @memo[key] = analyze(name, difference) } + end + AI.record_result(result) if result + result + end + + def analyze(name, difference) + raw = @backend.call( + name: name, + base: difference.original_image_path&.to_s, + current: difference.new_image_path&.to_s, + meta: difference.to_h + ).transform_keys(&:to_sym) + + # A backend's own :verdict outranks the shared thresholds, but + # only known verdicts are honored -- anything else becomes + # "unknown", which the fail-gate never suppresses. + raw[:verdict] = AI.verdict(raw[:similarity], **@thresholds) if raw[:verdict].nil? + raw[:verdict] = "unknown" unless AI::VERDICTS.include?(raw[:verdict]) + result = {name: name, backend: backend_name}.merge(raw.except(:name, :backend)) + log(result) + result + rescue => e + warn "[snap_diff:ai] Backend failed for #{name.inspect} (#{e.class}: #{e.message})" + nil + end + + def backend_name + @backend.respond_to?(:name) ? @backend.name : "custom" + end + + def log(r) + $stdout.puts "[snap_diff:ai] #{r[:name]}: #{AI.format(r)}" + $stdout.puts " pixels differ but semantics match -- candidate for skip_area or a tolerance bump" if r[:verdict] == "flaky" + end + end + end +end diff --git a/lib/snap_diff/reporters/html.rb b/lib/snap_diff/reporters/html.rb index 9f79a2f4..b0429e81 100644 --- a/lib/snap_diff/reporters/html.rb +++ b/lib/snap_diff/reporters/html.rb @@ -97,6 +97,7 @@ def summary end def render + attach_ai_annotations ERB.new(File.read(self.class.template_path)).result(binding) end @@ -123,7 +124,23 @@ def failure_entry_for(name, compare) diff_level: difference.ratio && (difference.ratio * 100).round(2), area_size: difference.region_area_size, max_color_distance: difference.meta[:max_color_distance]&.round(1) - } + }.compact + end + + # Advisory AI triage annotations, attached at RENDER time: HTML + # records before the AI reporter (auto-registration runs first) and + # fork-parallel merges land after record, so only the final render + # can see every verdict. HTML never requires the AI module -- the + # annotation appears iff the user opted into AI triage. + def attach_ai_annotations + return unless defined?(SnapDiff::AI) + + failures.each do |entry| + # || would create a nil :ai key on misses; keep the entry clean. + if (annotation = SnapDiff::AI[entry[:name]]) + entry[:ai] ||= annotation + end + end end def resolve_image(path) diff --git a/lib/snap_diff/reporters/templates/report.html.erb b/lib/snap_diff/reporters/templates/report.html.erb index 8c756e5a..e4ad5993 100644 --- a/lib/snap_diff/reporters/templates/report.html.erb +++ b/lib/snap_diff/reporters/templates/report.html.erb @@ -81,6 +81,16 @@ .thumb-badge-fail { background: var(--red-dim); color: var(--red); } .thumb-badge-pass { background: var(--green-dim); color: var(--green); } + /* ── AI triage ─────────────────────────────────────────────────── */ + .ai-badge { font-size: .625rem; font-family: var(--font-mono); font-weight: 600; padding: .1rem .375rem; border-radius: 10px; white-space: nowrap; flex-shrink: 0; } + .ai-real_bug { background: var(--red-dim); color: var(--red); } + .ai-intentional { background: rgba(251,146,60,.15); color: var(--orange); } + .ai-flaky { background: var(--green-dim); color: var(--green); } + .ai-unknown { background: var(--bg4); color: var(--text2); } + #ai-bar { display: none; align-items: center; gap: .5rem; padding: .3rem .75rem; background: var(--bg3); border-bottom: 1px solid var(--border); font-size: .6875rem; color: var(--text2); flex-shrink: 0; } + #ai-bar.visible { display: flex; } + #ai-text { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; min-width: 0; } + /* ── Main ────────────────────────────────────────────────────────── */ #main { flex: 1; overflow: hidden; display: flex; flex-direction: column; background: var(--bg); } @@ -202,6 +212,8 @@ +