schema_version: 1
comparison_set:
  - id: racc
    name: Racc
    aliases:
      - Racc
    state: evidence_pending
    pending_claims:
      - racc-error-ux-json-v1
      - racc-public-performance-2026-07-31
    version: 1.8.1
    revision: not_applicable
    reason: "Pending comparative claims: racc-error-ux-json-v1 [required subjective review incomplete]; racc-public-performance-2026-07-31 [missing evidence: Complete formal public-performance result artifact generated by the registered command]."
  - id: lrama
    name: Lrama
    aliases:
      - Lrama
    state: not_compared
    pending_claims: []
    version: unknown
    revision: unknown
    reason: No registered comparative claims.
  - id: bison
    name: GNU Bison
    aliases:
      - GNU Bison
      - Bison
      - GNU yacc
      - yacc
    state: not_compared
    pending_claims: []
    version: unknown
    revision: unknown
    reason: No registered comparative claims.
  - id: menhir
    name: Menhir
    aliases:
      - Menhir
    state: not_compared
    pending_claims: []
    version: unknown
    revision: unknown
    reason: No registered comparative claims.
  - id: tree_sitter
    name: Tree-sitter
    aliases:
      - Tree-sitter
      - Tree sitter
      - tree_sitter
    state: not_compared
    pending_claims: []
    version: unknown
    revision: unknown
    reason: No registered comparative claims.
  - id: antlr
    name: ANTLR
    aliases:
      - ANTLR
      - ANTLR4
    state: not_compared
    pending_claims: []
    version: unknown
    revision: unknown
    reason: No registered comparative claims.
claims:
  - id: racc-error-ux-json-v1
    state: review_pending
    title: JSON error callback and bounded-repair observations against Racc 1.8.1
    wording: >-
      At Ibex revision cc20c5eb799cc218ebea665df64261f10d030f75, on ten fixed malformed
      JSON inputs with the Ruby, OS, CPU, kernel, processor-count, and YJIT environment unrecorded,
      the committed result artifact records Ibex diagnostics,
      Racc 1.8.1's public on_error callback values, and a maintainer assessment of ten bounded repairs;
      independent subjective review is pending, so this is not a completed comparative UX claim.
    binding:
      path: docs/error-ux.md
      marker: racc-error-ux-json-v1
      kind: evidence
      required_text:
        - 8/10 (80%)
        - "| EUX-10 |"
      allowed_strength: []
      body_sha256: 8f96ed02f51364c2a71409123022dc7350b56f6583f517d157aa3ef623125ef4
    subjects:
      - tool: ibex
        identity_kind: repository
        version: 0.1.0
        revision: cc20c5eb799cc218ebea665df64261f10d030f75
      - tool: racc
        identity_kind: release
        version: 1.8.1
        revision: not_applicable
    public_command:
      executable: bundle
      argv:
        - exec
        - ruby
        - tool/error_ux_snapshot.rb
    corpus:
      - id: json-error-cases-v1
        path: test/fixtures/error_ux/json-errors-v1.json
        revision: cc20c5eb799cc218ebea665df64261f10d030f75
    environment:
      known:
        racc_version: 1.8.1
      unknown:
        - cpu_model
        - host_cpu
        - host_os
        - kernel_release
        - processors
        - ruby_engine
        - ruby_platform
        - ruby_version
        - yjit_enabled
    unsupported_semantics:
      - The Racc observation is limited to the public on_error callback without an application-specific message layer.
      - The repair usefulness labels are maintainer judgments, not measured Racc recovery behavior.
    subjective_review:
      required: true
      state: pending
      method: A third party whose canonical GitHub login is absent from the explicit maintainer roster reviews all ten case IDs with fixed diagnostic and repair labels, rationales, disagreements, and structured publication consent.
    validity:
      scope: Exact snapshot introduction revision and Racc 1.8.1 only.
      expires: No comparative UX conclusion may be published before independent review is recorded.
      review_when:
        - Any diagnostic, parser table, Racc version, corpus, or repair policy changes.
        - An independent assessment is added or disagrees with the maintainer assessment.
    evidence:
      - kind: method
        path: docs/error-ux-review-rubric-v1.md
        description: Versioned independent rubric, fixed labels, disagreement policy, publication consent, and import workflow.
      - kind: report
        path: docs/error-ux-review-status-v1.json
        description: Machine-readable HOLD state, immutable kit identity, and imported-record registry.
      - kind: method
        path: docs/error-ux.md
        description: Public method, case table, maintainer assessment, and explicit missing review.
      - kind: method
        path: schema/error-ux-review-v1.schema.json
        description: Closed independent-review record contract for identity, reproduction, publication, and ten assessments.
      - kind: result_artifact
        path: test/fixtures/error_ux/json-errors-v1.json
        description: Versioned normative observations for all ten inputs.
    limitations:
      - "Unknown environment fields: cpu_model, host_cpu, host_os, kernel_release, processors, ruby_engine, ruby_platform, ruby_version, yjit_enabled."
      - Independent subjective review is still missing, so the 8/10 usefulness assessment is not a completed comparison.
      - Independent human review remains subjective and cannot prove affiliation, conflict disclosure, coercion resistance, or general error UX quality.
    missing_evidence: []
  - id: racc-public-performance-2026-07-31
    state: evidence_pending
    title: Fixed-revision public Ruby-backend performance projection against Racc 1.8.1
    wording: >-
      At clean Ibex revision 984d4ae7a1c96db4c71e3077669842b7bbc3b4ea, the readiness
      projection for three pinned workloads on Ruby 4.0.0, arm64-darwin24, with YJIT disabled
      records slower cold generation and new-instance runtime than Racc 1.8.1's Ruby backend;
      the direct formal result artifact is absent, so this is evidence-pending and non-publishable
      as a comparative claim.
    binding:
      path: docs/release-readiness.md
      marker: racc-public-performance-2026-07-31
      kind: evidence
      required_text:
        - 0.039443/0.040048
        - 23.8–32.9% slower
      allowed_strength:
        - The harness's `target_met` field marks absolute parity when the interval's upper bound is no greater than 1.0.
        - "The results deliberately retain the unfavorable rows: cold generation is 23.8–32.9% slower, and new-instance parsing is 18.6–37.3% slower."
        - Reuse point estimates range from 1.5% faster to 5.2% slower.
        - Ibex allocates fewer objects for every reuse workload, while new-instance allocation ratios remain 1.051–1.190x.
        - Generated output is smaller for all three grammars.
        - These values are a release baseline, not portable scores or an assertion of parity.
      body_sha256: 44b2c95f676be051c56731b6b96a8a860001419c7a11cb99d8cd344181c7b604
    subjects:
      - tool: ibex
        identity_kind: repository
        version: 0.2.0
        revision: 984d4ae7a1c96db4c71e3077669842b7bbc3b4ea
      - tool: racc
        identity_kind: release
        version: 1.8.1
        revision: not_applicable
    public_command:
      executable: bundle
      argv:
        - exec
        - ruby
        - benchmark/public_comparison.rb
        - --checkout
        - namae=/path/to/namae
        - --checkout
        - bcdice_command=/path/to/bcdice
        - --checkout
        - nokogiri_css=/path/to/nokogiri
        - --runs
        - "10"
        - --warmup
        - "50"
        - --runtime-iterations
        - "250"
        - --behavior-probe-iterations
        - "5"
        - --bootstrap-samples
        - "10000"
        - --expected-racc-backend
        - ruby
        - --output
        - tmp/public-performance-comparison.json
    corpus:
      - id: bcdice-command
        path: benchmark/public_workloads.json
        revision: 21b4a03789bf2080ad41aaf31299b609ee7bda86
      - id: namae
        path: benchmark/public_workloads.json
        revision: d33875aaf1fc420a8dfe946a3b29cc3e19710061
      - id: nokogiri-css
        path: benchmark/public_workloads.json
        revision: 04a4c29c6a605ad40a78f4ce343ced0832a1805c
    environment:
      known:
        racc_backend: ruby
        ruby_engine: ruby
        ruby_platform: arm64-darwin24
        ruby_version: 4.0.0
        yjit_enabled: "false"
      unknown:
        - cpu_model
        - host_cpu
        - host_os
        - kernel_release
        - processors
    unsupported_semantics:
      - Only the three pinned Racc-compatible Ruby grammar workloads are covered.
      - Pretokenized parser-core timing is excluded from the public workload comparison.
      - Native Racc, YJIT, other Rubies, other operating systems, and other parser generators are excluded.
    subjective_review:
      required: false
      state: not_applicable
      method: Deterministic statistics and behavior digests are reviewed; no cross-category ordering is produced.
    validity:
      scope: Historical readiness projection for the exact Ibex and workload revisions on the recorded environment.
      expires: It cannot become a publishable comparative claim until the direct formal result artifact is available.
      review_when:
        - A complete formal result artifact becomes available.
        - Any subject version, workload revision, command, environment, or measured lifecycle changes.
    evidence:
      - kind: method
        path: benchmark/README.md
        description: Reproduction command, runtime scope, process isolation, and publication rules.
      - kind: corpus
        path: benchmark/public_workloads.json
        description: Pinned repositories, grammar paths, and fixed public inputs.
      - kind: report
        path: docs/release-readiness.md
        description: Historical reviewed projection with exact revision, environment, ratios, and unfavorable rows.
    limitations:
      - "Unknown environment fields: cpu_model, host_cpu, host_os, kernel_release, processors."
      - The direct formal result artifact is absent; the readiness report is a reviewed projection and cannot support a completed public comparative claim.
      - The projection does not describe the current revision, unmeasured workloads, other backends, or semantics beyond recorded behavior digests.
    missing_evidence:
      - Complete formal public-performance result artifact generated by the registered command.
  - id: racc-public-performance-2026-08-08
    state: measured
    title: Exact-revision public Ruby-backend performance comparison against Racc 1.8.1
    wording: >-
      At clean Ibex revision 26e94ce631bf5c075ecbca0205a34a8c231c31d9, on Ruby 4.0.0,
      arm64-darwin24, with YJIT disabled, the direct formal comparison of the pinned
      Namae, BCDice command, and Nokogiri CSS workloads against Racc 1.8.1 produced
      equivalent result sequences; Ibex recorded slower cold generation and new-instance
      parsing, fewer reuse allocations, and smaller generated output in every workload.
    binding:
      path: docs/release-readiness.md
      marker: racc-public-performance-2026-08-08
      kind: claim
      required_text:
        - "| Namae | 1.317830 | 0.996320 | 0.854462 | 1.209687 | 1.207559 | 0.861519 |"
        - "| BCDice command | 1.309690 | 1.010467 | 0.675012 | 1.331780 | 1.129299 | 0.981449 |"
        - "| Nokogiri CSS | 1.288118 | 0.994802 | 0.670484 | 1.193522 | 1.074838 | 0.936806 |"
      allowed_strength: []
      body_sha256: d15ee062a4e13e57e1395d1d6da776d9490375fd5e560831dc5f6f80057797c6
    subjects:
      - tool: ibex
        identity_kind: repository
        version: 0.2.0
        revision: 26e94ce631bf5c075ecbca0205a34a8c231c31d9
      - tool: racc
        identity_kind: release
        version: 1.8.1
        revision: not_applicable
    public_command:
      executable: bundle
      argv:
        - exec
        - ruby
        - benchmark/public_comparison.rb
        - --checkout
        - namae=/path/to/namae
        - --checkout
        - bcdice_command=/path/to/bcdice
        - --checkout
        - nokogiri_css=/path/to/nokogiri
        - --runs
        - "10"
        - --warmup
        - "50"
        - --runtime-iterations
        - "250"
        - --behavior-probe-iterations
        - "5"
        - --bootstrap-samples
        - "10000"
        - --expected-racc-backend
        - ruby
        - --output
        - tmp/public-performance-comparison.json
    corpus:
      - id: bcdice-command
        path: benchmark/public_workloads.json
        revision: 21b4a03789bf2080ad41aaf31299b609ee7bda86
      - id: namae
        path: benchmark/public_workloads.json
        revision: d33875aaf1fc420a8dfe946a3b29cc3e19710061
      - id: nokogiri-css
        path: benchmark/public_workloads.json
        revision: 04a4c29c6a605ad40a78f4ce343ced0832a1805c
    environment:
      known:
        host_cpu: arm64
        host_os: darwin24
        ruby_engine: ruby
        ruby_platform: arm64-darwin24
        ruby_version: 4.0.0
        yjit_enabled: "false"
      unknown:
        - cpu_model
        - kernel_release
        - processors
    unsupported_semantics:
      - Only the three pinned Racc-compatible Ruby grammar workloads are covered.
      - Pretokenized parser-core timing is excluded from the public workload comparison.
      - Native Racc, YJIT, other Rubies, other operating systems, and other parser generators are excluded.
    subjective_review:
      required: false
      state: not_applicable
      method: Deterministic statistics and behavior digests are reviewed; no cross-category ordering is produced.
    validity:
      scope: Exact Ibex revision, pinned workload revisions, Ruby 4.0.0 arm64-darwin24 environment, and Racc 1.8.1 Ruby backend.
      expires: Review whenever a subject version, workload revision, command, environment, or measured lifecycle changes.
      review_when:
        - Any subject version, workload revision, command, environment, or measured lifecycle changes.
    evidence:
      - kind: method
        path: benchmark/README.md
        description: Reproduction command, runtime scope, process isolation, and publication rules.
      - kind: corpus
        path: benchmark/public_workloads.json
        description: Pinned repositories, grammar paths, and fixed public inputs.
      - kind: result_artifact
        path: benchmark/results/public/2026-08-08-8c9cef999d09-ruby-4.0.0-arm64-darwin24.json
        description: Complete formal ten-run comparison with environment, commands, behavior digests, ratios, and bootstrap intervals.
      - kind: report
        path: docs/release-readiness.md
        description: Exact-revision result summary and unfavorable measurements.
    limitations:
      - Only the three pinned Racc-compatible Ruby grammar workloads are covered.
      - Pretokenized parser-core timing is excluded from the public workload comparison.
      - Native Racc, YJIT, other Rubies, other operating systems, and other parser generators are excluded.
      - "Unknown environment fields: cpu_model, kernel_release, processors."
    missing_evidence: []
