From e32246492304d58eafcee8c396abf02cb109d70d Mon Sep 17 00:00:00 2001 From: martinfrancois Date: Mon, 21 Sep 2026 06:06:02 +0200 Subject: [PATCH] test(evals): move repository metadata out of criteria.json Closes #89. tessl eval lint rejects the metadata key and warns on the per-item category field. criteria.json now holds only context, type, and the checklist items Tessl's schema knows; criteria-meta.json next to it carries this repository's metadata object and a categories map from checklist name to category. The validators merge the two, the classifier reads the sidecar, and a criteria.json that still carries metadata or categories fails validation. Pure metadata move: no task, criterion text, or max_score changed, so no hosted rerun is needed. --- CONTRIBUTING.md | 8 ++- docs/agents/evals.md | 13 ++-- .../05-parallel-cpu-review/criteria-meta.json | 16 +++++ .../05-parallel-cpu-review/criteria.json | 14 +---- .../criteria-meta.json | 17 +++++ .../08-primary-contact-review/criteria.json | 15 +---- .../criteria-meta.json | 19 ++++++ .../15-session-roster-indexes/criteria.json | 17 +---- .../criteria-meta.json | 16 +++++ .../criteria.json | 14 +---- .../criteria-meta.json | 16 +++++ .../criteria.json | 14 +---- .../criteria-meta.json | 18 ++++++ .../28-overdue-shipment-notices/criteria.json | 16 +---- .../criteria-meta.json | 14 +++++ .../01-permission-and-orders/criteria.json | 12 +--- .../criteria-meta.json | 14 +++++ .../02-product-list-transforms/criteria.json | 12 +--- .../criteria-meta.json | 13 ++++ .../04-primary-address-review/criteria.json | 11 +--- .../06-inventory-summary/criteria-meta.json | 16 +++++ .../06-inventory-summary/criteria.json | 14 +---- .../07-catalog-feed/criteria-meta.json | 16 +++++ .../07-catalog-feed/criteria.json | 14 +---- .../criteria-meta.json | 16 +++++ .../09-order-collector-report/criteria.json | 14 +---- .../criteria-meta.json | 15 +++++ .../10-packet-window-cleanup/criteria.json | 13 +--- .../criteria-meta.json | 15 +++++ .../criteria.json | 13 +--- .../criteria-meta.json | 16 +++++ .../13-training-and-packets/criteria.json | 14 +---- .../criteria-meta.json | 15 +++++ .../14-parallel-mutation-review/criteria.json | 13 +--- .../criteria-meta.json | 15 +++++ .../16-java11-report-review/criteria.json | 13 +--- .../criteria-meta.json | 16 +++++ .../criteria.json | 14 +---- .../criteria-meta.json | 15 +++++ .../18-priority-findany-review/criteria.json | 13 +--- .../criteria-meta.json | 15 +++++ .../19-null-collector-review/criteria.json | 13 +--- .../criteria-meta.json | 15 +++++ .../criteria.json | 13 +--- .../22-java8-version-scan/criteria-meta.json | 15 +++++ .../22-java8-version-scan/criteria.json | 13 +--- .../criteria-meta.json | 14 +++++ .../23-collector-order-scan/criteria.json | 12 +--- .../criteria-meta.json | 13 ++++ .../criteria.json | 11 +--- .../criteria-meta.json | 16 +++++ .../25-hard-stop-scan-audit/criteria.json | 14 +---- .../criteria-meta.json | 18 ++++++ .../criteria.json | 16 +---- .../criteria-meta.json | 18 ++++++ .../criteria.json | 16 +---- .../criteria-meta.json | 16 +++++ .../criteria.json | 14 +---- .../criteria-meta.json | 16 +++++ .../criteria.json | 14 +---- scripts/classify_eval_result.py | 6 +- scripts/eval_impact.py | 2 +- scripts/test_validate_eval_criteria.py | 5 +- scripts/validate_eval_criteria.py | 62 +++++++++++++++++-- 64 files changed, 563 insertions(+), 383 deletions(-) create mode 100644 evals-reference/05-parallel-cpu-review/criteria-meta.json create mode 100644 evals-reference/08-primary-contact-review/criteria-meta.json create mode 100644 evals-reference/15-session-roster-indexes/criteria-meta.json create mode 100644 evals-reference/26-uppercase-side-effect-review/criteria-meta.json create mode 100644 evals-reference/27-uppercase-names-implementation/criteria-meta.json create mode 100644 evals-reference/28-overdue-shipment-notices/criteria-meta.json create mode 100644 evals-regression/01-permission-and-orders/criteria-meta.json create mode 100644 evals-regression/02-product-list-transforms/criteria-meta.json create mode 100644 evals-regression/04-primary-address-review/criteria-meta.json create mode 100644 evals-regression/06-inventory-summary/criteria-meta.json create mode 100644 evals-regression/07-catalog-feed/criteria-meta.json create mode 100644 evals-regression/09-order-collector-report/criteria-meta.json create mode 100644 evals-regression/10-packet-window-cleanup/criteria-meta.json create mode 100644 evals-regression/11-null-category-price-index/criteria-meta.json create mode 100644 evals-regression/13-training-and-packets/criteria-meta.json create mode 100644 evals-regression/14-parallel-mutation-review/criteria-meta.json create mode 100644 evals-regression/16-java11-report-review/criteria-meta.json create mode 100644 evals-regression/17-java8-optional-prefix-review/criteria-meta.json create mode 100644 evals-regression/18-priority-findany-review/criteria-meta.json create mode 100644 evals-regression/19-null-collector-review/criteria-meta.json create mode 100644 evals-regression/20-parallel-shared-state-review/criteria-meta.json create mode 100644 evals-regression/22-java8-version-scan/criteria-meta.json create mode 100644 evals-regression/23-collector-order-scan/criteria-meta.json create mode 100644 evals-regression/24-mutable-batch-modernization/criteria-meta.json create mode 100644 evals-regression/25-hard-stop-scan-audit/criteria-meta.json create mode 100644 evals/01-offer-availability-mapconcurrent/criteria-meta.json create mode 100644 evals/02-delivery-appointments-mapconcurrent/criteria-meta.json create mode 100644 evals/03-payment-screening-gatherer-review/criteria-meta.json create mode 100644 evals/04-invoice-bounds-and-temperature-windows/criteria-meta.json diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index e4eaafc..30f92aa 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -171,9 +171,11 @@ claim. The main eval set should stay focused on realistic tasks where context should improve stream quality. It must include natural activation prompts and explicit invocation prompts. Natural scenarios must not mention `$java-streams` or ask to use the skill. Explicit scenarios may name the -skill and must be labeled as explicit in `criteria.json`. +skill and must be labeled as explicit in `criteria-meta.json`. -Every scenario directory must contain `task.md`, `criteria.json`, and `capability.txt`. Main eval +Every scenario directory must contain `task.md`, `criteria.json`, `criteria-meta.json`, and +`capability.txt` (`criteria.json` keeps Tessl's schema; `criteria-meta.json` carries this +repository's metadata and per-item categories). Main eval implementation criteria must include compile/artifact checks and behavior correctness checks as safety checks, but the main score should mainly measure stream-specific quality. Each main eval criterion must set `category` to `safety`, `stream_quality`, or `maintainability`. @@ -221,7 +223,7 @@ current benchmark claims until they are rerun against the current active suite m denominator, commit/ref, natural/explicit split, and pinned CLI behavior. Main eval weights should stay evidence-weighted: put more points on scenario families with larger observed missed-point reduction, keep ordinary 100-point main scenarios around 15 safety / 80 stream-quality / 5 -maintainability points, and document any `main_eval_weight_multiplier` in `criteria.json` metadata. +maintainability points, and document any `main_eval_weight_multiplier` in `criteria-meta.json`. When with-context is below 100%, keep the scenario wherever it already lives. Fix the skill or eval there, then rerun only that targeted scenario until it is clean before running broader suites. After diff --git a/docs/agents/evals.md b/docs/agents/evals.md index 375ed30..e254fd8 100644 --- a/docs/agents/evals.md +++ b/docs/agents/evals.md @@ -51,8 +51,12 @@ benchmark claims, or scoring rules. with-context and without-context, plus skill-context-dependent checks that are only fair as with-context regression coverage. These protect against regressions but should not be part of normal lift discovery runs. -- Every scenario directory must contain `task.md`, `criteria.json`, and `capability.txt`. -- Every `criteria.json` must classify `metadata.invocation` and `metadata.task_type`. +- Every scenario directory must contain `task.md`, `criteria.json`, `criteria-meta.json`, and + `capability.txt`. `criteria.json` holds only what Tessl's schema knows (`context`, `type`, and + checklist items with `name`, `description`, `max_score`), so `tessl eval lint` stays clean; + `criteria-meta.json` holds this repository's `metadata` object and a `categories` map from + checklist name to category. The validators read the merged view. +- Every `criteria-meta.json` must classify `metadata.invocation` and `metadata.task_type`. - Use `metadata.evidence_type` when scenario placement needs to be explicit: - `ordinary_lift`: an ordinary main or reference scenario where both variants are fair to compare. This value is invalid in `evals-regression/`, and it must not be used when the task overlaps @@ -138,7 +142,7 @@ benchmark claims, or scoring rules. - Normalize ordinary 100-point main scenarios around 15 safety, 80 stream-quality, and 5 maintainability points unless the scenario has a documented reason to differ. - Use `main_eval_weight_multiplier` only when a scenario family has stronger hosted delta or higher - benchmark importance; document why in `criteria.json` metadata and this file. + benchmark importance; document why in `criteria-meta.json` and this file. - Do not add or inflate weak-delta scenarios only to make coverage look balanced. - A 2x raw score ratio is useful only when earned by honest, realistic eval design. Don't suppress legitimate coverage just to improve lift. @@ -184,7 +188,8 @@ benchmark claims, or scoring rules. not the final all-suite requirement itself; it is rerunning required evidence after later edits to `skills/java-streams/SKILL.md` or bundled runtime references change the skill fingerprint. Do the local scenario/criteria crosswalk and obvious skill wording fixes before starting hosted runs. - - A pure suite move does not require a hosted rerun when `task.md`, `criteria.json`, and + - A pure suite move does not require a hosted rerun when `task.md`, `criteria.json`, + `criteria-meta.json`, and `capability.txt` content are unchanged except for suite-placement metadata or numbering notes. Run local validators and update suite totals/numbering instead. If the move also changes task wording, scoring criteria, capability text, runtime skill behavior, or benchmark claims, follow diff --git a/evals-reference/05-parallel-cpu-review/criteria-meta.json b/evals-reference/05-parallel-cpu-review/criteria-meta.json new file mode 100644 index 0000000..1b67bfb --- /dev/null +++ b/evals-reference/05-parallel-cpu-review/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "focused_reference", + "reference_selection": "Focused reference coverage for CPU-heavy stateless parallel stream advice.", + "runtime_reference_overlap_rationale": "Allowed only because this is reference-suite focused coverage, not ordinary broad lift." + }, + "categories": { + "Creates review artifact": "safety", + "Allows parallelism for this stream chain": "stream_quality", + "Mentions overhead and measurement": "stream_quality", + "Mentions common pool": "stream_quality", + "Avoids multi-line lambda examples": "maintainability" + } +} diff --git a/evals-reference/05-parallel-cpu-review/criteria.json b/evals-reference/05-parallel-cpu-review/criteria.json index 80e9366..730b081 100644 --- a/evals-reference/05-parallel-cpu-review/criteria.json +++ b/evals-reference/05-parallel-cpu-review/criteria.json @@ -4,40 +4,28 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear review decision." }, { "name": "Allows parallelism for this stream chain", - "category": "stream_quality", "max_score": 20, "description": "Recognizes the work is CPU-heavy, stateless, and has a primitive sum reduction, so parallelism can be reasonable." }, { "name": "Mentions overhead and measurement", - "category": "stream_quality", "max_score": 15, "description": "Says the performance should be measured because parallel stream overhead can outweigh benefits." }, { "name": "Mentions common pool", - "category": "stream_quality", "max_score": 10, "description": "Notes that parallel streams use the common fork-join pool by default and may contend with other work." }, { "name": "Avoids multi-line lambda examples", - "category": "maintainability", "max_score": 10, "description": "If the review includes replacement or optional example code, keeps lambda bodies on the same line as `->` or uses named helpers/method references instead of showing block lambdas, arrows whose body starts on the next line, or callback bodies that continue on later lines." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "focused_reference", - "reference_selection": "Focused reference coverage for CPU-heavy stateless parallel stream advice.", - "runtime_reference_overlap_rationale": "Allowed only because this is reference-suite focused coverage, not ordinary broad lift." - } + ] } diff --git a/evals-reference/08-primary-contact-review/criteria-meta.json b/evals-reference/08-primary-contact-review/criteria-meta.json new file mode 100644 index 0000000..828f918 --- /dev/null +++ b/evals-reference/08-primary-contact-review/criteria-meta.json @@ -0,0 +1,17 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "review", + "evidence_type": "focused_reference", + "reference_selection": "Focused reference coverage for preserving priority ordering when reviewing findFirst versus findAny changes.", + "runtime_reference_overlap_rationale": "Allowed only because this is reference-suite focused coverage, not ordinary broad lift." + }, + "categories": { + "Creates review artifact": "safety", + "Rejects behavior-changing proposal": "safety", + "Identifies priority ordering contract": "stream_quality", + "Flags unjustified parallelStream": "stream_quality", + "Suggests safe stream chain": "stream_quality", + "Review is concise and focused": "maintainability" + } +} diff --git a/evals-reference/08-primary-contact-review/criteria.json b/evals-reference/08-primary-contact-review/criteria.json index bfd6ca1..a3fd467 100644 --- a/evals-reference/08-primary-contact-review/criteria.json +++ b/evals-reference/08-primary-contact-review/criteria.json @@ -4,46 +4,33 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear review decision about the proposed change." }, { "name": "Rejects behavior-changing proposal", - "category": "safety", "max_score": 10, "description": "Clearly says the proposed change should not be accepted because it can return a different contact." }, { "name": "Identifies priority ordering contract", - "category": "stream_quality", "max_score": 35, "description": "Explains that sorted by Contact::priority followed by findFirst means the lowest-priority verified contact wins, and findAny does not preserve that contract." }, { "name": "Flags unjustified parallelStream", - "category": "stream_quality", "max_score": 25, "description": "Explains that parallelStream is not justified here and makes ordered selection less predictable or more expensive without evidence of CPU-bound work." }, { "name": "Suggests safe stream chain", - "category": "stream_quality", "max_score": 20, "description": "Suggests keeping sorted(...).filter(...).findFirst(), or filtering before sorting only if that preserves the same lowest-priority verified contact behavior." }, { "name": "Review is concise and focused", - "category": "maintainability", "max_score": 5, "description": "Keeps the review focused on stream ordering and parallelism without internal workflow labels such as hard stop, marker, scan, checklist, skill, rubric, or criteria. Award full credit for a brief note that the original chain can also be simplified only if it supports the decision and does not distract from it." } - ], - "metadata": { - "invocation": "natural", - "task_type": "review", - "evidence_type": "focused_reference", - "reference_selection": "Focused reference coverage for preserving priority ordering when reviewing findFirst versus findAny changes.", - "runtime_reference_overlap_rationale": "Allowed only because this is reference-suite focused coverage, not ordinary broad lift." - } + ] } diff --git a/evals-reference/15-session-roster-indexes/criteria-meta.json b/evals-reference/15-session-roster-indexes/criteria-meta.json new file mode 100644 index 0000000..dce10cf --- /dev/null +++ b/evals-reference/15-session-roster-indexes/criteria-meta.json @@ -0,0 +1,19 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "implementation", + "evidence_type": "focused_reference", + "reference_selection": "Focused reference coverage for nested session-registration collector behavior and helper extraction.", + "runtime_reference_overlap_rationale": "Allowed only because this is reference-suite focused coverage, not ordinary broad lift." + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Preserves requested roster and index behavior": "safety", + "Flattens nested conference data directly": "stream_quality", + "Uses collector semantics for track email lists": "stream_quality", + "Avoids multi-line lambdas": "maintainability", + "Handles duplicate rooms explicitly": "stream_quality", + "Short-circuits waitlisted existence": "stream_quality", + "Keeps implementation focused": "maintainability" + } +} diff --git a/evals-reference/15-session-roster-indexes/criteria.json b/evals-reference/15-session-roster-indexes/criteria.json index 769ee05..6a47a2b 100644 --- a/evals-reference/15-session-roster-indexes/criteria.json +++ b/evals-reference/15-session-roster-indexes/criteria.json @@ -4,58 +4,43 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 5, "description": "Creates SessionRosterIndexes.java with the requested methods, nested records, imports, and Java 17-compatible code." }, { "name": "Preserves requested roster and index behavior", - "category": "safety", "max_score": 10, "description": "Correctly handles null tracks, null rooms, null emails, opted-in filtering, duplicate emails, sorted email lists, room duplicate handling, encounter-order ties, and waitlisted existence." }, { "name": "Flattens nested conference data directly", - "category": "stream_quality", "max_score": 20, "description": "Traverses conferences to sessions to registrations with direct flatMap-style composition or an equivalently clear stream chain, without building nested temporary collections or mutating shared result maps inside forEach." }, { "name": "Uses collector semantics for track email lists", - "category": "stream_quality", "max_score": 20, "description": "Builds opted-in emails by track with collector semantics such as groupingBy plus downstream mapping/filtering/collection, producing sorted unique email lists without a post-hoc shared mutable accumulation pass." }, { "name": "Avoids multi-line lambdas", - "category": "maintainability", "max_score": 10, "description": "Keeps stream and collector lambdas as short same-line glue. Extracts nested registration streams, duplicate-room comparison logic, or other multi-step callback bodies into named helpers or method references instead of using block lambdas, arrows whose body starts on the next line, or lambda bodies that continue their stream chain on later lines." }, { "name": "Handles duplicate rooms explicitly", - "category": "stream_quality", "max_score": 20, "description": "Builds longestSessionByRoom with an explicit duplicate-room merge or grouping reduction that keeps the longer session and preserves the earlier encounter-order session on ties." }, { "name": "Short-circuits waitlisted existence", - "category": "stream_quality", "max_score": 10, "description": "Uses anyMatch or an equivalent early-exit check for hasWaitlistedRegistration rather than collecting or counting all waitlisted registrations first." }, { "name": "Keeps implementation focused", - "category": "maintainability", "max_score": 5, "description": "Keeps the implementation limited to the requested helpers without unrelated caching, concurrency, dependencies, API redesign, or broad style rewrites." } - ], - "metadata": { - "invocation": "natural", - "task_type": "implementation", - "evidence_type": "focused_reference", - "reference_selection": "Focused reference coverage for nested session-registration collector behavior and helper extraction.", - "runtime_reference_overlap_rationale": "Allowed only because this is reference-suite focused coverage, not ordinary broad lift." - } + ] } diff --git a/evals-reference/26-uppercase-side-effect-review/criteria-meta.json b/evals-reference/26-uppercase-side-effect-review/criteria-meta.json new file mode 100644 index 0000000..dfbff9e --- /dev/null +++ b/evals-reference/26-uppercase-side-effect-review/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "ordinary_lift", + "issue": "https://github.com/martinfrancois/java-streams-skill/issues/4" + }, + "categories": { + "Creates review artifact": "safety", + "Identifies external mutation": "stream_quality", + "Warns against unsafe parallelStream shortcut": "stream_quality", + "Suggests direct result-producing replacement": "stream_quality", + "Addresses performance honestly": "stream_quality", + "Keeps review focused": "maintainability" + } +} diff --git a/evals-reference/26-uppercase-side-effect-review/criteria.json b/evals-reference/26-uppercase-side-effect-review/criteria.json index 72fd946..eee5a4d 100644 --- a/evals-reference/26-uppercase-side-effect-review/criteria.json +++ b/evals-reference/26-uppercase-side-effect-review/criteria.json @@ -4,45 +4,33 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear recommendation." }, { "name": "Identifies external mutation", - "category": "stream_quality", "max_score": 25, "description": "Explains that forEach mutates the external ArrayList and that the stream should produce the result list directly instead." }, { "name": "Warns against unsafe parallelStream shortcut", - "category": "stream_quality", "max_score": 25, "description": "Does not recommend simply changing stream() to parallelStream() while keeping the external ArrayList mutation. Full credit may still be awarded when the review first makes the code side-effect-free, then prominently recommends benchmarking a pure parallelStream map plus collector/toList version for the 10-million-name case before relying on it for throughput." }, { "name": "Suggests direct result-producing replacement", - "category": "stream_quality", "max_score": 25, "description": "Shows or clearly recommends a replacement such as names.stream().map(String::toUpperCase).toList(), or collect(Collectors.toList()) when Java-version compatibility or mutability requires it." }, { "name": "Addresses performance honestly", - "category": "stream_quality", "max_score": 15, "description": "Distinguishes the correctness fix from the performance decision: does not claim direct collection alone is a guaranteed throughput win, strongly recommends benchmarking a pure parallel stream for the 10-million-name case, and warns that parallelStream can be slower for small lists or mostly-small call paths." }, { "name": "Keeps review focused", - "category": "maintainability", "max_score": 5, "description": "Keeps the review concise and focused on the stream/performance issue without unrelated rewrites, broad style commentary, or internal workflow labels such as hard stop, marker, scan, checklist, skill, rubric, or criteria." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "ordinary_lift", - "issue": "https://github.com/martinfrancois/java-streams-skill/issues/4" - } + ] } diff --git a/evals-reference/27-uppercase-names-implementation/criteria-meta.json b/evals-reference/27-uppercase-names-implementation/criteria-meta.json new file mode 100644 index 0000000..f00ecda --- /dev/null +++ b/evals-reference/27-uppercase-names-implementation/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "implementation", + "evidence_type": "ordinary_lift", + "issue": "https://github.com/martinfrancois/java-streams-skill/issues/4" + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Preserves requested behavior": "safety", + "Uses direct stream result production": "stream_quality", + "Avoids unsafe parallel shortcut": "stream_quality", + "Keeps performance scope focused": "stream_quality", + "Keeps API focused": "maintainability" + } +} diff --git a/evals-reference/27-uppercase-names-implementation/criteria.json b/evals-reference/27-uppercase-names-implementation/criteria.json index 8068355..227d311 100644 --- a/evals-reference/27-uppercase-names-implementation/criteria.json +++ b/evals-reference/27-uppercase-names-implementation/criteria.json @@ -4,45 +4,33 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 10, "description": "Creates UppercaseNames.java with the requested uppercaseNames(List names) method, necessary imports, and Java 17-compatible code." }, { "name": "Preserves requested behavior", - "category": "safety", "max_score": 15, "description": "Returns every input name converted with String::toUpperCase in the original encounter order and does not mutate the input list." }, { "name": "Uses direct stream result production", - "category": "stream_quality", "max_score": 35, "description": "Uses a direct stream chain such as names.stream().map(String::toUpperCase).toList(), or an equivalent direct collector, instead of creating a list and mutating it from forEach." }, { "name": "Avoids unsafe parallel shortcut", - "category": "stream_quality", "max_score": 25, "description": "Does not use parallelStream(), .parallel(), or manual concurrent workers together with shared mutable accumulation. Full credit may be awarded for either sequential direct collection or a pure side-effect-free parallelStream map plus ordered collector/toList implementation that preserves behavior for the large-list performance task." }, { "name": "Keeps performance scope focused", - "category": "stream_quality", "max_score": 10, "description": "Keeps the implementation O(n), avoids extra unnecessary passes, and does not add caching, batching frameworks, or unrelated memory-heavy transformations." }, { "name": "Keeps API focused", - "category": "maintainability", "max_score": 5, "description": "Keeps the implementation limited to the requested helper without unrelated methods, dependencies, or broad API redesign." } - ], - "metadata": { - "invocation": "natural", - "task_type": "implementation", - "evidence_type": "ordinary_lift", - "issue": "https://github.com/martinfrancois/java-streams-skill/issues/4" - } + ] } diff --git a/evals-reference/28-overdue-shipment-notices/criteria-meta.json b/evals-reference/28-overdue-shipment-notices/criteria-meta.json new file mode 100644 index 0000000..445bb47 --- /dev/null +++ b/evals-reference/28-overdue-shipment-notices/criteria-meta.json @@ -0,0 +1,18 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "implementation", + "evidence_type": "focused_reference", + "issue": "https://github.com/martinfrancois/java-streams-skill/issues/36", + "reference_selection": "Focused issue #36 behavior-delta coverage for extracting non-trivial stream lambda bodies into helpers.", + "runtime_reference_overlap_rationale": "Allowed only because this is reference-suite focused coverage, not ordinary broad lift or main-suite generalization evidence." + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Preserves filtering behavior": "safety", + "Computes notice fields correctly": "safety", + "Uses direct stream result production": "stream_quality", + "Avoids multi-line stream lambdas": "stream_quality", + "Keeps API focused": "maintainability" + } +} diff --git a/evals-reference/28-overdue-shipment-notices/criteria.json b/evals-reference/28-overdue-shipment-notices/criteria.json index 94d1610..fb102f0 100644 --- a/evals-reference/28-overdue-shipment-notices/criteria.json +++ b/evals-reference/28-overdue-shipment-notices/criteria.json @@ -4,47 +4,33 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 5, "description": "Creates OverdueShipmentNotices.java with the requested overdueNotices(List shipments, Clock clock) method, minimal Shipment and ShipmentNotice records, necessary imports, and Java 17-compatible code." }, { "name": "Preserves filtering behavior", - "category": "safety", "max_score": 5, "description": "Includes only shipments with no deliveredAt value and a dueDate before LocalDate.now(clock), and preserves the encounter order of the input list." }, { "name": "Computes notice fields correctly", - "category": "safety", "max_score": 5, "description": "Each notice contains the shipment id, customer email, ChronoUnit.DAYS days-late value from dueDate to today, and severity critical for at least 14 days late or late otherwise." }, { "name": "Uses direct stream result production", - "category": "stream_quality", "max_score": 3, "description": "Uses a direct stream chain that filters overdue shipments and collects or returns the mapped notices without external mutable accumulation, manual indexing, background workers, or unrelated caching." }, { "name": "Avoids multi-line stream lambdas", - "category": "stream_quality", "max_score": 80, "description": "Keeps lambdas passed to stream operations as concise one-expression glue or method references, with no block lambdas using braces inside the stream chain. Non-trivial overdue checks, days-late calculation, severity selection, or notice construction are extracted to named helper methods or equivalent reusable methods." }, { "name": "Keeps API focused", - "category": "maintainability", "max_score": 2, "description": "Keeps the solution limited to the requested class, method, records, and private helpers without broad public API redesign or unnecessary abstractions." } - ], - "metadata": { - "invocation": "natural", - "task_type": "implementation", - "evidence_type": "focused_reference", - "issue": "https://github.com/martinfrancois/java-streams-skill/issues/36", - "reference_selection": "Focused issue #36 behavior-delta coverage for extracting non-trivial stream lambda bodies into helpers.", - "runtime_reference_overlap_rationale": "Allowed only because this is reference-suite focused coverage, not ordinary broad lift or main-suite generalization evidence." - } + ] } diff --git a/evals-regression/01-permission-and-orders/criteria-meta.json b/evals-regression/01-permission-and-orders/criteria-meta.json new file mode 100644 index 0000000..5c030e5 --- /dev/null +++ b/evals-regression/01-permission-and-orders/criteria-meta.json @@ -0,0 +1,14 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "implementation", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Uses short-circuit permission lookup": "stream_quality", + "Uses IntStream range mapping": "stream_quality", + "Uses reduce for BigDecimal": "stream_quality", + "Uses primitive sum for radii": "stream_quality" + } +} diff --git a/evals-regression/01-permission-and-orders/criteria.json b/evals-regression/01-permission-and-orders/criteria.json index c8cec2c..20ccb38 100644 --- a/evals-regression/01-permission-and-orders/criteria.json +++ b/evals-regression/01-permission-and-orders/criteria.json @@ -4,38 +4,28 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 5, "description": "Creates StreamHelpers.java with requested methods and nested types." }, { "name": "Uses short-circuit permission lookup", - "category": "stream_quality", "max_score": 20, "description": "Checks permissions with a stream chain that short-circuits, such as flatMap(...).anyMatch(...), nested anyMatch/contains, or flatMap(...).filter(...).findAny().orElseThrow(...) when it preserves the same behavior clearly." }, { "name": "Uses IntStream range mapping", - "category": "stream_quality", "max_score": 15, "description": "Creates orders with IntStream.range(0, 50).mapToObj(...) or equivalent range stream." }, { "name": "Uses reduce for BigDecimal", - "category": "stream_quality", "max_score": 15, "description": "Sums order totals with map(Order::totalAmount).reduce(BigDecimal.ZERO, BigDecimal::add) or equivalent immutable reduction." }, { "name": "Uses primitive sum for radii", - "category": "stream_quality", "max_score": 20, "description": "Filters/casts Circle values safely and uses mapToDouble(Circle::radius).sum(), not boxed reduce." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "implementation", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/02-product-list-transforms/criteria-meta.json b/evals-regression/02-product-list-transforms/criteria-meta.json new file mode 100644 index 0000000..2532358 --- /dev/null +++ b/evals-regression/02-product-list-transforms/criteria-meta.json @@ -0,0 +1,14 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "implementation", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Uses filter and sorted for favorites": "stream_quality", + "Uses sorted then limit for top products": "stream_quality", + "Uses toSet for unique codes": "stream_quality", + "Uses partitioningBy for availability": "stream_quality" + } +} diff --git a/evals-regression/02-product-list-transforms/criteria.json b/evals-regression/02-product-list-transforms/criteria.json index c500ecc..753fdc7 100644 --- a/evals-regression/02-product-list-transforms/criteria.json +++ b/evals-regression/02-product-list-transforms/criteria.json @@ -4,38 +4,28 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 5, "description": "Creates ProductListTransforms.java with the requested methods and nested records." }, { "name": "Uses filter and sorted for favorites", - "category": "stream_quality", "max_score": 20, "description": "Filters in-stock favorites and sorts with Comparator.comparing(Product::name)." }, { "name": "Uses sorted then limit for top products", - "category": "stream_quality", "max_score": 20, "description": "Sorts by rating descending before limit(3), avoiding subList after full materialization." }, { "name": "Uses toSet for unique codes", - "category": "stream_quality", "max_score": 15, "description": "Maps orders to discount codes and collects directly to a set." }, { "name": "Uses partitioningBy for availability", - "category": "stream_quality", "max_score": 20, "description": "Uses Collectors.partitioningBy(product -> product.stock() > 0) or equivalent boolean partition that returns both keys." } - ], - "metadata": { - "invocation": "natural", - "task_type": "implementation", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/04-primary-address-review/criteria-meta.json b/evals-regression/04-primary-address-review/criteria-meta.json new file mode 100644 index 0000000..349889a --- /dev/null +++ b/evals-regression/04-primary-address-review/criteria-meta.json @@ -0,0 +1,13 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "review", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates review artifact": "safety", + "Rejects collect-then-first": "stream_quality", + "Chooses findFirst to preserve first-match behavior": "stream_quality", + "Names the findAny exception": "stream_quality" + } +} diff --git a/evals-regression/04-primary-address-review/criteria.json b/evals-regression/04-primary-address-review/criteria.json index f8e96cd..3150478 100644 --- a/evals-regression/04-primary-address-review/criteria.json +++ b/evals-regression/04-primary-address-review/criteria.json @@ -4,32 +4,23 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear recommendation." }, { "name": "Rejects collect-then-first", - "category": "stream_quality", "max_score": 20, "description": "Says collecting all primary addresses just to inspect emptiness/read the first result is unnecessary." }, { "name": "Chooses findFirst to preserve first-match behavior", - "category": "stream_quality", "max_score": 20, "description": "Recommends filter(Address::primary).findFirst().orElse(null) to preserve the current get(0) encounter-order behavior." }, { "name": "Names the findAny exception", - "category": "stream_quality", "max_score": 20, "description": "Says findAny is appropriate only if the domain explicitly says all matching primary addresses are equivalent and encounter order does not define which one should win." } - ], - "metadata": { - "invocation": "natural", - "task_type": "review", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/06-inventory-summary/criteria-meta.json b/evals-regression/06-inventory-summary/criteria-meta.json new file mode 100644 index 0000000..78ada5d --- /dev/null +++ b/evals-regression/06-inventory-summary/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "cleanup", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates coherent Java 8 artifact": "safety", + "Preserves summary behavior": "safety", + "Uses anyMatch for existence": "stream_quality", + "Uses count without unread materialization": "stream_quality", + "Uses max instead of sorted temporary list": "stream_quality", + "Uses Collectors.joining": "stream_quality", + "Keeps cleanup focused": "maintainability" + } +} diff --git a/evals-regression/06-inventory-summary/criteria.json b/evals-regression/06-inventory-summary/criteria.json index 20ecf81..c9d873f 100644 --- a/evals-regression/06-inventory-summary/criteria.json +++ b/evals-regression/06-inventory-summary/criteria.json @@ -4,50 +4,38 @@ "checklist": [ { "name": "Creates coherent Java 8 artifact", - "category": "safety", "max_score": 5, "description": "Creates InventorySummary.java with the requested class and no APIs newer than Java 8, such as Stream.toList(), Predicate.not(), records, or List.getFirst()." }, { "name": "Preserves summary behavior", - "category": "safety", "max_score": 10, "description": "Preserves true when any product has stock 0, exact unread count, null newest for an empty product list, newest by updatedAt otherwise, and product categories joined in encounter order with comma-space separators." }, { "name": "Uses anyMatch for existence", - "category": "stream_quality", "max_score": 20, "description": "Replaces the collected out-of-stock list with products.stream().anyMatch(product -> product.stock() == 0) or an equivalent direct match terminal." }, { "name": "Uses count without unread materialization", - "category": "stream_quality", "max_score": 15, "description": "Counts unread messages with filter(...).count() or equivalent direct counting without collecting unread messages only to read size." }, { "name": "Uses max instead of sorted temporary list", - "category": "stream_quality", "max_score": 25, "description": "Computes newest with max(Comparator.comparing(Product::updatedAt)).orElse(null) or an equivalent max terminal rather than sorting all products and reading index 0." }, { "name": "Uses Collectors.joining", - "category": "stream_quality", "max_score": 20, "description": "Builds the category string with products.stream().map(Product::category).collect(Collectors.joining(\", \")) or an equivalent joining collector." }, { "name": "Keeps cleanup focused", - "category": "maintainability", "max_score": 5, "description": "Keeps the existing class structure and avoids unrelated abstractions, dependencies, or behavior changes." } - ], - "metadata": { - "invocation": "natural", - "task_type": "cleanup", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/07-catalog-feed/criteria-meta.json b/evals-regression/07-catalog-feed/criteria-meta.json new file mode 100644 index 0000000..4fb4c5a --- /dev/null +++ b/evals-regression/07-catalog-feed/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "cleanup", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Preserves feed behavior and fixes stated edge cases": "safety", + "Flattens Stream of Optionals directly": "stream_quality", + "Uses distinct before sorted with null filtering": "stream_quality", + "Uses toMap with merge for duplicate categories": "stream_quality", + "Avoids unnecessary temporary collections": "stream_quality", + "Keeps cleanup focused": "maintainability" + } +} diff --git a/evals-regression/07-catalog-feed/criteria.json b/evals-regression/07-catalog-feed/criteria.json index 81f5802..d304410 100644 --- a/evals-regression/07-catalog-feed/criteria.json +++ b/evals-regression/07-catalog-feed/criteria.json @@ -4,50 +4,38 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 5, "description": "Creates CatalogFeed.java with the requested class, method names, records, imports, and Java 17-compatible code." }, { "name": "Preserves feed behavior and fixes stated edge cases", - "category": "safety", "max_score": 10, "description": "visibleCards skips failed normalization and filters active/non-discontinued cards; sortedDiscountCodes excludes nulls and sorts unique codes; cheapestByCategory handles duplicate categories by price." }, { "name": "Flattens Stream of Optionals directly", - "category": "stream_quality", "max_score": 25, "description": "Replaces filter(Optional::isPresent).map(Optional::get) with flatMap(Optional::stream) or an equivalent Java 9+ direct flattening stream chain." }, { "name": "Uses distinct before sorted with null filtering", - "category": "stream_quality", "max_score": 20, "description": "Computes sortedDiscountCodes with a stream that filters Objects::nonNull before natural sorting and removes duplicates with distinct or an equivalent set-plus-safe-sort approach." }, { "name": "Uses toMap with merge for duplicate categories", - "category": "stream_quality", "max_score": 25, "description": "Computes cheapestByCategory with Collectors.toMap and an explicit cheapest-product merge, or an equivalently concise collector-based grouping/min approach that cannot fail on duplicates." }, { "name": "Avoids unnecessary temporary collections", - "category": "stream_quality", "max_score": 10, "description": "Does not introduce list/set materialization just to inspect or unwrap values when a direct stream terminal or collector expresses the operation." }, { "name": "Keeps cleanup focused", - "category": "maintainability", "max_score": 5, "description": "Keeps normalize and records intact except for imports or small helper extraction; avoids unrelated API redesign." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "cleanup", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/09-order-collector-report/criteria-meta.json b/evals-regression/09-order-collector-report/criteria-meta.json new file mode 100644 index 0000000..6cd2461 --- /dev/null +++ b/evals-regression/09-order-collector-report/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "implementation", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Preserves requested report behavior": "safety", + "Uses toMap with merge for cheapest products": "stream_quality", + "Uses groupingBy with mapping for product names": "stream_quality", + "Uses flatMap and counting for item sales": "stream_quality", + "Uses summarizingInt for quantity stats": "stream_quality", + "Keeps implementation readable": "maintainability" + } +} diff --git a/evals-regression/09-order-collector-report/criteria.json b/evals-regression/09-order-collector-report/criteria.json index c2861b4..4be00b1 100644 --- a/evals-regression/09-order-collector-report/criteria.json +++ b/evals-regression/09-order-collector-report/criteria.json @@ -4,50 +4,38 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 5, "description": "Creates OrderCollectorReport.java with the requested build method, nested records, imports for BigDecimal, List, Map, IntSummaryStatistics, and Java 17-compatible code." }, { "name": "Preserves requested report behavior", - "category": "safety", "max_score": 10, "description": "Returns the four requested report values with correct handling for empty orders or products, duplicate categories, duplicate item names, and multiple line item quantities." }, { "name": "Uses toMap with merge for cheapest products", - "category": "stream_quality", "max_score": 25, "description": "Computes cheapestByCategory with a duplicate-safe collector, such as Collectors.toMap(Product::category, Function.identity(), BinaryOperator.minBy(...)) or groupingBy plus a minBy downstream, avoiding duplicate-key failure." }, { "name": "Uses groupingBy with mapping for product names", - "category": "stream_quality", "max_score": 20, "description": "Computes productNamesByCategory with Collectors.groupingBy(Product::category, Collectors.mapping(Product::name, Collectors.toList())) or equivalent downstream mapping." }, { "name": "Uses flatMap and counting for item sales", - "category": "stream_quality", "max_score": 20, "description": "Flattens orders to line items and computes salesByItemName with groupingBy(Item::name, counting()) or an equivalent collector, rather than nested manual map updates. Award full credit if the implementation flattens once into a local item list and reuses it for both salesByItemName and quantityStats." }, { "name": "Uses summarizingInt for quantity stats", - "category": "stream_quality", "max_score": 15, "description": "Computes quantityStats with a flattened item stream and Collectors.summarizingInt(Item::quantity) or an equivalent IntSummaryStatistics stream terminal." }, { "name": "Keeps implementation readable", - "category": "maintainability", "max_score": 5, "description": "Keeps helper code small if used and avoids unrelated abstractions or dependencies." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "implementation", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/10-packet-window-cleanup/criteria-meta.json b/evals-regression/10-packet-window-cleanup/criteria-meta.json new file mode 100644 index 0000000..d56f902 --- /dev/null +++ b/evals-regression/10-packet-window-cleanup/criteria-meta.json @@ -0,0 +1,15 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "cleanup", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Preserves before-prefix behavior": "safety", + "Preserves after-prefix behavior": "safety", + "Uses takeWhile or equivalent prefix-preserving loop": "stream_quality", + "Uses dropWhile or equivalent prefix-skipping loop": "stream_quality", + "Keeps refactor focused": "maintainability" + } +} diff --git a/evals-regression/10-packet-window-cleanup/criteria.json b/evals-regression/10-packet-window-cleanup/criteria.json index 28fddb2..9f5e337 100644 --- a/evals-regression/10-packet-window-cleanup/criteria.json +++ b/evals-regression/10-packet-window-cleanup/criteria.json @@ -4,44 +4,33 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 5, "description": "Creates PacketWindow.java with requested methods, record, imports, and Java 17-compatible code." }, { "name": "Preserves before-prefix behavior", - "category": "safety", "max_score": 5, "description": "beforeFirstLossSpike stops at the first packet above threshold and excludes later packets even if they are below threshold." }, { "name": "Preserves after-prefix behavior", - "category": "safety", "max_score": 5, "description": "afterInitialHealthyPrefix removes only the initial healthy prefix and includes all packets from the first spike onward." }, { "name": "Uses takeWhile or equivalent prefix-preserving loop", - "category": "stream_quality", "max_score": 40, "description": "Implements beforeFirstLossSpike with takeWhile(packet -> packet.loss() <= threshold) or keeps an equivalent break-based loop; does not use filter." }, { "name": "Uses dropWhile or equivalent prefix-skipping loop", - "category": "stream_quality", "max_score": 40, "description": "Implements afterInitialHealthyPrefix with dropWhile(packet -> packet.loss() <= threshold) or keeps an equivalent prefix-skipping loop; does not use filter." }, { "name": "Keeps refactor focused", - "category": "maintainability", "max_score": 5, "description": "Avoids unrelated state, sorting, or packet API changes." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "cleanup", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/11-null-category-price-index/criteria-meta.json b/evals-regression/11-null-category-price-index/criteria-meta.json new file mode 100644 index 0000000..1f198d8 --- /dev/null +++ b/evals-regression/11-null-category-price-index/criteria-meta.json @@ -0,0 +1,15 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "cleanup", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Preserves null and duplicate behavior": "safety", + "Filters null categories before collectors": "stream_quality", + "Uses explicit merge for cheapest product": "stream_quality", + "Uses groupingBy with mapping for names": "stream_quality", + "Keeps implementation readable": "maintainability" + } +} diff --git a/evals-regression/11-null-category-price-index/criteria.json b/evals-regression/11-null-category-price-index/criteria.json index 07125a5..98a0c9b 100644 --- a/evals-regression/11-null-category-price-index/criteria.json +++ b/evals-regression/11-null-category-price-index/criteria.json @@ -4,44 +4,33 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 5, "description": "Creates PriceIndex.java with requested methods, record, imports, and Java 17-compatible code." }, { "name": "Preserves null and duplicate behavior", - "category": "safety", "max_score": 10, "description": "Skips null categories, accepts duplicate categories, keeps cheapest products, and preserves encounter order of names in grouped lists." }, { "name": "Filters null categories before collectors", - "category": "stream_quality", "max_score": 25, "description": "Filters product.category() != null before toMap or groupingBy so the refactor preserves the original null-category skip behavior and avoids groupingBy null classifier failures." }, { "name": "Uses explicit merge for cheapest product", - "category": "stream_quality", "max_score": 30, "description": "Uses Collectors.toMap with a merge function such as BinaryOperator.minBy(Comparator.comparing(Product::price)), or an equivalent grouping/min collector, so duplicate categories cannot throw." }, { "name": "Uses groupingBy with mapping for names", - "category": "stream_quality", "max_score": 25, "description": "Uses groupingBy(Product::category, mapping(Product::name, toList())) or equivalent collector-based grouping without manual computeIfAbsent." }, { "name": "Keeps implementation readable", - "category": "maintainability", "max_score": 5, "description": "Keeps helpers small if used and avoids unrelated abstractions." } - ], - "metadata": { - "invocation": "natural", - "task_type": "cleanup", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/13-training-and-packets/criteria-meta.json b/evals-regression/13-training-and-packets/criteria-meta.json new file mode 100644 index 0000000..2d27b23 --- /dev/null +++ b/evals-regression/13-training-and-packets/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "implementation", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Preserves helper behavior": "safety", + "Flattens company employees directly": "stream_quality", + "Uses clean conditional developer emission": "stream_quality", + "Uses takeWhile for prefix semantics": "stream_quality", + "Uses appropriate collectors": "stream_quality", + "Keeps implementation compact": "maintainability" + } +} diff --git a/evals-regression/13-training-and-packets/criteria.json b/evals-regression/13-training-and-packets/criteria.json index c847665..255db4a 100644 --- a/evals-regression/13-training-and-packets/criteria.json +++ b/evals-regression/13-training-and-packets/criteria.json @@ -4,50 +4,38 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 5, "description": "Creates TrainingAndPackets.java with the requested methods, nested interface/records, imports, and Java 17-compatible code." }, { "name": "Preserves helper behavior", - "category": "safety", "max_score": 10, "description": "Returns only developers with incomplete training as a set of emails and returns only the packet prefix before the first loss spike." }, { "name": "Flattens company employees directly", - "category": "stream_quality", "max_score": 15, "description": "Uses companies.stream().flatMap(company -> company.employees().stream()) or equivalent direct flattening rather than collecting nested employee sets before flattening." }, { "name": "Uses clean conditional developer emission", - "category": "stream_quality", "max_score": 30, "description": "Uses Java 17 mapMulti with pattern matching for Developer, or gives full credit for an equivalently concise filter/cast/map stream chain, without unsafe casts before type checks. A cast after a verified instanceof/Class::isInstance filter is acceptable. Do not deduct merely because the safe equivalent does not use mapMulti, does not reuse a pattern variable, or uses an explicit safe cast in the following map step." }, { "name": "Uses takeWhile for prefix semantics", - "category": "stream_quality", "max_score": 30, "description": "Implements packetsBeforeFirstLossSpike with takeWhile(packet -> packet.loss() <= threshold) or an equivalent prefix-preserving loop, not filter." }, { "name": "Uses appropriate collectors", - "category": "stream_quality", "max_score": 5, "description": "Collects emails to a Set and packets to a List with Java 17-compatible collection terminals." }, { "name": "Keeps implementation compact", - "category": "maintainability", "max_score": 5, "description": "Avoids unrelated abstractions, mutable class state, or behavior not requested." } - ], - "metadata": { - "invocation": "natural", - "task_type": "implementation", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/14-parallel-mutation-review/criteria-meta.json b/evals-regression/14-parallel-mutation-review/criteria-meta.json new file mode 100644 index 0000000..e050f38 --- /dev/null +++ b/evals-regression/14-parallel-mutation-review/criteria-meta.json @@ -0,0 +1,15 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates review artifact": "safety", + "Rejects unsafe proposal": "safety", + "Identifies shared mutable state bug": "stream_quality", + "Explains parallelism is unjustified": "stream_quality", + "Suggests collector-owned accumulation": "stream_quality", + "Review is concise and focused": "maintainability" + } +} diff --git a/evals-regression/14-parallel-mutation-review/criteria.json b/evals-regression/14-parallel-mutation-review/criteria.json index d90f58c..024e0d1 100644 --- a/evals-regression/14-parallel-mutation-review/criteria.json +++ b/evals-regression/14-parallel-mutation-review/criteria.json @@ -4,44 +4,33 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear decision." }, { "name": "Rejects unsafe proposal", - "category": "safety", "max_score": 10, "description": "Clearly says the proposed parallelStream change should not be accepted." }, { "name": "Identifies shared mutable state bug", - "category": "stream_quality", "max_score": 35, "description": "Explains that mutating a shared HashMap from parallel forEach is a data race and can corrupt or lose entries." }, { "name": "Explains parallelism is unjustified", - "category": "stream_quality", "max_score": 20, "description": "Notes there is no evidence of CPU-heavy stateless work and that parallel overhead is not justified for this cache-building operation." }, { "name": "Suggests collector-owned accumulation", - "category": "stream_quality", "max_score": 25, "description": "Recommends products.stream().filter(Product::enabled).collect(Collectors.toMap(Product::id, Function.identity(), ...)) or another collector-owned accumulation approach rather than shared mutable forEach." }, { "name": "Review is concise and focused", - "category": "maintainability", "max_score": 5, "description": "Keeps the review focused on stream safety and the replacement stream chain." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/16-java11-report-review/criteria-meta.json b/evals-regression/16-java11-report-review/criteria-meta.json new file mode 100644 index 0000000..670555e --- /dev/null +++ b/evals-regression/16-java11-report-review/criteria-meta.json @@ -0,0 +1,15 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates review artifact": "safety", + "Rejects proposed change": "stream_quality", + "Catches Java baseline violation": "stream_quality", + "Catches mutability violation": "stream_quality", + "Suggests compatible mutable replacement": "stream_quality", + "Keeps review focused": "maintainability" + } +} diff --git a/evals-regression/16-java11-report-review/criteria.json b/evals-regression/16-java11-report-review/criteria.json index c3675df..1252729 100644 --- a/evals-regression/16-java11-report-review/criteria.json +++ b/evals-regression/16-java11-report-review/criteria.json @@ -4,44 +4,33 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear decision." }, { "name": "Rejects proposed change", - "category": "stream_quality", "max_score": 20, "description": "Clearly says the proposed refactor should not be accepted as written." }, { "name": "Catches Java baseline violation", - "category": "stream_quality", "max_score": 30, "description": "Explains that Stream.toList() is unavailable on the Java 11 baseline because it was added in Java 16." }, { "name": "Catches mutability violation", - "category": "stream_quality", "max_score": 30, "description": "Explains that Stream.toList() returns an unmodifiable list and the code later mutates it with add and sort." }, { "name": "Suggests compatible mutable replacement", - "category": "stream_quality", "max_score": 10, "description": "Recommends Collectors.toCollection(ArrayList::new), Collectors.toList(), or another Java 11-compatible mutable result." }, { "name": "Keeps review focused", - "category": "maintainability", "max_score": 5, "description": "Stays focused on the Java 11 Stream.toList incompatibility and mutable-result behavior. Award full credit when any extra notes are directly tied to this refactor and do not change the reject-as-written recommendation, including a brief note that the now-dead rank sort could be dropped or that imports could be simplified." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/17-java8-optional-prefix-review/criteria-meta.json b/evals-regression/17-java8-optional-prefix-review/criteria-meta.json new file mode 100644 index 0000000..8e7832d --- /dev/null +++ b/evals-regression/17-java8-optional-prefix-review/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates review artifact": "safety", + "Rejects proposed change": "stream_quality", + "Catches Optional.stream incompatibility": "stream_quality", + "Catches takeWhile incompatibility": "stream_quality", + "Catches Stream.toList incompatibility": "stream_quality", + "Preserves prefix semantics in alternative": "stream_quality", + "Keeps review concise": "maintainability" + } +} diff --git a/evals-regression/17-java8-optional-prefix-review/criteria.json b/evals-regression/17-java8-optional-prefix-review/criteria.json index de84e1e..2d7fb75 100644 --- a/evals-regression/17-java8-optional-prefix-review/criteria.json +++ b/evals-regression/17-java8-optional-prefix-review/criteria.json @@ -4,50 +4,38 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear decision." }, { "name": "Rejects proposed change", - "category": "stream_quality", "max_score": 15, "description": "Clearly says the proposed refactor should not be accepted on a Java 8 project." }, { "name": "Catches Optional.stream incompatibility", - "category": "stream_quality", "max_score": 20, "description": "Explains that Optional.stream() is Java 9+, not Java 8." }, { "name": "Catches takeWhile incompatibility", - "category": "stream_quality", "max_score": 25, "description": "Explains that Stream.takeWhile() is Java 9+, not Java 8." }, { "name": "Catches Stream.toList incompatibility", - "category": "stream_quality", "max_score": 20, "description": "Explains that Stream.toList() is Java 16+, not Java 8." }, { "name": "Preserves prefix semantics in alternative", - "category": "stream_quality", "max_score": 10, "description": "Suggests keeping the break-based loop or another Java 8-compatible implementation that stops at the first invisible event rather than filtering all events." }, { "name": "Keeps review concise", - "category": "maintainability", "max_score": 5, "description": "Avoids unrelated modernization feedback." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/18-priority-findany-review/criteria-meta.json b/evals-regression/18-priority-findany-review/criteria-meta.json new file mode 100644 index 0000000..39d9738 --- /dev/null +++ b/evals-regression/18-priority-findany-review/criteria-meta.json @@ -0,0 +1,15 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates review artifact": "safety", + "Rejects proposed change": "stream_quality", + "Identifies findAny priority bug": "stream_quality", + "Identifies parallel ordering risk": "stream_quality", + "Suggests ordered replacement": "stream_quality", + "Keeps review concise": "maintainability" + } +} diff --git a/evals-regression/18-priority-findany-review/criteria.json b/evals-regression/18-priority-findany-review/criteria.json index 72d7901..55e6a40 100644 --- a/evals-regression/18-priority-findany-review/criteria.json +++ b/evals-regression/18-priority-findany-review/criteria.json @@ -4,44 +4,33 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear decision." }, { "name": "Rejects proposed change", - "category": "stream_quality", "max_score": 20, "description": "Clearly says the proposed change should not be accepted because it changes ordering semantics." }, { "name": "Identifies findAny priority bug", - "category": "stream_quality", "max_score": 35, "description": "Explains that findAny may return any matching contact and does not preserve the first configured priority match." }, { "name": "Identifies parallel ordering risk", - "category": "stream_quality", "max_score": 20, "description": "Explains that parallelStream is not justified and makes the findAny nondeterminism especially visible for this ordered lookup." }, { "name": "Suggests ordered replacement", - "category": "stream_quality", "max_score": 15, "description": "Recommends contacts.stream().filter(...).findFirst() or keeping the loop." }, { "name": "Keeps review concise", - "category": "maintainability", "max_score": 5, "description": "Avoids unrelated API redesign." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/19-null-collector-review/criteria-meta.json b/evals-regression/19-null-collector-review/criteria-meta.json new file mode 100644 index 0000000..62dc20f --- /dev/null +++ b/evals-regression/19-null-collector-review/criteria-meta.json @@ -0,0 +1,15 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates review artifact": "safety", + "Rejects proposed change": "stream_quality", + "Catches null-key behavior change": "stream_quality", + "Catches duplicate-key failure": "stream_quality", + "Preserves cheapest merge semantics": "stream_quality", + "Keeps review concise": "maintainability" + } +} diff --git a/evals-regression/19-null-collector-review/criteria.json b/evals-regression/19-null-collector-review/criteria.json index ce06734..8f38c1a 100644 --- a/evals-regression/19-null-collector-review/criteria.json +++ b/evals-regression/19-null-collector-review/criteria.json @@ -4,44 +4,33 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear decision." }, { "name": "Rejects proposed change", - "category": "stream_quality", "max_score": 15, "description": "Clearly says the proposed toMap refactor should not be accepted as written." }, { "name": "Catches null-key behavior change", - "category": "stream_quality", "max_score": 25, "description": "Explains that null categories were skipped and must be filtered or handled before collecting." }, { "name": "Catches duplicate-key failure", - "category": "stream_quality", "max_score": 25, "description": "Explains that duplicate categories are expected and two-argument toMap throws without a merge function." }, { "name": "Preserves cheapest merge semantics", - "category": "stream_quality", "max_score": 25, "description": "Recommends filtering category != null and using a merge function such as BinaryOperator.minBy(Comparator.comparing(Product::price))." }, { "name": "Keeps review concise", - "category": "maintainability", "max_score": 5, "description": "Avoids unrelated collector commentary. Award full credit when the review stays focused on the accept/reject decision, null-key behavior, duplicate-key behavior, merge semantics, and directly related code clarity such as Function.identity() versus a lambda. Do not deduct merely because the review includes a short corrected collector snippet or a brief explanation of why it preserves the loop behavior." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/20-parallel-shared-state-review/criteria-meta.json b/evals-regression/20-parallel-shared-state-review/criteria-meta.json new file mode 100644 index 0000000..47fd6fc --- /dev/null +++ b/evals-regression/20-parallel-shared-state-review/criteria-meta.json @@ -0,0 +1,15 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates review artifact": "safety", + "Rejects proposed change": "stream_quality", + "Catches shared HashMap race": "stream_quality", + "Rejects unjustified parallelism": "stream_quality", + "Suggests collector-owned accumulation": "stream_quality", + "Keeps review concise": "maintainability" + } +} diff --git a/evals-regression/20-parallel-shared-state-review/criteria.json b/evals-regression/20-parallel-shared-state-review/criteria.json index 564ed36..dbba04f 100644 --- a/evals-regression/20-parallel-shared-state-review/criteria.json +++ b/evals-regression/20-parallel-shared-state-review/criteria.json @@ -4,44 +4,33 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear decision." }, { "name": "Rejects proposed change", - "category": "stream_quality", "max_score": 20, "description": "Clearly says the proposed parallelStream change should not be accepted." }, { "name": "Catches shared HashMap race", - "category": "stream_quality", "max_score": 35, "description": "Explains that parallel forEach mutates a shared HashMap and can race, corrupt state, or lose entries." }, { "name": "Rejects unjustified parallelism", - "category": "stream_quality", "max_score": 15, "description": "Notes that no CPU-heavy stateless work or measured benefit justifies parallelStream here." }, { "name": "Suggests collector-owned accumulation", - "category": "stream_quality", "max_score": 20, "description": "Recommends a sequential stream with Collectors.toMap(Shipment::id, Function.identity(), merge) or another collector-owned accumulation approach." }, { "name": "Keeps review concise", - "category": "maintainability", "max_score": 5, "description": "Avoids unrelated concurrency redesign." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/22-java8-version-scan/criteria-meta.json b/evals-regression/22-java8-version-scan/criteria-meta.json new file mode 100644 index 0000000..7f42ed8 --- /dev/null +++ b/evals-regression/22-java8-version-scan/criteria-meta.json @@ -0,0 +1,15 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "skill_context_dependent" + }, + "categories": { + "Creates review artifact": "safety", + "Includes skill-provided scan command": "stream_quality", + "Classifies Java 8 version drift": "stream_quality", + "Avoids multi-line lambda replacements": "maintainability", + "Classifies allowed count usage": "stream_quality", + "Keeps audit concise": "maintainability" + } +} diff --git a/evals-regression/22-java8-version-scan/criteria.json b/evals-regression/22-java8-version-scan/criteria.json index 318a533..ac69a6e 100644 --- a/evals-regression/22-java8-version-scan/criteria.json +++ b/evals-regression/22-java8-version-scan/criteria.json @@ -4,44 +4,33 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a Java 8 compatibility audit." }, { "name": "Includes skill-provided scan command", - "category": "stream_quality", "max_score": 20, "description": "Includes the scan header 'java-streams hard-stop scan v1' and the complete rg command from hard-stops.md, including the full marker regex and '' placeholder. Equivalent shell escaping is acceptable, but do not award this for a reduced, approximate, or hand-written scan command." }, { "name": "Classifies Java 8 version drift", - "category": "stream_quality", "max_score": 50, "description": "Flags Optional::stream, Stream.toList, Stream.ofNullable, takeWhile, dropWhile, Collectors.flatMapping, Collectors.teeing, and mapMulti as unavailable on Java 8 and gives Java 8-compatible directions." }, { "name": "Avoids multi-line lambda replacements", - "category": "maintainability", "max_score": 10, "description": "Keeps Java 8 replacement snippets free of multi-line lambdas. Uses helper methods, method references, or same-line one-expression lambdas instead of block lambdas, arrows whose body starts on the next line, or lambda bodies that continue a nested stream chain on later lines." }, { "name": "Classifies allowed count usage", - "category": "stream_quality", "max_score": 10, "description": "Mentions activeCount uses count() as the requested numeric result, not count() > 0, so it is acceptable Java 8 stream code even though it is not a hard-stop scan hit." }, { "name": "Keeps audit concise", - "category": "maintainability", "max_score": 5, "description": "Focuses on version drift and directly related Java 8 replacements." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "skill_context_dependent" - } + ] } diff --git a/evals-regression/23-collector-order-scan/criteria-meta.json b/evals-regression/23-collector-order-scan/criteria-meta.json new file mode 100644 index 0000000..459df78 --- /dev/null +++ b/evals-regression/23-collector-order-scan/criteria-meta.json @@ -0,0 +1,14 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "skill_context_dependent" + }, + "categories": { + "Creates review artifact": "safety", + "Includes skill-provided scan command": "stream_quality", + "Classifies required collector and ordering fixes": "stream_quality", + "Classifies acceptable markers": "stream_quality", + "Keeps audit concise": "maintainability" + } +} diff --git a/evals-regression/23-collector-order-scan/criteria.json b/evals-regression/23-collector-order-scan/criteria.json index 52101c9..2878949 100644 --- a/evals-regression/23-collector-order-scan/criteria.json +++ b/evals-regression/23-collector-order-scan/criteria.json @@ -4,38 +4,28 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a complete scan-hit classification." }, { "name": "Includes skill-provided scan command", - "category": "stream_quality", "max_score": 20, "description": "Includes the scan header 'java-streams hard-stop scan v1' and the complete rg command from hard-stops.md, including the full marker regex and '' placeholder. Equivalent shell escaping is acceptable, but do not award this for a reduced, approximate, or hand-written scan command." }, { "name": "Classifies required collector and ordering fixes", - "category": "stream_quality", "max_score": 45, "description": "Flags byDesk duplicate toMap, byDepartment nullable groupingBy, cheapestDesk sorted-findFirst, topFiveByScore limit before sort, sortedNicknames nullable natural sort, and stableDepartments sorted-before-distinct inefficiency." }, { "name": "Classifies acceptable markers", - "category": "stream_quality", "max_score": 25, "description": "Identifies byId toMap as acceptable due unique id, byStatus groupingBy as acceptable due non-null status and intended counting aggregate, and cpuScore parallelStream as acceptable due large CPU-heavy stateless non-blocking work with a safe sum terminal." }, { "name": "Keeps audit concise", - "category": "maintainability", "max_score": 5, "description": "Separates required fixes from acceptable markers without unrelated rewrites." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "skill_context_dependent" - } + ] } diff --git a/evals-regression/24-mutable-batch-modernization/criteria-meta.json b/evals-regression/24-mutable-batch-modernization/criteria-meta.json new file mode 100644 index 0000000..89b6778 --- /dev/null +++ b/evals-regression/24-mutable-batch-modernization/criteria-meta.json @@ -0,0 +1,13 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "cleanup", + "evidence_type": "solved_regression" + }, + "categories": { + "Creates requested artifact": "safety", + "Preserves mutable batch behavior": "stream_quality", + "Avoids Stream.toList mutability break": "stream_quality", + "Keeps cleanup focused": "maintainability" + } +} diff --git a/evals-regression/24-mutable-batch-modernization/criteria.json b/evals-regression/24-mutable-batch-modernization/criteria.json index 4fb6f77..de93933 100644 --- a/evals-regression/24-mutable-batch-modernization/criteria.json +++ b/evals-regression/24-mutable-batch-modernization/criteria.json @@ -4,32 +4,23 @@ "checklist": [ { "name": "Creates requested artifact", - "category": "safety", "max_score": 10, "description": "Creates ExportBatch.java with coherent Java 17-compatible content." }, { "name": "Preserves mutable batch behavior", - "category": "stream_quality", "max_score": 55, "description": "ExportBatch.java keeps only enabled rows, sorts by position before optional footer append, appends the footer only when requested, then sorts the final mutable result by id." }, { "name": "Avoids Stream.toList mutability break", - "category": "stream_quality", "max_score": 30, "description": "Does not use Stream.toList() for exportRows because the list is later mutated with add and sort; uses Collectors.toCollection(ArrayList::new), Collectors.toList(), or an equivalent mutable result." }, { "name": "Keeps cleanup focused", - "category": "maintainability", "max_score": 5, "description": "Avoids unrelated API redesign, dependencies, or broad rewrites." } - ], - "metadata": { - "invocation": "natural", - "task_type": "cleanup", - "evidence_type": "solved_regression" - } + ] } diff --git a/evals-regression/25-hard-stop-scan-audit/criteria-meta.json b/evals-regression/25-hard-stop-scan-audit/criteria-meta.json new file mode 100644 index 0000000..071beff --- /dev/null +++ b/evals-regression/25-hard-stop-scan-audit/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "evidence_type": "skill_context_dependent", + "main_eval_weight_multiplier": 1, + "reference_selection": "demoted from the main eval set because exact skill-provided scan command recall is not a fair without-context benchmark target" + }, + "categories": { + "Creates review artifact": "safety", + "Includes skill-provided scan command": "stream_quality", + "Classifies required customer fixes": "stream_quality", + "Classifies acceptable marker": "stream_quality", + "Keeps audit concise": "maintainability" + } +} diff --git a/evals-regression/25-hard-stop-scan-audit/criteria.json b/evals-regression/25-hard-stop-scan-audit/criteria.json index 917b8b0..53f58bc 100644 --- a/evals-regression/25-hard-stop-scan-audit/criteria.json +++ b/evals-regression/25-hard-stop-scan-audit/criteria.json @@ -4,40 +4,28 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 5, "description": "Creates review.md with a clear hard-stop audit." }, { "name": "Includes skill-provided scan command", - "category": "stream_quality", "max_score": 20, "description": "Includes the scan header 'java-streams hard-stop scan v1' and the rg command from hard-stops.md with complete marker coverage and the '' placeholder. Equivalent shell escaping or formatting is acceptable; do not require character-perfect parity if all marker categories are present, but do not award this for a reduced, approximate, or hand-written scan command." }, { "name": "Classifies required customer fixes", - "category": "stream_quality", "max_score": 60, "description": "Correctly flags the customer class issues: existence checks, sorted-findFirst newest lookup, collect-then-join, Stream.toList mutation, parallel findAny priority break, duplicate toMap, nullable groupingBy, nullable natural sort, limit-before-sort top-N, and Optional get pattern." }, { "name": "Classifies acceptable marker", - "category": "stream_quality", "max_score": 10, "description": "Explicitly identifies totalSpend's BigDecimal reduce as acceptable rather than forcing primitive aggregation, whether it is framed as an acceptable scan hit or as an intentional non-hit that should remain unchanged." }, { "name": "Keeps audit concise", - "category": "maintainability", "max_score": 5, "description": "Does not rewrite the class wholesale or add unrelated Java style feedback." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "evidence_type": "skill_context_dependent", - "main_eval_weight_multiplier": 1, - "reference_selection": "demoted from the main eval set because exact skill-provided scan command recall is not a fair without-context benchmark target" - } + ] } diff --git a/evals/01-offer-availability-mapconcurrent/criteria-meta.json b/evals/01-offer-availability-mapconcurrent/criteria-meta.json new file mode 100644 index 0000000..a04e01f --- /dev/null +++ b/evals/01-offer-availability-mapconcurrent/criteria-meta.json @@ -0,0 +1,18 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "implementation", + "evidence_type": "focused_main", + "main_eval_weight_multiplier": 5, + "main_eval_selection": "kept for bounded remote-check coverage; rerun hosted evals before updating benchmark claims", + "runtime_reference_overlap_rationale": "Focused main-suite coverage for bounded mapConcurrent service-stub usage; not ordinary broad lift." + }, + "categories": { + "Creates coherent Java 24 artifact": "safety", + "Preserves offer behavior": "safety", + "Uses bounded Gatherers.mapConcurrent": "stream_quality", + "Carries offer with availability decision": "stream_quality", + "Rejects weak concurrency choices": "stream_quality", + "Keeps implementation focused": "maintainability" + } +} diff --git a/evals/01-offer-availability-mapconcurrent/criteria.json b/evals/01-offer-availability-mapconcurrent/criteria.json index dd5ec1d..767d98d 100644 --- a/evals/01-offer-availability-mapconcurrent/criteria.json +++ b/evals/01-offer-availability-mapconcurrent/criteria.json @@ -4,47 +4,33 @@ "checklist": [ { "name": "Creates coherent Java 24 artifact", - "category": "safety", "max_score": 25, "description": "Creates OfferAvailability.java with the requested method, nested types, imports, and Java 24-compatible code." }, { "name": "Preserves offer behavior", - "category": "safety", "max_score": 50, "description": "Returns only offers whose AvailabilityApi.lookup decision has show() true and sorts the final returned offers by rank, then id." }, { "name": "Uses bounded Gatherers.mapConcurrent", - "category": "stream_quality", "max_score": 225, "description": "Uses Gatherers.mapConcurrent with an explicit bounded concurrency value of at most 8 to perform the blocking AvailabilityApi.lookup calls inside the stream chain." }, { "name": "Carries offer with availability decision", - "category": "stream_quality", "max_score": 100, "description": "Keeps each Offer associated with its Availability decision using a clear non-null carrier such as a small record, Map.Entry, or equivalent pair, then filters by Availability.show() before mapping back to Offer." }, { "name": "Rejects weak concurrency choices", - "category": "stream_quality", "max_score": 75, "description": "Does not use parallelStream(), .parallel(), unbounded CompletableFuture fan-out, or shared mutable state for collecting remote check results." }, { "name": "Keeps implementation focused", - "category": "maintainability", "max_score": 25, "description": "Avoids unrelated caching, retries, dependencies, API redesign, or broad concurrency framework changes." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "implementation", - "evidence_type": "focused_main", - "main_eval_weight_multiplier": 5, - "main_eval_selection": "kept for bounded remote-check coverage; rerun hosted evals before updating benchmark claims", - "runtime_reference_overlap_rationale": "Focused main-suite coverage for bounded mapConcurrent service-stub usage; not ordinary broad lift." - } + ] } diff --git a/evals/02-delivery-appointments-mapconcurrent/criteria-meta.json b/evals/02-delivery-appointments-mapconcurrent/criteria-meta.json new file mode 100644 index 0000000..ff1f4cd --- /dev/null +++ b/evals/02-delivery-appointments-mapconcurrent/criteria-meta.json @@ -0,0 +1,18 @@ +{ + "metadata": { + "invocation": "natural", + "task_type": "implementation", + "evidence_type": "focused_main", + "main_eval_weight_multiplier": 4, + "main_eval_selection": "active replacement for a runtime-reference-overlapping bounded remote-check scenario; rerun hosted evals before updating benchmark claims", + "runtime_reference_overlap_rationale": "Focused main-suite coverage for bounded mapConcurrent service-stub usage; not ordinary broad natural lift." + }, + "categories": { + "Creates coherent Java 24 artifact": "safety", + "Preserves appointment behavior": "safety", + "Uses bounded Gatherers.mapConcurrent": "stream_quality", + "Carries appointment with schedule result": "stream_quality", + "Rejects weak concurrency choices": "stream_quality", + "Keeps implementation focused": "maintainability" + } +} diff --git a/evals/02-delivery-appointments-mapconcurrent/criteria.json b/evals/02-delivery-appointments-mapconcurrent/criteria.json index 8d430b8..213851d 100644 --- a/evals/02-delivery-appointments-mapconcurrent/criteria.json +++ b/evals/02-delivery-appointments-mapconcurrent/criteria.json @@ -4,47 +4,33 @@ "checklist": [ { "name": "Creates coherent Java 24 artifact", - "category": "safety", "max_score": 20, "description": "Creates DeliveryAppointments.java with the requested method, nested types, imports, and Java 24-compatible code." }, { "name": "Preserves appointment behavior", - "category": "safety", "max_score": 40, "description": "Returns only appointments whose token passes CalendarService.canSchedule and sorts the final returned appointments by Appointment::startsAt, then Appointment::token." }, { "name": "Uses bounded Gatherers.mapConcurrent", - "category": "stream_quality", "max_score": 180, "description": "Uses Gatherers.mapConcurrent with an explicit max concurrency no greater than 8 to perform the blocking CalendarService.canSchedule calls inside the stream chain." }, { "name": "Carries appointment with schedule result", - "category": "stream_quality", "max_score": 80, "description": "Keeps each Appointment associated with its schedule-check result using a non-null holder, then filters by the boolean result before mapping back to Appointment." }, { "name": "Rejects weak concurrency choices", - "category": "stream_quality", "max_score": 60, "description": "Does not use parallelStream(), .parallel(), unbounded CompletableFuture fan-out, or shared mutable state for collecting remote check results." }, { "name": "Keeps implementation focused", - "category": "maintainability", "max_score": 20, "description": "Avoids unrelated caching, retries, dependencies, API redesign, or broad concurrency framework changes." } - ], - "metadata": { - "invocation": "natural", - "task_type": "implementation", - "evidence_type": "focused_main", - "main_eval_weight_multiplier": 4, - "main_eval_selection": "active replacement for a runtime-reference-overlapping bounded remote-check scenario; rerun hosted evals before updating benchmark claims", - "runtime_reference_overlap_rationale": "Focused main-suite coverage for bounded mapConcurrent service-stub usage; not ordinary broad natural lift." - } + ] } diff --git a/evals/03-payment-screening-gatherer-review/criteria-meta.json b/evals/03-payment-screening-gatherer-review/criteria-meta.json new file mode 100644 index 0000000..c0c2cc3 --- /dev/null +++ b/evals/03-payment-screening-gatherer-review/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "review", + "main_eval_weight_multiplier": 3, + "main_eval_selection": "active replacement for a runtime-reference-overlapping remote stock review; rerun hosted evals before updating benchmark claims" + }, + "categories": { + "Creates review artifact": "safety", + "Preserves required output behavior": "safety", + "Rejects parallelStream for blocking IO": "stream_quality", + "Recommends bounded Gatherers.mapConcurrent": "stream_quality", + "Shows a safe stream chain": "stream_quality", + "Review is concise and actionable": "maintainability" + } +} diff --git a/evals/03-payment-screening-gatherer-review/criteria.json b/evals/03-payment-screening-gatherer-review/criteria.json index 7a2f4dc..4ae86b2 100644 --- a/evals/03-payment-screening-gatherer-review/criteria.json +++ b/evals/03-payment-screening-gatherer-review/criteria.json @@ -4,45 +4,33 @@ "checklist": [ { "name": "Creates review artifact", - "category": "safety", "max_score": 15, "description": "Creates review.md with a clear decision and recommendation." }, { "name": "Preserves required output behavior", - "category": "safety", "max_score": 30, "description": "Recognizes that the final payment list must still contain only payments approved by FraudApi.approve and sorted by Payment::submittedAt." }, { "name": "Rejects parallelStream for blocking IO", - "category": "stream_quality", "max_score": 75, "description": "Rejects the proposed parallelStream change and explains that it uses the common fork-join pool, remains blocking, and is not the right default for remote API calls." }, { "name": "Recommends bounded Gatherers.mapConcurrent", - "category": "stream_quality", "max_score": 120, "description": "Recommends a Java 24 Gatherers.mapConcurrent stream chain with an explicit concurrency bound for the blocking fraud-screening calls." }, { "name": "Shows a safe stream chain", - "category": "stream_quality", "max_score": 45, "description": "Describes or sketches carrying each Payment with the fraud-screening result using a non-null holder, filtering by that result, mapping back to Payment, sorting by submittedAt, and collecting with toList." }, { "name": "Review is concise and actionable", - "category": "maintainability", "max_score": 15, "description": "Keeps the review focused on the stream/concurrency decision and gives a concrete path forward." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "review", - "main_eval_weight_multiplier": 3, - "main_eval_selection": "active replacement for a runtime-reference-overlapping remote stock review; rerun hosted evals before updating benchmark claims" - } + ] } diff --git a/evals/04-invoice-bounds-and-temperature-windows/criteria-meta.json b/evals/04-invoice-bounds-and-temperature-windows/criteria-meta.json new file mode 100644 index 0000000..c1e2da6 --- /dev/null +++ b/evals/04-invoice-bounds-and-temperature-windows/criteria-meta.json @@ -0,0 +1,16 @@ +{ + "metadata": { + "invocation": "explicit", + "task_type": "implementation", + "main_eval_weight_multiplier": 2, + "main_eval_selection": "active replacement for a runtime-reference-overlapping teeing/prefix scenario; rerun hosted evals before updating benchmark claims" + }, + "categories": { + "Creates coherent Java 17 artifact": "safety", + "Uses teeing for invoice bounds": "stream_quality", + "Handles empty invoice bounds": "safety", + "Uses takeWhile for before-overheat prefix": "stream_quality", + "Uses dropWhile for after-safe-run remainder": "stream_quality", + "Keeps implementation focused": "maintainability" + } +} diff --git a/evals/04-invoice-bounds-and-temperature-windows/criteria.json b/evals/04-invoice-bounds-and-temperature-windows/criteria.json index efa2463..386aa81 100644 --- a/evals/04-invoice-bounds-and-temperature-windows/criteria.json +++ b/evals/04-invoice-bounds-and-temperature-windows/criteria.json @@ -4,45 +4,33 @@ "checklist": [ { "name": "Creates coherent Java 17 artifact", - "category": "safety", "max_score": 10, "description": "Creates InvoiceBoundsAndTemperatures.java with requested methods and records." }, { "name": "Uses teeing for invoice bounds", - "category": "stream_quality", "max_score": 70, "description": "Uses Collectors.teeing with minBy and maxBy by Invoice::total, or an equivalent single-pass pair reduction." }, { "name": "Handles empty invoice bounds", - "category": "safety", "max_score": 20, "description": "Returns a Bounds whose low and high values are null for an empty invoice list." }, { "name": "Uses takeWhile for before-overheat prefix", - "category": "stream_quality", "max_score": 45, "description": "Uses takeWhile(reading -> reading.temperatureCelsius() <= maxSafeTemperature) or an equivalent prefix-preserving loop, not filter." }, { "name": "Uses dropWhile for after-safe-run remainder", - "category": "stream_quality", "max_score": 45, "description": "Uses dropWhile(reading -> reading.temperatureCelsius() <= maxSafeTemperature) or an equivalent prefix-skipping loop, not a filter that removes later safe readings." }, { "name": "Keeps implementation focused", - "category": "maintainability", "max_score": 10, "description": "Keeps the implementation focused on the requested stream helpers without unrelated abstractions, caching, concurrency, or API redesign." } - ], - "metadata": { - "invocation": "explicit", - "task_type": "implementation", - "main_eval_weight_multiplier": 2, - "main_eval_selection": "active replacement for a runtime-reference-overlapping teeing/prefix scenario; rerun hosted evals before updating benchmark claims" - } + ] } diff --git a/scripts/classify_eval_result.py b/scripts/classify_eval_result.py index ba78636..7bf3dbd 100755 --- a/scripts/classify_eval_result.py +++ b/scripts/classify_eval_result.py @@ -50,11 +50,11 @@ def scenario_text_from_dir(path: Path | None) -> str: def scenario_metadata_from_dir(path: Path | None) -> dict[str, Any]: if path is None: return {} - criteria_path = path / "criteria.json" - if not criteria_path.is_file(): + sidecar_path = path / "criteria-meta.json" + if not sidecar_path.is_file(): return {} try: - data = json.loads(criteria_path.read_text(encoding="utf-8")) + data = json.loads(sidecar_path.read_text(encoding="utf-8")) except json.JSONDecodeError: return {} metadata = data.get("metadata") diff --git a/scripts/eval_impact.py b/scripts/eval_impact.py index 1cac536..221a6e0 100755 --- a/scripts/eval_impact.py +++ b/scripts/eval_impact.py @@ -157,7 +157,7 @@ def changed_text(repo_root: Path, base_ref: str, head_ref: str, paths: list[Path def scenario_text(path: Path) -> str: parts: list[str] = [path.name] - for filename in ("task.md", "capability.txt", "criteria.json"): + for filename in ("task.md", "capability.txt", "criteria.json", "criteria-meta.json"): file_path = path / filename if not file_path.is_file(): continue diff --git a/scripts/test_validate_eval_criteria.py b/scripts/test_validate_eval_criteria.py index 4b08154..70ceeed 100644 --- a/scripts/test_validate_eval_criteria.py +++ b/scripts/test_validate_eval_criteria.py @@ -106,9 +106,12 @@ def write_scenario( "description": "Uses clear stream code.", }, ], - "metadata": metadata, } + categories = {item["name"]: item.pop("category") for item in criteria["checklist"] if "category" in item} (scenario / "criteria.json").write_text(json.dumps(criteria, indent=2) + "\n", encoding="utf-8") + (scenario / "criteria-meta.json").write_text( + json.dumps({"metadata": metadata, "categories": categories}, indent=2) + "\n", encoding="utf-8" + ) return scenario diff --git a/scripts/validate_eval_criteria.py b/scripts/validate_eval_criteria.py index 758bdb9..fcf7db5 100755 --- a/scripts/validate_eval_criteria.py +++ b/scripts/validate_eval_criteria.py @@ -167,6 +167,60 @@ def load_json(path: Path) -> tuple[dict[str, Any] | None, list[str]]: return data, [] +SIDECAR_NAME = "criteria-meta.json" + + +def load_criteria(criteria_file: Path) -> tuple[dict[str, Any] | None, list[str]]: + """Load criteria.json and merge the repository metadata kept in criteria-meta.json. + + Tessl's schema for criteria.json is context, type, and checklist items with name, + description, and max_score. Everything this repository adds (invocation, task type, + evidence type, selection notes, per-item categories) lives in criteria-meta.json next to + it, so `tessl eval lint` stays clean. The merged view is what every validator reads. + """ + data, failures = load_json(criteria_file) + if data is None: + return None, failures + if "metadata" in data: + failures.append(f"{criteria_file}: move the metadata object to {SIDECAR_NAME}") + checklist = data.get("checklist") + items = [item for item in checklist if isinstance(item, dict)] if isinstance(checklist, list) else [] + if any("category" in item for item in items): + failures.append(f"{criteria_file}: move checklist categories to {SIDECAR_NAME}") + sidecar_file = criteria_file.with_name(SIDECAR_NAME) + if not sidecar_file.is_file(): + failures.append(f"{sidecar_file}: missing; it must carry metadata and categories") + data.setdefault("metadata", {}) + return data, failures + sidecar, sidecar_failures = load_json(sidecar_file) + failures.extend(sidecar_failures) + if sidecar is None: + data.setdefault("metadata", {}) + return data, failures + unknown = sorted(set(sidecar) - {"metadata", "categories"}) + if unknown: + failures.append(f"{sidecar_file}: unknown keys {unknown}; use metadata and categories") + metadata = sidecar.get("metadata") + if not isinstance(metadata, dict): + failures.append(f"{sidecar_file}: metadata must be an object") + metadata = {} + categories = sidecar.get("categories") + if not isinstance(categories, dict): + failures.append(f"{sidecar_file}: categories must map checklist names to categories") + categories = {} + names = {str(item.get("name")) for item in items} + for name in sorted(set(categories) - names): + failures.append(f"{sidecar_file}: category for unknown checklist item {name!r}") + for item in items: + category = categories.get(str(item.get("name"))) + if category is None: + failures.append(f"{sidecar_file}: no category for checklist item {item.get('name')!r}") + else: + item["category"] = category + data["metadata"] = metadata + return data, failures + + def invocation_from_task(task_text: str) -> bool: return any(re.search(pattern, task_text, re.IGNORECASE) for pattern in EXPLICIT_INVOCATION_PATTERNS) @@ -290,7 +344,7 @@ def validate_scenario(scenario: Path, main_eval_root: Path | None) -> list[str]: if not criteria_file.is_file(): return failures - data, json_failures = load_json(criteria_file) + data, json_failures = load_criteria(criteria_file) failures.extend(json_failures) if data is None: return failures @@ -485,7 +539,7 @@ def validate_runtime_reference_overlap(dirs: list[Path]) -> list[str]: task_text = task_file.read_text(encoding="utf-8") metadata: dict[str, Any] = {} if criteria_file.exists(): - data, _ = load_json(criteria_file) + data, _ = load_criteria(criteria_file) if isinstance(data, dict) and isinstance(data.get("metadata"), dict): metadata = data["metadata"] @@ -646,7 +700,7 @@ def main() -> int: failures.extend(validate_scenario(scenario, main_eval_root)) criteria = scenario / "criteria.json" if criteria.exists(): - data, _ = load_json(criteria) + data, _ = load_criteria(criteria) if data and isinstance(data.get("metadata"), dict): invocation = data["metadata"].get("invocation") if invocation in invocations: @@ -657,7 +711,7 @@ def main() -> int: main_eval_invocations = {"natural": 0, "explicit": 0} main_eval_category_scores = {category: 0 for category in CRITERION_CATEGORIES} for scenario in main_eval_dirs: - data, _ = load_json(scenario / "criteria.json") + data, _ = load_criteria(scenario / "criteria.json") if data and isinstance(data.get("metadata"), dict): invocation = data["metadata"].get("invocation") if invocation in main_eval_invocations: