{"schema":"alhena-research-lab/policy-study-catalog-v1","publisher":"Alhena Research Lab","description":"Approved, separately versioned public-session studies. PCR, quality, speed and composite are distinct measures. Detailed evidence requires verified work email.","studies":[{"schema":"alhena-research-lab/policy-study-summary-v1","protocol":"policy-resolution-v1","slug":"alhena-gorgias-policy-resolution-2026-09-21","title":"Alhena and Gorgias: policy-compliant resolution","description":"A study of shopping and support on ten storefront deployments, with conversation evidence, public merchant policies, fixed quality criteria and separate response timing.","publishedAt":"2026-09-22T04:07:09.170Z","captureStartAt":"2026-09-21T14:13:43.224Z","captureEndAt":"2026-09-22T00:30:54.972Z","commissionedBy":"Alhena Research Lab","method":{"status":"final","sha256":"712b1e5753cd9b871881d328d79ecb20ccbb279c045fe9233891c6805441133b","sourceCommit":"19b1420d2520d48baa52be81ac33fc4b9bd0ff8b","differences":["Replaces the source automation classifier with policy-compliant resolution.","Every eligible checkpoint receives two fresh blind judgments; attainment requires agreement and valid evidence.","Cause-selected capture repairs retained: 12. The source study remains unchanged."]},"sample":{"plannedCoreContexts":100,"capturedCoreContexts":100,"guardrailContexts":10,"judgedCoreContexts":88,"pcrDecisions":836,"auditedPcrDecisions":836},"providers":[{"id":"alhena","name":"Alhena","website":"https://alhena.ai/","registeredStores":5,"shopping":{"policyResolution":{"value":64.8,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":83,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":53.7,"explanation":"Full-answer completion 11.8 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":68.4,"explanation":"40% policy-compliant resolution, 35% quality, 25% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":242,"observedCheckpoints":242,"submittedCheckpoints":242,"assessedCheckpoints":242,"unassessableCheckpoints":0,"attainedCheckpoints":158,"policyUnverifiedCheckpoints":64,"includedContexts":25,"excludedContexts":0,"includedStores":5,"excludedStores":0,"qualityEligibleContexts":25,"qualityEligibleStores":5,"originalCaptures":23,"repairedCaptures":2}},"support":{"policyResolution":{"value":72.8,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":95,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":70,"explanation":"Full-answer completion 8.7 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":81.4,"explanation":"50% policy-compliant resolution, 40% quality, 10% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":250,"observedCheckpoints":250,"submittedCheckpoints":250,"assessedCheckpoints":250,"unassessableCheckpoints":0,"attainedCheckpoints":182,"policyUnverifiedCheckpoints":63,"includedContexts":25,"excludedContexts":0,"includedStores":5,"excludedStores":0,"qualityEligibleContexts":25,"qualityEligibleStores":5,"originalCaptures":19,"repairedCaptures":6}},"overallComposite":{"value":74.9,"explanation":"Equal mean of the two unrounded lane composites; only available when both lanes meet coverage floors."}},{"id":"gorgias","name":"Gorgias","website":"https://www.gorgias.com/","registeredStores":5,"shopping":{"policyResolution":{"value":60.8,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":72,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":25.3,"explanation":"Full-answer completion 17.2 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":55.8,"explanation":"40% policy-compliant resolution, 35% quality, 25% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":187,"observedCheckpoints":187,"submittedCheckpoints":187,"assessedCheckpoints":164,"unassessableCheckpoints":23,"attainedCheckpoints":109,"policyUnverifiedCheckpoints":37,"includedContexts":20,"excludedContexts":5,"includedStores":4,"excludedStores":1,"qualityEligibleContexts":17,"qualityEligibleStores":4,"originalCaptures":23,"repairedCaptures":2}},"support":{"policyResolution":{"value":68,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":94,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":41.6,"explanation":"Full-answer completion 14.1 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":75.8,"explanation":"50% policy-compliant resolution, 40% quality, 10% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":205,"observedCheckpoints":205,"submittedCheckpoints":205,"assessedCheckpoints":180,"unassessableCheckpoints":25,"attainedCheckpoints":122,"policyUnverifiedCheckpoints":45,"includedContexts":18,"excludedContexts":7,"includedStores":4,"excludedStores":1,"qualityEligibleContexts":18,"qualityEligibleStores":4,"originalCaptures":23,"repairedCaptures":2}},"overallComposite":{"value":65.8,"explanation":"Equal mean of the two unrounded lane composites; only available when both lanes meet coverage floors."}}],"limitations":["Alhena commissioned and operates this study. Selected storefront deployments do not establish a universal provider ranking or independent certification.","This methodology was developed after reviewing the original automation results, then frozen before the new policy-resolution judgments.","The sessions were logged out. No real order, refund, account change or completed human resolution was independently verified.","Unknown delivery or actor attribution remains unassessable. Unsent questions are not successes or failures. Conditional scores must be read with planned and assessed coverage.","All ten JSHealth core contexts were excluded because a generative actor and delivery could not be confirmed. Two additional support contexts stopped at login gates.","Two retained dryrobe shopping contexts have only six measured replies. Twelve cause-selected capture repairs contain all 120 submitted questions; every repaired outcome is retained.","Merchant policy conflicts and gaps are retained. Product efficacy, review authenticity and downstream cart or refund completion were not independently established.","The repaired captures have gaps in live resource telemetry. Terminal state and artifact hashes were verified; continuous resource observations are not claimed.","Private cart, session or customer links are redacted from displayed evidence and quotations. Judgments used the unchanged original captures; these display redactions do not change scores."],"audit":{"description":"Every assessed policy-resolution checkpoint receives a primary and a fresh blind audit. Both must award valid evidence-backed attainment. Attainment disagreements receive zero verified credit and remain visible.","limitations":["Provider names were masked where practicable; full anonymity is not claimed.","Retained original quality judgments have separate historical audit coverage. The repair-quality audit sees the primary quality verdicts and full captured replies; it is distinct from the blind policy-resolution audit.","One initial scoring attempt failed in the response decoder with zero accepted decisions; upstream completion is unknown. The output was unavailable, and the documented infrastructure repair did not select among verdicts.","The first blind policy-resolution audit exhausted its 8,000-token per-model-turn output allowance without a structured verdict. Its unusable output was preserved and the identical input was audited once more under the reviewed 16,384-token per-model-turn allowance.","A later blind audit failed after about 602 seconds without a returned verdict. This is consistent with the original 600-second local timeout, but the inner error was not retained and provider completion is unknown. The unusable attempt was preserved and the identical input was retried once under a uniform 1,200-second future execution limit with a 1,250-second outer transport limit. All 36 previously successful stage outputs remain unchanged under their original timeout contracts.","The exact successful initial primary judgment for 30 checkpoints is retained with an 8,000-token per-model-turn allowance. Every subsequent policy-resolution and repair-quality call uses 16,384 per model turn, with the same frozen prompts, model, effort and criteria. New calls use fresh independent sessions under reviewed pool releases with 2 and 4 active slots; 4 distinct isolated slots were used. A CLI session can contain multiple internal steps, so its total output can exceed the per-turn allowance.","The original quality audit sampled only Alhena records. New policy-resolution audits cover both providers, but this does not retroactively expand historical quality-audit coverage."]},"url":"https://evals.alhena.ai/studies/alhena-gorgias-policy-resolution-2026-09-21","methodology":"https://evals.alhena.ai/studies/policy-resolution-v1"},{"schema":"alhena-research-lab/policy-study-summary-v1","protocol":"policy-resolution-v1","slug":"alhena-ai-e8955b5e31-vs-sierra-ai-98c6f6e402-549df402bc","title":"Alhena vs. Sierra: policy-resolution comparison","description":"Comparison assembled from published evaluations; no new conversations were run for this comparison.","publishedAt":"2026-09-23T00:59:21.323Z","captureStartAt":"2026-09-21T14:20:25.398Z","captureEndAt":"2026-09-23T00:53:06.466Z","commissionedBy":"Alhena Research Lab","method":{"status":"final","sha256":"712b1e5753cd9b871881d328d79ecb20ccbb279c045fe9233891c6805441133b","sourceCommit":"19b1420d2520d48baa52be81ac33fc4b9bd0ff8b","differences":["Replaces the source automation classifier with policy-compliant resolution.","Every eligible checkpoint receives two fresh blind judgments; attainment requires agreement and valid evidence.","New automated captures use the frozen protocol; historical study scores remain unchanged."]},"sample":{"plannedCoreContexts":100,"capturedCoreContexts":100,"guardrailContexts":10,"judgedCoreContexts":100,"pcrDecisions":917,"auditedPcrDecisions":917},"providers":[{"id":"tool-df07f051cc68e0cd","name":"Sierra","website":"https://sierra.ai/","registeredStores":5,"shopping":{"policyResolution":{"value":61.1,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":48,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":71.1,"explanation":"Full-answer completion 8.5 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":59,"explanation":"40% policy-compliant resolution, 35% quality, 25% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":183,"observedCheckpoints":183,"submittedCheckpoints":183,"assessedCheckpoints":183,"unassessableCheckpoints":0,"attainedCheckpoints":104,"policyUnverifiedCheckpoints":38,"includedContexts":25,"excludedContexts":0,"includedStores":5,"excludedStores":0,"qualityEligibleContexts":20,"qualityEligibleStores":4,"originalCaptures":24,"repairedCaptures":1}},"support":{"policyResolution":{"value":40,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":86,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":71.6,"explanation":"Full-answer completion 8.4 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":61.6,"explanation":"50% policy-compliant resolution, 40% quality, 10% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":242,"observedCheckpoints":242,"submittedCheckpoints":242,"assessedCheckpoints":242,"unassessableCheckpoints":0,"attainedCheckpoints":96,"policyUnverifiedCheckpoints":137,"includedContexts":25,"excludedContexts":0,"includedStores":5,"excludedStores":0,"qualityEligibleContexts":24,"qualityEligibleStores":5,"originalCaptures":25,"repairedCaptures":0}},"overallComposite":{"value":60.3,"explanation":"Equal mean of the two unrounded lane composites; only available when both lanes meet coverage floors."}},{"id":"alhena","name":"Alhena","website":"https://alhena.ai/","registeredStores":5,"shopping":{"policyResolution":{"value":64.8,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":83,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":53.7,"explanation":"Full-answer completion 11.8 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":68.4,"explanation":"40% policy-compliant resolution, 35% quality, 25% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":242,"observedCheckpoints":242,"submittedCheckpoints":242,"assessedCheckpoints":242,"unassessableCheckpoints":0,"attainedCheckpoints":158,"policyUnverifiedCheckpoints":64,"includedContexts":25,"excludedContexts":0,"includedStores":5,"excludedStores":0,"qualityEligibleContexts":25,"qualityEligibleStores":5,"originalCaptures":23,"repairedCaptures":2}},"support":{"policyResolution":{"value":72.8,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":95,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":70,"explanation":"Full-answer completion 8.7 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":81.4,"explanation":"50% policy-compliant resolution, 40% quality, 10% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":250,"observedCheckpoints":250,"submittedCheckpoints":250,"assessedCheckpoints":250,"unassessableCheckpoints":0,"attainedCheckpoints":182,"policyUnverifiedCheckpoints":63,"includedContexts":25,"excludedContexts":0,"includedStores":5,"excludedStores":0,"qualityEligibleContexts":25,"qualityEligibleStores":5,"originalCaptures":19,"repairedCaptures":6}},"overallComposite":{"value":74.9,"explanation":"Equal mean of the two unrounded lane composites; only available when both lanes meet coverage floors."}}],"derivedFrom":[{"slug":"sierra-ai-98c6f6e402-9cba991e","sha256":"52f0c2a935a09af65abf7eace014fcbd5a912bea4b1b4b513923bbda13e6c44d"},{"slug":"alhena-gorgias-policy-resolution-2026-09-21","sha256":"cd93a164ffc9f48301707ba8cb6f848a5160dec67276e8ca432b741c7707f7dc"}],"limitations":["Alhena commissioned and operates this study. Selected storefront deployments do not establish a universal provider ranking or independent certification.","This evaluation used the previously published method, frozen before capture and judging.","The sessions were logged out. No real order, refund, account change or completed human resolution was independently verified.","Unknown delivery or actor attribution remains unassessable. Unsent questions are not successes or failures. Conditional scores must be read with planned and assessed coverage.","Unresolved capture failures pause publication; completed studies do not represent every deployment of a provider.","1 earlier submission attempt(s) preceded operator-corrected captures. They are retained in the repair provenance and excluded from the selected-capture scoring denominators.","This methodology was developed after reviewing the original automation results, then frozen before the new policy-resolution judgments.","All ten JSHealth core contexts were excluded because a generative actor and delivery could not be confirmed. Two additional support contexts stopped at login gates.","Two retained dryrobe shopping contexts have only six measured replies. Twelve cause-selected capture repairs contain all 120 submitted questions; every repaired outcome is retained.","Merchant policy conflicts and gaps are retained. Product efficacy, review authenticity and downstream cart or refund completion were not independently established.","The repaired captures have gaps in live resource telemetry. Terminal state and artifact hashes were verified; continuous resource observations are not claimed.","Private cart, session or customer links are redacted from displayed evidence and quotations. Judgments used the unchanged original captures; these display redactions do not change scores.","Derived from separate published runs under policy-resolution-v1. Original capture dates and coverage are retained; execution environments may differ."],"audit":{"description":"This comparison retains the original judgments and audit coverage of each source study. No new judgment or audit is claimed for this derived comparison.","limitations":["Provider names were masked where practicable; full anonymity is not claimed.","Each new quality judgment has a separate adversarial audit. The quality audit sees the primary verdict; the policy-resolution audit does not.","Retained original quality judgments have separate historical audit coverage. The repair-quality audit sees the primary quality verdicts and full captured replies; it is distinct from the blind policy-resolution audit.","One initial scoring attempt failed in the response decoder with zero accepted decisions; upstream completion is unknown. The output was unavailable, and the documented infrastructure repair did not select among verdicts.","The first blind policy-resolution audit exhausted its 8,000-token per-model-turn output allowance without a structured verdict. Its unusable output was preserved and the identical input was audited once more under the reviewed 16,384-token per-model-turn allowance.","A later blind audit failed after about 602 seconds without a returned verdict. This is consistent with the original 600-second local timeout, but the inner error was not retained and provider completion is unknown. The unusable attempt was preserved and the identical input was retried once under a uniform 1,200-second future execution limit with a 1,250-second outer transport limit. All 36 previously successful stage outputs remain unchanged under their original timeout contracts.","The exact successful initial primary judgment for 30 checkpoints is retained with an 8,000-token per-model-turn allowance. Every subsequent policy-resolution and repair-quality call uses 16,384 per model turn, with the same frozen prompts, model, effort and criteria. New calls use fresh independent sessions under reviewed pool releases with 2 and 4 active slots; 4 distinct isolated slots were used. A CLI session can contain multiple internal steps, so its total output can exceed the per-turn allowance.","The original quality audit sampled only Alhena records. New policy-resolution audits cover both providers, but this does not retroactively expand historical quality-audit coverage.","Source methods: sierra-ai-98c6f6e402-9cba991e (712b1e5753cd9b871881d328d79ecb20ccbb279c045fe9233891c6805441133b); alhena-gorgias-policy-resolution-2026-09-21 (712b1e5753cd9b871881d328d79ecb20ccbb279c045fe9233891c6805441133b). Read each source method before interpreting differences."]},"url":"https://evals.alhena.ai/studies/alhena-ai-e8955b5e31-vs-sierra-ai-98c6f6e402-549df402bc","methodology":"https://evals.alhena.ai/studies/policy-resolution-v1"},{"schema":"alhena-research-lab/policy-study-summary-v1","protocol":"policy-resolution-v1","slug":"gorgias-com-b56a925583-vs-sierra-ai-98c6f6e402-633b8fd26e","title":"Gorgias vs. Sierra: policy-resolution comparison","description":"Comparison assembled from published evaluations; no new conversations were run for this comparison.","publishedAt":"2026-09-23T00:59:21.323Z","captureStartAt":"2026-09-21T14:13:43.224Z","captureEndAt":"2026-09-23T00:53:06.466Z","commissionedBy":"Alhena Research Lab","method":{"status":"final","sha256":"712b1e5753cd9b871881d328d79ecb20ccbb279c045fe9233891c6805441133b","sourceCommit":"19b1420d2520d48baa52be81ac33fc4b9bd0ff8b","differences":["Replaces the source automation classifier with policy-compliant resolution.","Every eligible checkpoint receives two fresh blind judgments; attainment requires agreement and valid evidence.","New automated captures use the frozen protocol; historical study scores remain unchanged."]},"sample":{"plannedCoreContexts":100,"capturedCoreContexts":100,"guardrailContexts":10,"judgedCoreContexts":88,"pcrDecisions":769,"auditedPcrDecisions":769},"providers":[{"id":"tool-df07f051cc68e0cd","name":"Sierra","website":"https://sierra.ai/","registeredStores":5,"shopping":{"policyResolution":{"value":61.1,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":48,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":71.1,"explanation":"Full-answer completion 8.5 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":59,"explanation":"40% policy-compliant resolution, 35% quality, 25% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":183,"observedCheckpoints":183,"submittedCheckpoints":183,"assessedCheckpoints":183,"unassessableCheckpoints":0,"attainedCheckpoints":104,"policyUnverifiedCheckpoints":38,"includedContexts":25,"excludedContexts":0,"includedStores":5,"excludedStores":0,"qualityEligibleContexts":20,"qualityEligibleStores":4,"originalCaptures":24,"repairedCaptures":1}},"support":{"policyResolution":{"value":40,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":86,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":71.6,"explanation":"Full-answer completion 8.4 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":61.6,"explanation":"50% policy-compliant resolution, 40% quality, 10% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":242,"observedCheckpoints":242,"submittedCheckpoints":242,"assessedCheckpoints":242,"unassessableCheckpoints":0,"attainedCheckpoints":96,"policyUnverifiedCheckpoints":137,"includedContexts":25,"excludedContexts":0,"includedStores":5,"excludedStores":0,"qualityEligibleContexts":24,"qualityEligibleStores":5,"originalCaptures":25,"repairedCaptures":0}},"overallComposite":{"value":60.3,"explanation":"Equal mean of the two unrounded lane composites; only available when both lanes meet coverage floors."}},{"id":"gorgias","name":"Gorgias","website":"https://www.gorgias.com/","registeredStores":5,"shopping":{"policyResolution":{"value":60.8,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":72,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":25.3,"explanation":"Full-answer completion 17.2 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":55.8,"explanation":"40% policy-compliant resolution, 35% quality, 25% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":187,"observedCheckpoints":187,"submittedCheckpoints":187,"assessedCheckpoints":164,"unassessableCheckpoints":23,"attainedCheckpoints":109,"policyUnverifiedCheckpoints":37,"includedContexts":20,"excludedContexts":5,"includedStores":4,"excludedStores":1,"qualityEligibleContexts":17,"qualityEligibleStores":4,"originalCaptures":23,"repairedCaptures":2}},"support":{"policyResolution":{"value":68,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":94,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":41.6,"explanation":"Full-answer completion 14.1 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":75.8,"explanation":"50% policy-compliant resolution, 40% quality, 10% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":205,"observedCheckpoints":205,"submittedCheckpoints":205,"assessedCheckpoints":180,"unassessableCheckpoints":25,"attainedCheckpoints":122,"policyUnverifiedCheckpoints":45,"includedContexts":18,"excludedContexts":7,"includedStores":4,"excludedStores":1,"qualityEligibleContexts":18,"qualityEligibleStores":4,"originalCaptures":23,"repairedCaptures":2}},"overallComposite":{"value":65.8,"explanation":"Equal mean of the two unrounded lane composites; only available when both lanes meet coverage floors."}}],"derivedFrom":[{"slug":"sierra-ai-98c6f6e402-9cba991e","sha256":"52f0c2a935a09af65abf7eace014fcbd5a912bea4b1b4b513923bbda13e6c44d"},{"slug":"alhena-gorgias-policy-resolution-2026-09-21","sha256":"cd93a164ffc9f48301707ba8cb6f848a5160dec67276e8ca432b741c7707f7dc"}],"limitations":["Alhena commissioned and operates this study. Selected storefront deployments do not establish a universal provider ranking or independent certification.","This evaluation used the previously published method, frozen before capture and judging.","The sessions were logged out. No real order, refund, account change or completed human resolution was independently verified.","Unknown delivery or actor attribution remains unassessable. Unsent questions are not successes or failures. Conditional scores must be read with planned and assessed coverage.","Unresolved capture failures pause publication; completed studies do not represent every deployment of a provider.","1 earlier submission attempt(s) preceded operator-corrected captures. They are retained in the repair provenance and excluded from the selected-capture scoring denominators.","This methodology was developed after reviewing the original automation results, then frozen before the new policy-resolution judgments.","All ten JSHealth core contexts were excluded because a generative actor and delivery could not be confirmed. Two additional support contexts stopped at login gates.","Two retained dryrobe shopping contexts have only six measured replies. Twelve cause-selected capture repairs contain all 120 submitted questions; every repaired outcome is retained.","Merchant policy conflicts and gaps are retained. Product efficacy, review authenticity and downstream cart or refund completion were not independently established.","The repaired captures have gaps in live resource telemetry. Terminal state and artifact hashes were verified; continuous resource observations are not claimed.","Private cart, session or customer links are redacted from displayed evidence and quotations. Judgments used the unchanged original captures; these display redactions do not change scores.","Derived from separate published runs under policy-resolution-v1. Original capture dates and coverage are retained; execution environments may differ."],"audit":{"description":"This comparison retains the original judgments and audit coverage of each source study. No new judgment or audit is claimed for this derived comparison.","limitations":["Provider names were masked where practicable; full anonymity is not claimed.","Each new quality judgment has a separate adversarial audit. The quality audit sees the primary verdict; the policy-resolution audit does not.","Retained original quality judgments have separate historical audit coverage. The repair-quality audit sees the primary quality verdicts and full captured replies; it is distinct from the blind policy-resolution audit.","One initial scoring attempt failed in the response decoder with zero accepted decisions; upstream completion is unknown. The output was unavailable, and the documented infrastructure repair did not select among verdicts.","The first blind policy-resolution audit exhausted its 8,000-token per-model-turn output allowance without a structured verdict. Its unusable output was preserved and the identical input was audited once more under the reviewed 16,384-token per-model-turn allowance.","A later blind audit failed after about 602 seconds without a returned verdict. This is consistent with the original 600-second local timeout, but the inner error was not retained and provider completion is unknown. The unusable attempt was preserved and the identical input was retried once under a uniform 1,200-second future execution limit with a 1,250-second outer transport limit. All 36 previously successful stage outputs remain unchanged under their original timeout contracts.","The exact successful initial primary judgment for 30 checkpoints is retained with an 8,000-token per-model-turn allowance. Every subsequent policy-resolution and repair-quality call uses 16,384 per model turn, with the same frozen prompts, model, effort and criteria. New calls use fresh independent sessions under reviewed pool releases with 2 and 4 active slots; 4 distinct isolated slots were used. A CLI session can contain multiple internal steps, so its total output can exceed the per-turn allowance.","The original quality audit sampled only Alhena records. New policy-resolution audits cover both providers, but this does not retroactively expand historical quality-audit coverage.","Source methods: sierra-ai-98c6f6e402-9cba991e (712b1e5753cd9b871881d328d79ecb20ccbb279c045fe9233891c6805441133b); alhena-gorgias-policy-resolution-2026-09-21 (712b1e5753cd9b871881d328d79ecb20ccbb279c045fe9233891c6805441133b). Read each source method before interpreting differences."]},"url":"https://evals.alhena.ai/studies/gorgias-com-b56a925583-vs-sierra-ai-98c6f6e402-633b8fd26e","methodology":"https://evals.alhena.ai/studies/policy-resolution-v1"},{"schema":"alhena-research-lab/policy-study-summary-v1","protocol":"policy-resolution-v1","slug":"sierra-ai-98c6f6e402-9cba991e","title":"Sierra: policy-resolution study","description":"Live storefront evaluation of policy-compliant resolution, answer quality and response speed.","publishedAt":"2026-09-23T00:59:21.323Z","captureStartAt":"2026-09-22T15:03:08.664Z","captureEndAt":"2026-09-23T00:53:06.466Z","commissionedBy":"Alhena Research Lab","method":{"status":"final","sha256":"712b1e5753cd9b871881d328d79ecb20ccbb279c045fe9233891c6805441133b","sourceCommit":"19b1420d2520d48baa52be81ac33fc4b9bd0ff8b","differences":["Replaces the source automation classifier with policy-compliant resolution.","Every eligible checkpoint receives two fresh blind judgments; attainment requires agreement and valid evidence.","New automated captures use the frozen protocol; historical study scores remain unchanged."]},"sample":{"plannedCoreContexts":50,"capturedCoreContexts":50,"guardrailContexts":5,"judgedCoreContexts":50,"pcrDecisions":425,"auditedPcrDecisions":425},"providers":[{"id":"tool-df07f051cc68e0cd","name":"Sierra","website":"https://sierra.ai/","registeredStores":5,"shopping":{"policyResolution":{"value":61.1,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":48,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":71.1,"explanation":"Full-answer completion 8.5 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":59,"explanation":"40% policy-compliant resolution, 35% quality, 25% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":183,"observedCheckpoints":183,"submittedCheckpoints":183,"assessedCheckpoints":183,"unassessableCheckpoints":0,"attainedCheckpoints":104,"policyUnverifiedCheckpoints":38,"includedContexts":25,"excludedContexts":0,"includedStores":5,"excludedStores":0,"qualityEligibleContexts":20,"qualityEligibleStores":4,"originalCaptures":24,"repairedCaptures":1}},"support":{"policyResolution":{"value":40,"explanation":"Verified correct answer or policy-prescribed next step, averaged equally by conversation and storefront. This does not measure downstream case completion."},"quality":{"value":86,"explanation":"The original fixed answer-quality rubric. Unchanged captures retain their original judgments; repaired captures receive a fresh judgment and audit."},"speed":{"value":71.6,"explanation":"Full-answer completion 8.4 seconds; 100 at 3 seconds, zero at 22 seconds, bounded between 0 and 100."},"composite":{"value":61.6,"explanation":"50% policy-compliant resolution, 40% quality, 10% speed."},"coverage":{"plannedCheckpoints":250,"attemptedCheckpoints":242,"observedCheckpoints":242,"submittedCheckpoints":242,"assessedCheckpoints":242,"unassessableCheckpoints":0,"attainedCheckpoints":96,"policyUnverifiedCheckpoints":137,"includedContexts":25,"excludedContexts":0,"includedStores":5,"excludedStores":0,"qualityEligibleContexts":24,"qualityEligibleStores":5,"originalCaptures":25,"repairedCaptures":0}},"overallComposite":{"value":60.3,"explanation":"Equal mean of the two unrounded lane composites; only available when both lanes meet coverage floors."}}],"limitations":["Alhena commissioned and operates this study. Selected storefront deployments do not establish a universal provider ranking or independent certification.","This evaluation used the previously published method, frozen before capture and judging.","The sessions were logged out. No real order, refund, account change or completed human resolution was independently verified.","Unknown delivery or actor attribution remains unassessable. Unsent questions are not successes or failures. Conditional scores must be read with planned and assessed coverage.","Unresolved capture failures pause publication; completed studies do not represent every deployment of a provider.","1 earlier submission attempt(s) preceded operator-corrected captures. They are retained in the repair provenance and excluded from the selected-capture scoring denominators."],"audit":{"description":"Every assessed policy-resolution checkpoint receives a primary and a fresh blind audit. Both must award valid evidence-backed attainment. Attainment disagreements receive zero verified credit and remain visible.","limitations":["Provider names were masked where practicable; full anonymity is not claimed.","Each new quality judgment has a separate adversarial audit. The quality audit sees the primary verdict; the policy-resolution audit does not."]},"url":"https://evals.alhena.ai/studies/sierra-ai-98c6f6e402-9cba991e","methodology":"https://evals.alhena.ai/studies/policy-resolution-v1"}]}