<?xml version="1.0" encoding="UTF-8"?>
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:title>Statistical Inference for Language Model Evaluations</dc:title>
  <dc:title>R package evaluatellm version 0.1.0</dc:title>
  <dc:description>Treats language model evaluations as statistical experiments and
    supplies the inference they require. Provides central limit theorem and
    cluster-robust standard errors for evaluation scores, paired and unpaired
    model comparisons, variance decomposition when several responses are drawn
    per question, control-variate variance reduction, multiplicity adjustment
    across benchmark suites, and power and minimum detectable effect
    calculations for planning evaluations, following Miller (2024)
    &lt;doi:10.48550/arXiv.2411.00640&gt;. For evaluations scored by a model judge,
    implements agreement statistics against a human gold standard and
    prediction-powered inference (Angelopoulos et al. 2023)
    &lt;doi:10.1126/science.adi6000&gt; with the power-tuned estimator of
    Angelopoulos, Bates and Jordan (2023) &lt;doi:10.48550/arXiv.2311.01453&gt;, so a
    small set of human labels debiases a large set of judge scores. Leaderboards
    are supported through bootstrap rank intervals and Bradley-Terry ratings
    (Bradley and Terry 1952) &lt;doi:10.2307/2334029&gt;. Accepts scores from any
    evaluation harness.</dc:description>
  <dc:type>Software</dc:type>
  <dc:relation>Depends: R (&gt;= 4.1.0)</dc:relation>
  <dc:relation>Imports: cli (&gt;= 3.6.0), graphics, grDevices, stats, utils</dc:relation>
  <dc:relation>Suggests: testthat (&gt;= 3.0.0), knitr, rmarkdown, sandwich</dc:relation>
  <dc:creator>Charles Coverdale &lt;charlesfcoverdale@gmail.com&gt;</dc:creator>
  <dc:publisher>Comprehensive R Archive Network (CRAN)</dc:publisher>
  <dc:contributor>Charles Coverdale [aut, cre, cph]</dc:contributor>
  <dc:rights>MIT + file LICENSE (https://CRAN.R-project.org/package=evaluatellm/LICENSE)</dc:rights>
  <dc:date>2026-09-15</dc:date>
  <dc:format>application/tgz</dc:format>
  <dc:identifier>https://CRAN.R-project.org/package=evaluatellm</dc:identifier>
  <dc:identifier>doi:10.32614/CRAN.package.evaluatellm</dc:identifier>
  <dc:language>en-GB</dc:language>
</oai_dc:dc>
