lucasmaystre/choixrecord 1
{
"id": "R01",
"kind": "repo",
"name": "lucasmaystre/choix",
"url": "https://github.com/lucasmaystre/choix",
"licence": "MIT",
"stars": 203,
"evidence_class": "observed",
"source": "gh api repos/lucasmaystre/choix",
"observed": "2026-08-27",
"claim": "Inference for Luce-choice-axiom models: Bradley-Terry, Plackett-Luce, Thurstone; MM, Lu-Boutilier, opt & EP methods on pairwise/top-1/ranking data.",
"limitations": "Small (203 stars), single maintainer, no active-query design — estimation only, you must choose the pairs yourself.",
"disposition": "adopt — the single best fit for fitting BT over 7 knobs from pairwise data in Python"
}
sublee/trueskillrecord 2
{
"id": "R02",
"kind": "repo",
"name": "sublee/trueskill",
"url": "https://github.com/sublee/trueskill",
"licence": "BSD-3-Clause CODE, but TrueSkill(TM) TRADEMARK is restricted",
"stars": 803,
"evidence_class": "observed",
"source": "gh api repos/sublee/trueskill/license — read LICENSE body verbatim",
"observed": "2026-08-27",
"claim": "Python TrueSkill: Bayesian factor-graph skill rating, Gaussian belief per player updated by match outcomes.",
"limitations": "LICENSE BODY CARRIES A COMMERCIAL BAR the badge hides: 'Microsoft permits only Xbox Live games or non-commercial projects to use TrueSkill(TM). If your project is commercial, you should find another rating system.' GitHub reports only NOASSERTION.",
"disposition": "REJECT for paid client work — use openskill.py instead"
}
vivekjoshy/openskill.pyrecord 3
{
"id": "R03",
"kind": "repo",
"name": "vivekjoshy/openskill.py",
"url": "https://github.com/vivekjoshy/openskill.py",
"licence": "MIT",
"stars": 368,
"evidence_class": "observed",
"source": "gh api repos/vivekjoshy/openskill.py",
"observed": "2026-08-27",
"claim": "Clean-room multiplayer Bayesian rating (Weng-Lin models: BradleyTerryFull/Part, PlackettLuce, ThurstoneMostellerFull/Part). No Microsoft trademark encumbrance.",
"limitations": "Designed for multiplayer match rating, not design-knob elicitation; needs adapting to treat designs as 'players'.",
"disposition": "adopt — the licence-safe TrueSkill substitute"
}
scttcper/ts-trueskillrecord 4
{
"id": "R04",
"kind": "repo",
"name": "scttcper/ts-trueskill",
"url": "https://github.com/scttcper/ts-trueskill",
"licence": "MIT (LICENSE body reads 'MIT License, Copyright (c) Scott Cooper')",
"stars": 84,
"evidence_class": "observed",
"source": "gh api repos/scttcper/ts-trueskill/license",
"observed": "2026-08-27",
"claim": "TypeScript port of the Python TrueSkill package — usable browser-side.",
"licence_note": "GitHub says NOASSERTION; the LICENSE body is plain MIT. Same TrueSkill(TM) trademark caution applies by derivation.",
"limitations": "Port of R02, inherits the trademark question even though the code licence is MIT.",
"disposition": "study — only if elicitation must run client-side in TS"
}
hturner/BradleyTerry2record 5
{
"id": "R05",
"kind": "repo",
"name": "hturner/BradleyTerry2",
"url": "https://github.com/hturner/BradleyTerry2",
"licence": "GPL (>= 2) — read from DESCRIPTION",
"stars": 25,
"evidence_class": "observed",
"source": "gh api repos/hturner/BradleyTerry2/contents/DESCRIPTION",
"observed": "2026-08-27",
"claim": "R: Bradley-Terry models incl. structured versions where abilities are a linear predictor of item covariates — the 'knob' formulation we actually need.",
"limitations": "GPL — copyleft, careful if linked into proprietary client deliverable. R, not Python.",
"disposition": "study — the covariate-structured BT is the right model form; reimplement rather than link"
}
EllaKaye/BradleyTerryScalablerecord 6
{
"id": "R06",
"kind": "repo",
"name": "EllaKaye/BradleyTerryScalable",
"url": "https://github.com/EllaKaye/BradleyTerryScalable",
"licence": "NONE declared in repo",
"stars": 24,
"evidence_class": "observed",
"source": "gh api repos/EllaKaye/BradleyTerryScalable",
"observed": "2026-08-27",
"claim": "R package fitting BT to large sparse comparison datasets.",
"limitations": "No LICENSE file = all rights reserved by default. Legally unusable without asking the author.",
"disposition": "reject on licence grounds"
}
hturner/PlackettLucerecord 7
{
"id": "R07",
"kind": "repo",
"name": "hturner/PlackettLuce",
"url": "https://github.com/hturner/PlackettLuce",
"licence": "GPL-3 — read from DESCRIPTION",
"stars": 27,
"evidence_class": "observed",
"source": "gh api repos/hturner/PlackettLuce/contents/DESCRIPTION",
"observed": "2026-08-27",
"claim": "R: Plackett-Luce models for rankings, handles ties and partial rankings.",
"limitations": "GPL-3. Relevant only if we move from pairs to full N-way rankings.",
"disposition": "study — matters if we choose N-way over pairwise"
}
cran/prefmodrecord 8
{
"id": "R08",
"kind": "repo",
"name": "cran/prefmod",
"url": "https://github.com/cran/prefmod",
"licence": "GPL (>= 2)",
"stars": 0,
"evidence_class": "observed",
"source": "crandb.r-pkg.org/prefmod (v0.8-37)",
"observed": "2026-08-27",
"claim": "R: fits paired-comparison preference models (Bradley-Terry loglinear), handles undecided/ties and subject covariates.",
"limitations": "Read-only CRAN mirror. R only, dated.",
"disposition": "study"
}
UDST/choicemodelsrecord 9
{
"id": "R09",
"kind": "repo",
"name": "UDST/choicemodels",
"url": "https://github.com/UDST/choicemodels",
"licence": "BSD-3-Clause",
"stars": 86,
"evidence_class": "observed",
"source": "gh api repos/UDST/choicemodels",
"observed": "2026-08-27",
"claim": "Python discrete choice modelling (MNL, large-alternative sampling) from the UrbanSim group.",
"limitations": "Urban-simulation oriented; low activity.",
"disposition": "study"
}
timothyb0912/pylogitrecord 10
{
"id": "R10",
"kind": "repo",
"name": "timothyb0912/pylogit",
"url": "https://github.com/timothyb0912/pylogit",
"licence": "BSD-3-Clause",
"stars": 204,
"evidence_class": "observed",
"source": "gh api repos/timothyb0912/pylogit",
"observed": "2026-08-27",
"claim": "Python conditional logit / MNL estimation for discrete choice experiments.",
"limitations": "Estimation only, no experimental design or adaptive querying.",
"disposition": "study"
}
arteagac/xlogitrecord 11
{
"id": "R11",
"kind": "repo",
"name": "arteagac/xlogit",
"url": "https://github.com/arteagac/xlogit",
"licence": "GPL-3.0",
"stars": 67,
"evidence_class": "observed",
"source": "gh api repos/arteagac/xlogit",
"observed": "2026-08-27",
"claim": "GPU-accelerated mixed logit estimation.",
"limitations": "GPL-3.0 — copyleft. Mixed logit is overkill for a single client's 7 knobs.",
"disposition": "reject — wrong scale, restrictive licence"
}
michelbierlaire/biogemerecord 12
{
"id": "R12",
"kind": "repo",
"name": "michelbierlaire/biogeme",
"url": "https://github.com/michelbierlaire/biogeme",
"licence": "CUSTOM — not a standard OSS licence",
"stars": 134,
"evidence_class": "observed",
"source": "gh api repos/michelbierlaire/biogeme/license — LICENSE body reads 'BIOGEME is distributed free of charge. We ask each user to register to Biogeme's users group, and to mention explicitly the use of the package when publishing results'",
"observed": "2026-08-27",
"claim": "Mature maximum-likelihood estimation of parametric discrete choice models.",
"limitations": "Badge says NOASSERTION; the body imposes registration + citation obligations. Not MIT/BSD-equivalent — read before any client use.",
"disposition": "study with licence caution"
}
gsbDBI/torch-choicerecord 13
{
"id": "R13",
"kind": "repo",
"name": "gsbDBI/torch-choice",
"url": "https://github.com/gsbDBI/torch-choice",
"licence": "MIT",
"stars": 59,
"evidence_class": "observed",
"source": "gh api repos/gsbDBI/torch-choice",
"observed": "2026-08-27",
"claim": "PyTorch choice modelling (logit, nested logit) — GPU, differentiable, composes with learned embeddings.",
"limitations": "Young, small community.",
"disposition": "adopt-candidate if the taste model must sit inside a torch pipeline"
}
meta-pytorch/botorchrecord 14
{
"id": "R14",
"kind": "repo",
"name": "meta-pytorch/botorch",
"url": "https://github.com/meta-pytorch/botorch",
"licence": "MIT",
"stars": 3589,
"evidence_class": "observed",
"source": "gh api repos/meta-pytorch/botorch (note: pytorch/botorch now 301-redirects here)",
"observed": "2026-08-27",
"claim": "Bayesian optimization in PyTorch. Ships PairwiseGP, PairwiseLaplaceMarginalLogLikelihood, and qEUBO — a production-grade preferential-BO stack.",
"limitations": "Heavy dependency (torch+gpytorch); PBO components are less documented than the standard BO path.",
"disposition": "ADOPT — the single most load-bearing repo for this lane"
}
facebook/Axrecord 15
{
"id": "R15",
"kind": "repo",
"name": "facebook/Ax",
"url": "https://github.com/facebook/Ax",
"licence": "MIT",
"stars": 2792,
"evidence_class": "observed",
"source": "gh api repos/facebook/Ax",
"observed": "2026-08-27",
"claim": "Adaptive experimentation platform over BoTorch; service API manages the ask/tell loop and trial state.",
"limitations": "Opinionated abstraction; preference trials need custom metric plumbing.",
"disposition": "adopt — use for the ask/tell loop rather than hand-rolling"
}
facebookresearch/qEUBOrecord 16
{
"id": "R16",
"kind": "repo",
"name": "facebookresearch/qEUBO",
"url": "https://github.com/facebookresearch/qEUBO",
"licence": "MIT",
"stars": 23,
"evidence_class": "observed",
"source": "gh api repos/facebookresearch/qEUBO",
"observed": "2026-08-27",
"claim": "Reproducible code for the AISTATS 2023 qEUBO paper — decision-theoretic acquisition function for preferential BO.",
"limitations": "Research reproduction code, 23 stars, not a maintained library.",
"disposition": "study — read it, then use the BoTorch-integrated version"
}
AaltoPML/PPBOrecord 17
{
"id": "R17",
"kind": "repo",
"name": "AaltoPML/PPBO",
"url": "https://github.com/AaltoPML/PPBO",
"licence": "MIT",
"stars": 20,
"evidence_class": "observed",
"source": "gh api repos/AaltoPML/PPBO",
"observed": "2026-08-27",
"claim": "Projective Preferential Bayesian Optimization — queries along projections, designed for human-in-the-loop high-dim preference elicitation.",
"limitations": "Small research repo.",
"disposition": "study — projection idea is directly relevant to 7 knobs"
}
CyberAgentAILab/preferentialBOrecord 18
{
"id": "R18",
"kind": "repo",
"name": "CyberAgentAILab/preferentialBO",
"url": "https://github.com/CyberAgentAILab/preferentialBO",
"licence": "MIT",
"stars": 17,
"evidence_class": "observed",
"source": "gh api repos/CyberAgentAILab/preferentialBO",
"observed": "2026-08-27",
"claim": "ICML 2023 'Towards Practical Preferential Bayesian Optimization' implementation.",
"limitations": "Research repo, small.",
"disposition": "study"
}
ja-thomas/pbmohporecord 19
{
"id": "R19",
"kind": "repo",
"name": "ja-thomas/pbmohpo",
"url": "https://github.com/ja-thomas/pbmohpo",
"licence": "LGPL-2.1",
"stars": 8,
"evidence_class": "observed",
"source": "gh api repos/ja-thomas/pbmohpo",
"observed": "2026-08-27",
"claim": "Preferential Bayesian multi-objective hyperparameter optimization.",
"limitations": "8 stars, LGPL, narrow HPO framing.",
"disposition": "drop"
}
yuki-koyama/sequential-line-searchrecord 20
{
"id": "R20",
"kind": "repo",
"name": "yuki-koyama/sequential-line-search",
"url": "https://github.com/yuki-koyama/sequential-line-search",
"licence": "MIT",
"stars": 57,
"evidence_class": "observed",
"source": "gh api repos/yuki-koyama/sequential-line-search",
"observed": "2026-08-27",
"claim": "C++/Python preferential BO library implementing SIGGRAPH 2017 sequential line search — the user picks a point on a 1D slider through a high-dim design space.",
"limitations": "Only 57 stars; C++ core with Python bindings; author-maintained.",
"disposition": "ADOPT/STEAL — closest published prior art to our exact problem"
}
yuki-koyama/sequential-galleryrecord 21
{
"id": "R21",
"kind": "repo",
"name": "yuki-koyama/sequential-gallery",
"url": "https://github.com/yuki-koyama/sequential-gallery",
"licence": "MIT",
"stars": 22,
"evidence_class": "observed",
"source": "gh api repos/yuki-koyama/sequential-gallery",
"observed": "2026-08-27",
"claim": "SIGGRAPH 2020 sequential-plane-search + adaptive grid gallery UI for visual design optimization.",
"limitations": "22 stars, desktop C++ app, research-grade.",
"disposition": "ADOPT/STEAL — the gallery interaction model is the design we want"
}
modAL-python/modALrecord 22
{
"id": "R22",
"kind": "repo",
"name": "modAL-python/modAL",
"url": "https://github.com/modAL-python/modAL",
"licence": "MIT",
"stars": 2363,
"evidence_class": "observed",
"source": "gh api repos/modAL-python/modAL",
"observed": "2026-08-27",
"claim": "Modular active learning framework over scikit-learn; uncertainty/query-by-committee strategies.",
"limitations": "Classification-oriented, not preference/pairwise native. Maintenance has slowed.",
"disposition": "study"
}
ntucllab/libactrecord 23
{
"id": "R23",
"kind": "repo",
"name": "ntucllab/libact",
"url": "https://github.com/ntucllab/libact",
"licence": "BSD-2-Clause",
"stars": 792,
"evidence_class": "observed",
"source": "gh api repos/ntucllab/libact",
"observed": "2026-08-27",
"claim": "Pool-based active learning incl. active-learning-by-learning bandit meta-strategy.",
"limitations": "Older codebase, C extensions, install friction.",
"disposition": "study"
}
NUAA-AL/ALiPyrecord 24
{
"id": "R24",
"kind": "repo",
"name": "NUAA-AL/ALiPy",
"url": "https://github.com/NUAA-AL/ALiPy",
"licence": "BSD-3-Clause",
"stars": 906,
"evidence_class": "observed",
"source": "gh api repos/NUAA-AL/ALiPy",
"observed": "2026-08-27",
"claim": "Active learning toolbox with ~20 query strategies and experiment harness.",
"limitations": "Classification-focused; no preference queries.",
"disposition": "study"
}
webis-de/small-textrecord 25
{
"id": "R25",
"kind": "repo",
"name": "webis-de/small-text",
"url": "https://github.com/webis-de/small-text",
"licence": "MIT",
"stars": 646,
"evidence_class": "observed",
"source": "gh api repos/webis-de/small-text",
"observed": "2026-08-27",
"claim": "Active learning for text classification.",
"limitations": "Text-only — wrong modality for visual taste.",
"disposition": "drop"
}
scikit-activeml/scikit-activemlrecord 26
{
"id": "R26",
"kind": "repo",
"name": "scikit-activeml/scikit-activeml",
"url": "https://github.com/scikit-activeml/scikit-activeml",
"licence": "BSD-3-Clause",
"stars": 201,
"evidence_class": "observed",
"source": "gh api repos/scikit-activeml/scikit-activeml",
"observed": "2026-08-27",
"claim": "sklearn-compatible active learning library, well-documented query strategies.",
"limitations": "Small; classification/regression framing.",
"disposition": "study"
}
SheffieldML/GPyrecord 27
{
"id": "R27",
"kind": "repo",
"name": "SheffieldML/GPy",
"url": "https://github.com/SheffieldML/GPy",
"licence": "BSD-3-Clause",
"stars": 2161,
"evidence_class": "observed",
"source": "gh api repos/SheffieldML/GPy",
"observed": "2026-08-27",
"claim": "Mature GP framework; preference/ordinal likelihoods available.",
"limitations": "Largely in maintenance mode; superseded by GPflow/GPyTorch.",
"disposition": "study"
}
GPflow/GPflowrecord 28
{
"id": "R28",
"kind": "repo",
"name": "GPflow/GPflow",
"url": "https://github.com/GPflow/GPflow",
"licence": "Apache-2.0",
"stars": 1916,
"evidence_class": "observed",
"source": "gh api repos/GPflow/GPflow",
"observed": "2026-08-27",
"claim": "GPs in TensorFlow with custom likelihoods — a Bernoulli-on-latent-difference preference likelihood is straightforward.",
"limitations": "TensorFlow dependency; BoTorch is the better torch-native path.",
"disposition": "study"
}
dragonfly/dragonflyrecord 29
{
"id": "R29",
"kind": "repo",
"name": "dragonfly/dragonfly",
"url": "https://github.com/dragonfly/dragonfly",
"licence": "MIT",
"stars": 893,
"evidence_class": "observed",
"source": "gh api repos/dragonfly/dragonfly",
"observed": "2026-08-27",
"claim": "Scalable Bayesian optimisation library.",
"limitations": "No preference/duel query support; low recent activity.",
"disposition": "drop"
}
EmuKit/emukitrecord 30
{
"id": "R30",
"kind": "repo",
"name": "EmuKit/emukit",
"url": "https://github.com/EmuKit/emukit",
"licence": "Apache-2.0",
"stars": 672,
"evidence_class": "observed",
"source": "gh api repos/EmuKit/emukit",
"observed": "2026-08-27",
"claim": "Decision-making under uncertainty toolkit: BO, experimental design, sensitivity analysis.",
"limitations": "Experimental-design module is generic, not preference-specific.",
"disposition": "study"
}
huawei-noah/HEBOrecord 31
{
"id": "R31",
"kind": "repo",
"name": "huawei-noah/HEBO",
"url": "https://github.com/huawei-noah/HEBO",
"licence": "MIT (read from HEBO/LICENSE — repo root has no LICENSE, so GitHub reports NONE)",
"stars": 2797,
"evidence_class": "observed",
"source": "gh api repos/huawei-noah/HEBO/contents/HEBO/LICENSE — body reads 'MIT License, Copyright (C) 2019. Huawei Technologies Co., Ltd.'",
"observed": "2026-08-27",
"claim": "Monorepo of BO and RL research; HEBO won the NeurIPS 2020 BBO challenge.",
"limitations": "Badge NONE is misleading — licence lives one level down and is per-subproject; verify per directory before use.",
"disposition": "study — licence is fine but check the specific subdirectory"
}
optuna/optunarecord 32
{
"id": "R32",
"kind": "repo",
"name": "optuna/optuna",
"url": "https://github.com/optuna/optuna",
"licence": "MIT",
"stars": 14713,
"evidence_class": "observed",
"source": "gh api repos/optuna/optuna",
"observed": "2026-08-27",
"claim": "Hyperparameter optimization framework (TPE, CMA-ES). Included for contrast: mature ask/tell ergonomics.",
"limitations": "Assumes a scalar objective you can compute — exactly what we do NOT have with human taste.",
"disposition": "contrast only — proves why a preference-native tool is needed"
}
Alanthink/banditpylibrecord 33
{
"id": "R33",
"kind": "repo",
"name": "Alanthink/banditpylib",
"url": "https://github.com/Alanthink/banditpylib",
"licence": "MIT",
"stars": 29,
"evidence_class": "observed",
"source": "gh api repos/Alanthink/banditpylib",
"observed": "2026-08-27",
"claim": "Lightweight bandit library including ordinary and best-arm-identification protocols.",
"limitations": "29 stars; dueling-specific coverage is thin.",
"disposition": "study"
}
HuasenWu/DuelingBanditsrecord 34
{
"id": "R34",
"kind": "repo",
"name": "HuasenWu/DuelingBandits",
"url": "https://github.com/HuasenWu/DuelingBandits",
"licence": "NONE",
"stars": 25,
"evidence_class": "observed",
"source": "gh api repos/HuasenWu/DuelingBandits",
"observed": "2026-08-27",
"claim": "Simulations for dueling bandit algorithms including Double Thompson Sampling.",
"limitations": "No LICENSE = all rights reserved. Research simulation code.",
"disposition": "reference only — do not vendor"
}
AkihikoWatanabe/DBGDrecord 35
{
"id": "R35",
"kind": "repo",
"name": "AkihikoWatanabe/DBGD",
"url": "https://github.com/AkihikoWatanabe/DBGD",
"licence": "MIT",
"stars": 24,
"evidence_class": "observed",
"source": "gh api repos/AkihikoWatanabe/DBGD",
"observed": "2026-08-27",
"claim": "Python implementation of Dueling Bandit Gradient Descent (Yue & Joachims).",
"limitations": "Tiny, unmaintained.",
"disposition": "study"
}
methi1999/dueling-banditsrecord 36
{
"id": "R36",
"kind": "repo",
"name": "methi1999/dueling-bandits",
"url": "https://github.com/methi1999/dueling-bandits",
"licence": "NONE",
"stars": 6,
"evidence_class": "observed",
"source": "gh api repos/methi1999/dueling-bandits",
"observed": "2026-08-27",
"claim": "Implementations of several popular dueling bandit algorithms.",
"limitations": "6 stars, no licence.",
"disposition": "drop"
}
cran/support.BWSrecord 37
{
"id": "R37",
"kind": "repo",
"name": "cran/support.BWS",
"url": "https://github.com/cran/support.BWS",
"licence": "GPL (>= 2)",
"stars": 1,
"evidence_class": "observed",
"source": "crandb.r-pkg.org/support.BWS (v0.4-6)",
"observed": "2026-08-27",
"claim": "R tools for Case 1 Best-Worst Scaling: constructs BWS questions from a BIBD and analyses responses.",
"limitations": "R; Case 1 (object scaling) only. CRAN mirror.",
"disposition": "ADOPT-METHOD — best-worst is the highest-information-per-question format for 7 knobs"
}
cran/support.BWS2record 38
{
"id": "R38",
"kind": "repo",
"name": "cran/support.BWS2",
"url": "https://github.com/cran/support.BWS2",
"licence": "GPL (>= 2)",
"stars": 0,
"evidence_class": "observed",
"source": "crandb.r-pkg.org/support.BWS2 (v0.4-0)",
"observed": "2026-08-27",
"claim": "R tools for Case 2 (profile-case) Best-Worst Scaling — best/worst ATTRIBUTE-LEVEL within a profile.",
"limitations": "R only.",
"disposition": "ADOPT-METHOD — Case 2 maps almost exactly onto 'which knob is most/least right here'"
}
cran/ChoiceModelRrecord 39
{
"id": "R39",
"kind": "repo",
"name": "cran/ChoiceModelR",
"url": "https://github.com/cran/ChoiceModelR",
"licence": "GPL (>= 3)",
"stars": 5,
"evidence_class": "observed",
"source": "crandb.r-pkg.org/ChoiceModelR (v1.3.1)",
"observed": "2026-08-27",
"claim": "Hierarchical Bayes multinomial logit for choice data — the shrinkage machinery that makes few-question conjoint work.",
"limitations": "R, GPL-3, dated.",
"disposition": "study — HB shrinkage is the key idea to port"
}
radiant-rstats/radiant.multivariaterecord 40
{
"id": "R40",
"kind": "repo",
"name": "radiant-rstats/radiant.multivariate",
"url": "https://github.com/radiant-rstats/radiant.multivariate",
"licence": "AGPL-3 family (GitHub: NOASSERTION)",
"stars": 6,
"evidence_class": "observed",
"source": "gh api repos/radiant-rstats/radiant.multivariate",
"observed": "2026-08-27",
"claim": "R/Shiny multivariate menu incl. conjoint analysis with a GUI.",
"limitations": "Teaching tool; AGPL-family licence is aggressive for client work.",
"disposition": "reference only"
}
cran/conjointrecord 41
{
"id": "R41",
"kind": "repo",
"name": "cran/conjoint",
"url": "https://github.com/cran/conjoint",
"licence": "GPL",
"stars": 1,
"evidence_class": "observed",
"source": "crandb.r-pkg.org/conjoint (v1.42)",
"observed": "2026-08-27",
"claim": "Traditional ratings-based conjoint analysis in R.",
"limitations": "Ratings-based, not choice-based; dated methodology.",
"disposition": "drop"
}
cran/idefixrecord 42
{
"id": "R42",
"kind": "repo",
"name": "cran/idefix",
"url": "https://github.com/cran/idefix",
"licence": "GPL-3",
"stars": 1,
"evidence_class": "observed",
"source": "crandb.r-pkg.org/idefix (v1.1.0)",
"observed": "2026-08-27",
"claim": "Efficient (D-optimal, Bayesian) designs for discrete choice experiments, INCLUDING adaptive individual-level designs.",
"limitations": "R; requires priors for Bayesian efficiency.",
"disposition": "ADOPT-METHOD — this is the optimal-design lane for choosing which comparisons to show"
}
cran/AlgDesignrecord 43
{
"id": "R43",
"kind": "repo",
"name": "cran/AlgDesign",
"url": "https://github.com/cran/AlgDesign",
"licence": "GPL (>= 2)",
"stars": 4,
"evidence_class": "observed",
"source": "crandb.r-pkg.org/AlgDesign (v1.2.1.2)",
"observed": "2026-08-27",
"claim": "Algorithmic experimental design — D/A/I-optimal exact designs via Federov exchange.",
"limitations": "Generic DOE; no preference model built in.",
"disposition": "study"
}
cran/skprrecord 44
{
"id": "R44",
"kind": "repo",
"name": "cran/skpr",
"url": "https://github.com/cran/skpr",
"licence": "GPL-3",
"stars": 0,
"evidence_class": "observed",
"source": "crandb.r-pkg.org/skpr (v1.9.2). NOTE: tidymodels/skpr and ttgump/skpr are 404; the live dev repo is tylermorganwall/skpr",
"observed": "2026-08-27",
"claim": "Generates and evaluates optimal designs with MONTE-CARLO POWER analysis — answers 'how many trials do I need' empirically.",
"limitations": "R; power analysis is for GLM-type responses, needs adapting to a BT likelihood.",
"disposition": "ADOPT-METHOD — the simulate-to-size approach is how we should actually answer the N question"
}
cran/DoE.baserecord 45
{
"id": "R45",
"kind": "repo",
"name": "cran/DoE.base",
"url": "https://github.com/cran/DoE.base",
"licence": "GPL (>= 2)",
"stars": 1,
"evidence_class": "observed",
"source": "crandb.r-pkg.org/DoE.base (v1.2-5)",
"observed": "2026-08-27",
"claim": "Full factorials, orthogonal arrays, base DOE utilities.",
"limitations": "R; orthogonal arrays are the non-adaptive baseline we expect to beat.",
"disposition": "study — use as the passive baseline to benchmark against"
}
pydoe/pydoerecord 46
{
"id": "R46",
"kind": "repo",
"name": "pydoe/pydoe",
"url": "https://github.com/pydoe/pydoe",
"licence": "BSD-3-Clause",
"stars": 335,
"evidence_class": "observed",
"source": "gh api repos/pydoe/pydoe",
"observed": "2026-08-27",
"claim": "Design of Experiments for Python — factorials, LHS, response surface designs.",
"limitations": "No optimal-design-under-a-choice-model support.",
"disposition": "study"
}
clicumu/pyDOE2record 47
{
"id": "R47",
"kind": "repo",
"name": "clicumu/pyDOE2",
"url": "https://github.com/clicumu/pyDOE2",
"licence": "BSD-3-Clause",
"stars": 166,
"evidence_class": "observed",
"source": "gh api repos/clicumu/pyDOE2",
"observed": "2026-08-27",
"claim": "Maintained fork of pyDOE with bug fixes.",
"limitations": "Same scope limits as pyDOE.",
"disposition": "adopt if a Python DOE primitive is needed"
}
huggingface/trlrecord 48
{
"id": "R48",
"kind": "repo",
"name": "huggingface/trl",
"url": "https://github.com/huggingface/trl",
"licence": "Apache-2.0",
"stars": 19158,
"evidence_class": "observed",
"source": "gh api repos/huggingface/trl",
"observed": "2026-08-27",
"claim": "RewardTrainer fits a Bradley-Terry reward model on pairwise preferences; DPOTrainer optimizes directly from pairs. Production-grade BT-on-preferences at scale.",
"limitations": "Built for LLM-scale data (thousands+ of pairs). Our regime is ~tens of comparisons — the machinery is the wrong size, but the loss formulation is exactly right.",
"disposition": "study — steal the BT loss, not the trainer"
}
OpenRLHF/OpenRLHFrecord 49
{
"id": "R49",
"kind": "repo",
"name": "OpenRLHF/OpenRLHF",
"url": "https://github.com/OpenRLHF/OpenRLHF",
"licence": "Apache-2.0",
"stars": 9957,
"evidence_class": "observed",
"source": "gh api repos/OpenRLHF/OpenRLHF",
"observed": "2026-08-27",
"claim": "Scalable RLHF framework (PPO/DAPO/REINFORCE) on Ray, with reward modelling from preference pairs.",
"limitations": "Distributed-training scale; irrelevant at 7 knobs.",
"disposition": "drop for this lane"
}
CarperAI/trlxrecord 50
{
"id": "R50",
"kind": "repo",
"name": "CarperAI/trlx",
"url": "https://github.com/CarperAI/trlx",
"licence": "MIT",
"stars": 4755,
"evidence_class": "observed",
"source": "gh api repos/CarperAI/trlx",
"observed": "2026-08-27",
"claim": "Distributed RLHF training library.",
"limitations": "Largely superseded by trl; low recent activity.",
"disposition": "drop"
}
argilla-io/argillarecord 51
{
"id": "R51",
"kind": "repo",
"name": "argilla-io/argilla",
"url": "https://github.com/argilla-io/argilla",
"licence": "Apache-2.0",
"stars": 5085,
"evidence_class": "observed",
"source": "gh api repos/argilla-io/argilla",
"observed": "2026-08-27",
"claim": "Collaboration tool for building high-quality labelled/preference datasets — ranking and rating question types, annotator agreement tracking.",
"limitations": "Text/LLM oriented; image-pair UI would need custom work.",
"disposition": "study — good model for the annotation-UI layer and agreement metrics"
}
argilla-io/distilabelrecord 52
{
"id": "R52",
"kind": "repo",
"name": "argilla-io/distilabel",
"url": "https://github.com/argilla-io/distilabel",
"licence": "Apache-2.0",
"stars": 3379,
"evidence_class": "observed",
"source": "gh api repos/argilla-io/distilabel",
"observed": "2026-08-27",
"claim": "Synthetic data + AI feedback pipelines, incl. UltraFeedback-style preference generation.",
"limitations": "AI-judge preferences, not human taste — a different construct.",
"disposition": "study — relevant only if we bootstrap with an AI judge"
}
lm-sys/FastChatrecord 53
{
"id": "R53",
"kind": "repo",
"name": "lm-sys/FastChat",
"url": "https://github.com/lm-sys/FastChat",
"licence": "Apache-2.0",
"stars": 39525,
"evidence_class": "observed",
"source": "gh api repos/lm-sys/FastChat",
"observed": "2026-08-27",
"claim": "Release repo for Vicuna and Chatbot Arena; contains the Arena Elo / Bradley-Terry rating computation used on real human pairwise votes at scale.",
"limitations": "The BT/Elo code is a small part of a very large serving repo; needs extraction.",
"disposition": "study — the arena BT implementation is a battle-tested reference"
}
lmarena/arena-hard-autorecord 54
{
"id": "R54",
"kind": "repo",
"name": "lmarena/arena-hard-auto",
"url": "https://github.com/lmarena/arena-hard-auto",
"licence": "Apache-2.0",
"stars": 1061,
"evidence_class": "observed",
"source": "gh api repos/lmarena/arena-hard-auto (note: lm-sys/arena-hard-auto redirects here)",
"observed": "2026-08-27",
"claim": "Automatic benchmark with BT modelling and bootstrap confidence intervals over pairwise judgments.",
"limitations": "LLM-benchmark specific.",
"disposition": "study — the bootstrap CI approach is directly reusable for reporting taste-model uncertainty"
}
christophschuhmann/improved-aesthetic-predictorrecord 55
{
"id": "R55",
"kind": "repo",
"name": "christophschuhmann/improved-aesthetic-predictor",
"url": "https://github.com/christophschuhmann/improved-aesthetic-predictor",
"licence": "Apache-2.0",
"stars": 1339,
"evidence_class": "observed",
"source": "gh api repos/christophschuhmann/improved-aesthetic-predictor",
"observed": "2026-08-27",
"claim": "CLIP+MLP aesthetic score predictor (LAION) trained on human aesthetic ratings.",
"limitations": "Predicts a GENERIC population aesthetic, not one client's taste. Useful as a prior/feature, not as the target.",
"disposition": "adopt as a FEATURE — warm-start prior, then learn the client's deviation from it"
}
zai-org/ImageRewardrecord 56
{
"id": "R56",
"kind": "repo",
"name": "zai-org/ImageReward",
"url": "https://github.com/zai-org/ImageReward",
"licence": "Apache-2.0",
"stars": 1702,
"evidence_class": "observed",
"source": "gh api repos/zai-org/ImageReward (THUDM/ImageReward redirects here)",
"observed": "2026-08-27",
"claim": "NeurIPS 2023 general-purpose text-to-image human preference reward model, trained on 137k expert comparisons.",
"limitations": "Text-to-image specific; population-level preference.",
"disposition": "study — evidence on how many comparisons a general preference model needed"
}
tgxs002/HPSv2record 57
{
"id": "R57",
"kind": "repo",
"name": "tgxs002/HPSv2",
"url": "https://github.com/tgxs002/HPSv2",
"licence": "Apache-2.0",
"stars": 678,
"evidence_class": "observed",
"source": "gh api repos/tgxs002/HPSv2",
"observed": "2026-08-27",
"claim": "Human Preference Score v2 — benchmark and model for human preferences on generated images.",
"limitations": "Population-level, generated-image domain.",
"disposition": "study"
}
yuvalkirstain/PickScorerecord 58
{
"id": "R58",
"kind": "repo",
"name": "yuvalkirstain/PickScore",
"url": "https://github.com/yuvalkirstain/PickScore",
"licence": "MIT",
"stars": 604,
"evidence_class": "observed",
"source": "gh api repos/yuvalkirstain/PickScore",
"observed": "2026-08-27",
"claim": "CLIP-based preference scorer trained on Pick-a-Pic (~500k pairwise human preferences over generated images).",
"limitations": "Population preference; MIT licence on code, dataset terms separate.",
"disposition": "study — the scale contrast (500k pairs for a population model) is itself the argument for personalising with a prior"
}
anthropics/hh-rlhfrecord 59
{
"id": "R59",
"kind": "repo",
"name": "anthropics/hh-rlhf",
"url": "https://github.com/anthropics/hh-rlhf",
"licence": "MIT",
"stars": 1856,
"evidence_class": "observed",
"source": "gh api repos/anthropics/hh-rlhf",
"observed": "2026-08-27",
"claim": "Human preference dataset for helpfulness/harmlessness — canonical pairwise preference data format.",
"limitations": "Dataset not code; text domain.",
"disposition": "reference — data schema only"
}
A law of comparative judgmentrecord 60
{
"id": "P01",
"kind": "paper",
"name": "A law of comparative judgment",
"url": "https://doi.org/10.1037/h0070288",
"authors": "L. L. Thurstone",
"year": 1927,
"venue": "Psychological Review 34(4)",
"evidence_class": "observed",
"source": "Crossref api.crossref.org DOI 10.1037/h0070288",
"observed": "2026-08-27",
"claim": "Founding model: comparative judgments arise from Gaussian discriminal processes; the difference of two latent normals gives P(A>B) — the Thurstone/probit choice model.",
"limitations": "Metadata verified, full text not read. Assumes a unidimensional latent continuum and Gaussian noise.",
"disposition": "cite as foundation"
}
Rank Analysis of Incomplete Block Designs: I. The Method of Paired Comparisonsrecord 61
{
"id": "P02",
"kind": "paper",
"name": "Rank Analysis of Incomplete Block Designs: I. The Method of Paired Comparisons",
"url": "https://doi.org/10.2307/2334029",
"authors": "Ralph A. Bradley; Milton E. Terry",
"year": 1952,
"venue": "Biometrika 39(3/4):324-345",
"evidence_class": "observed",
"source": "Crossref DOI 10.2307/2334029 (also 10.1093/biomet/39.3-4.324)",
"observed": "2026-08-27",
"claim": "The Bradley-Terry model: P(i beats j) = pi_i/(pi_i+pi_j) — logistic-link paired comparison, the workhorse for our knob scoring.",
"limitations": "Metadata verified only. Assumes transitivity and a single latent scale per item.",
"disposition": "CITE — the core model"
}
Die Berechnung der Turnier-Ergebnisse als ein Maximumproblem der Wahrscheinlichkeitsrechnungrecord 62
{
"id": "P03",
"kind": "paper",
"name": "Die Berechnung der Turnier-Ergebnisse als ein Maximumproblem der Wahrscheinlichkeitsrechnung",
"url": "https://doi.org/10.1007/bf01180541",
"authors": "Ernst Zermelo",
"year": 1929,
"venue": "Mathematische Zeitschrift 29",
"evidence_class": "observed",
"source": "Crossref DOI 10.1007/bf01180541",
"observed": "2026-08-27",
"claim": "Derived the BT model and its MLE 23 years before Bradley & Terry, with an iterative algorithm and convergence conditions for tournament data.",
"limitations": "German; metadata verified only.",
"disposition": "cite for priority"
}
Individual Choice Behavior: A Theoretical Analysisrecord 63
{
"id": "P04",
"kind": "paper",
"name": "Individual Choice Behavior: A Theoretical Analysis",
"url": "https://doi.org/10.1037/14396-000",
"authors": "R. Duncan Luce",
"year": 1959,
"venue": "Wiley (APA reprint ed. 2005)",
"evidence_class": "observed",
"source": "Crossref DOI 10.1037/14396-000; earlier tech report 10.21236/ad0130718 (1957)",
"observed": "2026-08-27",
"claim": "The choice axiom (independence from irrelevant alternatives) — the axiomatic basis for logit/softmax choice probabilities across N-way sets.",
"limitations": "Book, metadata verified only. IIA is empirically violated in exactly the similarity situations design options often exhibit.",
"disposition": "CITE — and flag IIA as a real risk for N-way galleries of similar designs"
}
The Analysis of Permutationsrecord 64
{
"id": "P05",
"kind": "paper",
"name": "The Analysis of Permutations",
"url": "https://doi.org/10.2307/2346567",
"authors": "R. L. Plackett",
"year": 1975,
"venue": "Journal of the Royal Statistical Society Series C (Applied Statistics) 24(2)",
"evidence_class": "observed",
"source": "Crossref DOI 10.2307/2346567",
"observed": "2026-08-27",
"claim": "The Plackett-Luce model for full/partial rankings — sequential Luce choices without replacement. Lets one N-way ranking substitute for many pairs.",
"limitations": "Metadata verified only. Inherits IIA.",
"disposition": "CITE — the model for N-way ranking questions"
}
MM algorithms for generalized Bradley-Terry modelsrecord 65
{
"id": "P06",
"kind": "paper",
"name": "MM algorithms for generalized Bradley-Terry models",
"url": "https://doi.org/10.1214/aos/1079120141",
"authors": "David R. Hunter",
"year": 2004,
"venue": "The Annals of Statistics 32(1):384-406",
"evidence_class": "observed",
"source": "Crossref DOI 10.1214/aos/1079120141",
"observed": "2026-08-27",
"claim": "Minorization-maximization algorithms for BT and generalizations (ties, home-field, Plackett-Luce), with convergence guarantees. The standard practical fitting method.",
"limitations": "Metadata verified only. Convergence needs a strongly-connected comparison graph — a real constraint on which pairs we show.",
"disposition": "CITE — the fitting algorithm, and the connectivity condition matters for our design"
}
Rank Centrality: Ranking from Pair-wise Comparisonsrecord 66
{
"id": "P07",
"kind": "paper",
"name": "Rank Centrality: Ranking from Pair-wise Comparisons",
"url": "https://arxiv.org/abs/1209.1688",
"authors": "Sahand Negahban; Sewoong Oh; Devavrat Shah",
"year": 2012,
"venue": "arXiv 1209.1688 (later Operations Research 2017)",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read",
"observed": "2026-08-27",
"claim": "Spectral rank aggregation via the stationary distribution of a random walk on the comparison graph; finite-sample error bounds for BTL scores. Sample requirement depends on the SPECTRAL GAP of the comparison graph; near order-optimal when each item is compared to randomly chosen others.",
"limitations": "Bound is graph-structure dependent, not a single clean number. Assumes BTL is the true model.",
"disposition": "CITE — the 'which comparisons' structure result"
}
Spectral MLE: Top-K Rank Aggregation from Pairwise Comparisonsrecord 67
{
"id": "P08",
"kind": "paper",
"name": "Spectral MLE: Top-K Rank Aggregation from Pairwise Comparisons",
"url": "https://arxiv.org/abs/1504.07218",
"authors": "Yuxin Chen; Changho Suh",
"year": 2015,
"venue": "ICML 2015",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read",
"observed": "2026-08-27",
"claim": "Characterizes minimax limits for top-K identification under BTL: minimum sample complexity scales INVERSELY with the separation between the Kth and (K+1)th scores, irrespective of other distribution metrics.",
"limitations": "Random non-adaptive sampling; top-K not full ranking.",
"disposition": "CITE — the separation-measure insight explains why near-tied knob settings cost disproportionately many comparisons"
}
Top-K Ranking from Pairwise Comparisons: When Spectral Ranking is Optimalrecord 68
{
"id": "P09",
"kind": "paper",
"name": "Top-K Ranking from Pairwise Comparisons: When Spectral Ranking is Optimal",
"url": "https://arxiv.org/abs/1603.04153",
"authors": "Minje Jang; Sunghyun Kim; Changho Suh; Sewoong Oh",
"year": 2016,
"venue": "arXiv 1603.04153",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read",
"observed": "2026-08-27",
"claim": "Upper and lower sample-size bounds for reliable top-K recovery; under a random comparison model the gap narrows to a constant factor, so a spectral method alone attains minimax-optimal sample complexity without a second MLE stage.",
"limitations": "'Slightly restricted regimes'; BTL assumed.",
"disposition": "cite"
}
Active Ranking from Pairwise Comparisons and when Parametric Assumptions Don't Helprecord 69
{
"id": "P10",
"kind": "paper",
"name": "Active Ranking from Pairwise Comparisons and when Parametric Assumptions Don't Help",
"url": "https://arxiv.org/abs/1606.08842",
"authors": "Reinhard Heckel; Nihar B. Shah; Kannan Ramchandran; Martin J. Wainwright",
"year": 2016,
"venue": "arXiv; published Annals of Statistics 47(6) 2019, DOI 10.1214/18-AOS1772",
"evidence_class": "observed",
"source": "arXiv Atom API full abstract + Crossref DOI 10.1214/18-aos1772",
"observed": "2026-08-27",
"claim": "Adaptive algorithm counting comparisons won, with confidence-interval stopping, achieves ranking with a number of comparisons OPTIMAL UP TO LOGARITHMIC FACTORS and requires NO structural assumption on the comparison-probability matrix. Crucially proves a lower bound for parametric models showing BTL/Thurstone assumptions buy AT MOST LOGARITHMIC GAINS for stochastic comparisons.",
"limitations": "Requires pairwise probabilities bounded away from zero. Ranks by 'probability of beating a random item', not a utility vector.",
"disposition": "CITE — HIGH IMPORTANCE: this is the honest counterweight to assuming BTL saves us a lot of questions"
}
Simple, Robust and Optimal Ranking from Pairwise Comparisonsrecord 70
{
"id": "P11",
"kind": "paper",
"name": "Simple, Robust and Optimal Ranking from Pairwise Comparisons",
"url": "https://arxiv.org/abs/1512.08949",
"authors": "Nihar B. Shah; Martin J. Wainwright",
"year": 2015,
"venue": "arXiv 1512.08949 (JMLR 2017)",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read",
"observed": "2026-08-27",
"claim": "The Copeland counting algorithm (rank by number of comparisons won) is optimal up to CONSTANT factors for top-k recovery, imposes no conditions on the comparison-probability matrix, and is orders of magnitude faster than prior methods.",
"limitations": "Top-k subset recovery; constant factors unspecified in the abstract.",
"disposition": "CITE — argues a trivially simple estimator may suffice; a strong simplicity check on our design"
}
Active Ranking using Pairwise Comparisonsrecord 71
{
"id": "P12",
"kind": "paper",
"name": "Active Ranking using Pairwise Comparisons",
"url": "https://arxiv.org/abs/1109.3701",
"authors": "Kevin G. Jamieson; Robert D. Nowak",
"year": 2011,
"venue": "NIPS 2011, pp. 2240-2248",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read verbatim",
"observed": "2026-08-27",
"claim": "THE CRITICAL RESULT. Objects embedded in d-dimensional Euclidean space, ranked by distance from a common reference point: the number of possible rankings grows like n^(2d), and an algorithm identifies a randomly selected ranking using JUST SLIGHTLY MORE THAN d log n ADAPTIVELY selected pairwise comparisons, ON AVERAGE. If comparisons are chosen at random instead, ALMOST ALL pairwise comparisons must be made. Also gives an error-tolerant variant requiring only that comparisons are probably correct.",
"limitations": "'On average' over a randomly selected ranking, NOT worst case — the paper notes that for d>=2 there exist object placements requiring at least n-1 queries. Requires the low-dimensional embedding assumption to actually hold. Adaptivity is essential.",
"disposition": "CITE FIRST — directly licenses the 'few comparisons over 7 knobs' claim, WITH its average-case caveat"
}
An Active Learning Algorithm for Ranking from Pairwise Preferences with an Almost Optimal Query Complexityrecord 72
{
"id": "P13",
"kind": "paper",
"name": "An Active Learning Algorithm for Ranking from Pairwise Preferences with an Almost Optimal Query Complexity",
"url": "https://arxiv.org/abs/1011.0108",
"authors": "Nir Ailon",
"year": 2010,
"venue": "arXiv 1011.0108 (JMLR 2012)",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read",
"observed": "2026-08-27",
"claim": "Active learning for ranking under possible NON-TRANSITIVITY from human error/irrationality; query bounds significantly beat non-active bounds for the same error guarantee, almost achieving the information-theoretic lower bound.",
"limitations": "Abstract states the comparative result but not a closed-form bound; minimizes disagreement loss rather than recovering a true order.",
"disposition": "CITE — the tolerance for intransitive human answers is essential for real taste data"
}
Noisy Sorting Without Resamplingrecord 73
{
"id": "P14",
"kind": "paper",
"name": "Noisy Sorting Without Resampling",
"url": "https://arxiv.org/abs/0707.1051",
"authors": "Mark Braverman; Elchanan Mossel",
"year": 2007,
"venue": "arXiv 0707.1051 (SODA 2008)",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read verbatim",
"observed": "2026-08-27",
"claim": "With each comparison correct with probability at least 1/2+gamma and NO re-querying of the same pair, an algorithm with running time n^O(gamma^-4) and SAMPLING COMPLEXITY O_gamma(n log n) recovers the order w.h.p. The recovered order sigma satisfies sum_i|sigma(i)-pi(i)| = Theta(n) and max_i|sigma(i)-pi(i)| = Theta(log n).",
"limitations": "No-resampling model (you cannot ask the same pair twice — realistic for humans). Exact recovery is impossible for gamma<1/2 at large n; only closeness is guaranteed.",
"disposition": "CITE — sets the noisy passive baseline of n log n that adaptivity must beat"
}
Minimax-optimal Inference from Partial Rankingsrecord 74
{
"id": "P15",
"kind": "paper",
"name": "Minimax-optimal Inference from Partial Rankings",
"url": "https://arxiv.org/abs/1406.5638",
"authors": "Bruce Hajek; Sewoong Oh; Jiaming Xu",
"year": 2014,
"venue": "NIPS 2014",
"evidence_class": "observed",
"source": "arXiv Atom API title/author/year verified",
"observed": "2026-08-27",
"claim": "Minimax error rates for estimating BTL/PL parameters from partial rankings; quantifies the information gained per ranking of a set of size k.",
"limitations": "Abstract confirmed via search result, full abstract not individually read. Asymptotic rates.",
"disposition": "cite — supports N-way ranking over pairs on information grounds"
}
When is it Better to Compare than to Score?record 75
{
"id": "P16",
"kind": "paper",
"name": "When is it Better to Compare than to Score?",
"url": "https://arxiv.org/abs/1406.6618",
"authors": "Nihar B. Shah; Sivaraman Balakrishnan; Joseph Bradley; Abhay Parekh; Kannan Ramchandran; Martin Wainwright",
"year": 2014,
"venue": "arXiv 1406.6618",
"evidence_class": "observed",
"source": "arXiv Atom API — title/authors/year verified",
"observed": "2026-08-27",
"claim": "Directly analyses the cardinal (rating) vs ordinal (comparison) tradeoff — when pairwise comparison beats direct scoring.",
"limitations": "I verified metadata but did not read the full abstract; do not quote a specific bound from this without reading it.",
"disposition": "CITE — directly on our 'should we ask for ratings or comparisons' decision; READ FULL TEXT BEFORE QUOTING"
}
On a Measure of the Information Provided by an Experimentrecord 76
{
"id": "P17",
"kind": "paper",
"name": "On a Measure of the Information Provided by an Experiment",
"url": "https://doi.org/10.1214/aoms/1177728069",
"authors": "D. V. Lindley",
"year": 1956,
"venue": "The Annals of Mathematical Statistics 27(4):986-1005",
"evidence_class": "observed",
"source": "Crossref DOI 10.1214/aoms/1177728069",
"observed": "2026-08-27",
"claim": "Expected information gain (expected KL from prior to posterior) as the criterion for experiment value — the formal basis for choosing the most informative next comparison.",
"limitations": "Metadata verified only. Myopic/one-step when applied greedily.",
"disposition": "CITE — the objective function for query selection"
}
Information-Based Objective Functions for Active Data Selectionrecord 77
{
"id": "P18",
"kind": "paper",
"name": "Information-Based Objective Functions for Active Data Selection",
"url": "https://doi.org/10.1162/neco.1992.4.4.590",
"authors": "David J. C. MacKay",
"year": 1992,
"venue": "Neural Computation 4(4):590-604",
"evidence_class": "observed",
"source": "Crossref DOI 10.1162/neco.1992.4.4.590",
"observed": "2026-08-27",
"claim": "Practical information-based objectives for active data selection in Bayesian models; the bridge from Lindley's theory to usable acquisition functions.",
"limitations": "Metadata verified only. Gaussian approximations.",
"disposition": "CITE"
}
Bayesian Experimental Design: A Reviewrecord 78
{
"id": "P19",
"kind": "paper",
"name": "Bayesian Experimental Design: A Review",
"url": "https://doi.org/10.1214/ss/1177009939",
"authors": "Kathryn Chaloner; Isabella Verdinelli",
"year": 1995,
"venue": "Statistical Science 10(3):273-304",
"evidence_class": "observed",
"source": "Crossref DOI 10.1214/ss/1177009939",
"observed": "2026-08-27",
"claim": "Canonical review unifying Bayesian design criteria (D-, A-, E-optimality) under a decision-theoretic utility framework.",
"limitations": "Metadata verified only; pre-dates modern computational methods.",
"disposition": "CITE — the reference for D-optimality framing"
}
Bayesian Active Learning for Classification and Preference Learningrecord 79
{
"id": "P20",
"kind": "paper",
"name": "Bayesian Active Learning for Classification and Preference Learning",
"url": "https://arxiv.org/abs/1112.5745",
"authors": "Neil Houlsby; Ferenc Huszar; Zoubin Ghahramani; Mate Lengyel",
"year": 2011,
"venue": "arXiv 1112.5745",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read",
"observed": "2026-08-27",
"claim": "BALD: expresses information gain in terms of PREDICTIVE ENTROPIES, making it tractable for GP classifiers; explicitly EXTENDS TO GAUSSIAN PROCESS PREFERENCE LEARNING by reformulating binary preference learning as classification. Equal or lower computational cost than competitors.",
"limitations": "Myopic one-step criterion.",
"disposition": "ADOPT — this is the acquisition function to implement for pairwise queries"
}
Preference learning with Gaussian processesrecord 80
{
"id": "P21",
"kind": "paper",
"name": "Preference learning with Gaussian processes",
"url": "https://doi.org/10.1145/1102351.1102369",
"authors": "Wei Chu; Zoubin Ghahramani",
"year": 2005,
"venue": "ICML 2005",
"evidence_class": "observed",
"source": "Crossref DOI 10.1145/1102351.1102369",
"observed": "2026-08-27",
"claim": "GP prior over a latent preference function with a Thurstone-style likelihood on pairwise comparisons; Laplace approximation for inference. The foundation of BoTorch's PairwiseGP.",
"limitations": "Metadata verified only. Laplace approximation quality degrades with very few observations.",
"disposition": "CITE — the model behind PairwiseGP"
}
Active Preference Learning with Discrete Choice Datarecord 81
{
"id": "P22",
"kind": "paper",
"name": "Active Preference Learning with Discrete Choice Data",
"url": "https://proceedings.neurips.cc/paper/2007/file/b6a1085a27ab7bff7550f8a3bd017df8-Paper.pdf",
"authors": "Eric Brochu; Nando de Freitas; Abhijeet Ghosh",
"year": 2007,
"venue": "NIPS 2007",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — NeurIPS proceedings PDF, 8pp, Table 1 extracted",
"observed": "2026-08-27",
"claim": "GP preference model + expected-improvement acquisition over gallery queries. USER STUDY (5 subjects, 50 trials each arm, 38 MERL BRDFs): maxEI needed a MEAN OF 8.56 +/- 5.23 clicks to find a target, versus 17.87 +/- 8.60 for max-variance and 18.40 +/- 7.87 for Latin hypercubes. Seeded with 4 predetermined queries. Authors state 'requiring more than 50 user queries in a real application would be unacceptable'.",
"limitations": "Target-finding task (a known target exists), NOT open-ended taste discovery — an easier problem than ours. In 6D 'neither algorithm succeeds well in finding the optimum'. n=5 subjects.",
"disposition": "CITE — best hard number for 'how many clicks', with the caveat that 6D already strained it"
}
Preferential Bayesian Optimizationrecord 82
{
"id": "P23",
"kind": "paper",
"name": "Preferential Bayesian Optimization",
"url": "https://arxiv.org/abs/1704.03651",
"authors": "Javier Gonzalez; Zhenwen Dai; Andreas Damianou; Neil D. Lawrence",
"year": 2017,
"venue": "ICML 2017",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read",
"observed": "2026-08-27",
"claim": "Formalises PBO: optimize a latent function queried only through pairwise duels, modelling duel outcomes with a GP + Bernoulli likelihood; introduces acquisition functions incl. the dueling-Thompson approach. Reports PBO 'needs drastically fewer comparisons for finding the optimum', with correlation modelling identified as the key advantage.",
"limitations": "'Drastically fewer' is comparative, NOT an absolute query count. Experiments on synthetic functions.",
"disposition": "CITE — the framework paper; do not quote a number from it"
}
qEUBO: A Decision-Theoretic Acquisition Function for Preferential Bayesian Optimizationrecord 83
{
"id": "P24",
"kind": "paper",
"name": "qEUBO: A Decision-Theoretic Acquisition Function for Preferential Bayesian Optimization",
"url": "https://arxiv.org/abs/2303.15746",
"authors": "Raul Astudillo; Zhiyuan Jerry Lin; Eytan Bakshy; Peter I. Frazier",
"year": 2023,
"venue": "AISTATS 2023",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read",
"observed": "2026-08-27",
"claim": "qEUBO (expected utility of the best option) is ONE-STEP BAYES OPTIMAL under noise-free responses (equivalent to knowledge gradient), has an additive-constant approximation guarantee under noisy responses, and its Bayesian simple regret converges to zero at rate o(1/n). Shows qEI can FAIL to converge for PBO.",
"limitations": "Regret rate is asymptotic in n, gives no finite-sample query count. 'Sufficient regularity conditions'.",
"disposition": "ADOPT — the acquisition function to use, already in BoTorch (R14/R16)"
}
Sequential Line Search for Efficient Visual Design Optimization by Crowdsrecord 84
{
"id": "P25",
"kind": "paper",
"name": "Sequential Line Search for Efficient Visual Design Optimization by Crowds",
"url": "https://doi.org/10.1145/3072959.3073598",
"authors": "Yuki Koyama; Issei Sato; Daisuke Sakurai; Takeo Igarashi",
"year": 2017,
"venue": "ACM TOG 36(4), SIGGRAPH 2017",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — koyama.xyz/project/sequential_line_search/download/preprint.pdf",
"observed": "2026-08-27",
"claim": "CLOSEST PRIOR ART #1. Preferential BO where each query is a 1D slider through high-dim design space. ALL REPORTED RESULTS USED 15 ITERATIONS, 7 crowd microtasks per iteration, proceeding once >=5 responses arrived (total crowd cost 5.25 USD per result). Convergence observation: 'the distances become small rapidly in the first 4 or 5 iterations'. Design spaces: 6D photo colour enhancement, 3D and 7D BRDF material design; synthetic tests at n in {2,6,20}.",
"limitations": "NO explicit iteration-vs-dimension scaling law is stated in the paper (verified absent by full-text reader). Crowd-aggregated, not single-client. 15 was a chosen budget, not a derived requirement.",
"disposition": "CITE FIRST among applied work — 7D is our exact dimensionality, and 'good by 4-5 iterations' is the most directly transferable number"
}
Sequential Gallery for Interactive Visual Design Optimizationrecord 85
{
"id": "P26",
"kind": "paper",
"name": "Sequential Gallery for Interactive Visual Design Optimization",
"url": "https://arxiv.org/abs/2005.04107",
"authors": "Yuki Koyama; Issei Sato; Masataka Goto",
"year": 2020,
"venue": "ACM TOG 39(4):88:1-88:12, SIGGRAPH 2020, DOI 10.1145/3386569.3392444",
"evidence_class": "observed",
"source": "arXiv Atom API abstract + FULL TEXT PDF independently re-derived by me (pdftotext on arxiv.org/pdf/2005.04107v1)",
"observed": "2026-08-27",
"claim": "CLOSEST PRIOR ART #2. Sequential plane search: user picks best from a 2D grid gallery each round. USER STUDY: 'The mean iteration count necessary for finding satisfactory results was 5.36 with SD = 2.69'; 'One plane-search subtask took 14.8 seconds on average'; participants were asked to continue for 15 iterations regardless. 5 of 6 participants hit satisfaction within 15. Synthetic functions at 5D, 15D, 10D, 20D. Grid: 25 points in a 5-by-5 lattice for photo enhancement, reduced to 3-by-3 for body-shape design on a 13-inch display.",
"limitations": "n=6 participants (5 students + 1 researcher) — a PRELIMINARY user study, not a powered experiment. 'Satisfactory' is self-reported, not a measured distance to a ground-truth optimum. Grid size is application-dependent — quoting one number would repeat the 728-vs-113 error class.",
"disposition": "CITE FIRST — 5.36 +/- 2.69 rounds at ~15s each is the single most quotable applied figure for our deliverable"
}
A Bayesian Interactive Optimization Approach to Procedural Animation Designrecord 86
{
"id": "P27",
"kind": "paper",
"name": "A Bayesian Interactive Optimization Approach to Procedural Animation Design",
"url": "http://haikufactory.com/files/sca2010.pdf",
"authors": "Eric Brochu; Tyson Brochu; Nando de Freitas",
"year": 2010,
"venue": "ACM SIGGRAPH/Eurographics Symposium on Computer Animation (SCA) 2010",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — Table 1 extracted verbatim",
"observed": "2026-08-27",
"claim": "Gallery-based interactive BO for animation design. Table 1: 4-gallery on 4 parameters needed 7.57 +/- 4.67 iterations; PFHP pairwise on 4 params 8.45 +/- 2.81; the full 12-parameter task with 4-gallery+manual needed 5.38 +/- 2.63 iterations versus 28.40-35.33 for manual tuning. A learned prior cut iterations 'from an average of 11.25 to just 6.5'. Authors adopted a 20-iteration budget 'because this is roughly the point at which users start to quit if they do not see significant improvement'.",
"limitations": "Animation domain; iteration counts are not directly error-normalised across rows (error column varies). Small session counts (20-30).",
"disposition": "CITE — the 20-iteration patience ceiling is a hard UX constraint for our design"
}
Interactively optimizing information retrieval systems as a dueling bandits problemrecord 87
{
"id": "P28",
"kind": "paper",
"name": "Interactively optimizing information retrieval systems as a dueling bandits problem",
"url": "https://doi.org/10.1145/1553374.1553527",
"authors": "Yisong Yue; Thorsten Joachims",
"year": 2009,
"venue": "ICML 2009",
"evidence_class": "observed",
"source": "Crossref DOI 10.1145/1553374.1553527",
"observed": "2026-08-27",
"claim": "Introduces Dueling Bandit Gradient Descent — online optimization from relative (pairwise) feedback only.",
"limitations": "Metadata verified only. Online regret framing, continuous operation rather than a fixed elicitation session.",
"disposition": "cite"
}
The K-armed dueling bandits problemrecord 88
{
"id": "P29",
"kind": "paper",
"name": "The K-armed dueling bandits problem",
"url": "https://doi.org/10.1016/j.jcss.2011.12.028",
"authors": "Yisong Yue; Josef Broder; Robert Kleinberg; Thorsten Joachims",
"year": 2012,
"venue": "Journal of Computer and System Sciences 78(5) (COLT 2009 journal version)",
"evidence_class": "observed",
"source": "Crossref DOI 10.1016/j.jcss.2011.12.028",
"observed": "2026-08-27",
"claim": "Formalises the K-armed dueling bandit problem and gives the Interleaved Filter algorithm with regret bounds under relative feedback.",
"limitations": "Metadata verified only. Regret-minimisation, not fixed-budget identification — a different objective from ours.",
"disposition": "cite"
}
Relative Upper Confidence Bound for the K-Armed Dueling Bandit Problemrecord 89
{
"id": "P30",
"kind": "paper",
"name": "Relative Upper Confidence Bound for the K-Armed Dueling Bandit Problem",
"url": "https://arxiv.org/abs/1312.3393",
"authors": "Masrour Zoghi; Shimon Whiteson; Remi Munos; Maarten de Rijke",
"year": 2013,
"venue": "ICML 2014",
"evidence_class": "observed",
"source": "arXiv Atom API — title/authors/year verified",
"observed": "2026-08-27",
"claim": "RUCB: the first dueling-bandit algorithm not requiring knowledge of the time horizon or an explicit exploration phase.",
"limitations": "Metadata verified, full abstract not read. Finite arm set.",
"disposition": "cite"
}
Advancements in Dueling Banditsrecord 90
{
"id": "P31",
"kind": "paper",
"name": "Advancements in Dueling Bandits",
"url": "https://doi.org/10.24963/ijcai.2018/776",
"authors": "Yanan Sui; Masrour Zoghi; Katja Hofmann; Yisong Yue",
"year": 2018,
"venue": "IJCAI 2018 survey track",
"evidence_class": "observed",
"source": "Crossref DOI 10.24963/ijcai.2018/776",
"observed": "2026-08-27",
"claim": "Survey of dueling bandit formulations, algorithms and applications — the orientation map for relative-feedback methods.",
"limitations": "Metadata verified only; 2018 cutoff.",
"disposition": "cite — read first for orientation"
}
TrueSkill: A Bayesian Skill Rating Systemrecord 91
{
"id": "P32",
"kind": "paper",
"name": "TrueSkill: A Bayesian Skill Rating System",
"url": "https://doi.org/10.7551/mitpress/7503.003.0076",
"authors": "Ralf Herbrich; Tom Minka; Thore Graepel",
"year": 2007,
"venue": "NIPS 2006 (MIT Press proceedings)",
"evidence_class": "observed",
"source": "Crossref DOI 10.7551/mitpress/7503.003.0076",
"observed": "2026-08-27",
"claim": "Factor-graph Bayesian skill rating with per-item Gaussian uncertainty updated by expectation propagation — gives calibrated uncertainty per design, which Elo does not.",
"limitations": "Metadata verified only. Designed for matches, and the TrueSkill name is trademark-encumbered (see R02).",
"disposition": "CITE — the uncertainty-tracking idea; implement via openskill.py not sublee/trueskill"
}
TrueSkill 2: An improved Bayesian skill rating systemrecord 92
{
"id": "P33",
"kind": "paper",
"name": "TrueSkill 2: An improved Bayesian skill rating system",
"url": "https://www.microsoft.com/en-us/research/publication/trueskill-2-improved-bayesian-skill-rating-system/",
"authors": "Tom Minka; Ryan Cleven; Yordan Zaykov",
"year": 2018,
"venue": "Microsoft Research Technical Report MSR-TR-2018-8",
"evidence_class": "hypothesis",
"source": "NOT VERIFIED — Crossref returned no matching record; I did not fetch the MSR page (WebFetch/WebSearch blocked by credit limit this session)",
"observed": "2026-08-27",
"claim": "Reported extension of TrueSkill handling team correlations and individual statistics.",
"limitations": "EXISTENCE NOT INDEPENDENTLY VERIFIED IN THIS SESSION. It is a technical report, not a DOI-indexed publication, which is why Crossref misses it. Do not cite in client-facing material until the MSR page is fetched.",
"disposition": "HOLD — verify before use"
}
Some probabilistic models of best, worst, and best-worst choicesrecord 93
{
"id": "P34",
"kind": "paper",
"name": "Some probabilistic models of best, worst, and best-worst choices",
"url": "https://doi.org/10.1016/j.jmp.2005.05.003",
"authors": "A. A. J. Marley; Jordan J. Louviere",
"year": 2005,
"venue": "Journal of Mathematical Psychology 49(6):464-480",
"evidence_class": "observed",
"source": "Crossref DOI 10.1016/j.jmp.2005.05.003",
"observed": "2026-08-27",
"claim": "Axiomatic probabilistic models for best-worst choice, relating BW choice probabilities to underlying scale values.",
"limitations": "Metadata verified only; theory paper without an information-gain number.",
"disposition": "CITE — the theoretical basis for MaxDiff"
}
Probabilistic models of set-dependent and attribute-level best-worst choicerecord 94
{
"id": "P35",
"kind": "paper",
"name": "Probabilistic models of set-dependent and attribute-level best-worst choice",
"url": "https://doi.org/10.1016/j.jmp.2008.02.002",
"authors": "A. A. J. Marley; Terry N. Flynn; Jordan J. Louviere",
"year": 2008,
"venue": "Journal of Mathematical Psychology 52(5)",
"evidence_class": "observed",
"source": "Crossref DOI 10.1016/j.jmp.2008.02.002",
"observed": "2026-08-27",
"claim": "Extends BW models to ATTRIBUTE-LEVEL (Case 2) best-worst — choosing the best and worst attribute-level within a single profile.",
"limitations": "Metadata verified only.",
"disposition": "CITE — Case 2 is the format that maps onto 'which knob is most/least right on this design'"
}
Best-worst scaling: What it can do for health care research and how to do itrecord 95
{
"id": "P36",
"kind": "paper",
"name": "Best-worst scaling: What it can do for health care research and how to do it",
"url": "https://doi.org/10.1016/j.jhealeco.2006.04.002",
"authors": "Terry N. Flynn; Jordan J. Louviere; Tim J. Peters; Joanna Coast",
"year": 2007,
"venue": "Journal of Health Economics 26(1):171-189",
"evidence_class": "observed",
"source": "Crossref DOI 10.1016/j.jhealeco.2006.04.002 (metadata; Crossref carries no abstract for this record)",
"observed": "2026-08-27",
"claim": "Practical guide to designing and analysing BWS studies.",
"limitations": "I could NOT verify a specific quantified 'information gain of best-worst vs single choice' from this record — Crossref returns no abstract and I did not obtain full text. The common claim that best-worst yields more information per question than a single choice is INTUITIVELY grounded (a best AND a worst pick implies more implied pairwise orderings than one pick) but I have NOT verified a numeric multiplier from any source.",
"disposition": "CITE for method; DO NOT quote an information-gain multiplier without reading full text"
}
Best-Worst Scaling: Theory, Methods and Applicationsrecord 96
{
"id": "P37",
"kind": "paper",
"name": "Best-Worst Scaling: Theory, Methods and Applications",
"url": "https://doi.org/10.1017/cbo9781107337855",
"authors": "Jordan J. Louviere; Terry N. Flynn; A. A. J. Marley",
"year": 2015,
"venue": "Cambridge University Press",
"evidence_class": "observed",
"source": "Crossref DOI 10.1017/cbo9781107337855",
"observed": "2026-08-27",
"claim": "The canonical book-length treatment of BWS Cases 1-3.",
"limitations": "Book; metadata verified only.",
"disposition": "cite — the reference for BWS design"
}
Conjoint Analysis in Consumer Research: Issues and Outlookrecord 97
{
"id": "P38",
"kind": "paper",
"name": "Conjoint Analysis in Consumer Research: Issues and Outlook",
"url": "https://doi.org/10.1086/208721",
"authors": "Paul E. Green; V. Srinivasan",
"year": 1978,
"venue": "Journal of Consumer Research 5(2):103-123",
"evidence_class": "observed",
"source": "Crossref DOI 10.1086/208721",
"observed": "2026-08-27",
"claim": "The foundational review establishing conjoint analysis — decomposing overall preference judgments into attribute part-worths. Our 7 knobs are a conjoint attribute set.",
"limitations": "Metadata verified only; 1978, pre-choice-based methods.",
"disposition": "CITE — the framing ancestor of our problem"
}
Fast Polyhedral Adaptive Conjoint Estimationrecord 98
{
"id": "P39",
"kind": "paper",
"name": "Fast Polyhedral Adaptive Conjoint Estimation",
"url": "https://mitsloan.mit.edu/shared/ods/documents/?PublicationDocumentID=5603",
"authors": "Olivier Toubia; Duncan I. Simester; John R. Hauser; Ely Dahan",
"year": 2003,
"venue": "Marketing Science 22(3):273-303 (SSRN DOI 10.2139/ssrn.374460)",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — MIT Sloan hosted typeset paper",
"observed": "2026-08-27",
"claim": "THE KEY ADAPTIVE-CONJOINT RESULT. Explicitly targets q < p (fewer questions than parameters): 'We anticipate that the polyhedral methods are particularly well-suited to solving problems in which there are a large number of parameters relative to the number of responses from each individual (q < p). Thus, we vary the number of questions from slightly less than the number of parameters (q = 8) to comfortably more than the number of parameters (q = 16).' With p = 10 parameters. Abstract: 'For low numbers of questions, polyhedral question design does best (or is tied for best) for all tested domains. For high numbers of questions, efficient fixed designs do better in some domains.' Field test: 330 respondents, 16 paired-comparison questions + 4 holdouts, 9 binary features + price.",
"limitations": "'Does best' is RELATIVE to competing designs, not an absolute accuracy guarantee. Advantage REVERSES at high question counts. Simulations use 500 synthetic respondents in 5 sets of 100 with assumed response-error levels. Metric paired comparisons, not the discrete choices we would use.",
"disposition": "CITE — the strongest published evidence that adaptive design works with FEWER questions than parameters"
}
Polyhedral Methods for Adaptive Choice-Based Conjoint Analysisrecord 99
{
"id": "P40",
"kind": "paper",
"name": "Polyhedral Methods for Adaptive Choice-Based Conjoint Analysis",
"url": "https://mitsloan.mit.edu/shared/ods/documents/?PublicationDocumentID=4040",
"authors": "Olivier Toubia; John R. Hauser; Duncan I. Simester",
"year": 2004,
"venue": "Journal of Marketing Research 41(1):116-131",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker",
"observed": "2026-08-27",
"claim": "Choice-based (not metric) polyhedral adaptive conjoint. Simulation used q = 16 questions on 4 features x 4 levels x 4 profiles (12 free part-worths), with q = 8 and q = 24 giving 'similar qualitative insights'; 1000 simulated respondents. Field application: 354 web respondents, 8 multi-level program features, and 'Pretests confirmed that respondents could comfortably answer 12 stated-choice questions'.",
"limitations": "The 12-question figure is a COMFORT/pretest finding, not an accuracy-derived requirement. Aggregate-level field validation.",
"disposition": "CITE — '12 choice questions for 8 multi-level features' is the most transferable applied number"
}
The magical number seven, plus or minus two: Some limits on our capacity for processing informationrecord 100
{
"id": "P41",
"kind": "paper",
"name": "The magical number seven, plus or minus two: Some limits on our capacity for processing information",
"url": "https://doi.org/10.1037/h0043158",
"authors": "George A. Miller",
"year": 1956,
"venue": "Psychological Review 63(2):81-97",
"evidence_class": "observed",
"source": "Crossref DOI 10.1037/h0043158",
"observed": "2026-08-27",
"claim": "The famous 7+/-2. WHAT IT ACTUALLY SUPPORTS: limits on ABSOLUTE JUDGMENT of unidimensional stimuli and on IMMEDIATE MEMORY span for chunks. Miller himself treats the recurrence of ~7 across these distinct paradigms with explicit irony.",
"limitations": "COMMONLY MISCITED. It is NOT a finding about how many options to display in an interface, nor about menu length, nor a working-memory capacity of 7 items for complex material (later work, e.g. Cowan, argues ~4). Do NOT use it to justify a 7-option gallery. Metadata verified; the interpretive caution above is standard in the literature but I did not re-read the 1956 text this session.",
"disposition": "CITE ONLY WITH THE CORRECTION — flagging the misuse is more valuable to the client than repeating it"
}
When choice is demotivating: Can one desire too much of a good thing?record 101
{
"id": "P42",
"kind": "paper",
"name": "When choice is demotivating: Can one desire too much of a good thing?",
"url": "https://doi.org/10.1037/0022-3514.79.6.995",
"authors": "Sheena S. Iyengar; Mark R. Lepper",
"year": 2000,
"venue": "Journal of Personality and Social Psychology 79(6):995-1006",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — faculty.washington.edu hosted copy, quotes extracted verbatim",
"observed": "2026-08-27",
"claim": "The jam study. Study 1: a tasting booth displayed either a limited (6) or extensive (24) set of jams. 'Nearly 30% (31) of the consumers in the limited-choice condition subsequently purchased a jar... in contrast, only 3% (4) of the consumers in the extensive-choice condition did so, X2(1, N = 249) = 32.34, p < .0001.' But ATTRACTION ran the opposite way: 60% of passers-by stopped at the extensive display vs 40% at the limited one. Study 2 used 6 vs 30 essay topics (the '30' in the abstract is NOT the jam study).",
"limitations": "MUST be reported alongside P43. Single-site field study; the 30%/3% denominators are those who stopped (N=249), while 60%/40% are of passers-by. The chocolate study figures (48% vs 12%) are a DIFFERENT study and are frequently conflated with the jam numbers.",
"disposition": "CITE ONLY PAIRED WITH P43 — citing this alone as a design principle will not survive scrutiny"
}
Can There Ever Be Too Many Options? A Meta-Analytic Review of Choice Overloadrecord 102
{
"id": "P43",
"kind": "paper",
"name": "Can There Ever Be Too Many Options? A Meta-Analytic Review of Choice Overload",
"url": "https://doi.org/10.1086/651235",
"authors": "Benjamin Scheibehenne; Rainer Greifeneder; Peter M. Todd",
"year": 2010,
"venue": "Journal of Consumer Research 37(3):409-425",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — scheibehenne.de author-hosted copy, quotes extracted verbatim",
"observed": "2026-08-27",
"claim": "DECISIVE REPLICATION EVIDENCE. 'In a meta-analysis of 63 conditions from 50 published and unpublished experiments (N = 5,036), we found a mean effect size of virtually zero but considerable variance between studies.' Exact: 'The mean effect size of choice overload across all 63 data points... is D = 0.02 (95% confidence interval [CI95] -0.09 to 0.12).' Trimmed 20%: 'Dtrimmed = 0.001 (CI95 -0.08 to 0.07).' Heterogeneity I2 = 68% untrimmed, 22% trimmed. Conclusion: 'no sufficient conditions could be identified' and 'adverse consequences due to having too much choice are not a robust phenomenon'. Also notes a slight publication bias.",
"limitations": "High heterogeneity untrimmed (I2=68%) means moderators may exist even though none were identified as sufficient. Meta-analysis to 2010.",
"disposition": "CITE — ANSWERS THE REPLICATION QUESTION: choice overload does NOT robustly replicate; gallery size should be chosen on information grounds, not overload fear"
}
The construction of preferencerecord 103
{
"id": "P44",
"kind": "paper",
"name": "The construction of preference",
"url": "https://doi.org/10.1037/0003-066x.50.5.364",
"authors": "Paul Slovic",
"year": 1995,
"venue": "American Psychologist 50(5):364-371",
"evidence_class": "observed",
"source": "Crossref DOI 10.1037/0003-066x.50.5.364",
"observed": "2026-08-27",
"claim": "Preferences are frequently CONSTRUCTED during elicitation rather than retrieved from a stable store — they are sensitive to framing, response mode and context.",
"limitations": "Metadata verified only. This is a theoretical/review synthesis.",
"disposition": "CITE — THE key epistemic caveat: the elicitation procedure partly CREATES the taste it measures, which bounds how much precision is even meaningful"
}
Attitudinal effects of mere exposurerecord 104
{
"id": "P45",
"kind": "paper",
"name": "Attitudinal effects of mere exposure",
"url": "https://doi.org/10.1037/h0025848",
"authors": "Robert B. Zajonc",
"year": 1968,
"venue": "Journal of Personality and Social Psychology 9(2, Pt.2):1-27",
"evidence_class": "observed",
"source": "Crossref DOI 10.1037/h0025848",
"observed": "2026-08-27",
"claim": "Repeated exposure to a stimulus increases liking for it — mere exposure.",
"limitations": "Metadata verified only.",
"disposition": "CITE — a genuine confound: designs shown repeatedly across elicitation rounds may gain preference from exposure alone, biasing later comparisons"
}
Processing Fluency and Aesthetic Pleasure: Is Beauty in the Perceiver's Processing Experience?record 105
{
"id": "P46",
"kind": "paper",
"name": "Processing Fluency and Aesthetic Pleasure: Is Beauty in the Perceiver's Processing Experience?",
"url": "https://doi.org/10.1207/s15327957pspr0804_3",
"authors": "Rolf Reber; Norbert Schwarz; Piotr Winkielman",
"year": 2004,
"venue": "Personality and Social Psychology Review 8(4):364-382",
"evidence_class": "observed",
"source": "Crossref DOI 10.1207/s15327957pspr0804_3",
"observed": "2026-08-27",
"claim": "Aesthetic pleasure is substantially a function of PROCESSING FLUENCY — stimuli that are easier to process are experienced as more beautiful; fluency is affected by symmetry, contrast, prototypicality and prior exposure.",
"limitations": "Metadata verified only; theoretical review.",
"disposition": "CITE — explains WHY the knobs are correlated and why a low-dimensional latent structure (the Jamieson-Nowak precondition) is plausible for visual taste"
}
The role of visual complexity and prototypicality regarding first impression of websitesrecord 106
{
"id": "P47",
"kind": "paper",
"name": "The role of visual complexity and prototypicality regarding first impression of websites",
"url": "https://doi.org/10.1016/j.ijhcs.2012.06.003",
"authors": "Alexandre N. Tuch; Eva E. Presslaber; Markus Stoecklin; Klaus Opwis; Javier A. Bargas-Avila",
"year": 2012,
"venue": "International Journal of Human-Computer Studies 70(11):794-811",
"evidence_class": "observed",
"source": "Crossref DOI 10.1016/j.ijhcs.2012.06.003",
"observed": "2026-08-27",
"claim": "Visual complexity and prototypicality jointly drive first impressions of websites, with effects detectable at very short exposures.",
"limitations": "Metadata verified only; I did NOT read the reported effect sizes or exposure durations. Do not quote numbers from this without full text.",
"disposition": "cite for the two-factor structure only"
}
Predicting users' first impressions of website aesthetics with a quantification of perceived visual complexity and colorfulnessrecord 107
{
"id": "P48",
"kind": "paper",
"name": "Predicting users' first impressions of website aesthetics with a quantification of perceived visual complexity and colorfulness",
"url": "https://doi.org/10.1145/2470654.2481281",
"authors": "Katharina Reinecke; Tom Yeh; Luke Miratrix; Rahmatri Mardiko; Yuechen Zhao; Jenny Liu; Krzysztof Z. Gajos",
"year": 2013,
"venue": "CHI 2013",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — eecs.harvard.edu/~kgajos hosted copy, quotes extracted verbatim",
"observed": "2026-08-27",
"claim": "EXPOSURE CONFIRMED AT 500ms. Computational models of colourfulness (r = .88, R2 = .78, adj R2 = .77) and visual complexity (r = .80, R2 = .65, adj R2 = .64) against mean human ratings of those constructs. The APPEAL model: 'The model accounts for 48% of the variation in aesthetic preferences (adj. R2 = .48)' after 500ms viewing. Data: 548 volunteers rating 450 websites (184 colourfulness / 122 complexity / 242 appeal); 421 pages analysed for colourfulness, 382 for complexity.",
"limitations": "CRITICAL CAVEAT from the full text: the 48% model INCLUDES DEMOGRAPHIC VARIABLES (age, education, gender interactions), not colourfulness+complexity alone — the paper's own phrasing is 'in combination with demographic variables'. Population-level, not individual taste. Websites, not general visual design.",
"disposition": "CITE — two computed features + demographics explain ~48% of first-impression variance; the residual ~52% is roughly the space a per-client taste model must earn its keep in"
}
Attention web designers: You have 50 milliseconds to make a good first impression!record 108
{
"id": "P49",
"kind": "paper",
"name": "Attention web designers: You have 50 milliseconds to make a good first impression!",
"url": "https://doi.org/10.1080/01449290500330448",
"authors": "Gitte Lindgaard; Gary Fernandes; Cathy Dudek; J. Brown",
"year": 2006,
"venue": "Behaviour & Information Technology 25(2):115-126",
"evidence_class": "observed",
"source": "ABSTRACT ONLY — verbatim abstract retrieved from OpenAlex and independently corroborated byte-identical from Semantic Scholar by delegated Opus worker; full text paywalled (10 retrieval routes failed)",
"observed": "2026-08-27",
"claim": "Three studies. Studies 1-2 presented homepages for 500ms; Study 3 added a 50ms condition on the same stimuli. Abstract states 'visual appeal ratings were highly correlated from one phase to the next as were the correlations between the 50 ms and 500 ms conditions', concluding visual appeal can be assessed within 50ms.",
"limitations": "THE NUMERIC CORRELATION IS NOT VERIFIED. The abstract gives only the qualitative phrase 'highly correlated' — no coefficient. Widely circulated values (e.g. r = .88) are secondary-source claims I could NOT confirm; obtaining the real r requires institutional access to B&IT 25(2). Do not put a specific r in client-facing material on my evidence.",
"disposition": "CITE THE QUALITATIVE CLAIM ONLY — explicitly do not quote a coefficient"
}
Deep reinforcement learning from human preferencesrecord 109
{
"id": "P50",
"kind": "paper",
"name": "Deep reinforcement learning from human preferences",
"url": "https://arxiv.org/abs/1706.03741",
"authors": "Paul Christiano; Jan Leike; Tom B. Brown; Miljan Martic; Shane Legg; Dario Amodei",
"year": 2017,
"venue": "NIPS 2017",
"evidence_class": "observed",
"source": "arXiv Atom API — title/authors/year verified",
"observed": "2026-08-27",
"claim": "Learns a reward model from human pairwise comparisons of trajectory segments, demonstrating complex behaviours from feedback on well under 1% of agent interactions.",
"limitations": "Metadata verified, full abstract not re-read this session. RL domain; the specific efficiency figures should be re-read before quoting.",
"disposition": "cite — the modern proof that pairwise human feedback is a viable training signal"
}
Sequential Gallery (Koyama, Sato, Goto 2020)record 110
{
"id": "P26-TOP10",
"kind": "paper",
"name": "Sequential Gallery (Koyama, Sato, Goto 2020)",
"url": "https://arxiv.org/abs/2005.04107",
"authors": "Yuki Koyama; Issei Sato; Masataka Goto",
"year": 2020,
"venue": "ACM TOG 39(4):88:1-88:12, SIGGRAPH 2020, DOI 10.1145/3386569.3392444",
"evidence_class": "observed",
"source": "arXiv Atom API abstract + FULL TEXT PDF independently re-derived by me (pdftotext on arxiv.org/pdf/2005.04107v1)",
"observed": "2026-08-27",
"claim": "CLOSEST PRIOR ART #2. Sequential plane search: user picks best from a 2D grid gallery each round. USER STUDY: 'The mean iteration count necessary for finding satisfactory results was 5.36 with SD = 2.69'; 'One plane-search subtask took 14.8 seconds on average'; participants were asked to continue for 15 iterations regardless. 5 of 6 participants hit satisfaction within 15. Synthetic functions at 5D, 15D, 10D, 20D. Grid: 25 points in a 5-by-5 lattice for photo enhancement, reduced to 3-by-3 for body-shape design on a 13-inch display.",
"limitations": "n=6 participants (5 students + 1 researcher) — a PRELIMINARY user study, not a powered experiment. 'Satisfactory' is self-reported, not a measured distance to a ground-truth optimum. Grid size is application-dependent — quoting one number would repeat the 728-vs-113 error class.",
"disposition": "CITE FIRST — 5.36 +/- 2.69 rounds at ~15s each is the single most quotable applied figure for our deliverable",
"rank": 1,
"rationale": "Closest published prior art to our exact problem, with a real number I re-derived from the PDF myself: mean 5.36 +/- 2.69 rounds to a satisfactory visual design, ~14.8s per round. Tested at 5-20D, bracketing our 7 knobs. This is the anchor for any client-facing round-count estimate.",
"refers_to": "P26"
}
Jamieson & Nowak, Active Ranking using Pairwise Comparisons (NIPS 2011)record 111
{
"id": "P12-TOP10",
"kind": "paper",
"name": "Jamieson & Nowak, Active Ranking using Pairwise Comparisons (NIPS 2011)",
"url": "https://arxiv.org/abs/1109.3701",
"authors": "Kevin G. Jamieson; Robert D. Nowak",
"year": 2011,
"venue": "NIPS 2011, pp. 2240-2248",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read verbatim",
"observed": "2026-08-27",
"claim": "THE CRITICAL RESULT. Objects embedded in d-dimensional Euclidean space, ranked by distance from a common reference point: the number of possible rankings grows like n^(2d), and an algorithm identifies a randomly selected ranking using JUST SLIGHTLY MORE THAN d log n ADAPTIVELY selected pairwise comparisons, ON AVERAGE. If comparisons are chosen at random instead, ALMOST ALL pairwise comparisons must be made. Also gives an error-tolerant variant requiring only that comparisons are probably correct.",
"limitations": "'On average' over a randomly selected ranking, NOT worst case — the paper notes that for d>=2 there exist object placements requiring at least n-1 queries. Requires the low-dimensional embedding assumption to actually hold. Adaptivity is essential.",
"disposition": "CITE FIRST — directly licenses the 'few comparisons over 7 knobs' claim, WITH its average-case caveat",
"rank": 2,
"rationale": "The theoretical licence for 'few comparisons over 7 knobs': slightly more than d log n ADAPTIVE comparisons on average when items live in a d-dim embedding, versus almost all n-choose-2 if chosen at random. It also supplies the caveat that keeps us honest (average-case, not worst-case).",
"refers_to": "P12"
}
meta-pytorch/botorch (MIT, 3589 stars)record 112
{
"id": "R14-TOP10",
"kind": "repo",
"name": "meta-pytorch/botorch (MIT, 3589 stars)",
"url": "https://github.com/meta-pytorch/botorch",
"licence": "MIT",
"stars": 3589,
"evidence_class": "observed",
"source": "gh api repos/meta-pytorch/botorch (note: pytorch/botorch now 301-redirects here)",
"observed": "2026-08-27",
"claim": "Bayesian optimization in PyTorch. Ships PairwiseGP, PairwiseLaplaceMarginalLogLikelihood, and qEUBO — a production-grade preferential-BO stack.",
"limitations": "Heavy dependency (torch+gpytorch); PBO components are less documented than the standard BO path.",
"disposition": "ADOPT — the single most load-bearing repo for this lane",
"rank": 3,
"rationale": "The only production-grade, permissively-licensed stack that already ships PairwiseGP, PairwiseLaplaceMarginalLogLikelihood and qEUBO together. Building the learner on this instead of from scratch is the single biggest implementation shortcut available.",
"refers_to": "R14"
}
Sequential Line Search (Koyama et al. SIGGRAPH 2017)record 113
{
"id": "P25-TOP10",
"kind": "paper",
"name": "Sequential Line Search (Koyama et al. SIGGRAPH 2017)",
"url": "https://doi.org/10.1145/3072959.3073598",
"authors": "Yuki Koyama; Issei Sato; Daisuke Sakurai; Takeo Igarashi",
"year": 2017,
"venue": "ACM TOG 36(4), SIGGRAPH 2017",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — koyama.xyz/project/sequential_line_search/download/preprint.pdf",
"observed": "2026-08-27",
"claim": "CLOSEST PRIOR ART #1. Preferential BO where each query is a 1D slider through high-dim design space. ALL REPORTED RESULTS USED 15 ITERATIONS, 7 crowd microtasks per iteration, proceeding once >=5 responses arrived (total crowd cost 5.25 USD per result). Convergence observation: 'the distances become small rapidly in the first 4 or 5 iterations'. Design spaces: 6D photo colour enhancement, 3D and 7D BRDF material design; synthetic tests at n in {2,6,20}.",
"limitations": "NO explicit iteration-vs-dimension scaling law is stated in the paper (verified absent by full-text reader). Crowd-aggregated, not single-client. 15 was a chosen budget, not a derived requirement.",
"disposition": "CITE FIRST among applied work — 7D is our exact dimensionality, and 'good by 4-5 iterations' is the most directly transferable number",
"rank": 4,
"rationale": "Applied preferential BO at 6D and 7D — our exact dimensionality — with the transferable convergence observation that distances shrink rapidly in the first 4-5 iterations. Paired with R20 (MIT) it is both evidence and usable code.",
"refers_to": "P25"
}
Scheibehenne, Greifeneder & Todd 2010 choice-overload meta-analysisrecord 114
{
"id": "P43-TOP10",
"kind": "paper",
"name": "Scheibehenne, Greifeneder & Todd 2010 choice-overload meta-analysis",
"url": "https://doi.org/10.1086/651235",
"authors": "Benjamin Scheibehenne; Rainer Greifeneder; Peter M. Todd",
"year": 2010,
"venue": "Journal of Consumer Research 37(3):409-425",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — scheibehenne.de author-hosted copy, quotes extracted verbatim",
"observed": "2026-08-27",
"claim": "DECISIVE REPLICATION EVIDENCE. 'In a meta-analysis of 63 conditions from 50 published and unpublished experiments (N = 5,036), we found a mean effect size of virtually zero but considerable variance between studies.' Exact: 'The mean effect size of choice overload across all 63 data points... is D = 0.02 (95% confidence interval [CI95] -0.09 to 0.12).' Trimmed 20%: 'Dtrimmed = 0.001 (CI95 -0.08 to 0.07).' Heterogeneity I2 = 68% untrimmed, 22% trimmed. Conclusion: 'no sufficient conditions could be identified' and 'adverse consequences due to having too much choice are not a robust phenomenon'. Also notes a slight publication bias.",
"limitations": "High heterogeneity untrimmed (I2=68%) means moderators may exist even though none were identified as sufficient. Meta-analysis to 2010.",
"disposition": "CITE — ANSWERS THE REPLICATION QUESTION: choice overload does NOT robustly replicate; gallery size should be chosen on information grounds, not overload fear",
"rank": 5,
"rationale": "Settles a question that would otherwise silently shape the UI: choice overload does NOT robustly replicate (D = 0.02, CI -0.09 to 0.12; trimmed 0.001; 50 experiments, N = 5,036). Frees us to size galleries on information gain rather than folklore, and prevents shipping the jam study as a principle.",
"refers_to": "P43"
}
Heckel, Shah, Ramchandran & Wainwright, Active Ranking / parametric assumptionsrecord 115
{
"id": "P10-TOP10",
"kind": "paper",
"name": "Heckel, Shah, Ramchandran & Wainwright, Active Ranking / parametric assumptions",
"url": "https://arxiv.org/abs/1606.08842",
"authors": "Reinhard Heckel; Nihar B. Shah; Kannan Ramchandran; Martin J. Wainwright",
"year": 2016,
"venue": "arXiv; published Annals of Statistics 47(6) 2019, DOI 10.1214/18-AOS1772",
"evidence_class": "observed",
"source": "arXiv Atom API full abstract + Crossref DOI 10.1214/18-aos1772",
"observed": "2026-08-27",
"claim": "Adaptive algorithm counting comparisons won, with confidence-interval stopping, achieves ranking with a number of comparisons OPTIMAL UP TO LOGARITHMIC FACTORS and requires NO structural assumption on the comparison-probability matrix. Crucially proves a lower bound for parametric models showing BTL/Thurstone assumptions buy AT MOST LOGARITHMIC GAINS for stochastic comparisons.",
"limitations": "Requires pairwise probabilities bounded away from zero. Ranks by 'probability of beating a random item', not a utility vector.",
"disposition": "CITE — HIGH IMPORTANCE: this is the honest counterweight to assuming BTL saves us a lot of questions",
"rank": 6,
"rationale": "The necessary counterweight to over-promising: assuming a BTL/Thurstone parametric model buys AT MOST LOGARITHMIC gains for stochastic comparisons. It stops us from claiming a large question-count saving purely from modelling assumptions.",
"refers_to": "P10"
}
Toubia et al., Fast Polyhedral Adaptive Conjoint Estimation (2003)record 116
{
"id": "P39-TOP10",
"kind": "paper",
"name": "Toubia et al., Fast Polyhedral Adaptive Conjoint Estimation (2003)",
"url": "https://mitsloan.mit.edu/shared/ods/documents/?PublicationDocumentID=5603",
"authors": "Olivier Toubia; Duncan I. Simester; John R. Hauser; Ely Dahan",
"year": 2003,
"venue": "Marketing Science 22(3):273-303 (SSRN DOI 10.2139/ssrn.374460)",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — MIT Sloan hosted typeset paper",
"observed": "2026-08-27",
"claim": "THE KEY ADAPTIVE-CONJOINT RESULT. Explicitly targets q < p (fewer questions than parameters): 'We anticipate that the polyhedral methods are particularly well-suited to solving problems in which there are a large number of parameters relative to the number of responses from each individual (q < p). Thus, we vary the number of questions from slightly less than the number of parameters (q = 8) to comfortably more than the number of parameters (q = 16).' With p = 10 parameters. Abstract: 'For low numbers of questions, polyhedral question design does best (or is tied for best) for all tested domains. For high numbers of questions, efficient fixed designs do better in some domains.' Field test: 330 respondents, 16 paired-comparison questions + 4 holdouts, 9 binary features + price.",
"limitations": "'Does best' is RELATIVE to competing designs, not an absolute accuracy guarantee. Advantage REVERSES at high question counts. Simulations use 500 synthetic respondents in 5 sets of 100 with assumed response-error levels. Metric paired comparisons, not the discrete choices we would use.",
"disposition": "CITE — the strongest published evidence that adaptive design works with FEWER questions than parameters",
"rank": 7,
"rationale": "The strongest published evidence that adaptive question design works when questions are FEWER than parameters (q=8 vs p=10), from the commercial-research tradition Cena's problem actually resembles. Includes the honest reversal: fixed efficient designs win at high question counts.",
"refers_to": "P39"
}
Houlsby et al., BALD (2011)record 117
{
"id": "P20-TOP10",
"kind": "paper",
"name": "Houlsby et al., BALD (2011)",
"url": "https://arxiv.org/abs/1112.5745",
"authors": "Neil Houlsby; Ferenc Huszar; Zoubin Ghahramani; Mate Lengyel",
"year": 2011,
"venue": "arXiv 1112.5745",
"evidence_class": "observed",
"source": "arXiv Atom API — full abstract read",
"observed": "2026-08-27",
"claim": "BALD: expresses information gain in terms of PREDICTIVE ENTROPIES, making it tractable for GP classifiers; explicitly EXTENDS TO GAUSSIAN PROCESS PREFERENCE LEARNING by reformulating binary preference learning as classification. Equal or lower computational cost than competitors.",
"limitations": "Myopic one-step criterion.",
"disposition": "ADOPT — this is the acquisition function to implement for pairwise queries",
"rank": 8,
"rationale": "The concrete acquisition function for choosing the next comparison, with an explicit extension to GP preference learning and tractable cost. This is what we would actually implement on top of rank 3.",
"refers_to": "P20"
}
Slovic, The Construction of Preference (1995)record 118
{
"id": "P44-TOP10",
"kind": "paper",
"name": "Slovic, The Construction of Preference (1995)",
"url": "https://doi.org/10.1037/0003-066x.50.5.364",
"authors": "Paul Slovic",
"year": 1995,
"venue": "American Psychologist 50(5):364-371",
"evidence_class": "observed",
"source": "Crossref DOI 10.1037/0003-066x.50.5.364",
"observed": "2026-08-27",
"claim": "Preferences are frequently CONSTRUCTED during elicitation rather than retrieved from a stable store — they are sensitive to framing, response mode and context.",
"limitations": "Metadata verified only. This is a theoretical/review synthesis.",
"disposition": "CITE — THE key epistemic caveat: the elicitation procedure partly CREATES the taste it measures, which bounds how much precision is even meaningful",
"rank": 9,
"rationale": "Bounds the whole exercise: preferences are partly constructed during elicitation, not retrieved. It sets a ceiling on meaningful precision and argues for measuring test-retest stability rather than chasing ever-tighter parameter estimates.",
"refers_to": "P44"
}
Brochu, de Freitas & Ghosh, Active Preference Learning with Discrete Choice Data (NIPS 2007)record 119
{
"id": "P22-TOP10",
"kind": "paper",
"name": "Brochu, de Freitas & Ghosh, Active Preference Learning with Discrete Choice Data (NIPS 2007)",
"url": "https://proceedings.neurips.cc/paper/2007/file/b6a1085a27ab7bff7550f8a3bd017df8-Paper.pdf",
"authors": "Eric Brochu; Nando de Freitas; Abhijeet Ghosh",
"year": 2007,
"venue": "NIPS 2007",
"evidence_class": "observed",
"source": "FULL TEXT PDF read by delegated Opus worker — NeurIPS proceedings PDF, 8pp, Table 1 extracted",
"observed": "2026-08-27",
"claim": "GP preference model + expected-improvement acquisition over gallery queries. USER STUDY (5 subjects, 50 trials each arm, 38 MERL BRDFs): maxEI needed a MEAN OF 8.56 +/- 5.23 clicks to find a target, versus 17.87 +/- 8.60 for max-variance and 18.40 +/- 7.87 for Latin hypercubes. Seeded with 4 predetermined queries. Authors state 'requiring more than 50 user queries in a real application would be unacceptable'.",
"limitations": "Target-finding task (a known target exists), NOT open-ended taste discovery — an easier problem than ours. In 6D 'neither algorithm succeeds well in finding the optimum'. n=5 subjects.",
"disposition": "CITE — best hard number for 'how many clicks', with the caveat that 6D already strained it",
"rank": 10,
"rationale": "The best hard human-click number in the literature (8.56 +/- 5.23 clicks vs ~18 for non-adaptive baselines), plus the sobering finding that 6D already strained the method — a direct empirical warning at our dimensionality.",
"refers_to": "P22"
}