{
 "dataset": "OWHS Instrument Evidence Base (seed, deep pilot)",
 "version": "0.1.0",
 "schema_version": "0.1",
 "generated": "2026-07-12",
 "maintained_under": "Open Workplace Health Standard (openworkplacehealth.org)",
 "status": "resource maintained under the standard, distinct from the normative specification",
 "licence": {
  "prose": "CC BY 4.0",
  "structure": "Apache 2.0"
 },
 "pass": "Deep pilot (7 of ~25 watchlist instruments). Schema locked pending Task 3 critique; remaining instruments are pass two.",
 "editorial_rules": [
  "Conservative sourcing; every claim carries a citation (DOI where available).",
  "Confidence-graded per measurement property (COSMIN-informed, 5-level).",
  "Null, weak and contradictory findings reported as content, not smoothed over.",
  "Absence of evidence recorded explicitly (notably test-retest).",
  "No reproduction of copyrighted item text.",
  "Neutral: no vendor or product endorsement; no recommendation beyond evidence grades.",
  "'Clinical' never used affirmatively of a workplace deployment context.",
  "British English; no em dashes in registry prose."
 ],
 "integrity": {
  "distinct_dois": 189,
  "dois_resolved": 189,
  "dois_non_resolving": 0,
  "dois_retracted": 0,
  "verification": "All cited DOIs resolved against CrossRef with doi.org fallback; none non-resolving, none retracted, this pass."
 },
 "records": [
  {
   "instrument_id": "ons-4",
   "display_name": "ONS-4 personal wellbeing questions",
   "identity": {
    "name": "ONS-4 (ONS4) personal well-being questions",
    "current_version": "Four questions as introduced by the UK Office for National Statistics in the Annual Population Survey in 2011; each item uses an 11-point 0 to 10 response scale. The wording has been stable since introduction, with ONS relabelling the set as 'personal well-being' following public focus groups in 2013.",
    "item_count": "4 (life satisfaction; worthwhile; happiness yesterday; anxiety yesterday), each administered and reported as a separate single item; ONS does not publish a summed total.",
    "original_citation": "Office for National Statistics (2011). Four personal well-being questions introduced in the Annual Population Survey. No primary journal article; the measure is defined and disseminated in ONS technical guidance and described in Dolan P, Metcalfe R (2012) Measuring Subjective Wellbeing: Recommendations on Measures for use by National Governments. Journal of Social Policy. DOI 10.1017/s0047279411000833",
    "steward_publisher": "UK Office for National Statistics (ONS)",
    "licence_status": "Free to use. The four questions are designated National Statistics and approved as a Government Statistical Service Harmonised Principle, and ONS explicitly encourages their use and adaptation across government, local government, charities and the private sector (Benson 2019, 10.1136/bmjoq-2018-000394). ONS-published material is Crown copyright released under the Open Government Licence; no licence fee or registration is required. Note that this open status attaches to the original ONS wording and 0 to 10 scale, not to modified derivatives."
   },
   "constructs_claimed": "The four questions deliberately span three or four conceptually distinct facets of subjective well-being rather than a single latent trait: evaluative well-being (life satisfaction), eudaimonic well-being (the sense that things done in life are worthwhile), and hedonic affect split into positive (happiness yesterday) and negative (anxiety yesterday) components [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). This tripartite or four-part conception follows the Stiglitz-Sen-Fitoussi and national-accounts-of-well-being tradition that ONS adopted [Dolan & Metcalfe 2012](https://doi.org/10.1017/s0047279411000833). The design intent is that each item is informative in its own right; the anxiety item is a negative-affect measure and is therefore worded and scored in the opposite direction to the other three.",
   "structural_validity": {
    "findings": "There is no confirmatory factor-analytic validation of the original ONS-4 as a unidimensional scale, and this is by design: ONS treats the four items as separate indicators of distinct constructs and publishes each separately rather than as a summed score [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). The one psychometric study that factor-analyses ONS-4-derived items examined a modified four-point short version (the Personal Wellbeing Score, PWS) embedded alongside other R-Outcomes measures; a scree plot suggested four or six factors and Kaiser's criterion four, with the four well-being items loading together and separately from co-administered health and experience measures, and inter-item correlations of r=0.51 to r=0.77 [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). That evidence supports the well-being items cohering as a set in that instrument, but it pertains to the reworded four-point PWS, not to the original 0 to 10 ONS-4 wording. Broader multi-country work confirms that evaluative, eudaimonic and affective well-being are empirically separable dimensions rather than one factor, which is consistent with ONS-4 being reported item by item [Ruggeri 2020](https://doi.org/10.1186/s12955-020-01423-y).",
    "confidence": "Low. Direct factor-analytic evidence exists only for a modified derivative (single study, one English social-prescribing sample); the original ONS-4 is not validated as a scale and is not intended to be one."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent evidence for the ONS-4 items is largely indirect, resting on the wider single-item life-satisfaction literature and on derivative instruments rather than on the ONS wording itself. Single-item life-satisfaction measures correlate with the multi-item Satisfaction With Life Scale at zero-order r=0.62 to 0.64, rising to r=0.78 to 0.80 after disattenuation, and reproduce the SWLS pattern of associations with health, domain satisfaction and affect almost exactly (mean absolute difference in correlations 0.015 to 0.042) [Cheung & Lucas 2014](https://doi.org/10.1007/s11136-014-0726-4). In the modified PWS derivative, each well-being item correlated with its summary score at r=0.83 to 0.88 and the two evaluative items correlated at r=0.77 [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). UK multi-instrument comparison datasets now allow ONS-4 to be mapped against WEMWBS, EQ-5D and ICECAP-A, supporting moderate cross-measure convergence while confirming the instruments are not interchangeable [Wickramasekera & Tsuchiya 2025](https://doi.org/10.1007/s11205-025-03728-1). Discriminant behaviour of the anxiety item is notable: it is the negative-affect component and shows the weakest association with the evaluative items, consistent with affect and evaluation being separable [Ruggeri 2020](https://doi.org/10.1186/s12955-020-01423-y).",
    "confidence": "Low. Convergent magnitudes are reasonable but are drawn mostly from single-item life-satisfaction studies in non-UK samples and from a modified derivative, not from the original ONS-4 items as fielded."
   },
   "criterion_validity": {
    "findings": "No study located in this pass tests ONS-4 against organisational or workplace outcomes such as sickness absence, staff turnover or productivity; criterion evidence for these items in an employment context is absent. Criterion evidence for the underlying constructs comes from the general subjective-well-being literature and predicts health and mortality rather than organisational endpoints: low self-reported life satisfaction predicted 20-year all-cause mortality in a Finnish cohort of 22,461 adults [Koivumaa-Honkanen 2000](https://doi.org/10.1093/aje/152.10.983), life satisfaction predicted all-cause mortality over 22 years in older adults [Gana 2016](https://doi.org/10.1016/j.jpsychores.2016.04.004), and a broad review links subjective well-being to health and longevity [Diener & Chan 2011](https://doi.org/10.1111/j.1758-0854.2010.01045.x). Mendelian randomisation provides mixed causal support, finding little robust causal effect of subjective well-being on cardiometabolic disease [Wootton 2018](https://doi.org/10.1136/bmj.k3788). A UK natural experiment used ONS-4-style items as outcomes when evaluating a change in the built environment, illustrating policy-outcome use rather than establishing criterion validity against a gold standard [Ram 2020](https://doi.org/10.1136/jech-2019-213591).",
    "confidence": "Very low. Workplace and organisational criterion validity is absent; the available criterion evidence is for the general constructs (health, mortality) in non-UK, non-workplace populations and does not use the ONS-4 items specifically."
   },
   "internal_consistency": {
    "findings": "Internal consistency is not a meaningful property of ONS-4 as ONS uses it, because the four items are reported separately and Cronbach's alpha cannot be computed for a single item [Cheung & Lucas 2014](https://doi.org/10.1007/s11136-014-0726-4). The only alpha located is for the modified four-point PWS summary score, where Cronbach's alpha was 0.90 in one English social-prescribing sample; the authors themselves had expected 0.7 to 0.9 to justify an aggregate score, and note this value sits at the top of that range [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). That coefficient applies to the reworded PWS, not to the original ONS-4, and a high alpha on four items partly reflects deliberate content breadth rather than redundancy. No omega is reported.",
    "confidence": "Low. Only a single alpha exists and it belongs to a modified derivative scale, not the original item set; for the original ONS-4 internal consistency is not applicable."
   },
   "test_retest_reliability": {
    "findings": "Test-retest reliability for the ONS-4 items as fielded is, on the evidence located in this pass, absent, and this is the clearest evidential gap in the record. The one instrument-specific validation study states explicitly that its anonymous, unlinked data did not permit test-retest reliability, inter-rater reliability or within-individual change to be estimated [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). The nearest available evidence is not classical test-retest but model-based reliability from the single-item life-satisfaction literature: using latent state-trait models on four national panels (combined N over 68,000), reliability estimates for single-item life satisfaction ranged from about 0.68 to 0.74 [Lucas & Donnellan 2012](https://doi.org/10.1007/s11205-011-9783-z). These estimates concern single-item life satisfaction generally, not the ONS worthwhile, happiness or anxiety items, and they are derived from longitudinal decomposition rather than a short-interval retest. No short-interval test-retest coefficient for any ONS-4 item was located.",
    "confidence": "Absent. No test-retest study of the ONS-4 items was located; the instrument-specific paper explicitly could not estimate it, and analogous single-item reliability comes from non-ONS life-satisfaction panels."
   },
   "measurement_invariance": {
    "findings": "No formal measurement-invariance analysis of ONS-4 (configural, metric or scalar, across sex, age, occupation, language or time) was located in this pass. ONS publishes personal well-being estimates broken down by age and sex, but publishing subgroup means is not a test of invariance and does not establish that the items function equivalently across groups. Evidence that mode of administration shifts subjective well-being scores is directly relevant to invariance across data-collection contexts: telephone respondents report systematically higher well-being than face-to-face or self-completion respondents [Dolan & Kavetsos 2016](https://doi.org/10.1007/s10902-015-9642-8), a concern the ONS-4 validation study also flags because some of its ratings were collected face-to-face and some by telephone [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). This implies non-trivial risk of non-invariance across survey modes that has not been formally quantified for ONS-4.",
    "confidence": "Absent. No invariance model for ONS-4 was located; the only directly relevant finding is evidence of mode-of-administration effects, which points to a specific untested invariance risk."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness and a minimal important change (MIC) have not been established for ONS-4. Responsiveness was named as a design goal for the derivative PWS, but the validation study was a cross-sectional secondary analysis with no linkage between pre- and post-intervention responses, so it could not estimate within-individual change or responsiveness [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). ONS-4-style items have been used as outcomes in a longitudinal natural experiment on the built environment, demonstrating that the items are used to detect change over time in practice, but that study did not derive an anchor-based or distribution-based MIC [Ram 2020](https://doi.org/10.1136/jech-2019-213591). No minimal-important-change threshold for any ONS-4 item was located.",
    "confidence": "Absent. No responsiveness statistic or MIC for ONS-4 was located; the instrument-specific study could not estimate change, and applied uses do not report an MIC."
   },
   "populations_languages_norms": {
    "findings": "ONS-4 was developed for and is fielded on the UK general adult population through the Annual Population Survey, and UK population norms and benchmarks are published by ONS from that survey and reported against standard threshold bands: for life satisfaction, worthwhile and happiness, responses of 0 to 4 are classed low, 5 to 6 medium, 7 to 8 high and 9 to 10 very high, while for anxiety (reverse direction) 0 to 1 is low anxiety, 2 to 3 medium, 4 to 5 high and 6 to 10 very high [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). Beyond the UK general population, instrument-specific validation exists only in one English social-prescribing sample using a modified four-point derivative [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394), and applied UK studies have used the items in specific subpopulations such as residents of a regenerated urban neighbourhood [Ram 2020](https://doi.org/10.1136/jech-2019-213591) and during COVID-19 among home-workers [Hensher & Beck 2023](https://doi.org/10.1016/j.tra.2022.103579). Multi-instrument UK comparison datasets now situate ONS-4 alongside other measures for benchmarking [Wickramasekera & Tsuchiya 2025](https://doi.org/10.1007/s11205-025-03728-1). No validated non-English translation of the ONS-4 wording was located in this pass; the closely related OECD core module, which adds an affect item, provides an internationally harmonised counterpart [Dolan & Metcalfe 2012](https://doi.org/10.1017/s0047279411000833).",
    "confidence": "Moderate for UK general-population norms (large, ongoing national survey published by the steward); Low for validated subpopulation and non-English use."
   },
   "criticisms_controversies": "The most consequential issue for a registry is the scale-versus-four-items question. ONS-4 was built as four separate single items measuring distinct constructs and ONS provides no summary score, so summing the items into a composite (as some deployments do) is a departure from the instrument's design and psychometric basis [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). A second issue is that the strongest instrument-specific psychometrics (alpha 0.90, factor structure) belong to a reworded four-point derivative rather than the original 0 to 10 items, so those coefficients should not be read as evidence about ONS-4 as ONS fields it [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). Third, the anxiety item is negatively worded and reverse-scored, its distribution is strongly skewed towards low anxiety, and it behaves differently from the three positive items, which complicates any attempt to combine the four [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). Fourth, mode-of-administration effects are documented: telephone interviewing inflates reported well-being relative to other modes, a threat to comparability across surveys and over time that is particularly relevant when the items are used for benchmarking [Dolan & Kavetsos 2016](https://doi.org/10.1007/s10902-015-9642-8). Finally, single-item measurement remains debated; while single-item life-satisfaction measures perform comparably to multi-item scales on validity and reliability in large studies [Cheung & Lucas 2014](https://doi.org/10.1007/s11136-014-0726-4) [Lucas & Donnellan 2012](https://doi.org/10.1007/s11205-011-9783-z), that literature concerns life satisfaction specifically and does not directly certify the worthwhile, happiness or anxiety items, and the causal standing of subjective well-being for downstream health outcomes is itself contested [Wootton 2018](https://doi.org/10.1136/bmj.k3788).",
   "citations": [
    {
     "key": "Benson2019",
     "authors": "Benson T, Sladen J, Liles A, Potts HWW",
     "year": "2019",
     "title": "Personal Wellbeing Score (PWS): a short version of ONS4: development and validation in social prescribing",
     "journal": "BMJ Open Quality",
     "doi": "10.1136/bmjoq-2018-000394",
     "url": "https://doi.org/10.1136/bmjoq-2018-000394"
    },
    {
     "key": "BensonCorr2019",
     "authors": "Benson T, Sladen J, Liles A, Potts HWW",
     "year": "2019",
     "title": "Correction: Personal Wellbeing Score (PWS): a short version of ONS4: development and validation in social prescribing",
     "journal": "BMJ Open Quality",
     "doi": "10.1136/bmjoq-2018-000394corr1",
     "url": "https://doi.org/10.1136/bmjoq-2018-000394corr1"
    },
    {
     "key": "DolanMetcalfe2012",
     "authors": "Dolan P, Metcalfe R",
     "year": "2012",
     "title": "Measuring Subjective Wellbeing: Recommendations on Measures for use by National Governments",
     "journal": "Journal of Social Policy",
     "doi": "10.1017/s0047279411000833",
     "url": "https://doi.org/10.1017/s0047279411000833"
    },
    {
     "key": "LucasDonnellan2012",
     "authors": "Lucas RE, Donnellan MB",
     "year": "2012",
     "title": "Estimating the Reliability of Single-Item Life Satisfaction Measures: Results from Four National Panel Studies",
     "journal": "Social Indicators Research",
     "doi": "10.1007/s11205-011-9783-z",
     "url": "https://doi.org/10.1007/s11205-011-9783-z"
    },
    {
     "key": "CheungLucas2014",
     "authors": "Cheung F, Lucas RE",
     "year": "2014",
     "title": "Assessing the validity of single-item life satisfaction measures: results from three large samples",
     "journal": "Quality of Life Research",
     "doi": "10.1007/s11136-014-0726-4",
     "url": "https://doi.org/10.1007/s11136-014-0726-4"
    },
    {
     "key": "DolanKavetsos2016",
     "authors": "Dolan P, Kavetsos G",
     "year": "2016",
     "title": "Happy Talk: Mode of Administration Effects on Subjective Well-Being",
     "journal": "Journal of Happiness Studies",
     "doi": "10.1007/s10902-015-9642-8",
     "url": "https://doi.org/10.1007/s10902-015-9642-8"
    },
    {
     "key": "Ruggeri2020",
     "authors": "Ruggeri K, Garcia-Garzon E, Maguire A, Matz S, Huppert FA",
     "year": "2020",
     "title": "Well-being is more than happiness and life satisfaction: a multidimensional analysis of 21 countries",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-020-01423-y",
     "url": "https://doi.org/10.1186/s12955-020-01423-y"
    },
    {
     "key": "Wickramasekera2025",
     "authors": "Wickramasekera N, Tsuchiya A",
     "year": "2025",
     "title": "A Large Scale Population Survey of Health and Wellbeing to Allow Comparisons Between Outcome Measures: the SIPHER-HWMIC Dataset",
     "journal": "Social Indicators Research",
     "doi": "10.1007/s11205-025-03728-1",
     "url": "https://doi.org/10.1007/s11205-025-03728-1"
    },
    {
     "key": "Koivumaa2000",
     "authors": "Koivumaa-Honkanen H, Honkanen R, Viinamaki H, Heikkila K, Kaprio J, Koskenvuo M",
     "year": "2000",
     "title": "Self-reported life satisfaction and 20-year mortality in healthy Finnish adults",
     "journal": "American Journal of Epidemiology",
     "doi": "10.1093/aje/152.10.983",
     "url": "https://doi.org/10.1093/aje/152.10.983"
    },
    {
     "key": "Gana2016",
     "authors": "Gana K, Broc G, Saada Y, Amieva H, Quintard B",
     "year": "2016",
     "title": "Subjective wellbeing and longevity: Findings from a 22-year cohort study",
     "journal": "Journal of Psychosomatic Research",
     "doi": "10.1016/j.jpsychores.2016.04.004",
     "url": "https://doi.org/10.1016/j.jpsychores.2016.04.004"
    },
    {
     "key": "Wootton2018",
     "authors": "Wootton RE, Lawn RB, Millard LAC, Davies NM, Taylor AE, et al.",
     "year": "2018",
     "title": "Evaluation of the causal effects between subjective wellbeing and cardiometabolic health: mendelian randomisation study",
     "journal": "BMJ",
     "doi": "10.1136/bmj.k3788",
     "url": "https://doi.org/10.1136/bmj.k3788"
    },
    {
     "key": "DienerChan2011",
     "authors": "Diener E, Chan MY",
     "year": "2011",
     "title": "Happy People Live Longer: Subjective Well-Being Contributes to Health and Longevity",
     "journal": "Applied Psychology: Health and Well-Being",
     "doi": "10.1111/j.1758-0854.2010.01045.x",
     "url": "https://doi.org/10.1111/j.1758-0854.2010.01045.x"
    },
    {
     "key": "Ram2020",
     "authors": "Ram B, Limb ES, Shankar A, Nightingale CM, Rudnicka AR, et al.",
     "year": "2020",
     "title": "Evaluating the effect of change in the built environment on mental health and subjective well-being: a natural experiment",
     "journal": "Journal of Epidemiology and Community Health",
     "doi": "10.1136/jech-2019-213591",
     "url": "https://doi.org/10.1136/jech-2019-213591"
    },
    {
     "key": "HensherBeck2023",
     "authors": "Hensher DA, Beck MJ",
     "year": "2023",
     "title": "Exploring how worthwhile the things that you do in life are during COVID-19 and links to well-being and working from home",
     "journal": "Transportation Research Part A: Policy and Practice",
     "doi": "10.1016/j.tra.2022.103579",
     "url": "https://doi.org/10.1016/j.tra.2022.103579"
    }
   ],
   "record_notes": "Overall confidence is Low, with two properties (test-retest, invariance, responsiveness/MIC) graded Absent. The dominant honesty problem this record surfaced is a substitution risk that the schema does not explicitly guard against: almost all instrument-specific psychometric coefficients (alpha 0.90, factor structure, item-total r=0.83 to 0.88) come from Benson 2019, but that study validated a modified four-point derivative (the Personal Wellbeing Score) with reworded items, a collapsed response scale and a reversed anxiety direction, not the original 0 to 10 ONS-4. I have flagged this inline everywhere a Benson coefficient appears, but a naive reader could still mistake these for original-ONS-4 properties; the schema would benefit from a field distinguishing 'evidence for this instrument as canonically fielded' from 'evidence for a named derivative'. Second, the scale-versus-single-items status is central and cuts across every property: ONS-4 is four separate single items with no ONS summary score, which makes internal consistency, structural validity and a single test-retest coefficient partly category errors when applied to the instrument as designed; I recorded these honestly as Low or Absent rather than forcing scale-level statistics. Third, for reliability and convergent validity I had to rely on the general single-item life-satisfaction literature (Lucas & Donnellan; Cheung & Lucas) because ONS-4-specific studies do not report these; this is indirect evidence (non-UK samples, life satisfaction only, not the worthwhile/happiness/anxiety items) and I downgraded accordingly. Fourth, key steward facts (UK norms from the Annual Population Survey, Open Government Licence, National Statistics designation) are documented in ONS technical guidance that does not carry a DOI; I grounded the citable elements (threshold bands, National Statistics status, ONS encouragement of reuse) in Benson 2019 and did not invent a DOI for ONS web guidance. No workplace criterion validity was located, which is a material gap given the registry's audience. All fourteen DOIs were checked for resolution and retraction status this session."
  },
  {
   "instrument_id": "who-5",
   "display_name": "WHO-5 Well-Being Index",
   "identity": {
    "name": "WHO-5 Well-Being Index (World Health Organization Five Well-Being Index)",
    "current_version": "WHO-5 (1998 version), the current standard short form derived from earlier WHO-10 and WHO-6/28-item well-being schedules",
    "item_count": "5 items, each positively worded, referring to the previous two weeks, rated 0 (at no time) to 5 (all of the time); raw score 0 to 25 conventionally multiplied by 4 to give 0 to 100",
    "original_citation": "Bech P, Olsen LR, Kjoller M, Rasmussen NK (2003) Measuring well-being rather than the absence of distress symptoms: a comparison of the SF-36 Mental Health subscale and the WHO-Five Well-Being Scale. International Journal of Methods in Psychiatric Research 12(2):85-91. https://doi.org/10.1002/mpr.145 (the 1998 five-item version; consolidated evidence reviewed by Topp et al. 2015, https://doi.org/10.1159/000376585)",
    "steward_publisher": "World Health Organization; the scale was developed under the WHO Regional Office for Europe and historically maintained by the Psychiatric Research Unit, Mental Health Centre North Zealand, Denmark (Bech and colleagues). Master versions and translations are distributed by WHO.",
    "licence_status": "Licence changed in 2024, and earlier literature must not be relied on for commercial use. WHO republished the WHO-5 in October 2024 (reference WHO-UCN-MSD-MHE-2024.01) under Creative Commons BY-NC-SA 3.0 IGO, a non-commercial, share-alike licence (https://www.who.int/publications/m/item/WHO-UCN-MSD-MHE-2024.01). Pre-2024 literature describes the scale as free to reproduce and translate with acknowledgement ([Topp 2015](https://doi.org/10.1159/000376585), [Lara-Cabrera 2020 protocol](https://doi.org/10.1111/jan.14445)); that position remains a fair description of non-commercial use, but commercial deployment now requires WHO permission or sits outside the licence terms. Users should obtain the authorised master version and validated translations from WHO rather than re-typing items. [Registry correction 2026-07-12: this field originally stated the pre-2024 position only.]"
   },
   "constructs_claimed": "The WHO-5 is presented as a unidimensional measure of subjective psychological (hedonic) well-being over the preceding two weeks, tapping positive mood, vitality and general interest. It is a measure of the presence of positive well-being rather than of symptom burden, although its origin and much of its validation lie in depression screening. It is a generic well-being index, not a workplace-specific or organisational instrument; workplace studies adopt it unchanged as a general well-being outcome.",
   "structural_validity": {
    "findings": "The WHO-5 is predominantly unidimensional: a single well-being factor is recovered across a wide range of populations and languages. The anchoring systematic review concluded high clinimetric validity and treated the scale as a coherent single dimension ([Topp 2015](https://doi.org/10.1159/000376585)). One-factor structures have since been confirmed by confirmatory factor analysis in adolescents with type 1 diabetes ([de Wit 2007](https://doi.org/10.2337/dc07-0447)), in type 1 and type 2 diabetes outpatients ([Hajos 2013](https://doi.org/10.1111/dme.12040)), in a Chinese university sample ([Fung 2022](https://doi.org/10.3389/fpubh.2022.872436)), in Chinese type 2 diabetes patients ([Du 2023](https://doi.org/10.1186/s12888-023-05381-9)), in a Sinhala community sample ([Perera 2020](https://doi.org/10.1186/s12955-020-01532-8)), in a Bangla general sample ([Faruk 2021](https://doi.org/10.1017/gmh.2021.26)) and in Portuguese adolescents ([Carvalho 2025](https://doi.org/10.1159/000543728)); principal component analysis in euthymic bipolar patients returned a single factor explaining 59.7% of variance ([Bonnín 2017](https://doi.org/10.1016/j.jad.2017.12.006)). Item response and Rasch analyses concur: the scale was unidimensional with no differential item functioning by age, sex or inpatient/outpatient status in schizophrenia spectrum disorders, although initial category disordering was only resolved by merging two middle response options ([Nielsen 2023](https://doi.org/10.1016/j.jpsychires.2023.12.028)), and a nationwide Norwegian post-discharge sample found a one-factor solution explaining 71.7% of variance ([Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)). A five-country clinimetric study (Italy, Poland, Denmark, China, Japan) using Mokken and Rasch analyses reported scalability coefficients ranging from 0.42 to 0.84 with most language versions fitting Rasch expectations, supporting unidimensionality across cultures ([Carrozzino 2022](https://doi.org/10.1016/j.jad.2022.05.111)). Two qualifications recur. First, in five-item models the RMSEA is frequently elevated (for example 0.23 in an Australian diabetes sample despite CFI 0.98 and loadings 0.78 to 0.92, [Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005); elevated again in the Norwegian sample, [Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)), an artefact of very low degrees of freedom rather than clear misfit. Second, and more substantively, in a 15-country adolescent study the full five-item WHO-5 did not achieve a good measurement model; a four-item version dropping the first item ('cheerful and in good spirits') was needed ([Cosma 2022](https://doi.org/10.3390/ijerph19169798)), a finding echoed by a Japanese child study that also set that item aside on cultural grounds ([Adachi 2025](https://doi.org/10.3389/fpubh.2025.1662332)).",
    "confidence": "High: many good-quality CFA/IRT studies, large total N, consistently unidimensional in adults, with a documented five-item-model RMSEA artefact and a contested first item in some adolescent/cross-cultural samples."
   },
   "convergent_discriminant_validity": {
    "findings": "The WHO-5 correlates strongly and negatively with depression measures and positively with other well-being measures, as expected for a well-being index that shares variance with low mood. Against depression screeners and symptom scales it shows large negative correlations: r = -0.73 with the PHQ-9 in Australian adults with diabetes ([Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005)), r = -0.694 with the PHQ-9 and r = -0.610 with the Hamilton Depression Rating Scale in Chinese type 2 diabetes patients ([Du 2023](https://doi.org/10.1186/s12888-023-05381-9)), and r = -0.67 with the CES-D in adolescents with type 1 diabetes ([de Wit 2007](https://doi.org/10.2337/dc07-0447)); moderate-to-strong correlations of 0.55 to 0.69 with PHQ, diabetes-distress and SF-12 mental component scores were reported in diabetes outpatients ([Hajos 2013](https://doi.org/10.1111/dme.12040)). It relates very strongly to depression severity even after controlling for anxiety, supporting some discriminant separation from anxiety ([Krieger 2013](https://doi.org/10.1016/j.jad.2013.12.015)). Convergent evidence with positive constructs includes r = 0.542 with the Warwick-Edinburgh Mental Well-Being Scale alongside a smaller divergent r = -0.443 with perceived stress ([Faruk 2021](https://doi.org/10.1017/gmh.2021.26)), and strong associations with life satisfaction and meaning in life against weaker links to physical health ([Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)). Correlations are more modest against broader distress or family-function measures: r = -0.45 with the PHQ-9 and -0.56 with the Kessler K10 in a Sinhala sample ([Perera 2020](https://doi.org/10.1186/s12955-020-01532-8)), and r = 0.35 with a family-function measure in older Peruvian adults ([Del Pilar Diaz-Nunez P 2025](https://doi.org/10.3389/fpubh.2025.1670429)). Convergent construct validity with well-being, self-efficacy and self-esteem measures was also supported in a Chinese sample ([Fung 2022](https://doi.org/10.3389/fpubh.2022.872436)).",
    "confidence": "High: numerous studies, consistent direction and plausible magnitudes; discriminant evidence against anxiety is thinner than convergent evidence against depression."
   },
   "criterion_validity": {
    "findings": "Criterion validity is well established against one target, depression case status, and is essentially absent against organisational outcomes. As a depression screener the WHO-5 performs well: against structured diagnostic interviews or established depression scales, reported areas under the ROC curve cluster around 0.81 to 0.88 (AUC 0.87 in Australian diabetes adults, [Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005); 0.823 against clinical interview in Iranian students, [Ghazisaeedi 2021](https://doi.org/10.1007/s11469-021-00483-5); 0.882 in Chinese healthcare students, [Yang 2023](https://doi.org/10.2147/PRBM.S437219); 0.838 in schizophrenia, [Fekih-Romdhane 2024](https://doi.org/10.1186/s12888-024-05814-z)). Sensitivity and specificity depend heavily on the cut-off: a cut-off below 13 gave 0.79/0.79 while below 8 gave 0.44/0.96 in the same Australian sample ([Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005)), a cut-off below 50 (on the 0 to 100 metric) gave 0.79 to 0.88 sensitivity in Dutch diabetes outpatients ([Hajos 2013](https://doi.org/10.1111/dme.12040)) and 0.89/0.86 in adolescents with type 1 diabetes ([de Wit 2007](https://doi.org/10.2337/dc07-0447)); the anchoring review summarised the scale as a sensitive and specific depression screen across fields ([Topp 2015](https://doi.org/10.1159/000376585)). Importantly, this criterion evidence was earned in diagnostic and clinical-population settings, which differ from workplace deployment where no diagnostic gold standard is applied. Against organisational and health outcomes proper (sickness absence, staff turnover, diagnosed conditions, productivity) no criterion-validity study was located in this pass; workplace papers instead report cross-sectional or prospective associations, for example poor WHO-5 well-being being more common with low workplace social capital ([Gao 2014](https://doi.org/10.1371/journal.pone.0085005)), with adverse psychosocial work factors across 34 European countries ([Schütte 2014](https://doi.org/10.1007/s00420-014-0930-0)) and prospectively in France ([Bertrais 2021](https://doi.org/10.1177/14034948211008385)), lower well-being among self-employed than salaried workers in small enterprises ([Park 2025](https://doi.org/10.3349/ymj.2024.0441)), and higher odds of poor well-being with lower-quality supervisor behaviour and workplace social capital across 35 European countries ([Kizuki 2020](https://doi.org/10.1093/occmed/kqaa070)). These are construct-relevant associations, not criterion validation against an organisational gold standard.",
    "confidence": "Moderate: criterion validity against depression is strong and consistent but almost entirely from clinical and disease populations; criterion validity against organisational outcomes is Absent (no such study located)."
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently good to excellent. Cronbach's alpha values cluster in the low-to-high 0.80s to low 0.90s across populations: alpha 0.90 in Australian adults with diabetes ([Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005)), 0.88 in Chinese type 2 diabetes patients ([Du 2023](https://doi.org/10.1186/s12888-023-05381-9)), 0.82 in adolescents with type 1 diabetes ([de Wit 2007](https://doi.org/10.2337/dc07-0447)), 0.83 in euthymic bipolar patients ([Bonnín 2017](https://doi.org/10.1016/j.jad.2017.12.006)), 0.85 in a Sinhala community sample ([Perera 2020](https://doi.org/10.1186/s12955-020-01532-8)), 0.81 to 0.90 across three countries of nurses ([Lara-Cabrera 2022](https://doi.org/10.3390/ijerph191610106)), 0.80 in Portuguese adolescents ([Carvalho 2025](https://doi.org/10.1159/000543728)) and 0.80 in Arabic-speaking schizophrenia patients ([Fekih-Romdhane 2024](https://doi.org/10.1186/s12888-024-05814-z)). Higher values around 0.91 to 0.94 appear in Iranian students ([Ghazisaeedi 2021](https://doi.org/10.1007/s11469-021-00483-5)), a Norwegian post-discharge sample ([Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)) and Chinese healthcare students ([Yang 2023](https://doi.org/10.2147/PRBM.S437219)). Where reported, McDonald's omega agrees closely with alpha (omega 0.84 in Luxembourg adolescents, [Brisson 2025](https://doi.org/10.1080/00223891.2025.2569138); omega 0.86 to 0.91 in Japanese children, [Adachi 2025](https://doi.org/10.3389/fpubh.2025.1662332); omega 0.908 to 0.935 in Chinese healthcare students, [Yang 2023](https://doi.org/10.2147/PRBM.S437219)). A lower alpha of 0.75 was reported for the Bangla version ([Faruk 2021](https://doi.org/10.1017/gmh.2021.26)). Coefficient omega was the reliability index in the updated German norms study ([Kliem 2025](https://doi.org/10.3389/fpsyg.2025.1592614)).",
    "confidence": "High: many studies, large total N, alpha and omega both reported and consistently adequate to excellent across diverse populations."
   },
   "test_retest_reliability": {
    "findings": "Test-retest reliability is the weakest-evidenced reliability property for the WHO-5 and deserves explicit flagging: dedicated studies are few, intervals are short, and the single study designed specifically to estimate it raised a measurement-error concern. The one purpose-designed test-retest and measurement-error study, in Danish patients with type 1 diabetes using a median five-day interval, found an intraclass correlation of 0.87 (95% CI 0.82 to 0.90) but a large minimal detectable change of 18.56 points on the 0 to 100 scale, which the authors judged a larger measurement error than desirable and flagged for further research ([Schougaard 2022](https://doi.org/10.1186/s41687-022-00505-3)). Other estimates are incidental to validation studies and mostly over about two weeks: Pearson r = 0.72 with ICC 0.82 over two weeks in a Sinhala sample ([Perera 2020](https://doi.org/10.1186/s12955-020-01532-8)), r = 0.83 over ten days in euthymic bipolar patients ([Bonnín 2017](https://doi.org/10.1016/j.jad.2017.12.006)), a retest coefficient of 0.71 in the Bangla validation ([Faruk 2021](https://doi.org/10.1017/gmh.2021.26)) and ICC 0.803 over about one week in Chinese healthcare students ([Yang 2023](https://doi.org/10.2147/PRBM.S437219)). No UK-based or workplace-specific test-retest estimate was located in this pass, and intervals long enough to separate stability from genuine change in well-being are not represented.",
    "confidence": "Low: only one purpose-designed study (which itself flagged a large minimal detectable change), remaining estimates incidental with short intervals, none from workplace or UK samples."
   },
   "measurement_invariance": {
    "findings": "Measurement invariance across sex and age is generally supported in adults, with more mixed results across countries and in adolescents. In adults, configural, metric and scalar invariance across sex, age and education was reported in a nationwide Norwegian post-discharge sample ([Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)); invariance across gender and age was supported in the representative German norms study ([Kliem 2025](https://doi.org/10.3389/fpsyg.2025.1592614)); configural, metric and scalar invariance across sex held in a Sinhala community sample ([Perera 2020](https://doi.org/10.1186/s12955-020-01532-8)); and equivalence across sex was shown in older Peruvian adults ([Del Pilar Diaz-Nunez P 2025](https://doi.org/10.3389/fpubh.2025.1670429)) and cross-sex invariance in Arabic-speaking schizophrenia patients ([Fekih-Romdhane 2024](https://doi.org/10.1186/s12888-024-05814-z)). In Chinese healthcare students the scale was invariant across eleven sociodemographic groupings and longitudinally over one week ([Yang 2023](https://doi.org/10.2147/PRBM.S437219)). Strong measurement invariance across the presence or absence of a current major depressive episode was also demonstrated ([Krieger 2013](https://doi.org/10.1016/j.jad.2013.12.015)). Cross-national and adolescent evidence is more qualified. Across 15 European countries the five-item model did not achieve acceptable cross-country invariance in adolescents, and a four-item version was required for valid comparison ([Cosma 2022](https://doi.org/10.3390/ijerph19169798)); across 43 countries an IRT analysis found many item parameters non-invariant although overall differential test functioning was only modest ([Sischka 2025](https://doi.org/10.1177/10731911241309452)). Full scalar invariance was reported across academic, birth-country, language, sex, socioeconomic and school subgroups in Luxembourg adolescents ([Brisson 2025](https://doi.org/10.1080/00223891.2025.2569138)) and across age and gender in Japanese children ([Adachi 2025](https://doi.org/10.3389/fpubh.2025.1662332)). No study located in this pass tested invariance across occupational groups or between working and non-working populations.",
    "confidence": "Moderate: scalar invariance across sex and age is repeatedly demonstrated in adults and large adolescent samples, but cross-country invariance of the five-item form is contested and occupation-based invariance is untested."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness is asserted more often than it is quantified with a defined minimal important change. The anchoring review concluded the WHO-5 is responsive and can serve as an outcome measure in controlled clinical trials, balancing wanted and unwanted treatment effects ([Topp 2015](https://doi.org/10.1159/000376585)), and a 2025 disease-area review across 552 studies reported that the scale detects treatment-related changes in well-being across many conditions ([Domenech 2025](https://doi.org/10.1007/s12325-025-03266-9)). Direct workplace responsiveness evidence is limited to small intervention studies: a phase-II stress-preventive leadership intervention in a German hospital reported improved WHO-5 well-being over three months ([Stuber 2022](https://doi.org/10.1136/bmjopen-2021-049951)). A formal minimal important change for the WHO-5 was not located in this pass; the closest quantitative anchor is the measurement-error study's minimal detectable change of 18.56 points on the 0 to 100 scale, which is a distribution-based detectable-change threshold rather than an anchor-based minimal important change and was itself flagged as large ([Schougaard 2022](https://doi.org/10.1186/s41687-022-00505-3)).",
    "confidence": "Low: responsiveness is supported qualitatively by two large reviews and small intervention studies, but a validated anchor-based minimal important change was not located, and workplace-specific responsiveness rests on small samples."
   },
   "populations_languages_norms": {
    "findings": "The WHO-5 has been translated into more than 30 languages and validated across an unusually broad range of populations ([Topp 2015](https://doi.org/10.1159/000376585)); a 2025 review catalogued its use across essentially all major disease areas in 552 studies ([Domenech 2025](https://doi.org/10.1007/s12325-025-03266-9)). Validation samples located in this pass span general community adults (German, [Kliem 2025](https://doi.org/10.3389/fpsyg.2025.1592614); Sinhala, [Perera 2020](https://doi.org/10.1186/s12955-020-01532-8); Bangla, [Faruk 2021](https://doi.org/10.1017/gmh.2021.26)), older adults ([Bonsignore 2001](https://doi.org/10.1007/BF03035123); [Del Pilar Diaz-Nunez P 2025](https://doi.org/10.3389/fpubh.2025.1670429)), adolescents and children (43-country and 15-country adolescent studies, [Sischka 2025](https://doi.org/10.1177/10731911241309452), [Cosma 2022](https://doi.org/10.3390/ijerph19169798); Luxembourg, [Brisson 2025](https://doi.org/10.1080/00223891.2025.2569138); Portugal, [Carvalho 2025](https://doi.org/10.1159/000543728); Japan, [Adachi 2025](https://doi.org/10.3389/fpubh.2025.1662332)), diabetes populations ([de Wit 2007](https://doi.org/10.2337/dc07-0447), [Hajos 2013](https://doi.org/10.1111/dme.12040), [Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005), [Du 2023](https://doi.org/10.1186/s12888-023-05381-9)), severe mental illness ([Nielsen 2023](https://doi.org/10.1016/j.jpsychires.2023.12.028), [Fekih-Romdhane 2024](https://doi.org/10.1186/s12888-024-05814-z), [Bonnín 2017](https://doi.org/10.1016/j.jad.2017.12.006), [Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)) and occupational groups (nurses, [Lara-Cabrera 2022](https://doi.org/10.3390/ijerph191610106); medical educators, [Chan 2022](https://doi.org/10.1080/10872981.2022.2044635); European employees, [Schütte 2014](https://doi.org/10.1007/s00420-014-0930-0)). Updated general-population norms exist for Germany ([Kliem 2025](https://doi.org/10.3389/fpsyg.2025.1592614)). No UK-specific validation, UK normative benchmark or UK workplace norm was located in this pass; practitioners needing a UK reference distribution would be relying on non-UK norms, most directly the German representative norms, which is an indirectness limitation for a UK workplace audience.",
    "confidence": "High for breadth of populations and languages; Low specifically for UK norms and workplace-specific benchmarks, which were not located."
   },
   "criticisms_controversies": "Four themes recur in the critical literature. First, the identity of the construct: although marketed as a positive well-being index, the WHO-5's validation base is dominated by depression screening, and it correlates so strongly with depression measures (for example r = -0.73, [Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005)) that some authors treat it as effectively a measure of the severity of depression ([Krieger 2013](https://doi.org/10.1016/j.jad.2013.12.015)). This matters for a workplace registry: the scale's criterion evidence was earned in diagnostic and disease settings, so using it to screen a workforce is deploying a clinically validated instrument outside the diagnostic context in which that validity was established, without a diagnostic gold standard in play. Second, the first item ('cheerful and in good spirits') is repeatedly identified as problematic in cross-cultural and adolescent samples, to the point that a four-item WHO-4 has been proposed for valid cross-country adolescent comparison ([Cosma 2022](https://doi.org/10.3390/ijerph19169798)) and adopted on cultural grounds elsewhere ([Adachi 2025](https://doi.org/10.3389/fpubh.2025.1662332)). Third, cut-off scores for likely depression vary substantially between studies and populations (for example optimal cut-offs corresponding to different thresholds across [Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005), [Ghazisaeedi 2021](https://doi.org/10.1007/s11469-021-00483-5), [Du 2023](https://doi.org/10.1186/s12888-023-05381-9), [Fekih-Romdhane 2024](https://doi.org/10.1186/s12888-024-05814-z)), so no single cut-off transfers automatically to a new setting. Fourth, structural analyses frequently report an elevated RMSEA in the five-item model and occasional response-category disordering ([Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005), [Nielsen 2023](https://doi.org/10.1016/j.jpsychires.2023.12.028), [Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)), and floor effects on some items in low-well-being clinical samples ([Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)). Measurement error over short retest intervals has been flagged as larger than desirable ([Schougaard 2022](https://doi.org/10.1186/s41687-022-00505-3)).",
   "record_notes": "Overall confidence in the WHO-5 as a well-being measure is High for structural validity, internal consistency, convergent validity and breadth of populations/languages; Moderate for measurement invariance (adults yes, cross-country five-item form contested); Low for test-retest reliability and for responsiveness with a defined minimal important change; and effectively Absent for criterion validity against organisational outcomes and for UK-specific norms. Schema stress-test notes: (1) The schema's single 'criterion_validity' field forced two very different evidence states into one cell, namely strong criterion evidence against depression versus no located evidence against organisational outcomes (absence, turnover, diagnosed conditions). I reported both explicitly and graded to the audience's actual need, but a registry might benefit from separating clinical-criterion from organisational-criterion validity. (2) The WHO-5 versus WHO-4 question sits awkwardly: the WHO-4 is a proposed reduced form, not a separate instrument, and evidence about item 1 belongs partly under structural validity, partly under invariance, and partly under criticisms; I have cross-referenced rather than duplicated. (3) The clinical-origin caveat required by the brief cuts across identity, criterion validity and criticisms; I have stated it in each place rather than confining it to one field. (4) Licensing 'free to use' is well supported in the secondary literature but I did not retrieve a formal WHO licence document this session; the claim rests on peer-reviewed statements, so I have qualified it accordingly. (5) No UK data of any kind (validation, norms, workplace) surfaced; for a UK workplace audience the entire evidence base is indirect on locale, which I have flagged under populations and as a downgrade rationale.",
   "citations": [
    {
     "authors": "Topp, Østergaard, Søndergaard, et al.",
     "year": "2015",
     "title": "The WHO-5 Well-Being Index: a systematic review of the literature.",
     "journal": "Psychotherapy and psychosomatics",
     "doi": "10.1159/000376585",
     "key": "topp2015",
     "url": "https://doi.org/10.1159/000376585"
    },
    {
     "authors": "Domenech, Kasujee, Koscielny, et al.",
     "year": "2025",
     "title": "Systematic Review of the Use of the WHO-5 Well-Being Index Across Different Disease Areas.",
     "journal": "Advances in therapy",
     "doi": "10.1007/s12325-025-03266-9",
     "key": "domenech2025",
     "url": "https://doi.org/10.1007/s12325-025-03266-9"
    },
    {
     "key": "bech2003",
     "authors": "Bech P, Olsen LR, Kjoller M, Rasmussen NK",
     "year": "2003",
     "title": "Measuring well-being rather than the absence of distress symptoms: a comparison of the SF-36 Mental Health subscale and the WHO-Five Well-Being Scale",
     "journal": "International Journal of Methods in Psychiatric Research",
     "doi": "10.1002/mpr.145",
     "url": "https://doi.org/10.1002/mpr.145"
    },
    {
     "authors": "Bonsignore, Barkow, Jessen, et al.",
     "year": "2001",
     "title": "Validity of the five-item WHO Well-Being Index (WHO-5) in an elderly population.",
     "journal": "European archives of psychiatry and clinical neuroscience",
     "doi": "10.1007/BF03035123",
     "key": "bonsignore2001",
     "url": "https://doi.org/10.1007/BF03035123"
    },
    {
     "authors": "de Wit, Pouwer, Gemke, et al.",
     "year": "2007",
     "title": "Validation of the WHO-5 Well-Being Index in adolescents with type 1 diabetes.",
     "journal": "Diabetes care",
     "doi": "10.2337/dc07-0447",
     "key": "dewit2007",
     "url": "https://doi.org/10.2337/dc07-0447"
    },
    {
     "authors": "Hajos, Pouwer, Skovlund, et al.",
     "year": "2013",
     "title": "Psychometric and screening properties of the WHO-5 well-being index in adult outpatients with Type 1 or Type 2 diabetes mellitus.",
     "journal": "Diabetic medicine : a journal of the British Diabetic Association",
     "doi": "10.1111/dme.12040",
     "key": "hajos2013",
     "url": "https://doi.org/10.1111/dme.12040"
    },
    {
     "authors": "Halliday, Hendrieckx, Busija, et al.",
     "year": "2017",
     "title": "Validation of the WHO-5 as a first-step screening instrument for depression in adults with diabetes: Results from Diabetes MILES - Australia.",
     "journal": "Diabetes research and clinical practice",
     "doi": "10.1016/j.diabres.2017.07.005",
     "key": "halliday2017",
     "url": "https://doi.org/10.1016/j.diabres.2017.07.005"
    },
    {
     "authors": "Ghazisaeedi, Mahmoodi, Arpaci, et al.",
     "year": "2021",
     "title": "Validity, Reliability, and Optimal Cut-off Scores of the WHO-5, PHQ-9, and PHQ-2 to Screen Depression Among University Students in Iran.",
     "journal": "International journal of mental health and addiction",
     "doi": "10.1007/s11469-021-00483-5",
     "key": "ghazisaeedi2021",
     "url": "https://doi.org/10.1007/s11469-021-00483-5"
    },
    {
     "authors": "Krieger, Zimmermann, Huffziger, et al.",
     "year": "2013",
     "title": "Measuring depression with a well-being index: further evidence for the validity of the WHO Well-Being Index (WHO-5) as a measure of the severity of depression.",
     "journal": "Journal of affective disorders",
     "doi": "10.1016/j.jad.2013.12.015",
     "key": "krieger2014",
     "url": "https://doi.org/10.1016/j.jad.2013.12.015"
    },
    {
     "authors": "Carrozzino, Christensen, Patierno, et al.",
     "year": "2022",
     "title": "Cross-cultural validity of the WHO-5 Well-Being Index and Euthymia Scale: A clinimetric analysis.",
     "journal": "Journal of affective disorders",
     "doi": "10.1016/j.jad.2022.05.111",
     "key": "carrozzino2022",
     "url": "https://doi.org/10.1016/j.jad.2022.05.111"
    },
    {
     "authors": "Nielsen, Lauridsen, Østergaard, et al.",
     "year": "2023",
     "title": "Structural validity of the 5-item World Health Organization Well-being Index (WHO-5) in patients with schizophrenia spectrum disorders.",
     "journal": "Journal of psychiatric research",
     "doi": "10.1016/j.jpsychires.2023.12.028",
     "key": "nielsen2023",
     "url": "https://doi.org/10.1016/j.jpsychires.2023.12.028"
    },
    {
     "authors": "Kliem, Lohmann, Fischer, et al.",
     "year": "2025",
     "title": "Psychometric evaluation and updated community norms of the WHO-5 well-being index, based on a representative German sample.",
     "journal": "Frontiers in psychology",
     "doi": "10.3389/fpsyg.2025.1592614",
     "key": "kliem2025",
     "url": "https://doi.org/10.3389/fpsyg.2025.1592614"
    },
    {
     "authors": "Cosma, Költő, Chzhen, et al.",
     "year": "2022",
     "title": "Measurement Invariance of the WHO-5 Well-Being Index: Evidence from 15 European Countries.",
     "journal": "International journal of environmental research and public health",
     "doi": "10.3390/ijerph19169798",
     "key": "cosma2022",
     "url": "https://doi.org/10.3390/ijerph19169798"
    },
    {
     "authors": "Sischka, Martin, Residori, et al.",
     "year": "2025",
     "title": "Cross-National Validation of the WHO-5 Well-Being Index Within Adolescent Populations: Findings From 43 Countries.",
     "journal": "Assessment",
     "doi": "10.1177/10731911241309452",
     "key": "sischka2025",
     "url": "https://doi.org/10.1177/10731911241309452"
    },
    {
     "authors": "Brisson",
     "year": "2025",
     "title": "Psychometric Evaluation and Sociodemographic Measurement Invariance of the WHO-5 Well-Being Index among Adolescents in Luxembourg.",
     "journal": "Journal of personality assessment",
     "doi": "10.1080/00223891.2025.2569138",
     "key": "brisson2025",
     "url": "https://doi.org/10.1080/00223891.2025.2569138"
    },
    {
     "authors": "Schougaard, Laurberg, Lomborg, et al.",
     "year": "2022",
     "title": "Test-retest reliability and measurement error of the WHO-5 Well-being Index and the Problem Areas in Diabetes questionnaire (PAID) used in telehealth among patients with type 1 diabetes.",
     "journal": "Journal of patient-reported outcomes",
     "doi": "10.1186/s41687-022-00505-3",
     "key": "schougaard2022",
     "url": "https://doi.org/10.1186/s41687-022-00505-3"
    },
    {
     "authors": "Bonnín, Yatham, Michalak, et al.",
     "year": "2017",
     "title": "Psychometric properties of the well-being index (WHO-5) spanish version in a sample of euthymic patients with bipolar disorder.",
     "journal": "Journal of affective disorders",
     "doi": "10.1016/j.jad.2017.12.006",
     "key": "bonnin2017",
     "url": "https://doi.org/10.1016/j.jad.2017.12.006"
    },
    {
     "authors": "Fung, Kong, Liu, et al.",
     "year": "2022",
     "title": "Validity and Psychometric Evaluation of the Chinese Version of the 5-Item WHO Well-Being Index.",
     "journal": "Frontiers in public health",
     "doi": "10.3389/fpubh.2022.872436",
     "key": "fung2022",
     "url": "https://doi.org/10.3389/fpubh.2022.872436"
    },
    {
     "authors": "Perera, Jayasuriya, Caldera, et al.",
     "year": "2020",
     "title": "Assessing mental well-being in a Sinhala speaking Sri Lankan population: validation of the WHO-5 well-being index.",
     "journal": "Health and quality of life outcomes",
     "doi": "10.1186/s12955-020-01532-8",
     "key": "perera2020",
     "url": "https://doi.org/10.1186/s12955-020-01532-8"
    },
    {
     "authors": "Faruk, Alam, Chowdhury, et al.",
     "year": "2021",
     "title": "Validation of the Bangla WHO-5 Well-being Index.",
     "journal": "Global mental health (Cambridge, England)",
     "doi": "10.1017/gmh.2021.26",
     "key": "faruk2021",
     "url": "https://doi.org/10.1017/gmh.2021.26"
    },
    {
     "authors": "Lara-Cabrera, Betancort, Muñoz-Rubilar, et al.",
     "year": "2022",
     "title": "Psychometric Properties of the WHO-5 Well-Being Index among Nurses during the COVID-19 Pandemic: A Cross-Sectional Study in Three Countries.",
     "journal": "International journal of environmental research and public health",
     "doi": "10.3390/ijerph191610106",
     "key": "laracabrera2022",
     "url": "https://doi.org/10.3390/ijerph191610106"
    },
    {
     "authors": "Yang, Ma, Huang, et al.",
     "year": "2023",
     "title": "Measurement Properties and Optimal Cutoff Point of the WHO-5 Among Chinese Healthcare Students.",
     "journal": "Psychology research and behavior management",
     "doi": "10.2147/PRBM.S437219",
     "key": "yang2023",
     "url": "https://doi.org/10.2147/PRBM.S437219"
    },
    {
     "authors": "Du, Jiang, Lloyd, et al.",
     "year": "2023",
     "title": "Validation of Chinese version of the 5-item WHO well-being index in type 2 diabetes mellitus patients.",
     "journal": "BMC psychiatry",
     "doi": "10.1186/s12888-023-05381-9",
     "key": "du2023",
     "url": "https://doi.org/10.1186/s12888-023-05381-9"
    },
    {
     "authors": "Fekih-Romdhane, Al Mouzakzak, Abilmona, et al.",
     "year": "2024",
     "title": "Validation and optimal cut-off score of the World Health Organization Well-being Index (WHO-5) as a screening tool for depression among patients with schizophrenia.",
     "journal": "BMC psychiatry",
     "doi": "10.1186/s12888-024-05814-z",
     "key": "fekih2024",
     "url": "https://doi.org/10.1186/s12888-024-05814-z"
    },
    {
     "authors": "Iversen, Kjøllesdal, Ellingsen-Dalskau, et al.",
     "year": "2025",
     "title": "Psychometric performance of the WHO-5 well-being index in a nationwide sample of inpatients discharged from specialised mental health care.",
     "journal": "Quality of life research : an international journal of quality of life aspects of treatment, care and rehabilitation",
     "doi": "10.1007/s11136-025-04104-9",
     "key": "iversen2025",
     "url": "https://doi.org/10.1007/s11136-025-04104-9"
    },
    {
     "authors": "Adachi, Takahashi, Mori, et al.",
     "year": "2025",
     "title": "Psychometric validation of the WHO-5 and WHO-4 well-being index scales for assessing psychological well-being and detecting depression in Japanese school-aged children: a community-based study.",
     "journal": "Frontiers in public health",
     "doi": "10.3389/fpubh.2025.1662332",
     "key": "adachi2025",
     "url": "https://doi.org/10.3389/fpubh.2025.1662332"
    },
    {
     "authors": "Gao, Weaver, Dai, et al.",
     "year": "2014",
     "title": "Workplace social capital and mental health among Chinese employees: a multi-level, cross-sectional study.",
     "journal": "PloS one",
     "doi": "10.1371/journal.pone.0085005",
     "key": "gao2014",
     "url": "https://doi.org/10.1371/journal.pone.0085005"
    },
    {
     "authors": "Schütte, Chastang, Malard, et al.",
     "year": "2014",
     "title": "Psychosocial working conditions and psychological well-being among employees in 34 European countries.",
     "journal": "International archives of occupational and environmental health",
     "doi": "10.1007/s00420-014-0930-0",
     "key": "schutte2014",
     "url": "https://doi.org/10.1007/s00420-014-0930-0"
    },
    {
     "authors": "Kizuki, Fujiwara",
     "year": "2020",
     "title": "Quality of supervisor behaviour, workplace social capital and psychological well-being.",
     "journal": "Occupational medicine (Oxford, England)",
     "doi": "10.1093/occmed/kqaa070",
     "key": "kizuki2020",
     "url": "https://doi.org/10.1093/occmed/kqaa070"
    },
    {
     "authors": "Bertrais, HÉRault, Chastang, et al.",
     "year": "2021",
     "title": "Multiple psychosocial work exposures and well-being among employees: prospective associations from the French national Working Conditions Survey.",
     "journal": "Scandinavian journal of public health",
     "doi": "10.1177/14034948211008385",
     "key": "bertrais2021",
     "url": "https://doi.org/10.1177/14034948211008385"
    },
    {
     "authors": "Stuber, Seifried-Dübon, Tsarouha, et al.",
     "year": "2022",
     "title": "Feasibility, psychological outcomes and practical use of a stress-preventive leadership intervention in the workplace hospital: the results of a mixed-method phase-II study.",
     "journal": "BMJ open",
     "doi": "10.1136/bmjopen-2021-049951",
     "key": "stuber2022",
     "url": "https://doi.org/10.1136/bmjopen-2021-049951"
    },
    {
     "authors": "Park, Kim, Sung",
     "year": "2025",
     "title": "Factors Affecting Subjective Well-Being in Workers at Small-Sized Enterprises: A Cross-Sectional Study from the 6th Korean Working Conditions Survey.",
     "journal": "Yonsei medical journal",
     "doi": "10.3349/ymj.2024.0441",
     "key": "park2025",
     "url": "https://doi.org/10.3349/ymj.2024.0441"
    },
    {
     "authors": "Chan, Liu, Lam, et al.",
     "year": "2022",
     "title": "Validation of the World Health Organization Well-Being Index (WHO-5) among medical educators in Hong Kong: a confirmatory factor analysis.",
     "journal": "Medical education online",
     "doi": "10.1080/10872981.2022.2044635",
     "key": "chan2022",
     "url": "https://doi.org/10.1080/10872981.2022.2044635"
    },
    {
     "authors": "Carvalho, Vieira Martins, Azevedo, et al.",
     "year": "2025",
     "title": "World Health Organization's Well-Being Index - WHO-5: Psychometric Performance of the Portuguese Version for Adolescents.",
     "journal": "Portuguese journal of public health",
     "doi": "10.1159/000543728",
     "key": "carvalho2025",
     "url": "https://doi.org/10.1159/000543728"
    },
    {
     "key": "delpilar2025",
     "authors": "Del Pilar Diaz-Nunez P, Dominguez-Lara S, et al.",
     "year": "2025",
     "title": "Psychometric evidence of the WHO-5 well-being index in a sample of participants from hospitals and older adults care centers in Peru",
     "journal": "Frontiers in Public Health",
     "doi": "10.3389/fpubh.2025.1670429",
     "url": "https://doi.org/10.3389/fpubh.2025.1670429"
    },
    {
     "key": "laracabreraprot2020",
     "authors": "Lara-Cabrera ML, Bjorngaard JH, Salvesen O, et al.",
     "year": "2020",
     "title": "Psychometric properties of the Five-item World Health Organization Well-being Index used in mental health services: Protocol for a systematic review",
     "journal": "Journal of Advanced Nursing",
     "doi": "10.1111/jan.14445",
     "url": "https://doi.org/10.1111/jan.14445"
    }
   ]
  },
  {
   "instrument_id": "wemwbs",
   "display_name": "Warwick-Edinburgh Mental Wellbeing Scale (WEMWBS) and Short Warwick-Edinburgh Mental Wellbeing Scale (SWEMWBS)",
   "identity": {
    "name": "Warwick-Edinburgh Mental Wellbeing Scale (WEMWBS); Short Warwick-Edinburgh Mental Wellbeing Scale (SWEMWBS)",
    "current_version": "WEMWBS (14-item, 2007) and SWEMWBS (7-item, Rasch-derived, 2009). Both use a five-category response frame (none of the time to all of the time) over a two-week recall window.",
    "item_count": "14 items (WEMWBS); 7 items (SWEMWBS). WEMWBS is summed to a 14 to 70 raw score; SWEMWBS is summed then converted to a Rasch interval-scale metric via a published transformation table.",
    "original_citation": "WEMWBS: Tennant R, Hiller L, Fishwick R, Platt S, Joseph S, Weich S, Parkinson J, Secker J, Stewart-Brown S (2007). The Warwick-Edinburgh Mental Well-being Scale (WEMWBS): development and UK validation. Health and Quality of Life Outcomes 5:63. DOI 10.1186/1477-7525-5-63. SWEMWBS: Stewart-Brown S, Tennant A, Tennant R, Platt S, Parkinson J, Weich S (2009). Internal construct validity of the WEMWBS: a Rasch analysis. Health and Quality of Life Outcomes 7:15. DOI 10.1186/1477-7525-7-15.",
    "steward_publisher": "University of Warwick (Warwick Medical School; licensing administered by Warwick Innovations), with copyright held jointly by NHS Health Scotland, the University of Warwick and the University of Edinburgh.",
    "licence_status": "Restricted-but-free-at-point-of-use for non-commercial users, with mandatory registration. Copyright is held jointly by NHS Health Scotland and the Universities of Warwick and Edinburgh, and permission via a registration form (administered by Warwick Innovations, University of Warwick) is required before use. Academic and non-profit users obtain a no-fee non-commercial licence; commercial users pay a tiered fee. The non-commercial licence does not permit onward distribution of the instrument to third parties, and the copyrighted item wording must be reproduced only under licence with the prescribed copyright statement. Fee terms have changed over time (Warwick has extended charging to some organisational users), so workplace deployers, especially for-profit employers and their vendors, should verify the current commercial-tier terms directly with Warwick Innovations before fielding. (Licence terms per University of Warwick / Warwick Innovations WEMWBS licensing pages; scale copyright statement per the original validation paper, Tennant 2007, DOI 10.1186/1477-7525-5-63.)"
   },
   "constructs_claimed": "WEMWBS claims to measure mental wellbeing as a single positive construct, deliberately covering both the hedonic (subjective happiness, positive affect, life satisfaction) and eudaimonic (positive psychological functioning, autonomy, competence, good relationships) traditions, using exclusively positively worded items and no symptom or deficit content ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). It is framed as a population-level monitoring instrument rather than an individual diagnostic or screening tool. The seven-item SWEMWBS retains a subset weighted more toward the functioning aspect of wellbeing than the feeling aspect ([Stewart-Brown 2009](https://doi.org/10.1186/1477-7525-7-15)).",
   "structural_validity": {
    "findings": "WEMWBS was designed as a unidimensional scale and the original UK validation reported that confirmatory factor analysis supported a single-factor solution ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). The picture is more nuanced under stricter modelling. A Rasch analysis of Scottish Health Education Population Survey data (n=779) found the full 14-item set did not fit the Rasch model; sequential removal of misfitting items produced a strictly unidimensional seven-item scale, SWEMWBS, that provides an interval-scale estimate of wellbeing, with the 14-item and 7-item raw scores correlating 0.954 ([Stewart-Brown 2009](https://doi.org/10.1186/1477-7525-7-15)). Independent Rasch work in the UK veterinary profession reproduced this pattern: the 14 items deviated significantly from Rasch expectations while a 7-item SWEMWBS achieved acceptable fit (person separation index 0.832) ([Bartram 2013](https://doi.org/10.1007/s11136-012-0144-4)). In classical CFA terms several validation studies report that a clean single factor emerges only after correlated residuals are allowed between items ([Smith 2017](https://doi.org/10.1186/s12888-017-1343-x), [Fung 2019](https://doi.org/10.1186/s12955-019-1113-1)), and the German Mental Health Surveillance validation could not confirm strict unidimensionality for either version, with a bifactor model (one general wellbeing factor plus grouping factors) fitting best ([Peitz 2024](https://doi.org/10.1186/s12955-024-02304-4)). Recent large surveillance datasets in the UK, Denmark and Catalonia ([Yadav 2025](https://doi.org/10.1136/bmjment-2024-301433)) and in Canada ([Capaldi 2026](https://doi.org/10.25318/82-003-x202600300002-eng)) converge on the scale being 'essentially unidimensional', typically via bifactor models, which supports summing to a single score while acknowledging minor multidimensionality. The consistent bottom line is that SWEMWBS has the stronger unidimensionality credentials, and that the 14-item WEMWBS is treated as effectively unidimensional rather than strictly so.",
    "confidence": "High: multiple large, good-quality Rasch and CFA studies across countries, consistent in showing SWEMWBS strict unidimensionality and WEMWBS essential (bifactor) unidimensionality."
   },
   "convergent_discriminant_validity": {
    "findings": "WEMWBS correlates strongly with other mental health and wellbeing measures and more weakly with general-health measures, the pattern predicted at development ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). In UK teenagers, WEMWBS correlated 0.65 with the Mental Health Continuum Short Form, 0.59 with the KIDSCREEN-27 psychological wellbeing domain and 0.57 with the WHO-5, and negatively (-0.44) with the Strengths and Difficulties Questionnaire ([Clarke 2011](https://doi.org/10.1186/1471-2458-11-487)). Against the national SWEMWBS norming data, SWEMWBS correlated 0.53 with a happiness index, -0.52 with the GHQ-12 and 0.40 with the EQ-VAS ([Ng 2017](https://doi.org/10.1007/s11136-016-1454-8)). Discriminant validity from distress measures is the contested point: SWEMWBS correlated 0.60 to 0.79 with the PHQ-9 and 0.63 to 0.74 with the GAD-7 in a primary care sample ([Shah 2021](https://doi.org/10.1186/s12955-021-01882-x)), and a Norwegian study modelling the latent correlation between the 14-item WEMWBS and the PHQ-9 found it approaching -0.80, with bifactor indices suggesting wellbeing and depression items were 'essentially unidimensional' jointly ([Aarø 2025](https://doi.org/10.1186/s12888-025-06922-0)). This raises a genuine question of whether WEMWBS and depression scales measure distinct constructs or opposite poles of one, which matters when both are fielded together in a workplace survey.",
    "confidence": "High: convergent correlations replicated across many samples; discriminant validity from distress deliberately reported including the unfavourable finding of very high wellbeing-distress overlap."
   },
   "criterion_validity": {
    "findings": "Direct criterion validity against hard organisational outcomes such as sickness absence, staff turnover or diagnosed conditions is essentially absent from the peer-reviewed WEMWBS psychometric literature retrieved here; the instrument's stewards explicitly state it was not designed for individual screening or diagnosis. What exists is known-groups and concurrent criterion evidence. WEMWBS discriminated between population subgroups in the directions expected from other UK surveys ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)), and SWEMWBS categories were associated with health behaviours (for example lower fruit and vegetable consumption predicting lower wellbeing) in the Health Survey for England norming study ([Ng 2017](https://doi.org/10.1007/s11136-016-1454-8)). Against mental-health criteria, WEMWBS separates carer and patient groups from general-population norms: family carers of people with psychosis scored on average 7.3 points below the Health Survey for England general population, more than double the 3-point minimum important difference ([Sin 2020](https://doi.org/10.1017/S2045796020001067)). A CES-D based threshold of a WEMWBS score at or below 40 has been proposed as indicating elevated depression risk, but the stewards caution the scale was not built for screening. No published workplace criterion study linking WEMWBS to absence or productivity was located in this pass, which is a material gap for the workplace audience.",
    "confidence": "Low: known-groups and concurrent evidence is reasonable, but criterion validity against organisational or occupational outcomes (absence, turnover, diagnosis) was not located; workplace-specific criterion evidence is absent."
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently high for the 14-item WEMWBS and adequate-to-high for SWEMWBS. The original UK validation reported Cronbach's alpha of 0.89 in the student sample and 0.91 in the population sample, the authors noting this suggests some item redundancy ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). UK teenagers gave alpha 0.87 ([Clarke 2011](https://doi.org/10.1186/1471-2458-11-487)). International WEMWBS alphas cluster around 0.90 or above: 0.91 in Slovenian nursing students ([Cilar 2020](https://doi.org/10.1111/jonm.13087)) and 0.93 in Chinese university students, with SWEMWBS at 0.88 in the same sample ([Fung 2019](https://doi.org/10.1186/s12955-019-1113-1)). Omega estimates from pooled UK, Danish and Catalan surveillance data reached 0.94 for WEMWBS ([Yadav 2025](https://doi.org/10.1136/bmjment-2024-301433)), and SWEMWBS internal consistency was 0.88 in a large US urban sample ([Millington 2026](https://doi.org/10.1007/s00127-026-03109-0)). Values above roughly 0.90 for the 14-item version are frequently read as mild redundancy rather than a fault.",
    "confidence": "High: many good-quality studies, large total N, alpha/omega consistently 0.87 to 0.94 for WEMWBS and around 0.84 to 0.88 for SWEMWBS."
   },
   "test_retest_reliability": {
    "findings": "Test-retest evidence exists but is notably thinner than the internal-consistency evidence, and this is the weakest-covered reliability property. The anchor value is from the original UK validation, where 14-item WEMWBS test-retest reliability at one week was reported as an intraclass correlation of 0.83 in a student sub-sample ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). Test-retest stability was also examined in the UK teenage validation ([Clarke 2011](https://doi.org/10.1186/1471-2458-11-487)), and a Polish adaptation reported a dedicated test-retest study, though on a very small retest sample of only 24 participants ([Konaszewski 2021](https://doi.org/10.1186/s12955-021-01716-w)). No test-retest coefficient specific to the 7-item SWEMWBS, and no UK workplace or occupational test-retest study, was located in this pass. Reported intervals are short (around one week), so stability over the weeks-to-months horizon typical of workplace re-survey cycles is effectively unquantified. The absence of a robust, replicated SWEMWBS test-retest estimate is a real gap that should be flagged to anyone using change scores at the individual level.",
    "confidence": "Low: a single well-known 14-item value (ICC 0.83 at one week) plus sparse, small-sample replications; no SWEMWBS-specific or workplace test-retest evidence located, and intervals are short."
   },
   "measurement_invariance": {
    "findings": "SWEMWBS has been tested for invariance more thoroughly than most wellbeing measures, mainly in youth and general-population samples. In a very large Welsh school sample (n=103,971), SWEMWBS was single-factor and loadings and thresholds were invariant across school-year (age) groups, though residual variances were not, indicating partial (metric and scalar/threshold) invariance by age ([Melendez-Torres 2019](https://doi.org/10.1186/s12955-019-1204-z)). In the same survey programme, SWEMWBS reached configural, metric and scalar invariance between young people in care and their peers ([Anthony 2022](https://doi.org/10.1007/s11136-021-02896-0)). Scalar invariance across sex and age group was reported for both versions in a Norwegian primary-care sample ([Smith 2017](https://doi.org/10.1186/s12888-017-1343-x)), and invariance across gender and age was supported in Canadian national surveillance data ([Capaldi 2026](https://doi.org/10.25318/82-003-x202600300002-eng)) and across three European populations at configural, metric and scalar levels ([Yadav 2025](https://doi.org/10.1136/bmjment-2024-301433)). The German MHS validation reported invariance across age and sex ([Peitz 2024](https://doi.org/10.1186/s12955-024-02304-4)). The main exception is age-related differential item functioning for the 'feeling optimistic about the future' item, found in a large US sample and in the original Rasch analysis ([Millington 2026](https://doi.org/10.1007/s00127-026-03109-0), [Stewart-Brown 2009](https://doi.org/10.1186/1477-7525-7-15)). Occupation-specific and longitudinal (over-time) invariance are less well evidenced, and no UK workplace invariance study was located.",
    "confidence": "Moderate: multiple large studies reach scalar invariance across sex, age and care status, but coverage is dominated by youth and general-population samples; occupational and longitudinal invariance are thin, with a recurring optimism-item DIF by age."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness of the 14-item WEMWBS is supported by a secondary analysis of twelve intervention studies, where standardised response means ranged up to 1.35 and the scale detected group-level change in most studies; the standard error of measurement was 2.4 to 3.1 points, and at the 2.77-SEM threshold WEMWBS flagged important individual improvement in 12.8 to 45.7 per cent of participants ([Maheswaran 2012](https://doi.org/10.1186/1477-7525-10-156)). SWEMWBS showed linear sensitivity to change over five therapy sessions in a primary-care common-mental-disorder sample, tracking alongside PHQ-9 and GAD-7 change ([Shah 2021](https://doi.org/10.1186/s12955-021-01882-x)). A minimum important change of around 3 points on the 14-item WEMWBS is widely cited (and used as the benchmark in, for example, the carer comparison of [Sin 2020](https://doi.org/10.1017/S2045796020001067)), and the stewards publish a 3-point change threshold for WEMWBS and a 1-to-3-point threshold for SWEMWBS. Formal anchor-based MIC estimation with external criteria of change is still limited; the responsiveness authors themselves called for further work using external change criteria, and no workplace-intervention MIC study was located in this pass.",
    "confidence": "Moderate: one strong multi-study responsiveness analysis for WEMWBS plus supportive SWEMWBS change data and a widely used 3-point MIC, but external-criterion MIC estimation is limited and not workplace-specific."
   },
   "populations_languages_norms": {
    "findings": "WEMWBS was developed and validated in the UK in student and general-population samples aged 16 and over ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)), with subsequent UK validation in teenagers aged 13 to 16 ([Clarke 2011](https://doi.org/10.1186/1471-2458-11-487)) and in Northern Ireland via the Continuous Household Survey (n=3,355) ([Lloyd 2012](https://doi.org/10.3109/09638237.2012.670883)). It has been translated and validated widely, including Norwegian, Chinese, Polish, Slovenian, German, Arabic, Italian and others, and has been adopted in national surveillance systems in Germany and Canada ([Smith 2017](https://doi.org/10.1186/s12888-017-1343-x), [Fung 2019](https://doi.org/10.1186/s12955-019-1113-1), [Konaszewski 2021](https://doi.org/10.1186/s12955-021-01716-w), [Cilar 2020](https://doi.org/10.1111/jonm.13087), [Peitz 2024](https://doi.org/10.1186/s12955-024-02304-4), [Capaldi 2026](https://doi.org/10.25318/82-003-x202600300002-eng)). UK population norms are the strongest feature: age- and sex-specific SWEMWBS norms were derived from the Health Survey for England 2010-2013 (n=27,169 adults aged 16+), giving mean SWEMWBS of about 23.7 for men and 23.2 for women ([Ng 2017](https://doi.org/10.1007/s11136-016-1454-8)), and Health Survey for England general-population values are used as the reference point in comparative UK studies ([Sin 2020](https://doi.org/10.1017/S2045796020001067)). WEMWBS is also embedded in the Scottish Health Survey and originated in Scottish population-survey data. A UK preference-based value set for SWEMWBS has been derived to support health-economic (utility) use ([Yiu 2023](https://doi.org/10.1016/j.socscimed.2023.115928)). Occupational norms are sparse; the veterinary-profession Rasch study is one of the few occupation-specific UK datasets ([Bartram 2013](https://doi.org/10.1007/s11136-012-0144-4)), and there is no consolidated UK workplace-sector norm set.",
    "confidence": "High for UK adult general-population norms (large nationally representative surveys) and broad language coverage; Low for occupation-specific or workplace-sector norms."
   },
   "criticisms_controversies": "Several recurring criticisms appear in the literature. First, the discriminant validity question: WEMWBS and SWEMWBS correlate very strongly (negatively) with depression and anxiety measures, with latent correlations approaching -0.80 against the PHQ-9 and joint bifactor models suggesting near-unidimensionality of wellbeing and distress items together ([Aarø 2025](https://doi.org/10.1186/s12888-025-06922-0), [Shah 2021](https://doi.org/10.1186/s12955-021-01882-x)), which challenges the claim that positive wellbeing is a construct distinct from the absence of symptoms. Second, ceiling and targeting problems: a Rasch analysis of a large Swedish general-population survey concluded SWEMWBS is an 'off-target' scale, skewed toward lower wellbeing with a ceiling effect and large measurement uncertainty for most respondents, and cautioned against using it to assess change or group differences in whole-population surveys ([Melin 2022](https://doi.org/10.1016/j.puhe.2021.10.009)). Third, dimensionality: strict unidimensionality holds for SWEMWBS but the 14-item WEMWBS repeatedly requires correlated residuals or a bifactor structure to fit, and at least one national validation could not confirm a single factor for either version ([Peitz 2024](https://doi.org/10.1186/s12955-024-02304-4)). Fourth, item-level bias: the 'optimism about the future' item shows differential functioning by age, and several items dropped from WEMWBS to form SWEMWBS had shown gender bias ([Stewart-Brown 2009](https://doi.org/10.1186/1477-7525-7-15), [Millington 2026](https://doi.org/10.1007/s00127-026-03109-0)). Fifth, high internal-consistency values (alpha above 0.90) are read by the developers themselves as indicating item redundancy in the 14-item form ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). Finally, a governance point relevant to workplace users: the licence is free only for registered non-commercial use, and Warwick has extended charging to commercial and some organisational users over time, so the cost basis for many workplace deployments is not zero and should be checked against current Warwick Innovations terms.",
   "citations": [
    {
     "key": "tennant2007",
     "authors": "Tennant R, Hiller L, Fishwick R, Platt S, Joseph S, Weich S, Parkinson J, Secker J, Stewart-Brown S",
     "year": "2007",
     "title": "The Warwick-Edinburgh Mental Well-being Scale (WEMWBS): development and UK validation",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/1477-7525-5-63",
     "url": "https://doi.org/10.1186/1477-7525-5-63"
    },
    {
     "key": "stewartbrown2009",
     "authors": "Stewart-Brown S, Tennant A, Tennant R, Platt S, Parkinson J, Weich S",
     "year": "2009",
     "title": "Internal construct validity of the Warwick-Edinburgh Mental Well-being Scale (WEMWBS): a Rasch analysis using data from the Scottish Health Education Population Survey",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/1477-7525-7-15",
     "url": "https://doi.org/10.1186/1477-7525-7-15"
    },
    {
     "key": "maheswaran2012",
     "authors": "Maheswaran H, Weich S, Powell J, Stewart-Brown S",
     "year": "2012",
     "title": "Evaluating the responsiveness of the Warwick Edinburgh Mental Well-Being Scale (WEMWBS): group and individual level analysis",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/1477-7525-10-156",
     "url": "https://doi.org/10.1186/1477-7525-10-156"
    },
    {
     "key": "clarke2011",
     "authors": "Clarke A, Friede T, Putz R, Ashdown J, Martin S, Blake A, Adi Y, Parkinson J, Flynn P, Platt S, Stewart-Brown S",
     "year": "2011",
     "title": "Warwick-Edinburgh Mental Well-being Scale (WEMWBS): validated for teenage school students in England and Scotland. A mixed methods assessment",
     "journal": "BMC Public Health",
     "doi": "10.1186/1471-2458-11-487",
     "url": "https://doi.org/10.1186/1471-2458-11-487"
    },
    {
     "key": "shah2021",
     "authors": "Shah N, Cader M, Andrews B, McCabe R, Stewart-Brown SL",
     "year": "2021",
     "title": "Short Warwick-Edinburgh Mental Well-being Scale (SWEMWBS): performance in a clinical sample in relation to PHQ-9 and GAD-7",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-021-01882-x",
     "url": "https://doi.org/10.1186/s12955-021-01882-x"
    },
    {
     "key": "melin2021",
     "authors": "Melin J, Nordin Å, et al.",
     "year": "2022",
     "title": "An off-target scale limits the utility of Short Warwick-Edinburgh Mental Well-Being Scale (SWEMWBS) in a general population survey",
     "journal": "Public Health",
     "doi": "10.1016/j.puhe.2021.10.009",
     "url": "https://doi.org/10.1016/j.puhe.2021.10.009"
    },
    {
     "key": "melendez2019",
     "authors": "Melendez-Torres GJ, Hewitt G, Hallingberg B, Anthony R, Collishaw S, Hall J, Murphy S, Moore G",
     "year": "2019",
     "title": "Measurement invariance properties and external construct validity of the short Warwick-Edinburgh mental wellbeing scale in a large national sample of secondary school students in Wales",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-019-1204-z",
     "url": "https://doi.org/10.1186/s12955-019-1204-z"
    },
    {
     "key": "anthony2021",
     "authors": "Anthony R, Moore G, Page N, Hewitt G, Murphy S, Melendez-Torres GJ",
     "year": "2022",
     "title": "Measurement invariance of the short Warwick-Edinburgh Mental Wellbeing Scale and latent mean differences (SWEMWBS) in young people by current care status",
     "journal": "Quality of Life Research",
     "doi": "10.1007/s11136-021-02896-0",
     "url": "https://doi.org/10.1007/s11136-021-02896-0"
    },
    {
     "key": "perera2025",
     "authors": "Perera G, et al.",
     "year": "2025",
     "title": "Psychometric properties of the Warwick Edinburgh Mental Well-being Scale: a systematic review",
     "journal": "Systematic Reviews",
     "doi": "10.1186/s13643-025-02897-x",
     "url": "https://doi.org/10.1186/s13643-025-02897-x"
    },
    {
     "key": "millington2026",
     "authors": "Millington E, et al.",
     "year": "2026",
     "title": "Construct validity of the Short Warwick-Edinburgh Mental Well-Being Scale in a diverse urban population",
     "journal": "Social Psychiatry and Psychiatric Epidemiology",
     "doi": "10.1007/s00127-026-03109-0",
     "url": "https://doi.org/10.1007/s00127-026-03109-0"
    },
    {
     "key": "sabin2020",
     "authors": "Sin J, Elkes J, Batchelor R, Henderson C, Gillard S, Woodham LA, Chen T, Aden A, Cornelius V",
     "year": "2020",
     "title": "Mental health and caregiving experiences of family carers supporting people with psychosis",
     "journal": "Epidemiology and Psychiatric Sciences",
     "doi": "10.1017/S2045796020001067",
     "url": "https://doi.org/10.1017/S2045796020001067"
    },
    {
     "key": "ngfat2017",
     "authors": "Ng Fat L, Scholes S, Boniface S, Mindell J, Stewart-Brown S",
     "year": "2017",
     "title": "Evaluating and establishing national norms for mental wellbeing using the short Warwick-Edinburgh Mental Well-being Scale (SWEMWBS): findings from the Health Survey for England",
     "journal": "Quality of Life Research",
     "doi": "10.1007/s11136-016-1454-8",
     "url": "https://doi.org/10.1007/s11136-016-1454-8"
    },
    {
     "key": "bartram2013",
     "authors": "Bartram DJ, Sinclair JMA, Baldwin DS",
     "year": "2013",
     "title": "Further validation of the Warwick-Edinburgh Mental Well-being Scale (WEMWBS) in the UK veterinary profession: Rasch analysis",
     "journal": "Quality of Life Research",
     "doi": "10.1007/s11136-012-0144-4",
     "url": "https://doi.org/10.1007/s11136-012-0144-4"
    },
    {
     "key": "lloyd2012",
     "authors": "Lloyd K, Devine P",
     "year": "2012",
     "title": "Psychometric properties of the Warwick-Edinburgh Mental Well-being Scale (WEMWBS) in Northern Ireland",
     "journal": "Journal of Mental Health",
     "doi": "10.3109/09638237.2012.670883",
     "url": "https://doi.org/10.3109/09638237.2012.670883"
    },
    {
     "key": "fung2019",
     "authors": "Fung SF",
     "year": "2019",
     "title": "Psychometric evaluation of the Warwick-Edinburgh Mental Well-being Scale (WEMWBS) with Chinese university students",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-019-1113-1",
     "url": "https://doi.org/10.1186/s12955-019-1113-1"
    },
    {
     "key": "peitz2024",
     "authors": "Peitz D, Kersjes C, Thom J, Hoelling H, Mauz E",
     "year": "2024",
     "title": "Validation of the Warwick-Edinburgh Mental Well-Being Scale for the Mental Health Surveillance (MHS) system in Germany",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-024-02304-4",
     "url": "https://doi.org/10.1186/s12955-024-02304-4"
    },
    {
     "key": "cilar2020",
     "authors": "Cilar L, Pajnkihar M, Štiglic G",
     "year": "2020",
     "title": "Validation of the Warwick-Edinburgh Mental Well-being Scale among nursing students in Slovenia",
     "journal": "Journal of Nursing Management",
     "doi": "10.1111/jonm.13087",
     "url": "https://doi.org/10.1111/jonm.13087"
    },
    {
     "key": "aaro2025",
     "authors": "Aarø LE, et al.",
     "year": "2025",
     "title": "Positive mental wellbeing or symptoms of depression? Discriminant validity of the Warwick-Edinburgh Mental Wellbeing Scale (WEMWBS) with respect to the PHQ-9",
     "journal": "BMC Psychiatry",
     "doi": "10.1186/s12888-025-06922-0",
     "url": "https://doi.org/10.1186/s12888-025-06922-0"
    },
    {
     "key": "yadav2025",
     "authors": "Yadav L, et al.",
     "year": "2025",
     "title": "Internal structure, reliability and cross-cultural validity of the Warwick-Edinburgh Mental Wellbeing Scale in three European populations",
     "journal": "BMJ Mental Health",
     "doi": "10.1136/bmjment-2024-301433",
     "url": "https://doi.org/10.1136/bmjment-2024-301433"
    },
    {
     "key": "yiu2023",
     "authors": "Yiu HHE, Buckell J, Petrou S, Stewart-Brown S, Madan J",
     "year": "2023",
     "title": "Derivation of a UK preference-based value set for the Short Warwick-Edinburgh Mental Well-being Scale (SWEMWBS)",
     "journal": "Social Science & Medicine",
     "doi": "10.1016/j.socscimed.2023.115928",
     "url": "https://doi.org/10.1016/j.socscimed.2023.115928"
    },
    {
     "key": "capaldi2026",
     "authors": "Capaldi CA, et al.",
     "year": "2026",
     "title": "Validating the Warwick-Edinburgh Mental Well-being Scale for the positive mental health surveillance of Canadian adults",
     "journal": "Health Reports",
     "doi": "10.25318/82-003-x202600300002-eng",
     "url": "https://doi.org/10.25318/82-003-x202600300002-eng"
    },
    {
     "key": "konaszewski2021",
     "authors": "Konaszewski K, Niesiobędzka M, Surzykiewicz J",
     "year": "2021",
     "title": "Factor structure and psychometric properties of a Polish adaptation of the Warwick-Edinburgh Mental Wellbeing Scale",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-021-01716-w",
     "url": "https://doi.org/10.1186/s12955-021-01716-w"
    },
    {
     "key": "smith2017",
     "authors": "Smith ORF, Alves DE, Knapstad M, Haug E, Aarø LE",
     "year": "2017",
     "title": "Measuring mental well-being in Norway: validation of the Warwick-Edinburgh Mental Well-being Scale (WEMWBS)",
     "journal": "BMC Psychiatry",
     "doi": "10.1186/s12888-017-1343-x",
     "url": "https://doi.org/10.1186/s12888-017-1343-x"
    }
   ],
   "record_notes": "The schema separates measurement properties cleanly, but WEMWBS forces two honesty caveats the fields do not naturally hold. (1) Almost all high-quality evidence is from general-population, student, youth and clinical/primary-care samples, not workplaces; convergent, invariance and responsiveness grades would drop a level if the audience's strict need is UK working-adult evidence, because that specific population is under-represented. I have graded on the overall evidence and flagged the workplace/occupational gap in each field rather than inflating or hiding it. (2) The 14-item and 7-item versions genuinely differ in their psychometrics (SWEMWBS is the Rasch interval scale; WEMWBS is essentially but not strictly unidimensional), so several fields had to carry two verdicts in one box. Test-retest is the weakest property: the well-known ICC 0.83 is a single 14-item, one-week, student estimate, with no replicated SWEMWBS or workplace value located, so change-score use at the individual level rests on responsiveness/MIC evidence rather than on retest stability. The word 'clinical' appears only in reference to the primary-care/CMD samples where SWEMWBS sensitivity to change was tested (Shah 2021); that is a statement about where evidence was earned, and workplace deployment is a different context with no equivalent criterion evidence located. Overall confidence in the record is High for structural validity, internal consistency and UK general-population norms; Moderate for invariance and responsiveness; Low for criterion validity and test-retest, chiefly because of the absence of workplace-specific and replicated stability data. Licensing is 'free' only in a qualified sense: registration is mandatory, and commercial and some organisational uses are chargeable under the Warwick licence tiers."
  },
  {
   "instrument_id": "hse-msit",
   "display_name": "HSE Management Standards Indicator Tool",
   "identity": {
    "name": "Health and Safety Executive (HSE) Management Standards Indicator Tool (also HSE-MS IT, MSIT, HSE-IT, Management Standards Revised Indicator Tool)",
    "current_version": "35-item revised Indicator Tool, seven subscales; the same instrument is described across studies as the standard form. A shorter 25-item form is sometimes referenced in practice but no psychometric validation of an official 25-item short form was located in this pass (see record_notes).",
    "item_count": "35 items across seven subscales: Demands (8), Control (6), Managerial Support (5), Peer Support (4), Relationships (4), Role (5), Change (3), per [Edwards 2008](https://doi.org/10.1080/02678370802166599).",
    "original_citation": "Cousins R, MacKay CJ, Clarke SD, Kelly C, Kelly PJ, McCaig RH (2004). 'Management Standards' work-related stress in the UK: practical development. Work & Stress, 18(2), 113-136. doi:10.1080/02678370410001734322; companion policy/science paper MacKay et al. (2004), doi:10.1080/02678370410001727474.",
    "steward_publisher": "UK Health and Safety Executive (HSE), the Great Britain national regulator for workplace health and safety. The tool sits within the HSE Management Standards approach to work-related stress ([Cousins 2004](https://doi.org/10.1080/02678370410001734322); [MacKay 2004](https://doi.org/10.1080/02678370410001727474)).",
    "licence_status": "Free to use. The Indicator Tool and its user manual are published by the HSE as a public sector resource under UK Crown copyright and made available without charge for organisational stress risk assessment ([Cousins 2004](https://doi.org/10.1080/02678370410001734322); provenance described in [Edwards 2008](https://doi.org/10.1080/02678370802166599)). A specific licence document was not retrieved this session, so the precise licence terms are stated from the tool's HSE provenance rather than a cited licence text."
   },
   "constructs_claimed": "The Indicator Tool is an exposure measure of psychosocial working conditions, not a wellbeing or health outcome measure. It captures employee perceptions of seven work-stressor dimensions that map onto the HSE Management Standards: Demands (workload, work patterns, environment), Control (autonomy over how work is done), Managerial Support (encouragement and resources from line management), Peer Support (support from colleagues), Relationships (conflict and unacceptable behaviour, e.g. bullying), Role (understanding of role and avoidance of conflicting roles) and Change (how organisational change is managed and communicated). Higher subscale scores denote more favourable (lower-risk) conditions. The framework derives from the demand-control-support tradition and UK epidemiological work on psychosocial hazards ([Cousins 2004](https://doi.org/10.1080/02678370410001734322); [MacKay 2004](https://doi.org/10.1080/02678370410001727474)).",
   "structural_validity": {
    "findings": "The seven-factor structure is well supported in UK data but the Managerial Support and Change dimensions are unstable in several non-UK adaptations. The instrument was developed from an item pool reduced by exploratory factor analysis to 35 items across seven subscales ([Cousins 2004](https://doi.org/10.1080/02678370410001734322)). The first confirmatory test on organisational-level UK data (39 organisations, N=26,382) found the original 35-item seven-factor first-order model gave an acceptable fit, and a second-order model was also acceptable, suggesting a possible higher-order single work-related stress dimension ([Edwards 2008](https://doi.org/10.1080/02678370802166599)). Cross-nationally, a seven-factor solution replicated and was equivalent across large UK (N=7,589) and Italian (N=1,298) private-sector samples in multiple-group CFA ([Toderi 2013](https://doi.org/10.1027/1015-5759/a000122)). However, several adaptations do not recover seven distinct factors: the Italian revised tool collapsed Managerial Support and Change into a single factor (termed 'elasticity'), retaining five to seven scales ([Magnavita 2012](https://doi.org/10.1093/occmed/kqs025)), the Irish ROI-MSIT (N=7,377) likewise merged Managerial Support and Change ([Boyd 2016](https://doi.org/10.1093/occmed/kqw163)), and an Argentine study retained only 24 items in six factors, discarding Change entirely ([Vaamonde 2023](https://doi.org/10.1093/occmed/kqad010)). A North Italian healthcare-worker study confirmed a seven-component structure but identified an additional factor relating to participation in work organisation ([Veronesi 2022](https://doi.org/10.3390/ijerph19159514)). A heavily revised Iranian version returned a nine-factor solution with new items, and so is best treated as a distinct instrument ([Zeinolabedini 2025](https://doi.org/10.1038/s41598-025-30714-x)).",
    "confidence": "Moderate: the seven-factor model has strong, consistent support in large UK samples and one large cross-national test, but the Managerial Support/Change distinction is not reproduced in several international adaptations, indicating the structure is not fully robust outside the UK."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent validity against other psychosocial work measures is generally adequate, with some weak spots in discriminant terms for specific subscales. In an Italian municipality sample (N=760), Indicator Tool scales showed moderate to strong correlations with the corresponding Job Content Questionnaire scales, and each scale added specific predictive contribution to self-reported stress, job satisfaction and job motivation ([Marcatto 2014](https://doi.org/10.1093/occmed/kqu038)). Subscales correlated in the expected directions with stress-related outcomes in the UK and Italian cross-cultural study ([Toderi 2013](https://doi.org/10.1027/1015-5759/a000122)). In the Argentine adaptation, discriminant validity between dimensions was adequate but convergent validity was a concern for Control, Role clarity and Relationships, where average variance extracted was at or below 0.50 ([Vaamonde 2023](https://doi.org/10.1093/occmed/kqad010)). Subscales also related meaningfully but with small effect sizes to burnout dimensions on the Maslach Burnout Inventory, with Demands and Role linked to emotional exhaustion ([Carpi 2021](https://doi.org/10.1093/occmed/kqab055)).",
    "confidence": "Moderate: several independent samples show expected-direction correlations with established measures (JCQ, MBI), but detailed convergent/discriminant metrics (e.g. AVE) are reported mainly in non-UK adaptations and are not uniformly strong."
   },
   "criterion_validity": {
    "findings": "Associations with health and job-attitude outcomes are consistently reported, but the evidence is overwhelmingly cross-sectional and against self-reported outcomes rather than objective organisational endpoints such as verified sickness absence or turnover. In a UK Health and Social Services Trust (N=707, 29% response), more favourable Management Standards scores were positively associated with job satisfaction and negatively with job-related anxiety, depression and witnessed errors/near misses ([Kerr 2009](https://doi.org/10.1093/occmed/kqp146)). In a UK call centre (N=304), only Demands (Spearman rho=-0.211) and Relationships (rho=-0.134) correlated significantly with GHQ-12 distress, while other dimensions did not, a mixed result that qualifies claims of uniform criterion validity ([Kazi 2013](https://doi.org/10.1093/occmed/kqt052)). In an Italian bank the tool related to GHQ-12 distress and Work Ability Index scores ([Guidi 2012](https://doi.org/10.1093/occmed/kqs021)), and in a UK prison-service sample (N=1,038) odds ratios linked poor psychosocial conditions to impaired psychological wellbeing, though the authors noted exposure scores alone were insufficient to set intervention priorities without an outcome measure ([Bevan 2010](https://doi.org/10.1093/occmed/kqq109)). A systematic review concluded there was a clear relationship between Indicator Tool scores and alternative wellbeing and stress measures ([Brookes 2013](https://doi.org/10.1093/occmed/kqt078)). Direct validation against objective sickness absence is weak: an early four-organisation study of the earlier filter-question version found the screening filters insensitive with low positive predictive value, and warned that using work absence as the measure of stress cost may substantially underestimate the true burden ([Main 2005](https://doi.org/10.1093/occmed/kqi044)).",
    "confidence": "Moderate: multiple studies link the tool to self-reported distress and job attitudes in expected directions, but findings are cross-sectional, at least one UK study found most subscales unrelated to GHQ-12, and criterion evidence against objective outcomes (verified absence, turnover, diagnosed conditions) is sparse and, for absence specifically, problematic."
   },
   "internal_consistency": {
    "findings": "Internal consistency is generally good to excellent across subscales and languages. In the anchor UK analysis, Cronbach's alpha was Demands 0.87, Control 0.82, Managerial Support 0.88, Peer Support 0.82, Relationships 0.78, Role 0.83 and Change 0.80, with the original development study ([Cousins 2004](https://doi.org/10.1080/02678370410001734322)) reporting a comparable range of about 0.78 to 0.89 ([Edwards 2008](https://doi.org/10.1080/02678370802166599)). The Italian revised tool reported alphas of 0.75 to 0.86 ([Magnavita 2012](https://doi.org/10.1093/occmed/kqs025)), the Irish ROI-MSIT 0.75 to 0.91 ([Boyd 2016](https://doi.org/10.1093/occmed/kqw163)), and the Argentine adaptation composite reliability of 0.70 to 0.82 ([Vaamonde 2023](https://doi.org/10.1093/occmed/kqad010)). The revised Iranian version reported an overall alpha of 0.949 and McDonald's omega of 0.739 to 0.894, though for a restructured nine-factor instrument ([Zeinolabedini 2025](https://doi.org/10.1038/s41598-025-30714-x)).",
    "confidence": "High: multiple good-quality studies across several countries and large samples consistently report subscale alphas at or above roughly 0.75, meeting conventional adequacy thresholds."
   },
   "test_retest_reliability": {
    "findings": "No test-retest (temporal stability) evidence for the standard UK 35-item Indicator Tool was located in this pass, and this is a genuine gap in the psychometric record. The reliability studies retrieved report internal consistency (alpha, omega) rather than repeat administration over time ([Edwards 2008](https://doi.org/10.1080/02678370802166599); [Magnavita 2012](https://doi.org/10.1093/occmed/kqs025); [Boyd 2016](https://doi.org/10.1093/occmed/kqw163)). The only intraclass correlation coefficient located (ICC=0.92) comes from the revised nine-factor Iranian version, where the abstract does not clearly establish it as a test-retest coefficient over a defined interval, and in any case applies to a modified instrument rather than the original tool ([Zeinolabedini 2025](https://doi.org/10.1038/s41598-025-30714-x)).",
    "confidence": "Absent: no test-retest reliability evidence for the standard tool was located this session; the absence is flagged explicitly."
   },
   "measurement_invariance": {
    "findings": "Invariance evidence exists across sector and across the UK/Italy language boundary, reaching at least metric level, but scalar invariance is not clearly demonstrated. Multiple-group CFA across large UK (N=7,589) and Italian (N=1,298) private-sector samples found the seven-factor solution equivalent, supporting metric equivalence together with factor variance and factor covariance equivalence ([Toderi 2013](https://doi.org/10.1027/1015-5759/a000122)). A dedicated UK study reported measurement invariance of the Indicator Tool across public and private sector organisations ([Edwards 2012](https://doi.org/10.1080/02678373.2012.688554)); the detailed invariance level (configural/metric/scalar) from that paper was not retrievable from the abstract in this pass. No formal invariance testing across sex, age or occupational group was located this session.",
    "confidence": "Low: cross-national evidence supports at least metric invariance and a UK study supports public/private invariance, but scalar (intercept) invariance is not clearly established and invariance by sex, age and occupation is untested in the retrieved literature."
   },
   "responsiveness_mic": {
    "findings": "No formal responsiveness or minimal important change (MIC) evidence was located in this pass. The tool has been used in pre/post and longitudinal designs, for example a longitudinal healthcare-worker study spanning the onset of the SARS-CoV-2 pandemic ([Veronesi 2022](https://doi.org/10.3390/ijerph19159514)) and a pre/post stress-management evaluation among Sierra Leone healthcare workers that observed changes in domain scores ([Jones 2020](https://doi.org/10.1136/bmjopen-2019-032929)), but neither established responsiveness statistics or a minimal important change threshold. HSE positions the tool for organisational monitoring and comparison rather than individual change detection.",
    "confidence": "Absent: no responsiveness or MIC estimates for the tool were located this session; longitudinal use exists but without formal responsiveness/MIC analysis."
   },
   "populations_languages_norms": {
    "findings": "The tool has been validated in the UK and adapted into several languages and settings, with UK national benchmark norms published at organisational level. UK evidence spans large multi-organisation datasets ([Edwards 2008](https://doi.org/10.1080/02678370802166599); [Cousins 2004](https://doi.org/10.1080/02678370410001734322)), the NHS and health and social care ([Kerr 2009](https://doi.org/10.1093/occmed/kqp146)), call centres ([Kazi 2013](https://doi.org/10.1093/occmed/kqt052)), the prison service ([Bevan 2010](https://doi.org/10.1093/occmed/kqq109)) and Ministry of Defence personnel, where content was judged too narrow without an added work-life balance scale ([Bridger 2016](https://doi.org/10.1080/00140139.2015.1057544)). Validated or adapted non-UK versions include Italian ([Magnavita 2012](https://doi.org/10.1093/occmed/kqs025); [Guidi 2012](https://doi.org/10.1093/occmed/kqs021); [Toderi 2013](https://doi.org/10.1027/1015-5759/a000122); [Veronesi 2022](https://doi.org/10.3390/ijerph19159514)), Irish ([Boyd 2016](https://doi.org/10.1093/occmed/kqw163)), Argentine ([Vaamonde 2023](https://doi.org/10.1093/occmed/kqad010)) and a revised Iranian version ([Zeinolabedini 2025](https://doi.org/10.1038/s41598-025-30714-x)); the Argentine study also notes prior validation in Italy, Iran and Malta. UK normative percentile benchmark tables derived from the organisational dataset are provided to let employers compare their organisational averages against national reference values ([Edwards 2008](https://doi.org/10.1080/02678370802166599)), consistent with the HSE benchmarking approach.",
    "confidence": "Moderate: broad UK and multi-country evidence with published UK organisational-level norms, but norms are for organisational means rather than individuals, and several non-UK versions differ structurally from the UK original."
   },
   "criticisms_controversies": "Several substantive criticisms recur. First, the seven-factor structure is not fully reproducible outside the UK: the Managerial Support and Change dimensions repeatedly collapse into one factor in Italian and Irish adaptations, and Change is sometimes dropped altogether, questioning the universality of the seven-domain model ([Magnavita 2012](https://doi.org/10.1093/occmed/kqs025); [Boyd 2016](https://doi.org/10.1093/occmed/kqw163); [Vaamonde 2023](https://doi.org/10.1093/occmed/kqad010)). Second, criterion evidence is largely cross-sectional and against self-reported outcomes, with at least one UK study finding most subscales unrelated to GHQ-12 distress ([Kazi 2013](https://doi.org/10.1093/occmed/kqt052)), and an early evaluation showing the screening filters were insensitive with poor positive predictive value and that sickness absence understates the true cost of psychosocial hazards ([Main 2005](https://doi.org/10.1093/occmed/kqi044)). Third, the tool measures exposure (perceived working conditions) rather than health or wellbeing outcomes, so exposure data alone are insufficient for prioritising interventions and are best paired with an outcome measure ([Bevan 2010](https://doi.org/10.1093/occmed/kqq109)). Fourth, content validity can be too narrow for specific contexts, for example the military, where a work-life balance dimension was needed ([Bridger 2016](https://doi.org/10.1080/00140139.2015.1057544)). Finally, test-retest reliability and formal responsiveness/MIC remain unestablished for the standard tool (this pass), and much criterion evidence shares common-method variance because exposure and outcome are both self-reported.",
   "citations": [
    {
     "key": "cousins2004",
     "authors": "Cousins R, MacKay CJ, Clarke SD, Kelly C, Kelly PJ, McCaig RH",
     "year": "2004",
     "title": "'Management Standards' work-related stress in the UK: practical development",
     "journal": "Work & Stress",
     "doi": "10.1080/02678370410001734322",
     "url": "https://doi.org/10.1080/02678370410001734322"
    },
    {
     "key": "mackay2004",
     "authors": "MacKay CJ, Cousins R, Kelly PJ, Lee S, McCaig RH",
     "year": "2004",
     "title": "'Management Standards' and work-related stress in the UK: policy background and science",
     "journal": "Work & Stress",
     "doi": "10.1080/02678370410001727474",
     "url": "https://doi.org/10.1080/02678370410001727474"
    },
    {
     "key": "edwards2008",
     "authors": "Edwards JA, Webster S, Van Laar D, Easton S",
     "year": "2008",
     "title": "Psychometric analysis of the UK Health and Safety Executive's Management Standards work-related stress Indicator Tool",
     "journal": "Work & Stress",
     "doi": "10.1080/02678370802166599",
     "url": "https://doi.org/10.1080/02678370802166599"
    },
    {
     "key": "edwards2012",
     "authors": "Edwards JA, Webster S",
     "year": "2012",
     "title": "Psychosocial risk assessment: measurement invariance of the UK Health and Safety Executive's Management Standards Indicator Tool across public and private sector organizations",
     "journal": "Work & Stress",
     "doi": "10.1080/02678373.2012.688554",
     "url": "https://doi.org/10.1080/02678373.2012.688554"
    },
    {
     "key": "toderi2013",
     "authors": "Toderi S, Balducci C, Edwards JA, Sarchielli G, Broccoli M, Mancini G",
     "year": "2013",
     "title": "Psychometric properties of the UK and Italian versions of the HSE Stress Indicator Tool: a cross-cultural investigation",
     "journal": "European Journal of Psychological Assessment",
     "doi": "10.1027/1015-5759/a000122",
     "url": "https://doi.org/10.1027/1015-5759/a000122"
    },
    {
     "key": "kerr2009",
     "authors": "Kerr R, McHugh M, McCrory M",
     "year": "2009",
     "title": "HSE Management Standards and stress-related work outcomes",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqp146",
     "url": "https://doi.org/10.1093/occmed/kqp146"
    },
    {
     "key": "marcatto2014",
     "authors": "Marcatto F, Colautti L, Larese Filon F, Luis O, Ferrante D",
     "year": "2014",
     "title": "The HSE Management Standards Indicator Tool: concurrent and construct validity",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqu038",
     "url": "https://doi.org/10.1093/occmed/kqu038"
    },
    {
     "key": "bevan2010",
     "authors": "Bevan A, Houdmont J, Menear N",
     "year": "2010",
     "title": "The Management Standards Indicator Tool and the estimation of risk",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqq109",
     "url": "https://doi.org/10.1093/occmed/kqq109"
    },
    {
     "key": "guidi2012",
     "authors": "Guidi S, Bagnara S, Fichera GP",
     "year": "2012",
     "title": "The HSE indicator tool, psychological distress and work ability",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqs021",
     "url": "https://doi.org/10.1093/occmed/kqs021"
    },
    {
     "key": "magnavita2012",
     "authors": "Magnavita N",
     "year": "2012",
     "title": "Validation of the Italian version of the HSE Indicator Tool",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqs025",
     "url": "https://doi.org/10.1093/occmed/kqs025"
    },
    {
     "key": "brookes2013",
     "authors": "Brookes K, Limbert C, Deacy C, O'Reilly A, Scott S, Thirlaway K",
     "year": "2013",
     "title": "Systematic review: work-related stress and the HSE Management Standards",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqt078",
     "url": "https://doi.org/10.1093/occmed/kqt078"
    },
    {
     "key": "kazi2013",
     "authors": "Kazi A, Haslam C",
     "year": "2013",
     "title": "Stress management standards: a warning indicator for employee health",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqt052",
     "url": "https://doi.org/10.1093/occmed/kqt052"
    },
    {
     "key": "boyd2016",
     "authors": "Boyd S, Kerr R, Murray P",
     "year": "2016",
     "title": "Psychometric properties of the Irish Management Standards Indicator Tool",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqw163",
     "url": "https://doi.org/10.1093/occmed/kqw163"
    },
    {
     "key": "carpi2021",
     "authors": "Carpi M, Bruschini M, Burla F",
     "year": "2021",
     "title": "HSE Management Standards and burnout dimensions among rehabilitation professionals",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqab055",
     "url": "https://doi.org/10.1093/occmed/kqab055"
    },
    {
     "key": "vaamonde2023",
     "authors": "Vaamonde JD, Giacobino AE",
     "year": "2023",
     "title": "Psychometric properties of the HSE Indicator Tool: evidence from Argentina",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqad010",
     "url": "https://doi.org/10.1093/occmed/kqad010"
    },
    {
     "key": "bridger2016",
     "authors": "Bridger RS, Dobson K, Davison H",
     "year": "2016",
     "title": "Using the HSE stress indicator tool in a military context",
     "journal": "Ergonomics",
     "doi": "10.1080/00140139.2015.1057544",
     "url": "https://doi.org/10.1080/00140139.2015.1057544"
    },
    {
     "key": "veronesi2022",
     "authors": "Veronesi G, Giusti EM, D'Amato A, et al.",
     "year": "2022",
     "title": "The North Italian Longitudinal Study Assessing the Mental Health Effects of SARS-CoV-2 Pandemic on Health Care Workers, Part I: study design and psychometric structural validity of the HSE Indicator Tool and Work Satisfaction Scale",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph19159514",
     "url": "https://doi.org/10.3390/ijerph19159514"
    },
    {
     "key": "jones2020",
     "authors": "Jones S, White S, Ormrod J, et al.",
     "year": "2020",
     "title": "Work-based risk factors and quality of life in health care workers providing maternal and newborn care during the Sierra Leone Ebola epidemic",
     "journal": "BMJ Open",
     "doi": "10.1136/bmjopen-2019-032929",
     "url": "https://doi.org/10.1136/bmjopen-2019-032929"
    },
    {
     "key": "zeinolabedini2025",
     "authors": "Zeinolabedini M, Motlagh ME, Heidarnia A, et al.",
     "year": "2025",
     "title": "Psychometric properties of the revised version of the Health and Safety Executive Management Standards Indicator Tool",
     "journal": "Scientific Reports",
     "doi": "10.1038/s41598-025-30714-x",
     "url": "https://doi.org/10.1038/s41598-025-30714-x"
    },
    {
     "key": "main2005",
     "authors": "Main CJ, Glozier N, Wright IA",
     "year": "2005",
     "title": "Validity of the HSE stress tool: an investigation within four organizations by the Corporate Health and Performance Group",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqi044",
     "url": "https://doi.org/10.1093/occmed/kqi044"
    }
   ],
   "record_notes": "Framing: the Indicator Tool is a work-stressor EXPOSURE measure of perceived psychosocial working conditions, not a wellbeing outcome; the schema's outcome-oriented property fields (criterion validity, responsiveness/MIC) were completed with that in mind and criterion evidence is reported as associations with separate outcome measures. Version ambiguity: the literature retrieved this session consistently describes and validates the 35-item seven-subscale form; I did not locate a psychometric validation of an official 25-item HSE short form, so the item_count and version fields report the 35-item form and flag the 25-item form as referenced but unverified in this pass. Two properties are genuinely ABSENT in the retrieved evidence and should not be read as oversights: test-retest reliability (no temporal-stability study for the standard tool was found; the only ICC comes from a restructured Iranian version and its interpretation as test-retest is unclear) and responsiveness/MIC (longitudinal use exists but no formal statistics). Measurement invariance is graded Low because cross-national and public/private evidence reaches at most metric level and scalar invariance is not clearly demonstrated; the Edwards 2012 abstract was not retrievable in full, so the exact invariance level from that paper is stated conservatively. Licence terms are inferred from the tool's HSE Crown-copyright provenance rather than a cited licence document. Coefficients: all alpha/omega/ICC/correlation values quoted come from abstracts or (for Edwards 2008) the full text retrieved this session. Overall confidence in the record is moderate: internal consistency is well established (High), structural, convergent and criterion validity are Moderate with real conflicts reported, while invariance is Low and test-retest and responsiveness are Absent."
  },
  {
   "instrument_id": "perma",
   "display_name": "Workplace PERMA-Profiler (and PERMA-Profiler)",
   "identity": {
    "name": "Workplace PERMA-Profiler (occupational adaptation of the PERMA-Profiler)",
    "current_version": "PERMA-Profiler (Butler & Kern, 2016); Workplace PERMA-Profiler (Kern), a work-contextualised reword of the same item set. No numbered revision is in circulation.",
    "item_count": "23 items total: 15 core items measuring the five PERMA domains (three per domain), plus 8 additional items covering overall well-being, negative emotion, loneliness and physical health. The Workplace form retains this structure with items reworded to the work context; short forms exist but are not the reference version.",
    "original_citation": "Butler J, Kern ML (2016). The PERMA-Profiler: A brief multidimensional measure of flourishing. International Journal of Wellbeing. DOI 10.5502/ijw.v6i3.526",
    "steward_publisher": "Instrument developed by Julie Butler and Margaret (Peggy) L. Kern; the PERMA framework derives from Seligman (2011). The Workplace adaptation is attributed to Kern. Distributed by the authors through Peggy Kern's academic website rather than a commercial publisher.",
    "licence_status": "Reported as free to use for research and non-commercial purposes, distributed without fee by the authors via Peggy Kern's website. No formal licence text or terms document was retrieved in this pass, so the exact licence basis (for example an explicit non-commercial clause) could not be verified from a citable source; the free-for-research status rests on the developers' distribution practice and the gold open-access status of the founding paper (Butler & Kern, 2016, DOI 10.5502/ijw.v6i3.526)."
   },
   "constructs_claimed": "The instrument operationalises Seligman's PERMA model of flourishing across five claimed domains: Positive emotion, Engagement, Relationships, Meaning and Accomplishment, each measured by three items ([Butler & Kern 2016](https://doi.org/10.5502/ijw.v6i3.526)). A composite overall well-being score is derived from the fifteen domain items together with a general happiness item. Ancillary items index negative emotion, loneliness and self-rated physical health but are not part of the five-domain model. The Workplace PERMA-Profiler claims to measure the same five constructs specifically as they are experienced at work, that is workplace flourishing ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA); [Choi et al. 2019](https://doi.org/10.35371/aoem.2019.31.e17)). The multidimensional PERMA measurement approach was first applied to a defined setting in schools by [Kern et al. 2014](https://doi.org/10.1080/17439760.2014.936962), which prefigured the workplace adaptation.",
   "structural_validity": {
    "findings": "The claimed five-factor structure is only partly and inconsistently supported, and the factor structure is the most contested aspect of the instrument. In occupational samples the correlated five-factor model has typically reached only marginal fit: the Japanese Workplace PERMA-Profiler gave CFI 0.892, TLI 0.858, RMSEA 0.105 and SRMR 0.051 ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)), the Korean version CFI 0.909, TLI 0.881, RMSEA 0.110 and SRMR 0.054 ([Choi et al. 2019](https://doi.org/10.35371/aoem.2019.31.e17)), and the Chinese workplace version CFI 0.887, TLI 0.855, RMSEA 0.114 and SRMR 0.060 ([Yang et al. 2024](https://doi.org/10.1186/s12889-024-18194-6)); all three show RMSEA above conventional thresholds. The Polish Workplace version is the clearest positive case, with the five-factor model reaching desired fit and outperforming a single-factor model in a large sample (N=1070) ([Fortuna et al. 2025](https://doi.org/10.1371/journal.pone.0319088)). Evidence from the general (non-workplace) PERMA-Profiler diverges further: an Australian adult study could replicate neither the theorised five-factor model nor a single-factor alternative and concluded the psychometric properties were insufficient ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932)); a US student-veteran sample yielded a two-factor solution (emotional versus performance character strengths) ([Umucu et al. 2019](https://doi.org/10.1080/07448481.2018.1546182)); and a Chinese haemodialysis sample extracted two dimensions ([Qian et al. 2025](https://doi.org/10.1186/s40359-025-02560-z)). Where the five-factor model does hold, it is often as five highly correlated first-order factors: in a large Chinese general sample the first-order five-factor solution fitted better than a higher-order well-being factor (CFI 0.943 after modification) ([Nie et al. 2024](https://doi.org/10.1016/j.actpsy.2024.104248)), and a French study confirmed both a correlated five-factor and a second-order model while its exploratory analysis suggested a more parsimonious three factors ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)). In autistic adults a bifactor model with a dominant general factor fitted better than the five-factor model ([Grosvenor et al. 2023](https://doi.org/10.1089/aut.2022.0049)). Five-factor support is stronger in adolescent adaptations ([Fernandes et al. 2024](https://doi.org/10.3389/fpsyg.2024.1415084); [Waigel & Lemos 2023](https://doi.org/10.21500/20112084.5737)). Across studies the Engagement domain is the recurrent weak point.",
    "confidence": "Low. Many studies but findings are inconsistent across populations and forms; five-factor fit is frequently marginal, competing two-factor, three-factor, single-factor and bifactor solutions recur, and the workplace form specifically has few confirmatory studies."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent and discriminant patterns are generally in the expected direction, though discriminant separation from distress measures is sometimes weak. For the Workplace form, well-being domains correlated moderately to strongly and positively with job satisfaction and work-related psychosocial factors and inversely with psychological distress ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)), and positively with work engagement and life mental well-being while correlating negatively with burnout, occupational stressors and stress responses ([Choi et al. 2019](https://doi.org/10.35371/aoem.2019.31.e17)). For the general form, a large Chinese study reported convergent correlations of r=0.53 to 0.85 and discriminant correlations of r=-0.19 to -0.38 ([Nie et al. 2024](https://doi.org/10.1016/j.actpsy.2024.104248)). A French study found strong positive associations with flourishing and perceived happiness (convergent) and weaker associations with anxiety, depression, loneliness and negative emotion (discriminant), while cautioning that some correlations with anxiety and depression nonetheless reached moderate to strong magnitude ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)). An Australian study found moderate correlations with depression, anxiety and stress (r=-0.374 to -0.645) but no meaningful association with objectively measured physical activity or sleep ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932)), and an autistic-adult study confirmed expected correlations with mental-health conditions and life satisfaction ([Grosvenor et al. 2023](https://doi.org/10.1089/aut.2022.0049)).",
    "confidence": "Moderate. Multiple studies across the general and workplace forms show consistent, theory-congruent convergent correlations; discriminant validity is adequate but repeatedly weaker against depression and anxiety, and no association with objective (non-self-report) markers."
   },
   "criterion_validity": {
    "findings": "Criterion evidence against organisational and health outcomes is limited to cross-sectional, self-reported proxies; no study retrieved in this pass links the instrument to hard organisational records such as sickness absence, staff turnover or medically diagnosed conditions. Workplace well-being domains were associated with job strain components: job control and supervisor and coworker support correlated with all five domains, while job demands related only to Engagement and Meaning ([Yang et al. 2021](https://doi.org/10.1097/JOM.0000000000002455)). Well-being domains, principally Positive emotion, Engagement and Relationships, were inversely associated with personal and work-related burnout in a Taiwanese multicentre sample (N=242) ([Wu et al. 2025](https://doi.org/10.1097/JOM.0000000000003318)), and Positive emotion was inversely associated with prolonged fatigue in a Chinese workforce sample ([Yang et al. 2024](https://doi.org/10.1186/s12889-024-18194-6)). In a university-workplace study, workplace well-being mediated the relationship between psychological distress and subjective quality of life and was lower among staff with suspected psychological health conditions ([Ng et al. 2025](https://doi.org/10.3389/fpsyg.2025.1598910)). Among rural veterans, well-being scores were positively associated with mental and physical health outcomes and were lower in those with service-connected disabilities ([Umucu et al. 2024](https://doi.org/10.3389/fpubh.2024.1500659)). All of these are concurrent associations rather than predictive criterion validity against objective outcomes.",
    "confidence": "Low. Several studies show theory-consistent associations with burnout, fatigue, job strain and distress, but all are cross-sectional self-report; there is no predictive validity against objective organisational or clinical outcomes and no UK workplace evidence."
   },
   "internal_consistency": {
    "findings": "Internal consistency is the instrument's strongest property and is generally acceptable to good, with the Engagement subscale the consistent exception. For the Workplace form, Cronbach's alpha ranged from 0.75 to 0.96 in Japanese workers ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)), 0.70 to 0.95 in Korean workers ([Choi et al. 2019](https://doi.org/10.35371/aoem.2019.31.e17)) and 0.69 to 0.93 in Chinese workers ([Yang et al. 2024](https://doi.org/10.1186/s12889-024-18194-6)); the Polish Workplace version met minimum reliability thresholds ([Fortuna et al. 2025](https://doi.org/10.1371/journal.pone.0319088)). For the general form, alphas were 0.79 to 0.88 in a large Chinese sample ([Nie et al. 2024](https://doi.org/10.1016/j.actpsy.2024.104248)), the total scale reached alpha 0.914 in a Chinese haemodialysis sample ([Qian et al. 2025](https://doi.org/10.1186/s40359-025-02560-z)), and an autistic-adult sample reported alpha 0.93 ([Grosvenor et al. 2023](https://doi.org/10.1089/aut.2022.0049)). In Australian adults subscale alphas ranged 0.80 to 0.93 for all domains except Engagement (alpha 0.66) ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932)); the French validation likewise noted lower reliability specifically for Engagement ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)). Omega and composite reliability were reported as adequate in an adolescent adaptation ([Fernandes et al. 2024](https://doi.org/10.3389/fpsyg.2024.1415084)). The three-item domains, especially Engagement, are the source of most reliability shortfalls.",
    "confidence": "High. Many studies across multiple populations, forms and languages report consistent alpha values in the acceptable to good range, with a reproducible and openly reported weakness for the Engagement subscale."
   },
   "test_retest_reliability": {
    "findings": "Test-retest reliability is thinly evidenced and this is a genuine gap, most acute for the Workplace form. The founding paper reports cross-time consistency for the general PERMA-Profiler but does not present a conventional retest coefficient in the abstract retrieved ([Butler & Kern 2016](https://doi.org/10.5502/ijw.v6i3.526)). The Japanese Workplace validation collected a one-month follow-up (baseline N=310, follow-up N=86) and reported intraclass correlation coefficients of 0.75 to 0.96, but these ICCs are presented alongside internal-consistency alphas and the abstract does not cleanly separate a retest ICC per domain ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)). The Chinese workplace version similarly reported ICCs of 0.70 to 0.92 in a reliability analysis rather than a dedicated retest design ([Yang et al. 2024](https://doi.org/10.1186/s12889-024-18194-6)). The clearest temporal-stability evidence comes from non-workplace forms: the Polish Workplace study included a repeated-measure subsample (N=66) and reported strong stability ([Fortuna et al. 2025](https://doi.org/10.1371/journal.pone.0319088)); a French study reported good temporal stability for the total score and most subscales ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)); and a Chinese haemodialysis study reported a retest correlation of r=0.764 ([Qian et al. 2025](https://doi.org/10.1186/s40359-025-02560-z)). No UK or occupational-UK retest estimate was located, and retest intervals, when reported, are short (around one month). The absence of a robust, well-powered retest study specifically for the Workplace PERMA-Profiler is a flagged evidence gap.",
    "confidence": "Low. A small number of short-interval studies with modest follow-up samples, mostly reporting ICCs bundled with internal consistency rather than purpose-designed retest coefficients; no workplace-specific, adequately powered, or UK retest evidence located. Flag: near-absent for the Workplace form."
   },
   "measurement_invariance": {
    "findings": "Invariance testing exists but is concentrated in non-workplace populations and rarely reaches full scalar invariance in occupational samples. A large Chinese general-population study established factorial invariance across genders ([Nie et al. 2024](https://doi.org/10.1016/j.actpsy.2024.104248)). A Brazilian adolescent adaptation demonstrated invariance across age groups and gender ([Fernandes et al. 2024](https://doi.org/10.3389/fpsyg.2024.1415084)), and an open-and-distance-learning adaptation in Botswana examined invariance within a five-factor framework ([Magare et al. 2022](https://doi.org/10.3390/ijerph192416886)). For work-specific instruments in the wider PERMA family, the Spanish PERMA+4 Positive Functioning at Work scale showed measurement invariance across educational levels ([Garcia-Selva et al. 2024](https://doi.org/10.7334/psicothema2023.341)). For the Workplace PERMA-Profiler itself, no study retrieved in this pass reports a full multigroup invariance analysis across sex, age, occupation or language, and the level reached (configural, metric or scalar) for occupational deployment is therefore not established. Cross-language equivalence is assumed rather than formally tested, since each translation is validated as a standalone instrument.",
    "confidence": "Low. Invariance is established (up to scalar in some cases) for general-population and adolescent forms, but is absent for the Workplace PERMA-Profiler specifically; occupational and cross-occupation invariance was not located."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness and a minimal important change threshold are essentially unestablished. No study retrieved in this pass reports a minimal important change, minimal detectable change tied to an anchor, or a formal responsiveness analysis for the Workplace PERMA-Profiler. Some validation studies computed measurement error and, in the Japanese case, measurement-error statistics alongside reliability ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)), but these are not responsiveness or MIC estimates. Intervention and longitudinal PERMA studies exist in the wider literature, yet a defensible, citable MIC value for interpreting change scores in a workplace deployment was not located.",
    "confidence": "Absent. No responsiveness or minimal-important-change evidence located for the Workplace PERMA-Profiler in this pass."
   },
   "populations_languages_norms": {
    "findings": "The Workplace PERMA-Profiler has been validated in occupational samples in Japanese ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)), Korean ([Choi et al. 2019](https://doi.org/10.35371/aoem.2019.31.e17)), Chinese ([Yang et al. 2024](https://doi.org/10.1186/s12889-024-18194-6)) and Polish ([Fortuna et al. 2025](https://doi.org/10.1371/journal.pone.0319088)) workers, with further occupational application in Taiwanese ([Wu et al. 2025](https://doi.org/10.1097/JOM.0000000000003318)) and university-staff ([Ng et al. 2025](https://doi.org/10.3389/fpsyg.2025.1598910)) samples. The general PERMA-Profiler has been validated far more widely, including English-language Australian adults ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932)), US student veterans ([Umucu et al. 2019](https://doi.org/10.1080/07448481.2018.1546182); [Umucu et al. 2024](https://doi.org/10.3389/fpubh.2024.1500659)), autistic adults ([Grosvenor et al. 2023](https://doi.org/10.1089/aut.2022.0049)), Chinese adults and patients ([Nie et al. 2024](https://doi.org/10.1016/j.actpsy.2024.104248); [Qian et al. 2025](https://doi.org/10.1186/s40359-025-02560-z)), French adults ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)), German adults ([Wammerl et al. 2019](https://doi.org/10.1007/s41543-019-00021-0)) and adolescents in Brazil and Argentina ([Fernandes et al. 2024](https://doi.org/10.3389/fpsyg.2024.1415084); [Waigel & Lemos 2023](https://doi.org/10.21500/20112084.5737)). No UK validation study, UK occupational norms or UK benchmark tables were located in this pass; the founding studies used large but predominantly online international-English samples ([Butler & Kern 2016](https://doi.org/10.5502/ijw.v6i3.526)), and no published normative reference specific to UK workplaces was found. Practitioners in the UK would therefore be relying on non-UK reference data.",
    "confidence": "Moderate for breadth of language validation; Low for workplace-specific and UK-specific norms. Many languages validated, but the Workplace form in only a handful, and no UK occupational norms located."
   },
   "criticisms_controversies": "The central controversy is whether PERMA is empirically distinguishable from general subjective well-being. Goodman and colleagues found that a PERMA composite and an established subjective well-being measure were near-indistinguishable (correlating very highly), arguing the five elements do not add measurable information beyond overall well-being and raising a jangle-fallacy concern ([Goodman et al. 2018](https://doi.org/10.1080/17439760.2017.1388434)). This is reinforced structurally by the recurrent finding that the five factors are so highly intercorrelated that single-factor, bifactor or reduced-factor solutions often fit as well as or better than the theorised five-factor model ([Grosvenor et al. 2023](https://doi.org/10.1089/aut.2022.0049); [Umucu et al. 2019](https://doi.org/10.1080/07448481.2018.1546182); [Qian et al. 2025](https://doi.org/10.1186/s40359-025-02560-z)), and by an Australian study that could not replicate the five-factor structure and judged the psychometric properties insufficient ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932)). A second, consistent criticism is the weak Engagement subscale, which repeatedly shows the lowest reliability (for example alpha 0.66) and poorest item performance ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932); [Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)). A third issue is discriminant validity against distress: correlations with anxiety and depression sometimes reach moderate-to-strong magnitude, blurring the separation between well-being and ill-being ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)). The proliferation of a work-focused successor framework, PERMA+4, which adds further blocks and has itself been flagged for a possible jangle fallacy with psychological safety, reflects unsettled debate about what the workplace model should contain ([Donaldson et al. 2022](https://doi.org/10.3389/fpsyg.2021.817244); [Lorenz et al. 2023](https://doi.org/10.3389/fpsyg.2023.1231299)). Finally, the Workplace form specifically rests on a much thinner evidence base than the general PERMA-Profiler: its occupational validations are few, mostly East Asian and European, generally reach only marginal confirmatory fit, and lack dedicated test-retest, invariance, responsiveness and UK evidence.",
   "citations": [
    {
     "key": "butler_kern_2016",
     "authors": "Butler J, Kern ML",
     "year": "2016",
     "title": "The PERMA-Profiler: A brief multidimensional measure of flourishing",
     "journal": "International Journal of Wellbeing",
     "doi": "10.5502/ijw.v6i3.526",
     "url": "https://doi.org/10.5502/ijw.v6i3.526"
    },
    {
     "key": "goodman_2018",
     "authors": "Goodman FR, Disabato DJ, Kashdan TB, Kauffman SB",
     "year": "2018",
     "title": "Measuring well-being: A comparison of subjective well-being and PERMA",
     "journal": "The Journal of Positive Psychology",
     "doi": "10.1080/17439760.2017.1388434",
     "url": "https://doi.org/10.1080/17439760.2017.1388434"
    },
    {
     "key": "kern_2014",
     "authors": "Kern ML, Waters LE, Adler A, White MA",
     "year": "2014",
     "title": "A multidimensional approach to measuring well-being in students: Application of the PERMA framework",
     "journal": "The Journal of Positive Psychology",
     "doi": "10.1080/17439760.2014.936962",
     "url": "https://doi.org/10.1080/17439760.2014.936962"
    },
    {
     "key": "wammerl_2019",
     "authors": "Wammerl M, Jaunig J, Mairunteregger T, Streit P",
     "year": "2019",
     "title": "The German Version of the PERMA-Profiler: Evidence for Construct and Convergent Validity",
     "journal": "International Journal of Applied Positive Psychology",
     "doi": "10.1007/s41543-019-00021-0",
     "url": "https://doi.org/10.1007/s41543-019-00021-0"
    },
    {
     "key": "donaldson_2022",
     "authors": "Donaldson SI, van Zyl LE, Donaldson SI",
     "year": "2022",
     "title": "PERMA+4: A Framework for Work-Related Wellbeing, Performance and Positive Organizational Psychology 2.0",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2021.817244",
     "url": "https://doi.org/10.3389/fpsyg.2021.817244"
    },
    {
     "key": "watanabe_2018",
     "authors": "Watanabe K, Kawakami N, Shiotani T et al.",
     "year": "2018",
     "title": "The Japanese Workplace PERMA-Profiler: A validation study among Japanese workers.",
     "journal": "Journal of occupational health",
     "doi": "10.1539/joh.2018-0050-OA",
     "url": "https://doi.org/10.1539/joh.2018-0050-OA"
    },
    {
     "key": "choi_2019",
     "authors": "Choi SP, Suh C, Yang JW et al.",
     "year": "2019",
     "title": "Korean translation and validation of the Workplace Positive emotion, Engagement, Relationships, Meaning, and Accomplishment (PERMA)-Profiler.",
     "journal": "Annals of occupational and environmental medicine",
     "doi": "10.35371/aoem.2019.31.e17",
     "url": "https://doi.org/10.35371/aoem.2019.31.e17"
    },
    {
     "key": "yang_2024",
     "authors": "Yang CC, Chen HT, Luo KH et al.",
     "year": "2024",
     "title": "The validation of Chinese version of workplace PERMA-profiler and the association between workplace well-being and fatigue.",
     "journal": "BMC public health",
     "doi": "10.1186/s12889-024-18194-6",
     "url": "https://doi.org/10.1186/s12889-024-18194-6"
    },
    {
     "key": "fortuna_2025",
     "authors": "Fortuna P, Czerw A, Ostafińska-Molik B et al.",
     "year": "2025",
     "title": "Flourishing at work: Psychometric properties of the Polish version of the Workplace PERMA-Profiler.",
     "journal": "PloS one",
     "doi": "10.1371/journal.pone.0319088",
     "url": "https://doi.org/10.1371/journal.pone.0319088"
    },
    {
     "key": "yang_2021",
     "authors": "Yang CC, Watanabe K, Kawakami N",
     "year": "2021",
     "title": "The Associations Between Job Strain, Workplace PERMA Profiler, and Work Engagement.",
     "journal": "Journal of occupational and environmental medicine",
     "doi": "10.1097/JOM.0000000000002455",
     "url": "https://doi.org/10.1097/JOM.0000000000002455"
    },
    {
     "key": "wu_2025",
     "authors": "Wu TW, Chuang HY, Lin CP et al.",
     "year": "2025",
     "title": "Is Well-being Associated With Burnout? From a Multicenter Cross-sectional Study in Taiwan.",
     "journal": "Journal of occupational and environmental medicine",
     "doi": "10.1097/JOM.0000000000003318",
     "url": "https://doi.org/10.1097/JOM.0000000000003318"
    },
    {
     "key": "umucu_2019",
     "authors": "Umucu E, Wu JR, Sanchez J et al.",
     "year": "2019",
     "title": "Psychometric validation of the PERMA-profiler as a well-being measure for student veterans.",
     "journal": "Journal of American college health : J of ACH",
     "doi": "10.1080/07448481.2018.1546182",
     "url": "https://doi.org/10.1080/07448481.2018.1546182"
    },
    {
     "key": "ryan_2019",
     "authors": "Ryan J, Curtis R, Olds T et al.",
     "year": "2019",
     "title": "Psychometric properties of the PERMA Profiler for measuring wellbeing in Australian adults.",
     "journal": "PloS one",
     "doi": "10.1371/journal.pone.0225932",
     "url": "https://doi.org/10.1371/journal.pone.0225932"
    },
    {
     "key": "nie_2024",
     "authors": "Nie YZ, Zhang X, Hong NW et al.",
     "year": "2024",
     "title": "Psychometric validation of the PERMA-profiler for well-being in Chinese adults.",
     "journal": "Acta psychologica",
     "doi": "10.1016/j.actpsy.2024.104248",
     "url": "https://doi.org/10.1016/j.actpsy.2024.104248"
    },
    {
     "key": "julie_2026",
     "authors": "Jourdan J, Serrand C, Chevallier T, et al.",
     "year": "2026",
     "title": "PERMA-Profiler as a multidimensional measure of well-being in a French context.",
     "journal": "Scientific reports",
     "doi": "10.1038/s41598-026-57354-z",
     "url": "https://doi.org/10.1038/s41598-026-57354-z"
    },
    {
     "key": "qian_2025",
     "authors": "Qian Y, Yan H, Zeng X et al.",
     "year": "2025",
     "title": "The Chinese version of the PERMA profiler: a validity and reliability study.",
     "journal": "BMC psychology",
     "doi": "10.1186/s40359-025-02560-z",
     "url": "https://doi.org/10.1186/s40359-025-02560-z"
    },
    {
     "key": "magare_2022",
     "authors": "Magare I, Graham MA, Eloff I",
     "year": "2022",
     "title": "An Assessment of the Reliability and Validity of the PERMA Well-Being Scale for Adult Undergraduate Students in an Open and Distance Learning Context.",
     "journal": "International journal of environmental research and public health",
     "doi": "10.3390/ijerph192416886",
     "url": "https://doi.org/10.3390/ijerph192416886"
    },
    {
     "key": "grosvenor_2023",
     "authors": "Grosvenor LP, Errichetti CL, Holingue C et al.",
     "year": "2023",
     "title": "Self-Report Measurement of Well-Being in Autistic Adults: Psychometric Properties of the PERMA Profiler.",
     "journal": "Autism in adulthood",
     "doi": "10.1089/aut.2022.0049",
     "url": "https://doi.org/10.1089/aut.2022.0049"
    },
    {
     "key": "umucu_2024",
     "authors": "Umucu E, Granger TA, Pan D et al.",
     "year": "2024",
     "title": "Initial validation of a short version of the PERMA profiler in a national sample of rural veterans.",
     "journal": "Frontiers in public health",
     "doi": "10.3389/fpubh.2024.1500659",
     "url": "https://doi.org/10.3389/fpubh.2024.1500659"
    },
    {
     "key": "fernandes_2024",
     "authors": "Fernandes I, Zanini DS, Peixoto EM",
     "year": "2024",
     "title": "PERMA-Profiler for adolescents: validity evidence based on internal structure and related constructs.",
     "journal": "Frontiers in psychology",
     "doi": "10.3389/fpsyg.2024.1415084",
     "url": "https://doi.org/10.3389/fpsyg.2024.1415084"
    },
    {
     "key": "waigel_2023",
     "authors": "Waigel NC, Lemos VN",
     "year": "2023",
     "title": "Psychometric Properties of PERMA Proﬁler Scale in Argentinian Adolescents.",
     "journal": "International journal of psychological research",
     "doi": "10.21500/20112084.5737",
     "url": "https://doi.org/10.21500/20112084.5737"
    },
    {
     "key": "ng_2025",
     "authors": "Ng XH, Chu J, Doshi K",
     "year": "2025",
     "title": "A PERMA-nent solution to understanding psychological wellbeing? Exploring the utility of the PERMA model in a university workplace.",
     "journal": "Frontiers in psychology",
     "doi": "10.3389/fpsyg.2025.1598910",
     "url": "https://doi.org/10.3389/fpsyg.2025.1598910"
    },
    {
     "key": "lorenz_2023",
     "authors": "Lorenz T, Ho J, Beyer M et al.",
     "year": "2023",
     "title": "Measuring PERMA+4: validation of the German version of the Positive Functioning at Work Scale.",
     "journal": "Frontiers in psychology",
     "doi": "10.3389/fpsyg.2023.1231299",
     "url": "https://doi.org/10.3389/fpsyg.2023.1231299"
    },
    {
     "key": "garciaselva_2024",
     "authors": "García-Selva A, Neipp MC, Solanes-Puchol Á et al.",
     "year": "2024",
     "title": "The PERMA+4 Positive Functioning at Work Scale: Spanish Adaptation and Validation.",
     "journal": "Psicothema",
     "doi": "10.7334/psicothema2023.341",
     "url": "https://doi.org/10.7334/psicothema2023.341"
    }
   ],
   "record_notes": "Overall confidence: the general PERMA-Profiler is moderately well studied, but the Workplace PERMA-Profiler specifically rests on a thin, geographically narrow occupational evidence base (chiefly Japanese, Korean, Chinese and Polish workers) with generally marginal confirmatory fit. Internal consistency is the one property that grades High; structural validity, criterion validity, test-retest and invariance all grade Low, and responsiveness/MIC is Absent. No UK validation, UK occupational norms or UK benchmark data were located. Schema stress-test notes: (1) The schema's single identity block does not cleanly separate the general PERMA-Profiler from the Workplace PERMA-Profiler, which have materially different evidence depth; the record repeatedly distinguishes them in prose but a structured field pairing each property to 'general form' versus 'workplace form' would record the asymmetry more honestly. (2) Test-retest was hard to record faithfully because several studies report ICCs that mix retest stability with internal consistency; the coefficients cited (for example 0.75 to 0.96, Watanabe 2018) should not be read as clean retest values. (3) The word 'clinical' is deliberately not applied to any workplace deployment here; where clinical screeners appear as validation comparators they are outcome measures, not the deployment context. (4) A dedicated COSMIN systematic review of the Workplace PERMA-Profiler was not located in this pass, so grades are synthesised from individual validation studies rather than an aggregated quality appraisal. (5) One candidate source (Persian older-adults validation, Payoun 2020) was excluded because no resolvable DOI was found; the Flourish Index paper surfaced by keyword overlap was excluded as it does not measure the PERMA-Profiler. The licence status is reported from the developers' distribution practice rather than a citable licence document, which the schema's licence_status field would ideally flag as evidence-grade metadata rather than a settled fact."
  },
  {
   "instrument_id": "uwes-9",
   "display_name": "Utrecht Work Engagement Scale, 9-item (UWES-9)",
   "identity": {
    "name": "Utrecht Work Engagement Scale, 9-item (UWES-9)",
    "current_version": "UWES-9 (9-item short form of the original 17-item UWES). An ultra-short 3-item form (UWES-3) also exists.",
    "item_count": "9 items, scored 0 to 6 (never to always/daily); three subscales of 3 items each (vigour, dedication, absorption), commonly summed to a single engagement score.",
    "original_citation": "Schaufeli WB, Bakker AB, Salanova M (2006). The Measurement of Work Engagement With a Short Questionnaire: A Cross-National Study. Educational and Psychological Measurement. https://doi.org/10.1177/0013164405282471 (short form derived from the 17-item UWES, Schaufeli et al. 2002, https://doi.org/10.1023/a:1015630930326)",
    "steward_publisher": "Wilmar Schaufeli and colleagues (Occupational Health Psychology Unit, Utrecht University); distributed via the author's website with an accompanying test manual.",
    "licence_status": "Free for non-commercial research and educational use in the widely-reported convention for this instrument; the scale and an accompanying preliminary test manual are distributed by the authors via the author's website ([UWES Test Manual, Schaufeli & Bakker 2003](https://www.wilmarschaufeli.nl/publications/Schaufeli/Test%20Manuals/Test_manual_UWES_English.pdf)). The manual itself was not opened and read in full in this pass, so the exact wording of any commercial-use or permission clause is not independently verified here; the instrument carries a copyright notice and no formal open licence (for example Creative Commons) is attached. Users should confirm current terms and any commercial-use conditions directly with the stewards before deployment."
   },
   "constructs_claimed": "The UWES-9 claims to measure work engagement, defined as a positive, fulfilling, work-related state of mind comprising three dimensions: vigour (energy and mental resilience while working), dedication (a sense of significance, enthusiasm and pride) and absorption (being fully concentrated and happily engrossed in work) ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471); [Schaufeli et al. 2002](https://doi.org/10.1023/a:1015630930326)). Engagement is theorised as conceptually distinct from, and in part the positive antipode of, burnout, and is embedded in the job demands-resources tradition ([Crawford et al. 2010](https://doi.org/10.1037/a0019364)).",
   "structural_validity": {
    "findings": "The dimensional structure of the UWES-9 is genuinely unresolved in the literature, and this is the instrument's central psychometric controversy. The developers reported that confirmatory factor analysis supported the intended three-factor structure (vigour, dedication, absorption) across ten countries (N = 14,521), while noting the three subscales are very highly intercorrelated ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). A dedicated review of 21 CFA studies of the UWES found no consensus: the three-factor structure was judged superior in 6 studies, a single general factor in a further 6, the one- and three-factor solutions were treated as equivalent in 8 studies, and 1 study confirmed neither, leading the author to warn that this ambiguity may challenge the three-factor conception of engagement itself ([Kulikowski 2017](https://doi.org/10.13075/ijomeh.1896.00947)). Primary studies since then remain split. A three-factor (or second-order) solution fit best in Norwegian occupational groups ([Nerstad et al. 2009](https://doi.org/10.1111/j.1467-9450.2009.00770.x)), Portuguese rescue workers ([Sinval et al. 2018](https://doi.org/10.3389/fpsyg.2017.02229)), a Vietnamese nurse sample (three-factor marginally better than one-factor, [Tran et al. 2020](https://doi.org/10.1002/1348-9585.12157)) and Spanish health workers (three correlated factors with correlated errors, [Dominguez-Salas et al. 2022](https://doi.org/10.2147/PRBM.S387242)). A one-factor solution was preferred in a Serbian sample ([Petrovic et al. 2017](https://doi.org/10.3389/fpsyg.2017.01799)) and in a German oncology-rehabilitation sample where a single factor explained 67% of variance ([Sautier et al. 2015](https://doi.org/10.1055/s-0035-1555912)), and unidimensionality was supported using ordinal methods in Scandinavian haemodialysis nurses ([Lindberg et al. 2025](https://doi.org/10.1186/s12912-025-03545-4)). Most starkly, a Swedish multi-occupational female sample (N = 702) obtained poor fit for one-, two- and three-factor models alike (RMSEA never below 0.166), i.e. no acceptable structure at all ([Willmer et al. 2019](https://doi.org/10.3389/fpsyg.2019.02771)). The recurring pattern is that inter-factor correlations are so high that the three subscales are difficult to separate empirically, which is why many authors recommend using the total score as a single engagement index ([Mills et al. 2011](https://doi.org/10.1007/s10902-011-9277-3)).",
    "confidence": "Moderate. Many studies with large total N, but findings are directly contradictory on the key question (1-factor vs 3-factor), so the evidence supports 'well studied but unresolved' rather than a settled structure."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent validity is reasonably supported: UWES scores correlate positively with job satisfaction, organisational commitment, meaning of work and perceived job resources, and negatively with burnout and turnover intention ([Sautier et al. 2015](https://doi.org/10.1055/s-0035-1555912); [Hallberg & Schaufeli 2006](https://doi.org/10.1027/1016-9040.11.2.119)). Discriminant validity is more contested. On the positive side, one CFA study concluded that work engagement, job involvement and organisational commitment are empirically distinct constructs reflecting different aspects of work attachment ([Hallberg & Schaufeli 2006](https://doi.org/10.1027/1016-9040.11.2.119)), and a meta-analytic review reported that engagement shows discriminant validity from, and incremental criterion validity over, established job attitudes ([Christian et al. 2011](https://doi.org/10.1111/j.1744-6570.2010.01203.x)). On the negative side, the sharpest challenge concerns overlap with burnout: a meta-analysis of 50 samples found that dimension-level correlations between burnout and engagement are high, that the two show a similar pattern of associations with correlates, and that controlling for burnout substantially reduced the effect sizes attributed to engagement, casting doubt on their functional distinctiveness ([Cole et al. 2012](https://doi.org/10.1177/0149206311415252)). The instrument's own developers frame engagement partly as the positive antipode of burnout, with a best-fitting two-factor burnout-engagement model in the original cross-national data ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)), which is consistent with, rather than resolving, the overlap concern. Discriminant validity against related well-being constructs (workaholism, job boredom) was more clearly demonstrated for the ultra-short UWES-3 across five national samples ([Schaufeli et al. 2019](https://doi.org/10.1027/1015-5759/a000430)).",
    "confidence": "Moderate. Convergent evidence is consistent and includes meta-analysis; discriminant evidence is genuinely mixed, with a credible meta-analytic challenge (burnout overlap) that is not fully answered."
   },
   "criterion_validity": {
    "findings": "Criterion validity against hard organisational and health outcomes is thinner than the volume of UWES research implies, and most evidence is cross-sectional or predictive of self-reported rather than registered outcomes. The strongest workplace-relevant evidence against a registered health outcome comes from a one-year prospective cohort of 4,921 employees: baseline UWES scores were negatively associated with register-recorded long-term sickness absence due to mental illness, but discrimination was only moderate (area under the ROC curve = 0.70) and below the pre-set threshold for practical screening use (0.75); crucially, UWES scores were NOT associated with sickness absence due to musculoskeletal or other somatic illness, so predictive validity was specific to mental-illness absence ([Roelen et al. 2014](https://doi.org/10.1007/s00420-014-0981-2)). For turnover, engagement measured with the UWES is repeatedly associated with lower turnover intention, but through cross-sectional or mediational designs rather than actual turnover: examples include Chinese nurses during COVID-19 ([Tang et al. 2022](https://doi.org/10.3389/fpubh.2022.1051895)) and surgical trainees' intention to leave training ([Dominguez et al. 2018](https://doi.org/10.1371/journal.pone.0197276)). At meta-analytic level, engagement relates positively to task and contextual performance and mediates demands/resources effects on performance, though these syntheses pool multiple engagement measures rather than the UWES-9 alone ([Christian et al. 2011](https://doi.org/10.1111/j.1744-6570.2010.01203.x)). A UK quality-improvement evaluation found a modest but statistically significant difference in UWES scores between intervention and control wards ([White et al. 2014](https://doi.org/10.1016/j.ijnurstu.2014.05.002)). Evidence linking UWES scores prospectively to diagnosed conditions, objective productivity or actual (not intended) turnover is sparse.",
    "confidence": "Low. One good prospective registry study exists (with an informative null for somatic absence); most other criterion evidence is cross-sectional, uses intentions rather than behaviour, or pools multiple engagement measures."
   },
   "internal_consistency": {
    "findings": "Internal consistency is the UWES-9's most consistently strong property. The developers reported the three subscale scores had good internal consistency across the cross-national samples ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). Total-score Cronbach's alpha is typically in the low-to-mid 0.90s: 0.93 (ordinal alpha) in Scandinavian haemodialysis nurses ([Lindberg et al. 2025](https://doi.org/10.1186/s12912-025-03545-4)), 0.93 for the total scale in Vietnamese nurses (subscales 0.86 vigour, 0.77 absorption, 0.90 dedication) ([Tran et al. 2020](https://doi.org/10.1002/1348-9585.12157)), and 0.94 in the German oncology-rehabilitation sample ([Sautier et al. 2015](https://doi.org/10.1055/s-0035-1555912)). Reliability was likewise satisfactory in Norwegian ([Nerstad et al. 2009](https://doi.org/10.1111/j.1467-9450.2009.00770.x)) and Serbian ([Petrovic et al. 2017](https://doi.org/10.3389/fpsyg.2017.01799)) validations. A caveat: alpha values in the 0.90s for a 9-item scale with very high inter-item correlation partly reflect redundancy, and high total-scale alpha does not adjudicate the one- versus three-factor question. Where the three-item subscales are used separately, the absorption subscale tends to be the weakest ([Tran et al. 2020](https://doi.org/10.1002/1348-9585.12157)).",
    "confidence": "High. Multiple good-quality studies across many languages and occupations, large total N, consistently reporting total-score alpha in the low-to-mid 0.90s."
   },
   "test_retest_reliability": {
    "findings": "Genuine test-retest reliability evidence for the UWES-9 is comparatively scarce, and this is the weakest-documented property relative to the instrument's popularity. The developers' original short-form paper states summarily that the three UWES-9 scores have 'good' test-retest reliability across the cross-national dataset, but reports no coefficients or retest intervals in the material available for this record ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). Among UWES-9 validations, one of the few to report a short-interval coefficient found an intraclass correlation of only 0.48 over roughly three months in a Vietnamese nurse subsample, which is modest and below conventional stability thresholds ([Tran et al. 2020](https://doi.org/10.1002/1348-9585.12157)). A multisample, longitudinal construct-validity study of the UWES exists and, by its title, addresses longitudinal/temporal evidence ([Seppala et al. 2009](https://doi.org/10.1007/s10902-008-9100-y)); however, its full text and abstract could not be retrieved in this pass, so no specific stability coefficient from it is asserted here, and any stability figures it reports pertain principally to the 17-item UWES rather than the 9-item form. Many national validation studies report internal consistency and factor structure but omit test-retest data entirely. Engagement is theorised as a relatively stable state, so a stronger, replicated test-retest evidence base would be expected for the 9-item form than currently exists.",
    "confidence": "Low. Test-retest evidence specific to the UWES-9 is sparse: the developer claim is summary and coefficient-free in the material accessed, the clearest located UWES-9 retest coefficient (ICC 0.48 over ~3 months) is modest, and longer-interval stability evidence is tied to the 17-item form and was not retrievable this session. The thinness of a replicated UWES-9 test-retest base is itself the finding."
   },
   "measurement_invariance": {
    "findings": "Measurement invariance has been tested repeatedly, most often across country/language and occupational group, with configural and metric invariance commonly achieved and scalar invariance less consistently so. Full-scale measurement invariance for the UWES-9 was obtained between Portuguese and Brazilian workers for both the three-factor first-order and second-order models, permitting direct mean comparisons ([Sinval et al. 2018](https://doi.org/10.3389/fpsyg.2018.00353)). In Portuguese rescue workers, the UWES-9 first-order model reached full (uniqueness) invariance across occupational groups, while the second-order model reached only partial (metric) second-order invariance ([Sinval et al. 2018](https://doi.org/10.3389/fpsyg.2017.02229)). Gender invariance has been supported in Spanish health-care professionals ([Dominguez-Salas et al. 2022](https://doi.org/10.2147/PRBM.S387242)), and factorial invariance across ten occupational groups was reported for the Norwegian UWES ([Nerstad et al. 2009](https://doi.org/10.1111/j.1467-9450.2009.00770.x)). The developers' cross-national work established the UWES-9 across ten countries but treated cross-national equivalence descriptively rather than through the full modern invariance hierarchy ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). Against this, a Rasch analysis of the 17-item UWES in Korea found that only 9 of 17 items differentiated adequately between men and women, i.e. differential item functioning by sex ([Song et al. 2020](https://doi.org/10.1177/0033294120922494)), a caution that invariance is not universal. Over-time (longitudinal) invariance is less frequently formally tested; the multisample longitudinal study is the main source addressing temporal stability of the structure ([Seppala et al. 2009](https://doi.org/10.1007/s10902-008-9100-y)).",
    "confidence": "Moderate. Several studies reach metric and sometimes scalar/full invariance across language and occupation, but scalar invariance is not universal, some DIF by sex is reported, and formal over-time invariance testing is limited."
   },
   "responsiveness_mic": {
    "findings": "Formal responsiveness (sensitivity to change) and a minimal important change (MIC) value have not been established for the UWES-9. No located study derived an anchor-based or distribution-based MIC, and the instrument was not developed as an outcome measure with defined change metrics. Indirect evidence of sensitivity to change comes from a UK evaluation of the 'Productive Ward' quality-improvement programme, where UWES scores were modestly but significantly higher in intervention wards than matched controls (4.33 vs 4.07, p = 0.013), which the authors interpreted as the UWES being able to detect programme-related differences ([White et al. 2014](https://doi.org/10.1016/j.ijnurstu.2014.05.002)). This is a between-group cross-sectional contrast rather than a within-person responsiveness or MIC analysis. No UK or other MIC benchmark was located in this pass.",
    "confidence": "Very low. No MIC established and no formal responsiveness study located; only indirect, between-group evidence that scores can differ with an intervention."
   },
   "populations_languages_norms": {
    "findings": "The UWES-9 has been validated in a very wide range of languages and occupations, giving broad but heterogeneous population coverage. The developers' short-form study drew on samples from ten countries (N = 14,521) ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). Located validations span Norwegian occupational groups ([Nerstad et al. 2009](https://doi.org/10.1111/j.1467-9450.2009.00770.x)), Serbian employees ([Petrovic et al. 2017](https://doi.org/10.3389/fpsyg.2017.01799)), Brazilian and Portuguese workers ([Sinval et al. 2018](https://doi.org/10.3389/fpsyg.2018.00353)), Portuguese rescue workers ([Sinval et al. 2018](https://doi.org/10.3389/fpsyg.2017.02229)), Vietnamese hospital nurses ([Tran et al. 2020](https://doi.org/10.1002/1348-9585.12157)), Spanish health-care professionals ([Dominguez-Salas et al. 2022](https://doi.org/10.2147/PRBM.S387242)), a German oncology-rehabilitation patient sample ([Sautier et al. 2015](https://doi.org/10.1055/s-0035-1555912)), a Swedish multi-occupational female sample ([Willmer et al. 2019](https://doi.org/10.3389/fpsyg.2019.02771)) and Scandinavian (Danish and Swedish) haemodialysis nurses ([Lindberg et al. 2025](https://doi.org/10.1186/s12912-025-03545-4)); an ultra-short UWES-3 was validated across Finland, Japan, the Netherlands, Belgium/Flanders and Spain ([Schaufeli et al. 2019](https://doi.org/10.1027/1015-5759/a000430)) and in Peru ([Merino-Soto et al. 2022](https://doi.org/10.3390/ijerph19020890)). UK-specific evidence is limited: the clearest UK-context deployment located is the English-language 'Productive Ward' nursing study (conducted in Ireland with a UK/Ireland health-service context), which used the UWES and reported mean scores around 4.1 to 4.3 on the 0 to 6 metric, but this is a study sample, not a representative UK norm ([White et al. 2014](https://doi.org/10.1016/j.ijnurstu.2014.05.002)). No representative UK normative dataset was located in this pass; the reference norm tables that exist are those in the authors' international test manual ([UWES Test Manual, Schaufeli & Bakker 2003](https://www.wilmarschaufeli.nl/publications/Schaufeli/Test%20Manuals/Test_manual_UWES_English.pdf)), which are international rather than UK-specific.",
    "confidence": "Moderate for general population/language coverage (extensive and multi-national); Low for UK-specific norms, which were not located as a representative benchmark in this pass."
   },
   "criticisms_controversies": "Three recurring criticisms appear in the literature. First, factor-structure instability: a review of 21 CFA studies found no consensus on whether the UWES is one-dimensional or three-dimensional, with roughly equal support for each and one study confirming neither, which the author argued could undermine the three-dimensional conception of engagement itself ([Kulikowski 2017](https://doi.org/10.13075/ijomeh.1896.00947)); an extreme case obtained no acceptable fit for any of the one-, two- or three-factor models ([Willmer et al. 2019](https://doi.org/10.3389/fpsyg.2019.02771)). Second, discriminant validity from burnout: a meta-analysis of 50 samples reported high dimension-level correlations, near-identical correlate patterns, and shrinking engagement effect sizes once burnout is controlled, questioning whether engagement and burnout are functionally distinct ([Cole et al. 2012](https://doi.org/10.1177/0149206311415252)). The developers' own framing of engagement as the positive antipode of burnout, with a best-fitting combined two-factor model, sits uneasily with claims of full independence ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). Third, redundancy and subscale separability: because the three subscales correlate so highly, several authors conclude the total score should be used and that the subscales add little, so reporting 'vigour/dedication/absorption' as three distinct measures may over-claim ([Mills et al. 2011](https://doi.org/10.1007/s10902-011-9277-3)). A related psychometric caution is that Cronbach's alpha in the 0.90s for such highly intercorrelated items partly reflects item redundancy rather than only reliability. Discriminant concerns against neighbouring attitudes (job involvement, organisational commitment) are, by contrast, comparatively reassuring ([Hallberg & Schaufeli 2006](https://doi.org/10.1027/1016-9040.11.2.119)).",
   "citations": [
    {
     "key": "Schaufeli2006",
     "authors": "Schaufeli WB, Bakker AB, Salanova M",
     "year": "2006",
     "title": "The Measurement of Work Engagement With a Short Questionnaire: A Cross-National Study",
     "journal": "Educational and Psychological Measurement",
     "doi": "10.1177/0013164405282471",
     "url": "https://doi.org/10.1177/0013164405282471"
    },
    {
     "key": "Schaufeli2002",
     "authors": "Schaufeli WB, Salanova M, González-Romá V, Bakker AB",
     "year": "2002",
     "title": "The Measurement of Engagement and Burnout: A Two Sample Confirmatory Factor Analytic Approach",
     "journal": "Journal of Happiness Studies",
     "doi": "10.1023/a:1015630930326",
     "url": "https://doi.org/10.1023/a:1015630930326"
    },
    {
     "key": "Seppala2009",
     "authors": "Seppälä P, Mauno S, Feldt T, Hakanen J, Kinnunen U, Tolvanen A, Schaufeli W",
     "year": "2009",
     "title": "The Construct Validity of the Utrecht Work Engagement Scale: Multisample and Longitudinal Evidence",
     "journal": "Journal of Happiness Studies",
     "doi": "10.1007/s10902-008-9100-y",
     "url": "https://doi.org/10.1007/s10902-008-9100-y"
    },
    {
     "key": "Schaufeli2019",
     "authors": "Schaufeli WB, Shimazu A, Hakanen J, Salanova M, De Witte H",
     "year": "2019",
     "title": "An Ultra-Short Measure for Work Engagement: The UWES-3",
     "journal": "European Journal of Psychological Assessment",
     "doi": "10.1027/1015-5759/a000430",
     "url": "https://doi.org/10.1027/1015-5759/a000430"
    },
    {
     "key": "Kulikowski2017",
     "authors": "Kulikowski K",
     "year": "2017",
     "title": "Do we all agree on how to measure work engagement? Factorial validity of Utrecht Work Engagement Scale as a standard measurement tool: A literature review",
     "journal": "International Journal of Occupational Medicine and Environmental Health",
     "doi": "10.13075/ijomeh.1896.00947",
     "url": "https://doi.org/10.13075/ijomeh.1896.00947"
    },
    {
     "key": "Nerstad2009",
     "authors": "Nerstad CGL, Richardsen AM, Martinussen M",
     "year": "2009",
     "title": "Factorial validity of the Utrecht Work Engagement Scale (UWES) across occupational groups in Norway",
     "journal": "Scandinavian Journal of Psychology",
     "doi": "10.1111/j.1467-9450.2009.00770.x",
     "url": "https://doi.org/10.1111/j.1467-9450.2009.00770.x"
    },
    {
     "key": "Willmer2019",
     "authors": "Willmer M, Westerberg Jacobson J, Lindberg M",
     "year": "2019",
     "title": "Exploratory and Confirmatory Factor Analysis of the 9-Item Utrecht Work Engagement Scale in a Multi-Occupational Female Sample",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2019.02771",
     "url": "https://doi.org/10.3389/fpsyg.2019.02771"
    },
    {
     "key": "Lindberg2025",
     "authors": "Lindberg M, Knudsen K, Lindberg M",
     "year": "2025",
     "title": "Factor structure of the Utrecht Work Engagement Scale in a sample of Danish and Swedish haemodialysis nurses",
     "journal": "BMC Nursing",
     "doi": "10.1186/s12912-025-03545-4",
     "url": "https://doi.org/10.1186/s12912-025-03545-4"
    },
    {
     "key": "SinvalRescue2018",
     "authors": "Sinval J, Marques-Pinto A, Queirós C, Marôco J",
     "year": "2018",
     "title": "Work Engagement among Rescue Workers: Psychometric Properties of the Portuguese UWES",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2017.02229",
     "url": "https://doi.org/10.3389/fpsyg.2017.02229"
    },
    {
     "key": "Petrovic2017",
     "authors": "Petrović IB, Vukelić M, Čizmić S",
     "year": "2017",
     "title": "Work Engagement in Serbia: Psychometric Properties of the Serbian Version of the Utrecht Work Engagement Scale (UWES)",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2017.01799",
     "url": "https://doi.org/10.3389/fpsyg.2017.01799"
    },
    {
     "key": "SinvalBP2018",
     "authors": "Sinval J, Pasian S, Queirós C, Marôco J",
     "year": "2018",
     "title": "Brazil-Portugal Transcultural Adaptation of the UWES-9: Internal Consistency, Dimensionality, and Measurement Invariance",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2018.00353",
     "url": "https://doi.org/10.3389/fpsyg.2018.00353"
    },
    {
     "key": "Mills2011",
     "authors": "Mills MJ, Culbertson SS, Fullagar CJ",
     "year": "2011",
     "title": "Conceptualizing and Measuring Engagement: An Analysis of the Utrecht Work Engagement Scale",
     "journal": "Journal of Happiness Studies",
     "doi": "10.1007/s10902-011-9277-3",
     "url": "https://doi.org/10.1007/s10902-011-9277-3"
    },
    {
     "key": "Sautier2015",
     "authors": "Sautier LP, Scherwath A, Weis J, Sarkar S, Bosbach M, Schendel M, Ladehoff N, Koch U, Mehnert A",
     "year": "2015",
     "title": "Assessment of Work Engagement in Patients with Hematological Malignancies: Psychometric Properties of the German Version of the UWES-9",
     "journal": "Die Rehabilitation",
     "doi": "10.1055/s-0035-1555912",
     "url": "https://doi.org/10.1055/s-0035-1555912"
    },
    {
     "key": "Song2020",
     "authors": "Song HD, Hong AJ, Jo Y",
     "year": "2020",
     "title": "Psychometric Investigation of the Utrecht Work Engagement Scale-17 Using the Rasch Measurement Model",
     "journal": "Psychological Reports",
     "doi": "10.1177/0033294120922494",
     "url": "https://doi.org/10.1177/0033294120922494"
    },
    {
     "key": "Hallberg2006",
     "authors": "Hallberg UE, Schaufeli WB",
     "year": "2006",
     "title": "\"Same Same\" But Different? Can Work Engagement Be Discriminated from Job Involvement and Organizational Commitment?",
     "journal": "European Psychologist",
     "doi": "10.1027/1016-9040.11.2.119",
     "url": "https://doi.org/10.1027/1016-9040.11.2.119"
    },
    {
     "key": "Cole2012",
     "authors": "Cole MS, Walter F, Bedeian AG, O'Boyle EH",
     "year": "2012",
     "title": "Job Burnout and Employee Engagement: A Meta-Analytic Examination of Construct Proliferation",
     "journal": "Journal of Management",
     "doi": "10.1177/0149206311415252",
     "url": "https://doi.org/10.1177/0149206311415252"
    },
    {
     "key": "Christian2011",
     "authors": "Christian MS, Garza AS, Slaughter JE",
     "year": "2011",
     "title": "Work Engagement: A Quantitative Review and Test of Its Relations with Task and Contextual Performance",
     "journal": "Personnel Psychology",
     "doi": "10.1111/j.1744-6570.2010.01203.x",
     "url": "https://doi.org/10.1111/j.1744-6570.2010.01203.x"
    },
    {
     "key": "Crawford2010",
     "authors": "Crawford ER, LePine JA, Rich BL",
     "year": "2010",
     "title": "Linking job demands and resources to employee engagement and burnout: A theoretical extension and meta-analytic test",
     "journal": "Journal of Applied Psychology",
     "doi": "10.1037/a0019364",
     "url": "https://doi.org/10.1037/a0019364"
    },
    {
     "key": "Roelen2014",
     "authors": "Roelen CAM, van Hoffen MFA, Groothoff JW, de Bruin J, Schaufeli WB, van Rhenen W",
     "year": "2014",
     "title": "Can the Maslach Burnout Inventory and Utrecht Work Engagement Scale be used to screen for risk of long-term sickness absence?",
     "journal": "International Archives of Occupational and Environmental Health",
     "doi": "10.1007/s00420-014-0981-2",
     "url": "https://doi.org/10.1007/s00420-014-0981-2"
    },
    {
     "key": "White2014",
     "authors": "White M, Wells JS, Butterworth T",
     "year": "2014",
     "title": "The impact of a large-scale quality improvement programme on work engagement: preliminary results from a national cross-sectional survey of the 'Productive Ward'",
     "journal": "International Journal of Nursing Studies",
     "doi": "10.1016/j.ijnurstu.2014.05.002",
     "url": "https://doi.org/10.1016/j.ijnurstu.2014.05.002"
    },
    {
     "key": "Tang2022",
     "authors": "Tang Y, Dias Martins LM, Wang SB, He QX, Huang HH",
     "year": "2022",
     "title": "The impact of nurses' sense of security on turnover intention during the normalization of COVID-19 epidemic: The mediating role of work engagement",
     "journal": "Frontiers in Public Health",
     "doi": "10.3389/fpubh.2022.1051895",
     "url": "https://doi.org/10.3389/fpubh.2022.1051895"
    },
    {
     "key": "Dominguez2018",
     "authors": "Dominguez LC, Stassen L, de Grave W, Sanabria A, Alfonso E, Dolmans D",
     "year": "2018",
     "title": "Taking control: Is job crafting related to the intention to leave surgical training?",
     "journal": "PLoS ONE",
     "doi": "10.1371/journal.pone.0197276",
     "url": "https://doi.org/10.1371/journal.pone.0197276"
    },
    {
     "key": "Tran2020",
     "authors": "Tran TTT, Watanabe K, Imamura K, Nguyen HT, Sasaki N, Kuribayashi K, Sakuraya A, et al.",
     "year": "2020",
     "title": "Reliability and validity of the Vietnamese version of the 9-item Utrecht Work Engagement Scale",
     "journal": "Journal of Occupational Health",
     "doi": "10.1002/1348-9585.12157",
     "url": "https://doi.org/10.1002/1348-9585.12157"
    },
    {
     "key": "DominguezSalas2022",
     "authors": "Domínguez-Salas S, Rodríguez-Domínguez C, Arcos-Romero AI, Allande-Cussó R, et al.",
     "year": "2022",
     "title": "Psychometric Properties of the Utrecht Work Engagement Scale (UWES-9) in a Sample of Active Health Care Professionals in Spain",
     "journal": "Psychology Research and Behavior Management",
     "doi": "10.2147/PRBM.S387242",
     "url": "https://doi.org/10.2147/PRBM.S387242"
    },
    {
     "key": "MerinoSoto2022",
     "authors": "Merino-Soto C, Lozano-Huamán M, Lima-Mendoza S, Calderón de la Cruz G, Juárez-García A, Toledano-Toledano F",
     "year": "2022",
     "title": "Ultrashort Version of the Utrecht Work Engagement Scale (UWES-3): A Psychometric Assessment",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph19020890",
     "url": "https://doi.org/10.3390/ijerph19020890"
    },
    {
     "key": "SchaufeliBAT2020",
     "authors": "Schaufeli WB, Desart S, De Witte H",
     "year": "2020",
     "title": "Burnout Assessment Tool (BAT): Development, Validity, and Reliability",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph17249495",
     "url": "https://doi.org/10.3390/ijerph17249495"
    },
    {
     "key": "UWESManual",
     "authors": "Schaufeli WB, Bakker AB",
     "year": "2003",
     "title": "UWES Utrecht Work Engagement Scale: Preliminary Manual (Version 1)",
     "journal": "Occupational Health Psychology Unit, Utrecht University (test manual, grey literature)",
     "doi": "",
     "url": "https://www.wilmarschaufeli.nl/publications/Schaufeli/Test%20Manuals/Test_manual_UWES_English.pdf"
    }
   ],
   "record_notes": "Overall confidence: internal consistency is High; structural validity, convergent/discriminant validity and measurement invariance are Moderate but each carries a genuine, cited contradiction rather than clean support; criterion validity is Low; test-retest is Low with the absence itself a finding; responsiveness/MIC is Very low (effectively absent). Points the schema made hard to record honestly: (1) The one-factor versus three-factor debate is not a defect to be resolved to a single 'confidence' but a standing feature of the evidence; the 'structural_validity' field forces a single grade onto directly contradictory findings, so 'Moderate' here means 'extensively studied but unresolved', not 'moderately good'. (2) 'constructs_claimed' presents three subscales, yet a substantial strand of evidence argues the subscales are not empirically separable and only the total should be used; the field cannot easily flag that the claimed structure is itself contested. (3) Several strong stability and invariance findings pertain to the 17-item UWES, not the 9-item form; the schema does not distinguish evidence transferred from the parent instrument from evidence earned by the UWES-9 directly, and I have flagged this inline where it applies. (4) Test-retest and MIC are near-absent for the 9-item form specifically despite the instrument's popularity, which is a more important finding than a coefficient would have been. (5) UK-specific norms were not located as a representative benchmark; the 'Productive Ward' study is UK/Ireland health-service context but is a study sample, not a norm, and the only reference norms are the authors' international (non-UK) test manual. (6) Per the registry rule on 'clinical': the UWES-9 has been deployed in patient-rehabilitation samples (for example haematological malignancy patients), but it is a work-engagement instrument, not a clinical screener, and none of its validation was earned in a diagnostic setting; workplace deployment should be read as a non-clinical context throughout."
  },
  {
   "instrument_id": "phq-9",
   "display_name": "Patient Health Questionnaire-9 (PHQ-9)",
   "identity": {
    "name": "Patient Health Questionnaire-9 (PHQ-9)",
    "current_version": "PHQ-9 (2001); abbreviated variants PHQ-8 (omits item 9) and PHQ-2 (first two items) are in wide use",
    "item_count": "9 items, each scored 0 to 3 (total 0 to 27); severity bands 5, 10, 15, 20 map to mild, moderate, moderately severe and severe",
    "original_citation": "Kroenke K, Spitzer RL, Williams JB (2001) The PHQ-9: validity of a brief depression severity measure. Journal of General Internal Medicine. 10.1046/j.1525-1497.2001.016009606.x",
    "steward_publisher": "Developed by Spitzer, Williams and Kroenke as the depression module of the PRIME-MD PHQ, under an educational grant from Pfizer Inc; the PHQ suite was subsequently released by Pfizer for free public access.",
    "licence_status": "Free / public domain. Pfizer, which held the original copyright, released the PHQ and GAD-7 without copyright restriction and at no charge; the official form states no permission is required to reproduce, translate, display or distribute it. Attribution to Kroenke et al (2001) is expected ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)). No DOI-bearing primary source for the licence itself was retrieved this session; the release is documented via the developers' and steward's public statements."
   },
   "constructs_claimed": "Severity of depressive symptoms over the preceding two weeks and, secondarily, provisional detection of major depressive disorder. The nine items map directly onto the nine DSM-IV/DSM-5 symptom criteria for a major depressive episode, so the instrument claims both a continuous severity construct and a criterion-referenced screening function ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)). It is a depression-specific measure, not a general wellbeing or distress index, and was designed for primary-care and clinical use rather than for occupational or workplace-wellbeing measurement.",
   "structural_validity": {
    "findings": "The dimensionality of the PHQ-9 is genuinely contested, but the practical reading across studies is that it behaves as an essentially unidimensional depression severity scale even where a two-factor model fits marginally better. The original development treated the total score as a single severity dimension ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)). Within the individual participant data (IPD) programme, a comparison of unidimensional, two-dimensional (cognitive/affective versus somatic) and bifactor latent models found that all fitted reasonably and that scoring by the more complex latent models improved sensitivity only marginally (by about 0.04 to 0.05) while reducing specificity, relative to the sum score at a cut-off of 10 or above ([Fischer et al 2021](https://doi.org/10.1017/S0033291721000131)). Several population studies reach the same conclusion from different angles: in Korean nationally representative data a single-factor model fitted well (CFI 0.944) with satisfactory internal consistency ([Lee et al 2023](https://doi.org/10.3389/fpsyg.2023.1217038)); in Brazilian university students both one- and two-dimensional models fitted, with the authors favouring the unidimensional solution ([Rufino et al 2024](https://doi.org/10.1016/j.jad.2024.01.051)); and after traumatic brain injury a general factor explained about 85% of the variance despite a slight bifactor improvement ([Teymoori et al 2020](https://doi.org/10.3390/jcm9030873)). Where a two-factor (somatic versus cognitive-affective) structure is reported, the factors tend to be very highly correlated: in a stroke sample the two-factor model fitted slightly better than the one-factor model (CFI 0.984 versus 0.974) but the inter-factor correlation of 0.866 pointed back to unidimensionality ([Blake et al 2024](https://doi.org/10.1016/j.jpsychores.2024.111983)), and an Italian coronary heart disease sample reported a bi-dimensional somatic/cognitive solution ([Di et al 2025](https://doi.org/10.1097/JCN.0000000000001178)). The disagreement is therefore about whether the somatic items form a distinguishable subfactor, not about whether a single severity score is defensible; the weight of evidence supports scoring and interpreting a single total.",
    "confidence": "High: many good-quality factor-analytic studies across large and varied samples, converging on essentially unidimensional use despite a recurring, well-characterised somatic/cognitive two-factor debate."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent evidence is consistent in direction but uneven in magnitude, and discriminant separation from anxiety is imperfect. In development, higher PHQ-9 scores tracked substantial decrements across all six SF-20 functional-status subscales and greater symptom-related difficulty, supporting construct validity ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)). In the Chinese general population the PHQ-9 correlated negatively with SF-36 subscales (r from -0.11 to -0.47) as expected, but its correlation with the Zung Self-Rating Depression Scale was unexpectedly weak (r = 0.29), an inconsistency worth noting against the usual assumption of strong convergence with other depression measures ([Wang et al 2014](https://doi.org/10.1016/j.genhosppsych.2014.05.021)). Convergent relationships with sleep quality, alcohol use and physical activity have also been reported in students ([Rufino et al 2024](https://doi.org/10.1016/j.jad.2024.01.051)). On discriminant validity, the PHQ-9 and the GAD-7 anxiety scale share a large common factor: after traumatic brain injury a general distress factor dominated both scales (about 85% of variance) even though the instruments related differently to SF-36 subscales ([Teymoori et al 2020](https://doi.org/10.3390/jcm9030873)), and in coronary heart disease PHQ-9 and GAD-7 scores were significantly positively correlated ([Di et al 2025](https://doi.org/10.1097/JCN.0000000000001178)). Depression and anxiety symptoms measured this way are statistically distinguishable but strongly overlapping.",
    "confidence": "Moderate: multiple studies establish expected convergent and functional-status correlations, but magnitudes vary (including one weak convergent correlation) and discriminant separation from anxiety is only partial."
   },
   "criterion_validity": {
    "findings": "Criterion validity against a diagnostic interview for major depression is the PHQ-9's best-evidenced property, and it is strong; criterion validity against organisational or workplace outcomes is, by contrast, not established in the retrieved literature. In the original clinical validation a cut-off of 10 or above gave 88% sensitivity and 88% specificity against an independent mental-health-professional interview ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)). The individual participant data meta-analysis programme is the anchor here. The 2019 IPD meta-analysis (58 studies, n = 17,357, 2,312 major depression cases) found combined sensitivity and specificity maximised at a cut-off of 10 or above against semistructured interviews (sensitivity 0.88, 95% CI 0.83 to 0.92; specificity 0.85, 0.82 to 0.88) ([Levis et al 2019](https://doi.org/10.1136/bmj.l1476)). The 2021 update (100 studies, n = 44,503) reproduced this at the same cut-off (sensitivity 0.85, 0.79 to 0.89; specificity 0.85, 0.82 to 0.87) ([Negeri et al 2021](https://doi.org/10.1136/bmj.n2183)). A crucial nuance is reference-standard dependence: sensitivity against semistructured clinician interviews ran markedly higher than against fully structured lay interviews (median difference around 21%) or the MINI (around 11%), while specificity was similar across standards ([Levis et al 2019](https://doi.org/10.1136/bmj.l1476), [Negeri et al 2021](https://doi.org/10.1136/bmj.n2183)). The diagnostic algorithm approach performed worse than the cut-off (algorithm sensitivity around 0.57 to 0.61 versus 0.88 for the cut-off against semistructured interviews) ([He et al 2019](https://doi.org/10.1159/000502294)), and the PHQ-2 with cut-off 3 or above (sensitivity 0.72, specificity 0.85) is a reasonable first-stage screen followed by the PHQ-9 ([Levis et al 2020](https://doi.org/10.1001/jama.2020.6504)). For the health-related associations closest to organisational outcomes, the development study showed that self-reported sick days and health-care utilisation rose with PHQ-9 severity ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)), but no study retrieved this session tested PHQ-9 scores against measured sickness absence, staff turnover, productivity loss or other workplace outcomes as criteria. Criterion validity against organisational endpoints should therefore be treated as unestablished.",
    "confidence": "High for diagnostic criterion validity against clinical interviews (large, consistent IPD evidence); Absent for criterion validity against organisational/workplace outcomes (no such studies retrieved). The overall grade is split by criterion."
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently high across populations. A reliability generalisation meta-analysis of 60 studies (232,147 participants) estimated a pooled Cronbach's alpha of 0.86 (95% CI 0.85 to 0.87), with self-administered formats slightly higher (alpha 0.87) than face-to-face interview administration (alpha 0.80), though between-study heterogeneity was very large (I-squared 99.3%) ([Ajele 2025](https://doi.org/10.1007/s44192-025-00181-x)). Single-study estimates sit in the same range: alpha 0.86 in the Chinese general population ([Wang et al 2014](https://doi.org/10.1016/j.genhosppsych.2014.05.021)) and omega 0.812 in a Korean nationally representative sample ([Lee et al 2023](https://doi.org/10.3389/fpsyg.2023.1217038)). A systematic review of the wider PHQ family likewise reported good internal consistency for the PHQ-9 ([Kroenke 2010](https://doi.org/10.1016/j.genhosppsych.2010.03.006)), and a meta-analysis of the Spanish-language versions evaluated internal consistency alongside accuracy ([Martinez et al 2023](https://doi.org/10.1001/jamanetworkopen.2023.36529)). Reported coefficients of omega tend to align with alpha where both are given, but omega is reported far less often than alpha.",
    "confidence": "High: a large reliability-generalisation meta-analysis plus multiple primary studies converge on alpha around 0.86, albeit with substantial heterogeneity and sparse omega reporting."
   },
   "test_retest_reliability": {
    "findings": "Test-retest reliability exists but is markedly under-studied relative to the vast diagnostic-accuracy literature, and this asymmetry is itself the finding. The reliability generalisation meta-analysis could pool a test-retest estimate from only 8 of its 60 studies, yielding 0.82 (95% CI 0.74 to 0.90) ([Ajele 2025](https://doi.org/10.1007/s44192-025-00181-x)). Primary estimates are favourable where they exist: a two-week retest correlation of 0.86 in the Chinese general population ([Wang et al 2014](https://doi.org/10.1016/j.genhosppsych.2014.05.021)), test-retest described as excellent over a seven-day interval in a late-life depression treatment sample ([Lowe et al 2004](https://doi.org/10.1097/00005650-200412000-00006)), and intraclass agreement assessed for self- versus telephone-administration in Spanish primary care ([Pinto-Meza et al 2005](https://doi.org/10.1111/j.1525-1497.2005.0144.x)). No test-retest study in an occupational or workplace sample, and none using a UK working population, was located in this pass. The property is present and reassuring in the settings studied, but the evidence base is thin and skewed toward clinical and general-population samples over short intervals.",
    "confidence": "Moderate: several favourable estimates (roughly 0.82 to 0.86) exist and a meta-analytic pooled value is available, but from few studies (n = 8 pooled), short intervals, and no workplace or UK occupational data. Not absent, but comparatively neglected."
   },
   "measurement_invariance": {
    "findings": "Invariance has been tested piecemeal and mostly reaches metric to scalar level within the groups examined, but coverage of occupational and UK working populations is absent. In a Korean nationally representative sample the one-factor PHQ-9 showed equivalent structure, factor loadings and item intercepts across age groups, i.e. up to scalar invariance across ages ([Lee et al 2023](https://doi.org/10.3389/fpsyg.2023.1217038)). Invariance across sex and across age (65 and over versus under 65) was tested in an Italian coronary heart disease cohort ([Di et al 2025](https://doi.org/10.1097/JCN.0000000000001178)). Item response theory analysis in a large Danish implantable-defibrillator cohort found no differential item functioning across educational level, age, clinical indication or heart-failure severity, with only a single item showing DIF by gender ([Pedersen et al 2016](https://doi.org/10.1016/j.jpsychores.2016.09.010)). This points to broadly stable measurement across sex and age in clinical and general populations. However, the studies are confined to clinical or general-population samples; invariance across occupations, across employed versus unemployed status, over repeated workplace administrations, and within UK working populations was not established in retrieved evidence. Longitudinal (over-time) invariance is likewise weakly evidenced for the PHQ-9 specifically.",
    "confidence": "Low to Moderate: scalar invariance is demonstrated across age (one strong study) and DIF is minimal in another, but evidence is scattered across clinical populations, occupation and UK-workplace invariance is untested, and over-time invariance is weak."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness to change is well supported, and a minimal important change has been proposed, though the MIC rests largely on one study. In the IMPACT late-life depression trial (n = 434 intervention participants) the PHQ-9 was responsive to treatment, with an effect size (about -1.3 at three months) exceeding the SCL-20 depression scale at three months and matching it at six months; change scores discriminated persistent depression, partial remission and full remission against structured diagnostic interviews ([Lowe et al 2004](https://doi.org/10.1097/00005650-200412000-00006)). That study estimated a minimal clinically important difference for individual change of about 5 points on the 0 to 27 scale, derived as two standard errors of measurement ([Lowe et al 2004](https://doi.org/10.1097/00005650-200412000-00006)). A systematic review of the PHQ family concluded that sensitivity to change is well established for the PHQ-9 ([Kroenke 2010](https://doi.org/10.1016/j.genhosppsych.2010.03.006)). Routine-outcome use in stepped-care services also relies implicitly on responsiveness, with large pre-post effect sizes reported for depression ([Richards 2009](https://doi.org/10.1348/014466509X405178)). The MIC of 5 points is widely cited but should be understood as a single-derivation estimate rather than a triangulated consensus value.",
    "confidence": "Moderate: responsiveness is demonstrated in a good treatment study and endorsed by a review, but the 5-point MIC derives essentially from one study and one method (2 SEM), and no MIC has been established in a workplace context."
   },
   "populations_languages_norms": {
    "findings": "The PHQ-9 has been validated across a wide range of clinical, general-population and disease-specific samples and in many languages, but formal population norms in the classical sense are scarce; the UK benchmark is embedded in a national service dataset rather than a norm table. The IPD meta-analyses aggregate around 100 primary studies from many countries ([Negeri et al 2021](https://doi.org/10.1136/bmj.n2183), [Levis et al 2020](https://doi.org/10.1001/jama.2020.6504)). Validated non-English versions retrieved this session include Chinese ([Wang et al 2014](https://doi.org/10.1016/j.genhosppsych.2014.05.021)) and Spanish, the latter via a dedicated systematic review and meta-analysis of the Spanish-language PHQ-2 and PHQ-9 ([Martinez et al 2023](https://doi.org/10.1001/jamanetworkopen.2023.36529)), with further evaluations in Korean ([Lee et al 2023](https://doi.org/10.3389/fpsyg.2023.1217038)), Danish ([Pedersen et al 2016](https://doi.org/10.1016/j.jpsychores.2016.09.010)) and Italian ([Di et al 2025](https://doi.org/10.1097/JCN.0000000000001178)) samples. For the UK specifically, the PHQ-9 is the routine depression outcome measure of the English NHS Talking Therapies programme (formerly Improving Access to Psychological Therapies, IAPT), where a score of 10 or above defines 'caseness' and movement below it underpins recovery reporting; large UK service cohorts and a UK randomised trial have used it in exactly this way ([Richards 2009](https://doi.org/10.1348/014466509X405178), [Barkham et al 2021](https://doi.org/10.1016/S2215-0366%2821%2900083-3)). UK reference values therefore live in national service reporting rather than in a published normative sample. General-population norms exist chiefly through large national surveys used in the invariance literature rather than as a dedicated UK norm set.",
    "confidence": "Moderate: extensive multi-language and multi-population validation, and a clear de facto UK benchmark via NHS Talking Therapies, but classical published norms (especially UK general-population and occupational norms) are limited."
   },
   "criticisms_controversies": "Several substantive controversies recur in the literature. First, cut-point and accuracy estimation. Studies that selectively report only well-performing cut-offs bias meta-analytic accuracy: for the PHQ-9, published results underestimated sensitivity below a cut-off of 10 (median difference about -0.06) and overestimated it above 10 (median about +0.07) ([Neupane et al 2021](https://doi.org/10.1002/mpr.1873)). Relatedly, using small datasets to simultaneously choose an optimal cut-off and estimate its accuracy is biased; in resampling from the IPD database the population-level optimal cut-off was actually 8 or above, and small studies scattered widely around it (only about 17% of 100-participant studies recovered the true optimum) ([Levis et al 2024](https://doi.org/10.1001/jamanetworkopen.2024.29630)). This questions the near-universal reliance on a cut-off of exactly 10. Second, reference-standard dependence: reported sensitivity is substantially higher against semistructured clinician interviews than against fully structured lay interviews or the MINI, so headline accuracy figures are conditional on the comparator ([Levis et al 2019](https://doi.org/10.1136/bmj.l1476), [Negeri et al 2021](https://doi.org/10.1136/bmj.n2183), [He et al 2019](https://doi.org/10.1159/000502294)). Third, somatic-item confounding in physically ill populations. In systemic sclerosis, somatic items accounted for a larger share of the total score than in matched healthy respondents, inflating scores by roughly 1.0 to 1.4 points (Hedges g 0.38 to 0.55) ([Leavens et al 2012](https://doi.org/10.1002/acr.21675)); in stroke, summed scoring moderately overestimated depression relative to a comparison sample (Cohen d about 0.43), driven by items such as tiredness and appetite ([Blake et al 2024](https://doi.org/10.1016/j.jpsychores.2024.111983)). This matters wherever respondents have physical illness or fatigue. Fourth, item 9 (thoughts of death or self-harm) is often misread as a suicidality measure, yet most positive responses are not associated with suicidality; the PHQ-8, which omits item 9, correlates almost perfectly with the PHQ-9 (r = 0.996) and performs almost identically for detecting depression ([Wu et al 2019](https://doi.org/10.1017/S0033291719001314)). Fifth, complexity does not pay: latent-variable and machine-learning scoring add negligible accuracy over the simple sum score with a cut-off ([Fischer et al 2021](https://doi.org/10.1017/S0033291721000131), [Hong et al 2022](https://doi.org/10.1016/j.genhosppsych.2022.04.011)). Finally, and central to the OWHS use case, the PHQ-9 is a clinical depression screener whose criterion validity was earned in primary-care and clinical populations against diagnostic interviews. Workplace deployment is a different context: screening in a lower-prevalence, largely non-help-seeking workforce reduces positive predictive value, the somatic-confounding problem is relevant to occupational groups with physical demands or illness, and no retrieved study validates the PHQ-9 against workplace criteria. The instrument's clinical provenance must not be read as endorsement of clinical-grade performance in a workplace-wellbeing programme; deployments that used it in worker samples treated it as an off-the-shelf symptom measure rather than validating it there ([Doki et al 2024](https://doi.org/10.1265/ehpm.23-00372)).",
   "record_notes": "Overall confidence: the PHQ-9 is the confidence-grading high-water mark. Structural validity, diagnostic criterion validity and internal consistency are High, anchored on the Levis/Thombs individual participant data programme and a reliability-generalisation meta-analysis; responsiveness and the somatic-confounding and cut-point critiques are well evidenced. The genuinely weaker or absent cells are test-retest (present but from few studies, none occupational), measurement invariance (scattered, no occupational or UK-workplace coverage, weak over-time evidence), criterion validity against organisational outcomes (absent), and workplace-specific psychometrics (absent). Schema stress-test notes: (1) The single 'criterion_validity' field conflates two very different evidence bases, diagnostic criterion validity (High) versus criterion validity against organisational outcomes (Absent). I graded it as split and said so in the findings, but a schema that forced one grade would misrepresent the instrument; the field should ideally be divided. (2) 'confidence' is a single scalar per property, yet for several properties the honest grade differs by sub-question (e.g. invariance is scalar-level across age but untested across occupation). I encoded the dominant grade and qualified it in the justification. (3) The clinical-origin versus workplace-deployment gap is the most important caveat for this registry and does not have a dedicated field; I carried it in criticisms_controversies and flagged it in constructs_claimed and criterion_validity, but it risks being lost if a reader scans only the property grades. (4) Licence status is factually clear (public domain, Pfizer-released) but I could not attach a DOI-bearing primary source to the licensing act itself, only to the original validation paper; I flagged this rather than attach a non-resolvable citation. (5) 'Absent' was used strictly for organisational criterion validity and workplace psychometrics; test-retest was deliberately NOT graded Absent because evidence exists, only sparsely, which the scale's wording ('barely-studied') made a close call between Low and Moderate.",
   "citations": [
    {
     "key": "kroenke2001",
     "authors": "Kroenke K, Spitzer RL, Williams JB",
     "year": "2001",
     "title": "The PHQ-9: validity of a brief depression severity measure",
     "journal": "Journal of General Internal Medicine",
     "doi": "10.1046/j.1525-1497.2001.016009606.x",
     "url": "https://doi.org/10.1046/j.1525-1497.2001.016009606.x"
    },
    {
     "key": "kroenke2010",
     "authors": "Kroenke K, Spitzer RL, Williams JB, Lowe B",
     "year": "2010",
     "title": "The Patient Health Questionnaire Somatic, Anxiety, and Depressive Symptom Scales: a systematic review",
     "journal": "General Hospital Psychiatry",
     "doi": "10.1016/j.genhosppsych.2010.03.006",
     "url": "https://doi.org/10.1016/j.genhosppsych.2010.03.006"
    },
    {
     "key": "levis2019",
     "authors": "Levis B, Benedetti A, Thombs BD, et al",
     "year": "2019",
     "title": "Accuracy of Patient Health Questionnaire-9 (PHQ-9) for screening to detect major depression: individual participant data meta-analysis",
     "journal": "BMJ",
     "doi": "10.1136/bmj.l1476",
     "url": "https://doi.org/10.1136/bmj.l1476"
    },
    {
     "key": "negeri2021",
     "authors": "Negeri ZF, Levis B, Sun Y, et al",
     "year": "2021",
     "title": "Accuracy of the Patient Health Questionnaire-9 for screening to detect major depression: updated systematic review and individual participant data meta-analysis",
     "journal": "BMJ",
     "doi": "10.1136/bmj.n2183",
     "url": "https://doi.org/10.1136/bmj.n2183"
    },
    {
     "key": "wu2019",
     "authors": "Wu Y, Levis B, Riehm KE, et al",
     "year": "2019",
     "title": "Equivalency of the diagnostic accuracy of the PHQ-8 and PHQ-9: a systematic review and individual participant data meta-analysis",
     "journal": "Psychological Medicine",
     "doi": "10.1017/S0033291719001314",
     "url": "https://doi.org/10.1017/S0033291719001314"
    },
    {
     "key": "he2019",
     "authors": "He C, Levis B, Riehm KE, et al",
     "year": "2019",
     "title": "The Accuracy of the Patient Health Questionnaire-9 Algorithm for Screening to Detect Major Depression: An Individual Participant Data Meta-Analysis",
     "journal": "Psychotherapy and Psychosomatics",
     "doi": "10.1159/000502294",
     "url": "https://doi.org/10.1159/000502294"
    },
    {
     "key": "levis2020",
     "authors": "Levis B, Sun Y, He C, et al",
     "year": "2020",
     "title": "Accuracy of the PHQ-2 Alone and in Combination With the PHQ-9 for Screening to Detect Major Depression",
     "journal": "JAMA",
     "doi": "10.1001/jama.2020.6504",
     "url": "https://doi.org/10.1001/jama.2020.6504"
    },
    {
     "key": "neupane2021",
     "authors": "Neupane D, Levis B, Bhandari PM, et al",
     "year": "2021",
     "title": "Selective cutoff reporting in studies of the accuracy of the Patient Health Questionnaire-9 and Edinburgh Postnatal Depression Scale",
     "journal": "International Journal of Methods in Psychiatric Research",
     "doi": "10.1002/mpr.1873",
     "url": "https://doi.org/10.1002/mpr.1873"
    },
    {
     "key": "levis2024",
     "authors": "Levis B, Bhandari PM, Neupane D, et al",
     "year": "2024",
     "title": "Data-Driven Cutoff Selection for the Patient Health Questionnaire-9 Depression Screening Tool",
     "journal": "JAMA Network Open",
     "doi": "10.1001/jamanetworkopen.2024.29630",
     "url": "https://doi.org/10.1001/jamanetworkopen.2024.29630"
    },
    {
     "key": "fischer2021",
     "authors": "Fischer F, Levis B, Falk C, et al",
     "year": "2021",
     "title": "Comparison of different scoring methods based on latent variable models of the PHQ-9: an individual participant data meta-analysis",
     "journal": "Psychological Medicine",
     "doi": "10.1017/S0033291721000131",
     "url": "https://doi.org/10.1017/S0033291721000131"
    },
    {
     "key": "hong2022",
     "authors": "Hong ZM, Williams J, Bulloch A, et al",
     "year": "2022",
     "title": "Alternative scoring of the Patient Health Questionnaire-9 in neurological populations: an approach based on a predictive algorithm deriving from individual item scores",
     "journal": "General Hospital Psychiatry",
     "doi": "10.1016/j.genhosppsych.2022.04.011",
     "url": "https://doi.org/10.1016/j.genhosppsych.2022.04.011"
    },
    {
     "key": "rufino2024",
     "authors": "Rufino JV, Rodrigues R, Birolim MM, et al",
     "year": "2024",
     "title": "Analysis of the dimensional structure of the Patient Health Questionnaire-9 (PHQ-9) in undergraduate students at a public university in Brazil",
     "journal": "Journal of Affective Disorders",
     "doi": "10.1016/j.jad.2024.01.051",
     "url": "https://doi.org/10.1016/j.jad.2024.01.051"
    },
    {
     "key": "blake2024",
     "authors": "Blake JJ, Munyombwe T, Fischer F, et al",
     "year": "2024",
     "title": "The factor structure of the Patient Health Questionnaire-9 in stroke: A comparison with a non-stroke population",
     "journal": "Journal of Psychosomatic Research",
     "doi": "10.1016/j.jpsychores.2024.111983",
     "url": "https://doi.org/10.1016/j.jpsychores.2024.111983"
    },
    {
     "key": "teymoori2020",
     "authors": "Teymoori A, Gorbunova A, Haghish FE, et al",
     "year": "2020",
     "title": "Factorial Structure and Validity of Depression (PHQ-9) and Anxiety (GAD-7) Scales after Traumatic Brain Injury",
     "journal": "Journal of Clinical Medicine",
     "doi": "10.3390/jcm9030873",
     "url": "https://doi.org/10.3390/jcm9030873"
    },
    {
     "key": "wang2014",
     "authors": "Wang W, Bian Q, Zhao Y, et al",
     "year": "2014",
     "title": "Reliability and validity of the Chinese version of the Patient Health Questionnaire (PHQ-9) in the general population",
     "journal": "General Hospital Psychiatry",
     "doi": "10.1016/j.genhosppsych.2014.05.021",
     "url": "https://doi.org/10.1016/j.genhosppsych.2014.05.021"
    },
    {
     "key": "lowe2004",
     "authors": "Lowe B, Unutzer J, Callahan CM, et al",
     "year": "2004",
     "title": "Monitoring depression treatment outcomes with the Patient Health Questionnaire-9",
     "journal": "Medical Care",
     "doi": "10.1097/00005650-200412000-00006",
     "url": "https://doi.org/10.1097/00005650-200412000-00006"
    },
    {
     "key": "leavens2012",
     "authors": "Leavens A, Patten SB, Hudson M, et al",
     "year": "2012",
     "title": "Influence of somatic symptoms on Patient Health Questionnaire-9 depression scores among patients with systemic sclerosis compared to a healthy general population sample",
     "journal": "Arthritis Care & Research",
     "doi": "10.1002/acr.21675",
     "url": "https://doi.org/10.1002/acr.21675"
    },
    {
     "key": "lee2023",
     "authors": "Lee EH, Kang EH, Kang HJ, et al",
     "year": "2023",
     "title": "Measurement invariance of the patient health questionnaire-9 depression scale in a nationally representative population-based sample",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2023.1217038",
     "url": "https://doi.org/10.3389/fpsyg.2023.1217038"
    },
    {
     "key": "dimatteo2025",
     "authors": "Di Matteo R, Bolgeo T, Simonelli N, et al",
     "year": "2025",
     "title": "Psychometric Properties and Measurement Invariance of the Patient Health Questionnaire 9 in an Italian Coronary Heart Disease Population",
     "journal": "Journal of Cardiovascular Nursing",
     "doi": "10.1097/JCN.0000000000001178",
     "url": "https://doi.org/10.1097/JCN.0000000000001178"
    },
    {
     "key": "pedersen2016",
     "authors": "Pedersen SS, Mathiasen K, Christensen KB, et al",
     "year": "2016",
     "title": "Psychometric analysis of the Patient Health Questionnaire in Danish patients with an implantable cardioverter defibrillator (The DEFIB-WOMEN study)",
     "journal": "Journal of Psychosomatic Research",
     "doi": "10.1016/j.jpsychores.2016.09.010",
     "url": "https://doi.org/10.1016/j.jpsychores.2016.09.010"
    },
    {
     "key": "ajele2025",
     "authors": "Ajele KW, Idemudia ES",
     "year": "2025",
     "title": "Charting the course of depression care: a meta-analysis of reliability generalization of the Patient Health Questionnaire (PHQ-9) as the measure",
     "journal": "Discover Mental Health",
     "doi": "10.1007/s44192-025-00181-x",
     "url": "https://doi.org/10.1007/s44192-025-00181-x"
    },
    {
     "key": "martinez2023",
     "authors": "Martinez A, Teklu SM, Tahir P, et al",
     "year": "2023",
     "title": "Validity of the Spanish-Language Patient Health Questionnaires 2 and 9: A Systematic Review and Meta-Analysis",
     "journal": "JAMA Network Open",
     "doi": "10.1001/jamanetworkopen.2023.36529",
     "url": "https://doi.org/10.1001/jamanetworkopen.2023.36529"
    },
    {
     "key": "pintomeza2005",
     "authors": "Pinto-Meza A, Serrano-Blanco A, Penarrubia MT, et al",
     "year": "2005",
     "title": "Assessing depression in primary care with the PHQ-9: can it be carried out over the telephone?",
     "journal": "Journal of General Internal Medicine",
     "doi": "10.1111/j.1525-1497.2005.0144.x",
     "url": "https://doi.org/10.1111/j.1525-1497.2005.0144.x"
    },
    {
     "key": "barkham2021",
     "authors": "Barkham M, Saxon D, Hardy GE, et al",
     "year": "2021",
     "title": "Person-centred experiential therapy versus cognitive behavioural therapy delivered in the English Improving Access to Psychological Therapies service for the treatment of moderate or severe depression (PRaCTICED)",
     "journal": "Lancet Psychiatry",
     "doi": "10.1016/S2215-0366(21)00083-3",
     "url": "https://doi.org/10.1016/S2215-0366%2821%2900083-3"
    },
    {
     "key": "richards2009",
     "authors": "Richards DA, Suckling R",
     "year": "2009",
     "title": "Improving access to psychological therapies: phase IV prospective cohort study",
     "journal": "British Journal of Clinical Psychology",
     "doi": "10.1348/014466509X405178",
     "url": "https://doi.org/10.1348/014466509X405178"
    },
    {
     "key": "doki2024",
     "authors": "Doki S, Hori D, Takahashi T, et al",
     "year": "2024",
     "title": "Designing a test battery for workers' well-being: the first wave of the Tsukuba Salutogenic Occupational Cohort Study",
     "journal": "Environmental Health and Preventive Medicine",
     "doi": "10.1265/ehpm.23-00372",
     "url": "https://doi.org/10.1265/ehpm.23-00372"
    }
   ]
  }
 ],
 "corrections": [
  {
   "id": "C-0001",
   "date": "2026-07-12",
   "record": "who-5",
   "field": "identity.licence_status",
   "change": "Licence status updated to reflect the October 2024 WHO republication under CC BY-NC-SA 3.0 IGO (non-commercial); the original entry stated only the pre-2024 free-with-acknowledgement position.",
   "source": "https://www.who.int/publications/m/item/WHO-UCN-MSD-MHE-2024.01"
  }
 ]
}