{
 "dataset": "OWHS Instrument Evidence Base",
 "version": "0.2.1",
 "schema_version": "0.2",
 "generated": "2026-07-12",
 "maintained_under": "Open Workplace Health Standard (openworkplacehealth.org)",
 "status": "resource maintained under the standard, distinct from the normative specification",
 "licence": {
  "prose": "CC BY 4.0",
  "structure": "Apache 2.0"
 },
 "coverage": "Full v0.1 pilot (7) upgraded to v0.2, plus pass-two (20). 27 records across the priority watchlist.",
 "editorial_rules": [
  "Criterion validity split into reference-standard and organisational (never merged).",
  "Per-property evidence_form provenance (canonical/derivative/parent).",
  "Status enumerator (well-established/contested/thin/untested) orthogonal to grade.",
  "Instrument-type governs applicability; single items and item-sets mark scale properties not-applicable.",
  "Test-retest is a structured sub-object; bundled ICCs are not accepted as retest evidence.",
  "Indirectness flagged on every grade; record-level deployment-context caveat.",
  "Licence verified against the steward's current terms with a date; founding papers are never a licence source.",
  "British English; no em dashes; every claim cited with a DOI where available.",
  "Null, weak and contradictory findings reported as content; absence of evidence recorded as a finding."
 ],
 "corrections_log": [
  {
   "id": "C-0001",
   "instrument_id": "who-5",
   "description": "WHO-5 licence mis-stated in pass one by trusting 2015/2020 review literature and missing WHO's October 2024 republication of the WHO-5 master version under a formal Creative Commons licence.",
   "old_value": "Free to use; distributed at no charge, no licence fee or royalty; may be reproduced and translated with acknowledgement of source (free availability sourced to Topp 2015 / Lara-Cabrera 2020 review literature). No formal open licence stated.",
   "new_value": "Copyright World Health Organization 2024; the WHO-5 master version is available under the Creative Commons Attribution-NonCommercial-ShareAlike 3.0 IGO licence (CC BY-NC-SA 3.0 IGO). Document ref WHO/UCN/MSD/MHE/2024.1. Non-commercial use and adaptation permitted with attribution and share-alike; commercial (including employer/vendor) use requires WHO permission.",
   "source": "https://cdn.who.int/media/docs/default-source/mental-health/who-5_english-original4da539d6ed4b49389e3afe47cda2326a.pdf",
   "date": "2026-07-12"
  },
  {
   "id": "C-0002",
   "instrument_id": "wemwbs",
   "description": "WEMWBS licence-currency update: pass one implied NHS/non-profit users obtain a no-fee non-commercial licence. Warwick Innovations introduced charges for NHS organisations from 1 December 2024.",
   "old_value": "Academic and non-profit users obtain a no-fee non-commercial licence; commercial users pay a tiered fee.",
   "new_value": "Registration still required; non-commercial and commercial licences separated. From 1 December 2024, charges apply to NHS organisations (NHS trusts, GP surgeries and NHS-funded bodies) on a published tiered scale (e.g. up to 30 participants GBP 45, up to 100 GBP 125, up to 1000 GBP 600). Non-NHS academic/non-profit non-commercial registration remains available; commercial users pay fees. Verify current tier with Warwick Innovations before fielding.",
   "source": "https://warwick.ac.uk/services/innovations/wemwbs/licenses/",
   "date": "2026-07-12"
  },
  {
   "id": "C-0003",
   "instrument_id": "phq-9",
   "description": "PHQ-9 licence wording precision: pass one described the status as 'public domain'. The steward terms grant a copyright exemption and free use rather than asserting a formal public-domain dedication.",
   "old_value": "Free / public domain. Pfizer released the PHQ and GAD-7 without copyright restriction.",
   "new_value": "Free for download and use; content on the PHQ Screeners site is 'expressly exempted from Pfizer's general copyright restrictions'. No permission or fee required, attribution to Kroenke et al 2001 expected. This is a free-use copyright exemption, not a formally asserted public-domain status.",
   "source": "https://www.phqscreeners.com/terms",
   "date": "2026-07-12"
  }
 ],
 "integrity": {
  "pass_two_distinct_dois": 268,
  "pass_two_dois_resolved": 268,
  "pass_two_dois_non_resolving": 0,
  "note": "All pass-two DOIs resolve (the single doi.org-normalisation flag, 10.1027//1015-5759.19.1.12, confirmed genuine via CrossRef). Pass-one integrity verified separately in v0.1. Pass-one records upgraded to v0.2 with status enumerators populated (derived from findings); full 27-record conformance re-verified (criterion split, grade/status/evidence_form/indirectness on every graded property, structured test-retest)."
 },
 "records": [
  {
   "instrument_id": "ons-4",
   "display_name": "ONS-4 personal wellbeing questions",
   "instrument_type": "item-set",
   "schema_version": "0.2",
   "identity": {
    "name": "ONS-4 (ONS4) personal well-being questions",
    "current_version": "Four questions as introduced by the UK Office for National Statistics in the Annual Population Survey in 2011; each item uses an 11-point 0 to 10 response scale. The wording has been stable since introduction, with ONS relabelling the set as 'personal well-being' following public focus groups in 2013.",
    "item_count": "4 (life satisfaction; worthwhile; happiness yesterday; anxiety yesterday), each administered and reported as a separate single item; ONS does not publish a summed total.",
    "original_citation": "Office for National Statistics (2011). Four personal well-being questions introduced in the Annual Population Survey. No primary journal article; the measure is defined and disseminated in ONS technical guidance and described in Dolan P, Metcalfe R (2012) Measuring Subjective Wellbeing: Recommendations on Measures for use by National Governments. Journal of Social Policy. DOI 10.1017/s0047279411000833",
    "steward_publisher": "UK Office for National Statistics (ONS)",
    "licence_status": "CONFIRMED. ONS states the four personal well-being questions are part of the government harmonised standards and 'they are freely available to others and their use is encouraged'. All ONS website content is published under the Open Government Licence v3.0 (Crown copyright). No fee, no registration; open status attaches to the original ONS wording and 0-to-10 scale, not to modified derivatives.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://www.ons.gov.uk/peoplepopulationandcommunity/wellbeing/methodologies/surveysusingthe4officefornationalstatisticspersonalwellbeingquestions (use encouraged; freely available) and OGL v3.0 site-wide footer; corroborated at https://analysisfunction.civilservice.gov.uk/policy-store/personal-well-being/"
   },
   "constructs_claimed": "The four questions deliberately span three or four conceptually distinct facets of subjective well-being rather than a single latent trait: evaluative well-being (life satisfaction), eudaimonic well-being (the sense that things done in life are worthwhile), and hedonic affect split into positive (happiness yesterday) and negative (anxiety yesterday) components [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). This tripartite or four-part conception follows the Stiglitz-Sen-Fitoussi and national-accounts-of-well-being tradition that ONS adopted [Dolan & Metcalfe 2012](https://doi.org/10.1017/s0047279411000833). The design intent is that each item is informative in its own right; the anxiety item is a negative-affect measure and is therefore worded and scored in the opposite direction to the other three.",
   "deployment_context_caveat": "",
   "structural_validity": {
    "findings": "There is no confirmatory factor-analytic validation of the original ONS-4 as a unidimensional scale, and this is by design: ONS treats the four items as separate indicators of distinct constructs and publishes each separately rather than as a summed score [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). The one psychometric study that factor-analyses ONS-4-derived items examined a modified four-point short version (the Personal Wellbeing Score, PWS) embedded alongside other R-Outcomes measures; a scree plot suggested four or six factors and Kaiser's criterion four, with the four well-being items loading together and separately from co-administered health and experience measures, and inter-item correlations of r=0.51 to r=0.77 [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). That evidence supports the well-being items cohering as a set in that instrument, but it pertains to the reworded four-point PWS, not to the original 0 to 10 ONS-4 wording. Broader multi-country work confirms that evaluative, eudaimonic and affective well-being are empirically separable dimensions rather than one factor, which is consistent with ONS-4 being reported item by item [Ruggeri 2020](https://doi.org/10.1186/s12955-020-01423-y).",
    "grade": "Low",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low. Direct factor-analytic evidence exists only for a modified derivative (single study, one English social-prescribing sample); the original ONS-4 is not validated as a scale and is not intended to be one."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent evidence for the ONS-4 items is largely indirect, resting on the wider single-item life-satisfaction literature and on derivative instruments rather than on the ONS wording itself. Single-item life-satisfaction measures correlate with the multi-item Satisfaction With Life Scale at zero-order r=0.62 to 0.64, rising to r=0.78 to 0.80 after disattenuation, and reproduce the SWLS pattern of associations with health, domain satisfaction and affect almost exactly (mean absolute difference in correlations 0.015 to 0.042) [Cheung & Lucas 2014](https://doi.org/10.1007/s11136-014-0726-4). In the modified PWS derivative, each well-being item correlated with its summary score at r=0.83 to 0.88 and the two evaluative items correlated at r=0.77 [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). UK multi-instrument comparison datasets now allow ONS-4 to be mapped against WEMWBS, EQ-5D and ICECAP-A, supporting moderate cross-measure convergence while confirming the instruments are not interchangeable [Wickramasekera & Tsuchiya 2025](https://doi.org/10.1007/s11205-025-03728-1). Discriminant behaviour of the anxiety item is notable: it is the negative-affect component and shows the weakest association with the evaluative items, consistent with affect and evaluation being separable [Ruggeri 2020](https://doi.org/10.1186/s12955-020-01423-y).",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low. Convergent magnitudes are reasonable but are drawn mostly from single-item life-satisfaction studies in non-UK samples and from a modified derivative, not from the original ONS-4 items as fielded."
   },
   "criterion_validity_reference_standard": {
    "findings": "No study located in this pass tests ONS-4 against organisational or workplace outcomes such as sickness absence, staff turnover or productivity; criterion evidence for these items in an employment context is absent. Criterion evidence for the underlying constructs comes from the general subjective-well-being literature and predicts health and mortality rather than organisational endpoints: low self-reported life satisfaction predicted 20-year all-cause mortality in a Finnish cohort of 22,461 adults [Koivumaa-Honkanen 2000](https://doi.org/10.1093/aje/152.10.983), life satisfaction predicted all-cause mortality over 22 years in older adults [Gana 2016](https://doi.org/10.1016/j.jpsychores.2016.04.004), and a broad review links subjective well-being to health and longevity [Diener & Chan 2011](https://doi.org/10.1111/j.1758-0854.2010.01045.x). Mendelian randomisation provides mixed causal support, finding little robust causal effect of subjective well-being on cardiometabolic disease [Wootton 2018](https://doi.org/10.1136/bmj.k3788). A UK natural experiment used ONS-4-style items as outcomes when evaluating a change in the built environment, illustrating policy-outcome use rather than establishing criterion validity against a gold standard [Ram 2020](https://doi.org/10.1136/jech-2019-213591).",
    "grade": "Very low",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Very low. Workplace and organisational criterion validity is absent; the available criterion evidence is for the general constructs (health, mortality) in non-UK, non-workplace populations and does not use the ONS-4 items specifically."
   },
   "criterion_validity_organisational": {
    "findings": "Organisational criterion evidence (sickness absence, turnover, performance, diagnosed conditions in a work context): see the criterion findings; graded from the pass-one record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Very low. Workplace and organisational criterion validity is absent; the available criterion evidence is for the general constructs (health, mortality) in non-UK, non-workplace populations and does not use the ONS-4 items specifically."
   },
   "internal_consistency": {
    "findings": "Internal consistency is not a meaningful property of ONS-4 as ONS uses it, because the four items are reported separately and Cronbach's alpha cannot be computed for a single item [Cheung & Lucas 2014](https://doi.org/10.1007/s11136-014-0726-4). The only alpha located is for the modified four-point PWS summary score, where Cronbach's alpha was 0.90 in one English social-prescribing sample; the authors themselves had expected 0.7 to 0.9 to justify an aggregate score, and note this value sits at the top of that range [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). That coefficient applies to the reworded PWS, not to the original ONS-4, and a high alpha on four items partly reflects deliberate content breadth rather than redundancy. No omega is reported.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low. Only a single alpha exists and it belongs to a modified derivative scale, not the original item set; for the original ONS-4 internal consistency is not applicable."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "untested",
    "indirectness": "see summary",
    "summary": "Test-retest reliability for the ONS-4 items as fielded is, on the evidence located in this pass, absent, and this is the clearest evidential gap in the record. The one instrument-specific validation study states explicitly that its anonymous, unlinked data did not permit test-retest reliability, inter-rater reliability or within-individual change to be estimated [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). The nearest available evidence is not classical test-retest but model-based reliability from the single-item life-satisfaction literature: using latent state-trait models on four national panels (combined N over 68,000), reliability estimates for single-item life satisfaction ranged from about 0.68 to 0.74 [Lucas & Donnellan 2012](https://doi.org/10.1007/s11205-011-9783-z). These estimates concern single-item life satisfaction generally, not the ONS worthwhile, happiness or anxiety items, and they are derived from longitudinal decomposition rather than a short-interval retest. No short-interval test-retest coefficient for any ONS-4 item was located.",
    "confidence_note": "Absent. No test-retest study of the ONS-4 items was located; the instrument-specific paper explicitly could not estimate it, and analogous single-item reliability comes from non-ONS life-satisfaction panels."
   },
   "measurement_invariance": {
    "findings": "No formal measurement-invariance analysis of ONS-4 (configural, metric or scalar, across sex, age, occupation, language or time) was located in this pass. ONS publishes personal well-being estimates broken down by age and sex, but publishing subgroup means is not a test of invariance and does not establish that the items function equivalently across groups. Evidence that mode of administration shifts subjective well-being scores is directly relevant to invariance across data-collection contexts: telephone respondents report systematically higher well-being than face-to-face or self-completion respondents [Dolan & Kavetsos 2016](https://doi.org/10.1007/s10902-015-9642-8), a concern the ONS-4 validation study also flags because some of its ratings were collected face-to-face and some by telephone [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). This implies non-trivial risk of non-invariance across survey modes that has not been formally quantified for ONS-4.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Absent. No invariance model for ONS-4 was located; the only directly relevant finding is evidence of mode-of-administration effects, which points to a specific untested invariance risk."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness and a minimal important change (MIC) have not been established for ONS-4. Responsiveness was named as a design goal for the derivative PWS, but the validation study was a cross-sectional secondary analysis with no linkage between pre- and post-intervention responses, so it could not estimate within-individual change or responsiveness [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). ONS-4-style items have been used as outcomes in a longitudinal natural experiment on the built environment, demonstrating that the items are used to detect change over time in practice, but that study did not derive an anchor-based or distribution-based MIC [Ram 2020](https://doi.org/10.1136/jech-2019-213591). No minimal-important-change threshold for any ONS-4 item was located.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Absent. No responsiveness statistic or MIC for ONS-4 was located; the instrument-specific study could not estimate change, and applied uses do not report an MIC."
   },
   "populations_languages_norms": {
    "findings": "ONS-4 was developed for and is fielded on the UK general adult population through the Annual Population Survey, and UK population norms and benchmarks are published by ONS from that survey and reported against standard threshold bands: for life satisfaction, worthwhile and happiness, responses of 0 to 4 are classed low, 5 to 6 medium, 7 to 8 high and 9 to 10 very high, while for anxiety (reverse direction) 0 to 1 is low anxiety, 2 to 3 medium, 4 to 5 high and 6 to 10 very high [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). Beyond the UK general population, instrument-specific validation exists only in one English social-prescribing sample using a modified four-point derivative [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394), and applied UK studies have used the items in specific subpopulations such as residents of a regenerated urban neighbourhood [Ram 2020](https://doi.org/10.1136/jech-2019-213591) and during COVID-19 among home-workers [Hensher & Beck 2023](https://doi.org/10.1016/j.tra.2022.103579). Multi-instrument UK comparison datasets now situate ONS-4 alongside other measures for benchmarking [Wickramasekera & Tsuchiya 2025](https://doi.org/10.1007/s11205-025-03728-1). No validated non-English translation of the ONS-4 wording was located in this pass; the closely related OECD core module, which adds an affect item, provides an internationally harmonised counterpart [Dolan & Metcalfe 2012](https://doi.org/10.1017/s0047279411000833).",
    "grade": "Moderate",
    "indirectness": "see findings"
   },
   "criticisms_controversies": "The most consequential issue for a registry is the scale-versus-four-items question. ONS-4 was built as four separate single items measuring distinct constructs and ONS provides no summary score, so summing the items into a composite (as some deployments do) is a departure from the instrument's design and psychometric basis [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). A second issue is that the strongest instrument-specific psychometrics (alpha 0.90, factor structure) belong to a reworded four-point derivative rather than the original 0 to 10 items, so those coefficients should not be read as evidence about ONS-4 as ONS fields it [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). Third, the anxiety item is negatively worded and reverse-scored, its distribution is strongly skewed towards low anxiety, and it behaves differently from the three positive items, which complicates any attempt to combine the four [Benson 2019](https://doi.org/10.1136/bmjoq-2018-000394). Fourth, mode-of-administration effects are documented: telephone interviewing inflates reported well-being relative to other modes, a threat to comparability across surveys and over time that is particularly relevant when the items are used for benchmarking [Dolan & Kavetsos 2016](https://doi.org/10.1007/s10902-015-9642-8). Finally, single-item measurement remains debated; while single-item life-satisfaction measures perform comparably to multi-item scales on validity and reliability in large studies [Cheung & Lucas 2014](https://doi.org/10.1007/s11136-014-0726-4) [Lucas & Donnellan 2012](https://doi.org/10.1007/s11205-011-9783-z), that literature concerns life satisfaction specifically and does not directly certify the worthwhile, happiness or anxiety items, and the causal standing of subjective well-being for downstream health outcomes is itself contested [Wootton 2018](https://doi.org/10.1136/bmj.k3788).",
   "citations": [
    {
     "key": "Benson2019",
     "authors": "Benson T, Sladen J, Liles A, Potts HWW",
     "year": "2019",
     "title": "Personal Wellbeing Score (PWS): a short version of ONS4: development and validation in social prescribing",
     "journal": "BMJ Open Quality",
     "doi": "10.1136/bmjoq-2018-000394",
     "url": "https://doi.org/10.1136/bmjoq-2018-000394"
    },
    {
     "key": "BensonCorr2019",
     "authors": "Benson T, Sladen J, Liles A, Potts HWW",
     "year": "2019",
     "title": "Correction: Personal Wellbeing Score (PWS): a short version of ONS4: development and validation in social prescribing",
     "journal": "BMJ Open Quality",
     "doi": "10.1136/bmjoq-2018-000394corr1",
     "url": "https://doi.org/10.1136/bmjoq-2018-000394corr1"
    },
    {
     "key": "DolanMetcalfe2012",
     "authors": "Dolan P, Metcalfe R",
     "year": "2012",
     "title": "Measuring Subjective Wellbeing: Recommendations on Measures for use by National Governments",
     "journal": "Journal of Social Policy",
     "doi": "10.1017/s0047279411000833",
     "url": "https://doi.org/10.1017/s0047279411000833"
    },
    {
     "key": "LucasDonnellan2012",
     "authors": "Lucas RE, Donnellan MB",
     "year": "2012",
     "title": "Estimating the Reliability of Single-Item Life Satisfaction Measures: Results from Four National Panel Studies",
     "journal": "Social Indicators Research",
     "doi": "10.1007/s11205-011-9783-z",
     "url": "https://doi.org/10.1007/s11205-011-9783-z"
    },
    {
     "key": "CheungLucas2014",
     "authors": "Cheung F, Lucas RE",
     "year": "2014",
     "title": "Assessing the validity of single-item life satisfaction measures: results from three large samples",
     "journal": "Quality of Life Research",
     "doi": "10.1007/s11136-014-0726-4",
     "url": "https://doi.org/10.1007/s11136-014-0726-4"
    },
    {
     "key": "DolanKavetsos2016",
     "authors": "Dolan P, Kavetsos G",
     "year": "2016",
     "title": "Happy Talk: Mode of Administration Effects on Subjective Well-Being",
     "journal": "Journal of Happiness Studies",
     "doi": "10.1007/s10902-015-9642-8",
     "url": "https://doi.org/10.1007/s10902-015-9642-8"
    },
    {
     "key": "Ruggeri2020",
     "authors": "Ruggeri K, Garcia-Garzon E, Maguire A, Matz S, Huppert FA",
     "year": "2020",
     "title": "Well-being is more than happiness and life satisfaction: a multidimensional analysis of 21 countries",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-020-01423-y",
     "url": "https://doi.org/10.1186/s12955-020-01423-y"
    },
    {
     "key": "Wickramasekera2025",
     "authors": "Wickramasekera N, Tsuchiya A",
     "year": "2025",
     "title": "A Large Scale Population Survey of Health and Wellbeing to Allow Comparisons Between Outcome Measures: the SIPHER-HWMIC Dataset",
     "journal": "Social Indicators Research",
     "doi": "10.1007/s11205-025-03728-1",
     "url": "https://doi.org/10.1007/s11205-025-03728-1"
    },
    {
     "key": "Koivumaa2000",
     "authors": "Koivumaa-Honkanen H, Honkanen R, Viinamaki H, Heikkila K, Kaprio J, Koskenvuo M",
     "year": "2000",
     "title": "Self-reported life satisfaction and 20-year mortality in healthy Finnish adults",
     "journal": "American Journal of Epidemiology",
     "doi": "10.1093/aje/152.10.983",
     "url": "https://doi.org/10.1093/aje/152.10.983"
    },
    {
     "key": "Gana2016",
     "authors": "Gana K, Broc G, Saada Y, Amieva H, Quintard B",
     "year": "2016",
     "title": "Subjective wellbeing and longevity: Findings from a 22-year cohort study",
     "journal": "Journal of Psychosomatic Research",
     "doi": "10.1016/j.jpsychores.2016.04.004",
     "url": "https://doi.org/10.1016/j.jpsychores.2016.04.004"
    },
    {
     "key": "Wootton2018",
     "authors": "Wootton RE, Lawn RB, Millard LAC, Davies NM, Taylor AE, et al.",
     "year": "2018",
     "title": "Evaluation of the causal effects between subjective wellbeing and cardiometabolic health: mendelian randomisation study",
     "journal": "BMJ",
     "doi": "10.1136/bmj.k3788",
     "url": "https://doi.org/10.1136/bmj.k3788"
    },
    {
     "key": "DienerChan2011",
     "authors": "Diener E, Chan MY",
     "year": "2011",
     "title": "Happy People Live Longer: Subjective Well-Being Contributes to Health and Longevity",
     "journal": "Applied Psychology: Health and Well-Being",
     "doi": "10.1111/j.1758-0854.2010.01045.x",
     "url": "https://doi.org/10.1111/j.1758-0854.2010.01045.x"
    },
    {
     "key": "Ram2020",
     "authors": "Ram B, Limb ES, Shankar A, Nightingale CM, Rudnicka AR, et al.",
     "year": "2020",
     "title": "Evaluating the effect of change in the built environment on mental health and subjective well-being: a natural experiment",
     "journal": "Journal of Epidemiology and Community Health",
     "doi": "10.1136/jech-2019-213591",
     "url": "https://doi.org/10.1136/jech-2019-213591"
    },
    {
     "key": "HensherBeck2023",
     "authors": "Hensher DA, Beck MJ",
     "year": "2023",
     "title": "Exploring how worthwhile the things that you do in life are during COVID-19 and links to well-being and working from home",
     "journal": "Transportation Research Part A: Policy and Practice",
     "doi": "10.1016/j.tra.2022.103579",
     "url": "https://doi.org/10.1016/j.tra.2022.103579"
    }
   ],
   "record_notes": "[Upgraded from v0.1 to v0.2 structure in pass two; criterion field split, licence re-verified 2026-07-12.] Overall confidence is Low, with two properties (test-retest, invariance, responsiveness/MIC) graded Absent. The dominant honesty problem this record surfaced is a substitution risk that the schema does not explicitly guard against: almost all instrument-specific psychometric coefficients (alpha 0.90, factor structure, item-total r=0.83 to 0.88) come from Benson 2019, but that study validated a modified four-point derivative (the Personal Wellbeing Score) with reworded items, a collapsed response scale and a reversed anxiety direction, not the original 0 to 10 ONS-4. I have flagged this inline everywhere a Benson coefficient appears, but a naive reader could still mistake these for original-ONS-4 properties; the schema would benefit from a field distinguishing 'evidence for this instrument as canonically fielded' from 'evidence for a named derivative'. Second, the scale-versus-single-items status is central and cuts across every property: ONS-4 is four separate single items with no ONS summary score, which makes internal consistency, structural validity and a single test-retest coefficient partly category errors when applied to the instrument as designed; I recorded these honestly as Low or Absent rather than forcing scale-level statistics. Third, for reliability and convergent validity I had to rely on the general single-item life-satisfaction literature (Lucas & Donnellan; Cheung & Lucas) because ONS-4-specific studies do not report these; this is indirect evidence (non-UK samples, life satisfaction only, not the worthwhile/happiness/anxiety items) and I downgraded accordingly. Fourth, key steward facts (UK norms from the Annual Population Survey, Open Government Licence, National Statistics designation) are documented in ONS technical guidance that does not carry a DOI; I grounded the citable elements (threshold bands, National Statistics status, ONS encouragement of reuse) in Benson 2019 and did not invent a DOI for ONS web guidance. No workplace criterion validity was located, which is a material gap given the registry's audience. All fourteen DOIs were checked for resolution and retraction status this session. [0.2.1] ONS-4 item split implemented per schema v0.2 rule 4: four first-class item records (see items list) with graded evidence retained at set level; items cross-linked to the question bank.",
   "items": [
    "ons-4-life-satisfaction",
    "ons-4-worthwhile",
    "ons-4-happiness",
    "ons-4-anxiety"
   ]
  },
  {
   "instrument_id": "ons-4-life-satisfaction",
   "display_name": "ONS-4 item: life satisfaction",
   "instrument_type": "single-item",
   "schema_version": "0.2",
   "parent_id": "ons-4",
   "identity": {
    "name": "ONS-4 personal well-being item: life satisfaction",
    "current_version": "Four questions as introduced by the UK Office for National Statistics in the Annual Population Survey in 2011; each item uses an 11-point 0 to 10 response scale. The wording has been stable since introduction, with ONS relabelling the set as 'personal well-being' following public focus groups in 2013.",
    "item_count": "1 item, administered and reported on its own 0 to 10 scale (no composite is formed by ONS).",
    "original_citation": "Office for National Statistics (2011). Four personal well-being questions introduced in the Annual Population Survey. No primary journal article; the measure is defined and disseminated in ONS technical guidance and described in Dolan P, Metcalfe R (2012) Measuring Subjective Wellbeing: Recommendations on Measures for use by National Governments. Journal of Social Policy. DOI 10.1017/s0047279411000833",
    "steward_publisher": "UK Office for National Statistics (ONS)",
    "licence_status": "CONFIRMED. ONS states the four personal well-being questions are part of the government harmonised standards and 'they are freely available to others and their use is encouraged'. All ONS website content is published under the Open Government Licence v3.0 (Crown copyright). No fee, no registration; open status attaches to the original ONS wording and 0-to-10 scale, not to modified derivatives.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://www.ons.gov.uk/peoplepopulationandcommunity/wellbeing/methodologies/surveysusingthe4officefornationalstatisticspersonalwellbeingquestions (use encouraged; freely available) and OGL v3.0 site-wide footer; corroborated at https://analysisfunction.civilservice.gov.uk/policy-store/personal-well-being/"
   },
   "constructs_claimed": "Evaluative wellbeing: overall satisfaction with life nowadays, self-rated 0 to 10.",
   "deployment_context_caveat": "",
   "question_bank_ref": {
    "resource": "question-bank",
    "item_id": "aps-life-satisfaction",
    "note": "The question bank holds the canonical description of this item (topic, verified wording under OGL, response scale, benchmark); this record holds the evidence view."
   },
   "citations": [],
   "record_notes": "Created 2026-07-12 by the ONS-4 item split (schema v0.2 rule 4). Graded evidence lives on the ONS-4 parent record; this item record exists so the set's items are first-class and cross-linked one-to-one with the question bank.",
   "structural_validity": {
    "findings": "Not applicable at item level: this is a single item, so internal consistency and factor structure are category errors rather than gaps (schema v0.2 rule 4).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "not applicable at item level"
   },
   "convergent_discriminant_validity": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "criterion_validity_reference_standard": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Very low",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "criterion_validity_organisational": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "internal_consistency": {
    "findings": "Not applicable at item level: this is a single item, so internal consistency and factor structure are category errors rather than gaps (schema v0.2 rule 4).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "not applicable at item level"
   },
   "test_retest_reliability": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "measurement_invariance": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "responsiveness_mic": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   }
  },
  {
   "instrument_id": "ons-4-worthwhile",
   "display_name": "ONS-4 item: worthwhile",
   "instrument_type": "single-item",
   "schema_version": "0.2",
   "parent_id": "ons-4",
   "identity": {
    "name": "ONS-4 personal well-being item: worthwhile",
    "current_version": "Four questions as introduced by the UK Office for National Statistics in the Annual Population Survey in 2011; each item uses an 11-point 0 to 10 response scale. The wording has been stable since introduction, with ONS relabelling the set as 'personal well-being' following public focus groups in 2013.",
    "item_count": "1 item, administered and reported on its own 0 to 10 scale (no composite is formed by ONS).",
    "original_citation": "Office for National Statistics (2011). Four personal well-being questions introduced in the Annual Population Survey. No primary journal article; the measure is defined and disseminated in ONS technical guidance and described in Dolan P, Metcalfe R (2012) Measuring Subjective Wellbeing: Recommendations on Measures for use by National Governments. Journal of Social Policy. DOI 10.1017/s0047279411000833",
    "steward_publisher": "UK Office for National Statistics (ONS)",
    "licence_status": "CONFIRMED. ONS states the four personal well-being questions are part of the government harmonised standards and 'they are freely available to others and their use is encouraged'. All ONS website content is published under the Open Government Licence v3.0 (Crown copyright). No fee, no registration; open status attaches to the original ONS wording and 0-to-10 scale, not to modified derivatives.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://www.ons.gov.uk/peoplepopulationandcommunity/wellbeing/methodologies/surveysusingthe4officefornationalstatisticspersonalwellbeingquestions (use encouraged; freely available) and OGL v3.0 site-wide footer; corroborated at https://analysisfunction.civilservice.gov.uk/policy-store/personal-well-being/"
   },
   "constructs_claimed": "Eudaimonic wellbeing: the extent to which the things done in life feel worthwhile, self-rated 0 to 10.",
   "deployment_context_caveat": "",
   "question_bank_ref": {
    "resource": "question-bank",
    "item_id": "aps-worthwhile",
    "note": "The question bank holds the canonical description of this item (topic, verified wording under OGL, response scale, benchmark); this record holds the evidence view."
   },
   "citations": [],
   "record_notes": "Created 2026-07-12 by the ONS-4 item split (schema v0.2 rule 4). Graded evidence lives on the ONS-4 parent record; this item record exists so the set's items are first-class and cross-linked one-to-one with the question bank.",
   "structural_validity": {
    "findings": "Not applicable at item level: this is a single item, so internal consistency and factor structure are category errors rather than gaps (schema v0.2 rule 4).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "not applicable at item level"
   },
   "convergent_discriminant_validity": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "criterion_validity_reference_standard": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Very low",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "criterion_validity_organisational": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "internal_consistency": {
    "findings": "Not applicable at item level: this is a single item, so internal consistency and factor structure are category errors rather than gaps (schema v0.2 rule 4).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "not applicable at item level"
   },
   "test_retest_reliability": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "measurement_invariance": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "responsiveness_mic": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   }
  },
  {
   "instrument_id": "ons-4-happiness",
   "display_name": "ONS-4 item: happiness yesterday",
   "instrument_type": "single-item",
   "schema_version": "0.2",
   "parent_id": "ons-4",
   "identity": {
    "name": "ONS-4 personal well-being item: happiness yesterday",
    "current_version": "Four questions as introduced by the UK Office for National Statistics in the Annual Population Survey in 2011; each item uses an 11-point 0 to 10 response scale. The wording has been stable since introduction, with ONS relabelling the set as 'personal well-being' following public focus groups in 2013.",
    "item_count": "1 item, administered and reported on its own 0 to 10 scale (no composite is formed by ONS).",
    "original_citation": "Office for National Statistics (2011). Four personal well-being questions introduced in the Annual Population Survey. No primary journal article; the measure is defined and disseminated in ONS technical guidance and described in Dolan P, Metcalfe R (2012) Measuring Subjective Wellbeing: Recommendations on Measures for use by National Governments. Journal of Social Policy. DOI 10.1017/s0047279411000833",
    "steward_publisher": "UK Office for National Statistics (ONS)",
    "licence_status": "CONFIRMED. ONS states the four personal well-being questions are part of the government harmonised standards and 'they are freely available to others and their use is encouraged'. All ONS website content is published under the Open Government Licence v3.0 (Crown copyright). No fee, no registration; open status attaches to the original ONS wording and 0-to-10 scale, not to modified derivatives.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://www.ons.gov.uk/peoplepopulationandcommunity/wellbeing/methodologies/surveysusingthe4officefornationalstatisticspersonalwellbeingquestions (use encouraged; freely available) and OGL v3.0 site-wide footer; corroborated at https://analysisfunction.civilservice.gov.uk/policy-store/personal-well-being/"
   },
   "constructs_claimed": "Positive affect: happiness felt yesterday, self-rated 0 to 10.",
   "deployment_context_caveat": "",
   "question_bank_ref": {
    "resource": "question-bank",
    "item_id": "aps-happiness",
    "note": "The question bank holds the canonical description of this item (topic, verified wording under OGL, response scale, benchmark); this record holds the evidence view."
   },
   "citations": [],
   "record_notes": "Created 2026-07-12 by the ONS-4 item split (schema v0.2 rule 4). Graded evidence lives on the ONS-4 parent record; this item record exists so the set's items are first-class and cross-linked one-to-one with the question bank.",
   "structural_validity": {
    "findings": "Not applicable at item level: this is a single item, so internal consistency and factor structure are category errors rather than gaps (schema v0.2 rule 4).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "not applicable at item level"
   },
   "convergent_discriminant_validity": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "criterion_validity_reference_standard": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Very low",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "criterion_validity_organisational": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "internal_consistency": {
    "findings": "Not applicable at item level: this is a single item, so internal consistency and factor structure are category errors rather than gaps (schema v0.2 rule 4).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "not applicable at item level"
   },
   "test_retest_reliability": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "measurement_invariance": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "responsiveness_mic": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   }
  },
  {
   "instrument_id": "ons-4-anxiety",
   "display_name": "ONS-4 item: anxiety yesterday",
   "instrument_type": "single-item",
   "schema_version": "0.2",
   "parent_id": "ons-4",
   "identity": {
    "name": "ONS-4 personal well-being item: anxiety yesterday",
    "current_version": "Four questions as introduced by the UK Office for National Statistics in the Annual Population Survey in 2011; each item uses an 11-point 0 to 10 response scale. The wording has been stable since introduction, with ONS relabelling the set as 'personal well-being' following public focus groups in 2013.",
    "item_count": "1 item, administered and reported on its own 0 to 10 scale (no composite is formed by ONS).",
    "original_citation": "Office for National Statistics (2011). Four personal well-being questions introduced in the Annual Population Survey. No primary journal article; the measure is defined and disseminated in ONS technical guidance and described in Dolan P, Metcalfe R (2012) Measuring Subjective Wellbeing: Recommendations on Measures for use by National Governments. Journal of Social Policy. DOI 10.1017/s0047279411000833",
    "steward_publisher": "UK Office for National Statistics (ONS)",
    "licence_status": "CONFIRMED. ONS states the four personal well-being questions are part of the government harmonised standards and 'they are freely available to others and their use is encouraged'. All ONS website content is published under the Open Government Licence v3.0 (Crown copyright). No fee, no registration; open status attaches to the original ONS wording and 0-to-10 scale, not to modified derivatives.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://www.ons.gov.uk/peoplepopulationandcommunity/wellbeing/methodologies/surveysusingthe4officefornationalstatisticspersonalwellbeingquestions (use encouraged; freely available) and OGL v3.0 site-wide footer; corroborated at https://analysisfunction.civilservice.gov.uk/policy-store/personal-well-being/"
   },
   "constructs_claimed": "Negative affect: anxiety felt yesterday, self-rated 0 to 10 and interpreted in the opposite direction to the three positive items.",
   "deployment_context_caveat": "",
   "question_bank_ref": {
    "resource": "question-bank",
    "item_id": "aps-anxiety",
    "note": "The question bank holds the canonical description of this item (topic, verified wording under OGL, response scale, benchmark); this record holds the evidence view."
   },
   "citations": [],
   "record_notes": "Created 2026-07-12 by the ONS-4 item split (schema v0.2 rule 4). Graded evidence lives on the ONS-4 parent record; this item record exists so the set's items are first-class and cross-linked one-to-one with the question bank.",
   "structural_validity": {
    "findings": "Not applicable at item level: this is a single item, so internal consistency and factor structure are category errors rather than gaps (schema v0.2 rule 4).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "not applicable at item level"
   },
   "convergent_discriminant_validity": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "criterion_validity_reference_standard": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Very low",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "criterion_validity_organisational": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "internal_consistency": {
    "findings": "Not applicable at item level: this is a single item, so internal consistency and factor structure are category errors rather than gaps (schema v0.2 rule 4).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "not applicable at item level"
   },
   "test_retest_reliability": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "measurement_invariance": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   },
   "responsiveness_mic": {
    "findings": "Evidence for this property is held at the item-set level and is not duplicated here; see the ONS-4 parent record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "parent",
    "indirectness": "see the ONS-4 parent record"
   }
  },
  {
   "instrument_id": "who-5",
   "display_name": "WHO-5 Well-Being Index",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "WHO-5 Well-Being Index (World Health Organization Five Well-Being Index)",
    "current_version": "WHO-5 (1998 version), the current standard short form derived from earlier WHO-10 and WHO-6/28-item well-being schedules",
    "item_count": "5 items, each positively worded, referring to the previous two weeks, rated 0 (at no time) to 5 (all of the time); raw score 0 to 25 conventionally multiplied by 4 to give 0 to 100",
    "original_citation": "Bech P, Olsen LR, Kjoller M, Rasmussen NK (2003) Measuring well-being rather than the absence of distress symptoms: a comparison of the SF-36 Mental Health subscale and the WHO-Five Well-Being Scale. International Journal of Methods in Psychiatric Research 12(2):85-91. https://doi.org/10.1002/mpr.145 (the 1998 five-item version; consolidated evidence reviewed by Topp et al. 2015, https://doi.org/10.1159/000376585)",
    "steward_publisher": "World Health Organization; the scale was developed under the WHO Regional Office for Europe and historically maintained by the Psychiatric Research Unit, Mental Health Centre North Zealand, Denmark (Bech and colleagues). Master versions and translations are distributed by WHO.",
    "licence_status": "Copyright World Health Organization 2024; the WHO-5 master version is available under the Creative Commons Attribution-NonCommercial-ShareAlike 3.0 IGO licence (CC BY-NC-SA 3.0 IGO). Document ref WHO/UCN/MSD/MHE/2024.1. Non-commercial use and adaptation permitted with attribution and share-alike; commercial (including employer/vendor) use requires WHO permission.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://cdn.who.int/media/docs/default-source/mental-health/who-5_english-original4da539d6ed4b49389e3afe47cda2326a.pdf"
   },
   "constructs_claimed": "The WHO-5 is presented as a unidimensional measure of subjective psychological (hedonic) well-being over the preceding two weeks, tapping positive mood, vitality and general interest. It is a measure of the presence of positive well-being rather than of symptom burden, although its origin and much of its validation lie in depression screening. It is a generic well-being index, not a workplace-specific or organisational instrument; workplace studies adopt it unchanged as a general well-being outcome.",
   "deployment_context_caveat": "WHO-5 originated and was validated substantially as a wellbeing and depression-screening measure in clinical and disease populations; workplace deployment is a different context, and its criterion evidence against organisational outcomes is absent.",
   "structural_validity": {
    "findings": "The WHO-5 is predominantly unidimensional: a single well-being factor is recovered across a wide range of populations and languages. The anchoring systematic review concluded high clinimetric validity and treated the scale as a coherent single dimension ([Topp 2015](https://doi.org/10.1159/000376585)). One-factor structures have since been confirmed by confirmatory factor analysis in adolescents with type 1 diabetes ([de Wit 2007](https://doi.org/10.2337/dc07-0447)), in type 1 and type 2 diabetes outpatients ([Hajos 2013](https://doi.org/10.1111/dme.12040)), in a Chinese university sample ([Fung 2022](https://doi.org/10.3389/fpubh.2022.872436)), in Chinese type 2 diabetes patients ([Du 2023](https://doi.org/10.1186/s12888-023-05381-9)), in a Sinhala community sample ([Perera 2020](https://doi.org/10.1186/s12955-020-01532-8)), in a Bangla general sample ([Faruk 2021](https://doi.org/10.1017/gmh.2021.26)) and in Portuguese adolescents ([Carvalho 2025](https://doi.org/10.1159/000543728)); principal component analysis in euthymic bipolar patients returned a single factor explaining 59.7% of variance ([Bonnín 2017](https://doi.org/10.1016/j.jad.2017.12.006)). Item response and Rasch analyses concur: the scale was unidimensional with no differential item functioning by age, sex or inpatient/outpatient status in schizophrenia spectrum disorders, although initial category disordering was only resolved by merging two middle response options ([Nielsen 2023](https://doi.org/10.1016/j.jpsychires.2023.12.028)), and a nationwide Norwegian post-discharge sample found a one-factor solution explaining 71.7% of variance ([Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)). A five-country clinimetric study (Italy, Poland, Denmark, China, Japan) using Mokken and Rasch analyses reported scalability coefficients ranging from 0.42 to 0.84 with most language versions fitting Rasch expectations, supporting unidimensionality across cultures ([Carrozzino 2022](https://doi.org/10.1016/j.jad.2022.05.111)). Two qualifications recur. First, in five-item models the RMSEA is frequently elevated (for example 0.23 in an Australian diabetes sample despite CFI 0.98 and loadings 0.78 to 0.92, [Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005); elevated again in the Norwegian sample, [Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)), an artefact of very low degrees of freedom rather than clear misfit. Second, and more substantively, in a 15-country adolescent study the full five-item WHO-5 did not achieve a good measurement model; a four-item version dropping the first item ('cheerful and in good spirits') was needed ([Cosma 2022](https://doi.org/10.3390/ijerph19169798)), a finding echoed by a Japanese child study that also set that item aside on cultural grounds ([Adachi 2025](https://doi.org/10.3389/fpubh.2025.1662332)).",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High: many good-quality CFA/IRT studies, large total N, consistently unidimensional in adults, with a documented five-item-model RMSEA artefact and a contested first item in some adolescent/cross-cultural samples."
   },
   "convergent_discriminant_validity": {
    "findings": "The WHO-5 correlates strongly and negatively with depression measures and positively with other well-being measures, as expected for a well-being index that shares variance with low mood. Against depression screeners and symptom scales it shows large negative correlations: r = -0.73 with the PHQ-9 in Australian adults with diabetes ([Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005)), r = -0.694 with the PHQ-9 and r = -0.610 with the Hamilton Depression Rating Scale in Chinese type 2 diabetes patients ([Du 2023](https://doi.org/10.1186/s12888-023-05381-9)), and r = -0.67 with the CES-D in adolescents with type 1 diabetes ([de Wit 2007](https://doi.org/10.2337/dc07-0447)); moderate-to-strong correlations of 0.55 to 0.69 with PHQ, diabetes-distress and SF-12 mental component scores were reported in diabetes outpatients ([Hajos 2013](https://doi.org/10.1111/dme.12040)). It relates very strongly to depression severity even after controlling for anxiety, supporting some discriminant separation from anxiety ([Krieger 2013](https://doi.org/10.1016/j.jad.2013.12.015)). Convergent evidence with positive constructs includes r = 0.542 with the Warwick-Edinburgh Mental Well-Being Scale alongside a smaller divergent r = -0.443 with perceived stress ([Faruk 2021](https://doi.org/10.1017/gmh.2021.26)), and strong associations with life satisfaction and meaning in life against weaker links to physical health ([Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)). Correlations are more modest against broader distress or family-function measures: r = -0.45 with the PHQ-9 and -0.56 with the Kessler K10 in a Sinhala sample ([Perera 2020](https://doi.org/10.1186/s12955-020-01532-8)), and r = 0.35 with a family-function measure in older Peruvian adults ([Del Pilar Diaz-Nunez P 2025](https://doi.org/10.3389/fpubh.2025.1670429)). Convergent construct validity with well-being, self-efficacy and self-esteem measures was also supported in a Chinese sample ([Fung 2022](https://doi.org/10.3389/fpubh.2022.872436)).",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High: numerous studies, consistent direction and plausible magnitudes; discriminant evidence against anxiety is thinner than convergent evidence against depression."
   },
   "criterion_validity_reference_standard": {
    "findings": "Criterion validity is well established against one target, depression case status, and is essentially absent against organisational outcomes. As a depression screener the WHO-5 performs well: against structured diagnostic interviews or established depression scales, reported areas under the ROC curve cluster around 0.81 to 0.88 (AUC 0.87 in Australian diabetes adults, [Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005); 0.823 against clinical interview in Iranian students, [Ghazisaeedi 2021](https://doi.org/10.1007/s11469-021-00483-5); 0.882 in Chinese healthcare students, [Yang 2023](https://doi.org/10.2147/PRBM.S437219); 0.838 in schizophrenia, [Fekih-Romdhane 2024](https://doi.org/10.1186/s12888-024-05814-z)). Sensitivity and specificity depend heavily on the cut-off: a cut-off below 13 gave 0.79/0.79 while below 8 gave 0.44/0.96 in the same Australian sample ([Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005)), a cut-off below 50 (on the 0 to 100 metric) gave 0.79 to 0.88 sensitivity in Dutch diabetes outpatients ([Hajos 2013](https://doi.org/10.1111/dme.12040)) and 0.89/0.86 in adolescents with type 1 diabetes ([de Wit 2007](https://doi.org/10.2337/dc07-0447)); the anchoring review summarised the scale as a sensitive and specific depression screen across fields ([Topp 2015](https://doi.org/10.1159/000376585)). Importantly, this criterion evidence was earned in diagnostic and clinical-population settings, which differ from workplace deployment where no diagnostic gold standard is applied. Against organisational and health outcomes proper (sickness absence, staff turnover, diagnosed conditions, productivity) no criterion-validity study was located in this pass; workplace papers instead report cross-sectional or prospective associations, for example poor WHO-5 well-being being more common with low workplace social capital ([Gao 2014](https://doi.org/10.1371/journal.pone.0085005)), with adverse psychosocial work factors across 34 European countries ([Schütte 2014](https://doi.org/10.1007/s00420-014-0930-0)) and prospectively in France ([Bertrais 2021](https://doi.org/10.1177/14034948211008385)), lower well-being among self-employed than salaried workers in small enterprises ([Park 2025](https://doi.org/10.3349/ymj.2024.0441)), and higher odds of poor well-being with lower-quality supervisor behaviour and workplace social capital across 35 European countries ([Kizuki 2020](https://doi.org/10.1093/occmed/kqaa070)). These are construct-relevant associations, not criterion validation against an organisational gold standard.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: criterion validity against depression is strong and consistent but almost entirely from clinical and disease populations; criterion validity against organisational outcomes is Absent (no such study located)."
   },
   "criterion_validity_organisational": {
    "findings": "Organisational criterion evidence (sickness absence, turnover, performance, diagnosed conditions in a work context): see the criterion findings; graded from the pass-one record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: criterion validity against depression is strong and consistent but almost entirely from clinical and disease populations; criterion validity against organisational outcomes is Absent (no such study located)."
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently good to excellent. Cronbach's alpha values cluster in the low-to-high 0.80s to low 0.90s across populations: alpha 0.90 in Australian adults with diabetes ([Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005)), 0.88 in Chinese type 2 diabetes patients ([Du 2023](https://doi.org/10.1186/s12888-023-05381-9)), 0.82 in adolescents with type 1 diabetes ([de Wit 2007](https://doi.org/10.2337/dc07-0447)), 0.83 in euthymic bipolar patients ([Bonnín 2017](https://doi.org/10.1016/j.jad.2017.12.006)), 0.85 in a Sinhala community sample ([Perera 2020](https://doi.org/10.1186/s12955-020-01532-8)), 0.81 to 0.90 across three countries of nurses ([Lara-Cabrera 2022](https://doi.org/10.3390/ijerph191610106)), 0.80 in Portuguese adolescents ([Carvalho 2025](https://doi.org/10.1159/000543728)) and 0.80 in Arabic-speaking schizophrenia patients ([Fekih-Romdhane 2024](https://doi.org/10.1186/s12888-024-05814-z)). Higher values around 0.91 to 0.94 appear in Iranian students ([Ghazisaeedi 2021](https://doi.org/10.1007/s11469-021-00483-5)), a Norwegian post-discharge sample ([Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)) and Chinese healthcare students ([Yang 2023](https://doi.org/10.2147/PRBM.S437219)). Where reported, McDonald's omega agrees closely with alpha (omega 0.84 in Luxembourg adolescents, [Brisson 2025](https://doi.org/10.1080/00223891.2025.2569138); omega 0.86 to 0.91 in Japanese children, [Adachi 2025](https://doi.org/10.3389/fpubh.2025.1662332); omega 0.908 to 0.935 in Chinese healthcare students, [Yang 2023](https://doi.org/10.2147/PRBM.S437219)). A lower alpha of 0.75 was reported for the Bangla version ([Faruk 2021](https://doi.org/10.1017/gmh.2021.26)). Coefficient omega was the reliability index in the updated German norms study ([Kliem 2025](https://doi.org/10.3389/fpsyg.2025.1592614)).",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High: many studies, large total N, alpha and omega both reported and consistently adequate to excellent across diverse populations."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Low",
    "status": "thin",
    "indirectness": "see summary",
    "summary": "Test-retest reliability is the weakest-evidenced reliability property for the WHO-5 and deserves explicit flagging: dedicated studies are few, intervals are short, and the single study designed specifically to estimate it raised a measurement-error concern. The one purpose-designed test-retest and measurement-error study, in Danish patients with type 1 diabetes using a median five-day interval, found an intraclass correlation of 0.87 (95% CI 0.82 to 0.90) but a large minimal detectable change of 18.56 points on the 0 to 100 scale, which the authors judged a larger measurement error than desirable and flagged for further research ([Schougaard 2022](https://doi.org/10.1186/s41687-022-00505-3)). Other estimates are incidental to validation studies and mostly over about two weeks: Pearson r = 0.72 with ICC 0.82 over two weeks in a Sinhala sample ([Perera 2020](https://doi.org/10.1186/s12955-020-01532-8)), r = 0.83 over ten days in euthymic bipolar patients ([Bonnín 2017](https://doi.org/10.1016/j.jad.2017.12.006)), a retest coefficient of 0.71 in the Bangla validation ([Faruk 2021](https://doi.org/10.1017/gmh.2021.26)) and ICC 0.803 over about one week in Chinese healthcare students ([Yang 2023](https://doi.org/10.2147/PRBM.S437219)). No UK-based or workplace-specific test-retest estimate was located in this pass, and intervals long enough to separate stability from genuine change in well-being are not represented.",
    "confidence_note": "Low: only one purpose-designed study (which itself flagged a large minimal detectable change), remaining estimates incidental with short intervals, none from workplace or UK samples."
   },
   "measurement_invariance": {
    "findings": "Measurement invariance across sex and age is generally supported in adults, with more mixed results across countries and in adolescents. In adults, configural, metric and scalar invariance across sex, age and education was reported in a nationwide Norwegian post-discharge sample ([Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)); invariance across gender and age was supported in the representative German norms study ([Kliem 2025](https://doi.org/10.3389/fpsyg.2025.1592614)); configural, metric and scalar invariance across sex held in a Sinhala community sample ([Perera 2020](https://doi.org/10.1186/s12955-020-01532-8)); and equivalence across sex was shown in older Peruvian adults ([Del Pilar Diaz-Nunez P 2025](https://doi.org/10.3389/fpubh.2025.1670429)) and cross-sex invariance in Arabic-speaking schizophrenia patients ([Fekih-Romdhane 2024](https://doi.org/10.1186/s12888-024-05814-z)). In Chinese healthcare students the scale was invariant across eleven sociodemographic groupings and longitudinally over one week ([Yang 2023](https://doi.org/10.2147/PRBM.S437219)). Strong measurement invariance across the presence or absence of a current major depressive episode was also demonstrated ([Krieger 2013](https://doi.org/10.1016/j.jad.2013.12.015)). Cross-national and adolescent evidence is more qualified. Across 15 European countries the five-item model did not achieve acceptable cross-country invariance in adolescents, and a four-item version was required for valid comparison ([Cosma 2022](https://doi.org/10.3390/ijerph19169798)); across 43 countries an IRT analysis found many item parameters non-invariant although overall differential test functioning was only modest ([Sischka 2025](https://doi.org/10.1177/10731911241309452)). Full scalar invariance was reported across academic, birth-country, language, sex, socioeconomic and school subgroups in Luxembourg adolescents ([Brisson 2025](https://doi.org/10.1080/00223891.2025.2569138)) and across age and gender in Japanese children ([Adachi 2025](https://doi.org/10.3389/fpubh.2025.1662332)). No study located in this pass tested invariance across occupational groups or between working and non-working populations.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: scalar invariance across sex and age is repeatedly demonstrated in adults and large adolescent samples, but cross-country invariance of the five-item form is contested and occupation-based invariance is untested."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness is asserted more often than it is quantified with a defined minimal important change. The anchoring review concluded the WHO-5 is responsive and can serve as an outcome measure in controlled clinical trials, balancing wanted and unwanted treatment effects ([Topp 2015](https://doi.org/10.1159/000376585)), and a 2025 disease-area review across 552 studies reported that the scale detects treatment-related changes in well-being across many conditions ([Domenech 2025](https://doi.org/10.1007/s12325-025-03266-9)). Direct workplace responsiveness evidence is limited to small intervention studies: a phase-II stress-preventive leadership intervention in a German hospital reported improved WHO-5 well-being over three months ([Stuber 2022](https://doi.org/10.1136/bmjopen-2021-049951)). A formal minimal important change for the WHO-5 was not located in this pass; the closest quantitative anchor is the measurement-error study's minimal detectable change of 18.56 points on the 0 to 100 scale, which is a distribution-based detectable-change threshold rather than an anchor-based minimal important change and was itself flagged as large ([Schougaard 2022](https://doi.org/10.1186/s41687-022-00505-3)).",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low: responsiveness is supported qualitatively by two large reviews and small intervention studies, but a validated anchor-based minimal important change was not located, and workplace-specific responsiveness rests on small samples."
   },
   "populations_languages_norms": {
    "findings": "The WHO-5 has been translated into more than 30 languages and validated across an unusually broad range of populations ([Topp 2015](https://doi.org/10.1159/000376585)); a 2025 review catalogued its use across essentially all major disease areas in 552 studies ([Domenech 2025](https://doi.org/10.1007/s12325-025-03266-9)). Validation samples located in this pass span general community adults (German, [Kliem 2025](https://doi.org/10.3389/fpsyg.2025.1592614); Sinhala, [Perera 2020](https://doi.org/10.1186/s12955-020-01532-8); Bangla, [Faruk 2021](https://doi.org/10.1017/gmh.2021.26)), older adults ([Bonsignore 2001](https://doi.org/10.1007/BF03035123); [Del Pilar Diaz-Nunez P 2025](https://doi.org/10.3389/fpubh.2025.1670429)), adolescents and children (43-country and 15-country adolescent studies, [Sischka 2025](https://doi.org/10.1177/10731911241309452), [Cosma 2022](https://doi.org/10.3390/ijerph19169798); Luxembourg, [Brisson 2025](https://doi.org/10.1080/00223891.2025.2569138); Portugal, [Carvalho 2025](https://doi.org/10.1159/000543728); Japan, [Adachi 2025](https://doi.org/10.3389/fpubh.2025.1662332)), diabetes populations ([de Wit 2007](https://doi.org/10.2337/dc07-0447), [Hajos 2013](https://doi.org/10.1111/dme.12040), [Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005), [Du 2023](https://doi.org/10.1186/s12888-023-05381-9)), severe mental illness ([Nielsen 2023](https://doi.org/10.1016/j.jpsychires.2023.12.028), [Fekih-Romdhane 2024](https://doi.org/10.1186/s12888-024-05814-z), [Bonnín 2017](https://doi.org/10.1016/j.jad.2017.12.006), [Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)) and occupational groups (nurses, [Lara-Cabrera 2022](https://doi.org/10.3390/ijerph191610106); medical educators, [Chan 2022](https://doi.org/10.1080/10872981.2022.2044635); European employees, [Schütte 2014](https://doi.org/10.1007/s00420-014-0930-0)). Updated general-population norms exist for Germany ([Kliem 2025](https://doi.org/10.3389/fpsyg.2025.1592614)). No UK-specific validation, UK normative benchmark or UK workplace norm was located in this pass; practitioners needing a UK reference distribution would be relying on non-UK norms, most directly the German representative norms, which is an indirectness limitation for a UK workplace audience.",
    "grade": "High",
    "indirectness": "see findings"
   },
   "criticisms_controversies": "Four themes recur in the critical literature. First, the identity of the construct: although marketed as a positive well-being index, the WHO-5's validation base is dominated by depression screening, and it correlates so strongly with depression measures (for example r = -0.73, [Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005)) that some authors treat it as effectively a measure of the severity of depression ([Krieger 2013](https://doi.org/10.1016/j.jad.2013.12.015)). This matters for a workplace registry: the scale's criterion evidence was earned in diagnostic and disease settings, so using it to screen a workforce is deploying a clinically validated instrument outside the diagnostic context in which that validity was established, without a diagnostic gold standard in play. Second, the first item ('cheerful and in good spirits') is repeatedly identified as problematic in cross-cultural and adolescent samples, to the point that a four-item WHO-4 has been proposed for valid cross-country adolescent comparison ([Cosma 2022](https://doi.org/10.3390/ijerph19169798)) and adopted on cultural grounds elsewhere ([Adachi 2025](https://doi.org/10.3389/fpubh.2025.1662332)). Third, cut-off scores for likely depression vary substantially between studies and populations (for example optimal cut-offs corresponding to different thresholds across [Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005), [Ghazisaeedi 2021](https://doi.org/10.1007/s11469-021-00483-5), [Du 2023](https://doi.org/10.1186/s12888-023-05381-9), [Fekih-Romdhane 2024](https://doi.org/10.1186/s12888-024-05814-z)), so no single cut-off transfers automatically to a new setting. Fourth, structural analyses frequently report an elevated RMSEA in the five-item model and occasional response-category disordering ([Halliday 2017](https://doi.org/10.1016/j.diabres.2017.07.005), [Nielsen 2023](https://doi.org/10.1016/j.jpsychires.2023.12.028), [Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)), and floor effects on some items in low-well-being clinical samples ([Iversen 2025](https://doi.org/10.1007/s11136-025-04104-9)). Measurement error over short retest intervals has been flagged as larger than desirable ([Schougaard 2022](https://doi.org/10.1186/s41687-022-00505-3)).",
   "citations": [
    {
     "authors": "Topp, Østergaard, Søndergaard, et al.",
     "year": "2015",
     "title": "The WHO-5 Well-Being Index: a systematic review of the literature.",
     "journal": "Psychotherapy and psychosomatics",
     "doi": "10.1159/000376585",
     "key": "topp2015",
     "url": "https://doi.org/10.1159/000376585"
    },
    {
     "authors": "Domenech, Kasujee, Koscielny, et al.",
     "year": "2025",
     "title": "Systematic Review of the Use of the WHO-5 Well-Being Index Across Different Disease Areas.",
     "journal": "Advances in therapy",
     "doi": "10.1007/s12325-025-03266-9",
     "key": "domenech2025",
     "url": "https://doi.org/10.1007/s12325-025-03266-9"
    },
    {
     "key": "bech2003",
     "authors": "Bech P, Olsen LR, Kjoller M, Rasmussen NK",
     "year": "2003",
     "title": "Measuring well-being rather than the absence of distress symptoms: a comparison of the SF-36 Mental Health subscale and the WHO-Five Well-Being Scale",
     "journal": "International Journal of Methods in Psychiatric Research",
     "doi": "10.1002/mpr.145",
     "url": "https://doi.org/10.1002/mpr.145"
    },
    {
     "authors": "Bonsignore, Barkow, Jessen, et al.",
     "year": "2001",
     "title": "Validity of the five-item WHO Well-Being Index (WHO-5) in an elderly population.",
     "journal": "European archives of psychiatry and clinical neuroscience",
     "doi": "10.1007/BF03035123",
     "key": "bonsignore2001",
     "url": "https://doi.org/10.1007/BF03035123"
    },
    {
     "authors": "de Wit, Pouwer, Gemke, et al.",
     "year": "2007",
     "title": "Validation of the WHO-5 Well-Being Index in adolescents with type 1 diabetes.",
     "journal": "Diabetes care",
     "doi": "10.2337/dc07-0447",
     "key": "dewit2007",
     "url": "https://doi.org/10.2337/dc07-0447"
    },
    {
     "authors": "Hajos, Pouwer, Skovlund, et al.",
     "year": "2013",
     "title": "Psychometric and screening properties of the WHO-5 well-being index in adult outpatients with Type 1 or Type 2 diabetes mellitus.",
     "journal": "Diabetic medicine : a journal of the British Diabetic Association",
     "doi": "10.1111/dme.12040",
     "key": "hajos2013",
     "url": "https://doi.org/10.1111/dme.12040"
    },
    {
     "authors": "Halliday, Hendrieckx, Busija, et al.",
     "year": "2017",
     "title": "Validation of the WHO-5 as a first-step screening instrument for depression in adults with diabetes: Results from Diabetes MILES - Australia.",
     "journal": "Diabetes research and clinical practice",
     "doi": "10.1016/j.diabres.2017.07.005",
     "key": "halliday2017",
     "url": "https://doi.org/10.1016/j.diabres.2017.07.005"
    },
    {
     "authors": "Ghazisaeedi, Mahmoodi, Arpaci, et al.",
     "year": "2021",
     "title": "Validity, Reliability, and Optimal Cut-off Scores of the WHO-5, PHQ-9, and PHQ-2 to Screen Depression Among University Students in Iran.",
     "journal": "International journal of mental health and addiction",
     "doi": "10.1007/s11469-021-00483-5",
     "key": "ghazisaeedi2021",
     "url": "https://doi.org/10.1007/s11469-021-00483-5"
    },
    {
     "authors": "Krieger, Zimmermann, Huffziger, et al.",
     "year": "2013",
     "title": "Measuring depression with a well-being index: further evidence for the validity of the WHO Well-Being Index (WHO-5) as a measure of the severity of depression.",
     "journal": "Journal of affective disorders",
     "doi": "10.1016/j.jad.2013.12.015",
     "key": "krieger2014",
     "url": "https://doi.org/10.1016/j.jad.2013.12.015"
    },
    {
     "authors": "Carrozzino, Christensen, Patierno, et al.",
     "year": "2022",
     "title": "Cross-cultural validity of the WHO-5 Well-Being Index and Euthymia Scale: A clinimetric analysis.",
     "journal": "Journal of affective disorders",
     "doi": "10.1016/j.jad.2022.05.111",
     "key": "carrozzino2022",
     "url": "https://doi.org/10.1016/j.jad.2022.05.111"
    },
    {
     "authors": "Nielsen, Lauridsen, Østergaard, et al.",
     "year": "2023",
     "title": "Structural validity of the 5-item World Health Organization Well-being Index (WHO-5) in patients with schizophrenia spectrum disorders.",
     "journal": "Journal of psychiatric research",
     "doi": "10.1016/j.jpsychires.2023.12.028",
     "key": "nielsen2023",
     "url": "https://doi.org/10.1016/j.jpsychires.2023.12.028"
    },
    {
     "authors": "Kliem, Lohmann, Fischer, et al.",
     "year": "2025",
     "title": "Psychometric evaluation and updated community norms of the WHO-5 well-being index, based on a representative German sample.",
     "journal": "Frontiers in psychology",
     "doi": "10.3389/fpsyg.2025.1592614",
     "key": "kliem2025",
     "url": "https://doi.org/10.3389/fpsyg.2025.1592614"
    },
    {
     "authors": "Cosma, Költő, Chzhen, et al.",
     "year": "2022",
     "title": "Measurement Invariance of the WHO-5 Well-Being Index: Evidence from 15 European Countries.",
     "journal": "International journal of environmental research and public health",
     "doi": "10.3390/ijerph19169798",
     "key": "cosma2022",
     "url": "https://doi.org/10.3390/ijerph19169798"
    },
    {
     "authors": "Sischka, Martin, Residori, et al.",
     "year": "2025",
     "title": "Cross-National Validation of the WHO-5 Well-Being Index Within Adolescent Populations: Findings From 43 Countries.",
     "journal": "Assessment",
     "doi": "10.1177/10731911241309452",
     "key": "sischka2025",
     "url": "https://doi.org/10.1177/10731911241309452"
    },
    {
     "authors": "Brisson",
     "year": "2025",
     "title": "Psychometric Evaluation and Sociodemographic Measurement Invariance of the WHO-5 Well-Being Index among Adolescents in Luxembourg.",
     "journal": "Journal of personality assessment",
     "doi": "10.1080/00223891.2025.2569138",
     "key": "brisson2025",
     "url": "https://doi.org/10.1080/00223891.2025.2569138"
    },
    {
     "authors": "Schougaard, Laurberg, Lomborg, et al.",
     "year": "2022",
     "title": "Test-retest reliability and measurement error of the WHO-5 Well-being Index and the Problem Areas in Diabetes questionnaire (PAID) used in telehealth among patients with type 1 diabetes.",
     "journal": "Journal of patient-reported outcomes",
     "doi": "10.1186/s41687-022-00505-3",
     "key": "schougaard2022",
     "url": "https://doi.org/10.1186/s41687-022-00505-3"
    },
    {
     "authors": "Bonnín, Yatham, Michalak, et al.",
     "year": "2017",
     "title": "Psychometric properties of the well-being index (WHO-5) spanish version in a sample of euthymic patients with bipolar disorder.",
     "journal": "Journal of affective disorders",
     "doi": "10.1016/j.jad.2017.12.006",
     "key": "bonnin2017",
     "url": "https://doi.org/10.1016/j.jad.2017.12.006"
    },
    {
     "authors": "Fung, Kong, Liu, et al.",
     "year": "2022",
     "title": "Validity and Psychometric Evaluation of the Chinese Version of the 5-Item WHO Well-Being Index.",
     "journal": "Frontiers in public health",
     "doi": "10.3389/fpubh.2022.872436",
     "key": "fung2022",
     "url": "https://doi.org/10.3389/fpubh.2022.872436"
    },
    {
     "authors": "Perera, Jayasuriya, Caldera, et al.",
     "year": "2020",
     "title": "Assessing mental well-being in a Sinhala speaking Sri Lankan population: validation of the WHO-5 well-being index.",
     "journal": "Health and quality of life outcomes",
     "doi": "10.1186/s12955-020-01532-8",
     "key": "perera2020",
     "url": "https://doi.org/10.1186/s12955-020-01532-8"
    },
    {
     "authors": "Faruk, Alam, Chowdhury, et al.",
     "year": "2021",
     "title": "Validation of the Bangla WHO-5 Well-being Index.",
     "journal": "Global mental health (Cambridge, England)",
     "doi": "10.1017/gmh.2021.26",
     "key": "faruk2021",
     "url": "https://doi.org/10.1017/gmh.2021.26"
    },
    {
     "authors": "Lara-Cabrera, Betancort, Muñoz-Rubilar, et al.",
     "year": "2022",
     "title": "Psychometric Properties of the WHO-5 Well-Being Index among Nurses during the COVID-19 Pandemic: A Cross-Sectional Study in Three Countries.",
     "journal": "International journal of environmental research and public health",
     "doi": "10.3390/ijerph191610106",
     "key": "laracabrera2022",
     "url": "https://doi.org/10.3390/ijerph191610106"
    },
    {
     "authors": "Yang, Ma, Huang, et al.",
     "year": "2023",
     "title": "Measurement Properties and Optimal Cutoff Point of the WHO-5 Among Chinese Healthcare Students.",
     "journal": "Psychology research and behavior management",
     "doi": "10.2147/PRBM.S437219",
     "key": "yang2023",
     "url": "https://doi.org/10.2147/PRBM.S437219"
    },
    {
     "authors": "Du, Jiang, Lloyd, et al.",
     "year": "2023",
     "title": "Validation of Chinese version of the 5-item WHO well-being index in type 2 diabetes mellitus patients.",
     "journal": "BMC psychiatry",
     "doi": "10.1186/s12888-023-05381-9",
     "key": "du2023",
     "url": "https://doi.org/10.1186/s12888-023-05381-9"
    },
    {
     "authors": "Fekih-Romdhane, Al Mouzakzak, Abilmona, et al.",
     "year": "2024",
     "title": "Validation and optimal cut-off score of the World Health Organization Well-being Index (WHO-5) as a screening tool for depression among patients with schizophrenia.",
     "journal": "BMC psychiatry",
     "doi": "10.1186/s12888-024-05814-z",
     "key": "fekih2024",
     "url": "https://doi.org/10.1186/s12888-024-05814-z"
    },
    {
     "authors": "Iversen, Kjøllesdal, Ellingsen-Dalskau, et al.",
     "year": "2025",
     "title": "Psychometric performance of the WHO-5 well-being index in a nationwide sample of inpatients discharged from specialised mental health care.",
     "journal": "Quality of life research : an international journal of quality of life aspects of treatment, care and rehabilitation",
     "doi": "10.1007/s11136-025-04104-9",
     "key": "iversen2025",
     "url": "https://doi.org/10.1007/s11136-025-04104-9"
    },
    {
     "authors": "Adachi, Takahashi, Mori, et al.",
     "year": "2025",
     "title": "Psychometric validation of the WHO-5 and WHO-4 well-being index scales for assessing psychological well-being and detecting depression in Japanese school-aged children: a community-based study.",
     "journal": "Frontiers in public health",
     "doi": "10.3389/fpubh.2025.1662332",
     "key": "adachi2025",
     "url": "https://doi.org/10.3389/fpubh.2025.1662332"
    },
    {
     "authors": "Gao, Weaver, Dai, et al.",
     "year": "2014",
     "title": "Workplace social capital and mental health among Chinese employees: a multi-level, cross-sectional study.",
     "journal": "PloS one",
     "doi": "10.1371/journal.pone.0085005",
     "key": "gao2014",
     "url": "https://doi.org/10.1371/journal.pone.0085005"
    },
    {
     "authors": "Schütte, Chastang, Malard, et al.",
     "year": "2014",
     "title": "Psychosocial working conditions and psychological well-being among employees in 34 European countries.",
     "journal": "International archives of occupational and environmental health",
     "doi": "10.1007/s00420-014-0930-0",
     "key": "schutte2014",
     "url": "https://doi.org/10.1007/s00420-014-0930-0"
    },
    {
     "authors": "Kizuki, Fujiwara",
     "year": "2020",
     "title": "Quality of supervisor behaviour, workplace social capital and psychological well-being.",
     "journal": "Occupational medicine (Oxford, England)",
     "doi": "10.1093/occmed/kqaa070",
     "key": "kizuki2020",
     "url": "https://doi.org/10.1093/occmed/kqaa070"
    },
    {
     "authors": "Bertrais, HÉRault, Chastang, et al.",
     "year": "2021",
     "title": "Multiple psychosocial work exposures and well-being among employees: prospective associations from the French national Working Conditions Survey.",
     "journal": "Scandinavian journal of public health",
     "doi": "10.1177/14034948211008385",
     "key": "bertrais2021",
     "url": "https://doi.org/10.1177/14034948211008385"
    },
    {
     "authors": "Stuber, Seifried-Dübon, Tsarouha, et al.",
     "year": "2022",
     "title": "Feasibility, psychological outcomes and practical use of a stress-preventive leadership intervention in the workplace hospital: the results of a mixed-method phase-II study.",
     "journal": "BMJ open",
     "doi": "10.1136/bmjopen-2021-049951",
     "key": "stuber2022",
     "url": "https://doi.org/10.1136/bmjopen-2021-049951"
    },
    {
     "authors": "Park, Kim, Sung",
     "year": "2025",
     "title": "Factors Affecting Subjective Well-Being in Workers at Small-Sized Enterprises: A Cross-Sectional Study from the 6th Korean Working Conditions Survey.",
     "journal": "Yonsei medical journal",
     "doi": "10.3349/ymj.2024.0441",
     "key": "park2025",
     "url": "https://doi.org/10.3349/ymj.2024.0441"
    },
    {
     "authors": "Chan, Liu, Lam, et al.",
     "year": "2022",
     "title": "Validation of the World Health Organization Well-Being Index (WHO-5) among medical educators in Hong Kong: a confirmatory factor analysis.",
     "journal": "Medical education online",
     "doi": "10.1080/10872981.2022.2044635",
     "key": "chan2022",
     "url": "https://doi.org/10.1080/10872981.2022.2044635"
    },
    {
     "authors": "Carvalho, Vieira Martins, Azevedo, et al.",
     "year": "2025",
     "title": "World Health Organization's Well-Being Index - WHO-5: Psychometric Performance of the Portuguese Version for Adolescents.",
     "journal": "Portuguese journal of public health",
     "doi": "10.1159/000543728",
     "key": "carvalho2025",
     "url": "https://doi.org/10.1159/000543728"
    },
    {
     "key": "delpilar2025",
     "authors": "Del Pilar Diaz-Nunez P, Dominguez-Lara S, et al.",
     "year": "2025",
     "title": "Psychometric evidence of the WHO-5 well-being index in a sample of participants from hospitals and older adults care centers in Peru",
     "journal": "Frontiers in Public Health",
     "doi": "10.3389/fpubh.2025.1670429",
     "url": "https://doi.org/10.3389/fpubh.2025.1670429"
    },
    {
     "key": "laracabreraprot2020",
     "authors": "Lara-Cabrera ML, Bjorngaard JH, Salvesen O, et al.",
     "year": "2020",
     "title": "Psychometric properties of the Five-item World Health Organization Well-being Index used in mental health services: Protocol for a systematic review",
     "journal": "Journal of Advanced Nursing",
     "doi": "10.1111/jan.14445",
     "url": "https://doi.org/10.1111/jan.14445"
    }
   ],
   "record_notes": "[Upgraded from v0.1 to v0.2 structure in pass two; criterion field split, licence re-verified 2026-07-12.] Overall confidence in the WHO-5 as a well-being measure is High for structural validity, internal consistency, convergent validity and breadth of populations/languages; Moderate for measurement invariance (adults yes, cross-country five-item form contested); Low for test-retest reliability and for responsiveness with a defined minimal important change; and effectively Absent for criterion validity against organisational outcomes and for UK-specific norms. Schema stress-test notes: (1) The schema's single 'criterion_validity' field forced two very different evidence states into one cell, namely strong criterion evidence against depression versus no located evidence against organisational outcomes (absence, turnover, diagnosed conditions). I reported both explicitly and graded to the audience's actual need, but a registry might benefit from separating clinical-criterion from organisational-criterion validity. (2) The WHO-5 versus WHO-4 question sits awkwardly: the WHO-4 is a proposed reduced form, not a separate instrument, and evidence about item 1 belongs partly under structural validity, partly under invariance, and partly under criticisms; I have cross-referenced rather than duplicated. (3) The clinical-origin caveat required by the brief cuts across identity, criterion validity and criticisms; I have stated it in each place rather than confining it to one field. (4) Licensing 'free to use' is well supported in the secondary literature but I did not retrieve a formal WHO licence document this session; the claim rests on peer-reviewed statements, so I have qualified it accordingly. (5) No UK data of any kind (validation, norms, workplace) surfaced; for a UK workplace audience the entire evidence base is indirect on locale, which I have flagged under populations and as a downgrade rationale."
  },
  {
   "instrument_id": "wemwbs",
   "display_name": "Warwick-Edinburgh Mental Wellbeing Scale (WEMWBS) and Short Warwick-Edinburgh Mental Wellbeing Scale (SWEMWBS)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Warwick-Edinburgh Mental Wellbeing Scale (WEMWBS); Short Warwick-Edinburgh Mental Wellbeing Scale (SWEMWBS)",
    "current_version": "WEMWBS (14-item, 2007) and SWEMWBS (7-item, Rasch-derived, 2009). Both use a five-category response frame (none of the time to all of the time) over a two-week recall window.",
    "item_count": "14 items (WEMWBS); 7 items (SWEMWBS). WEMWBS is summed to a 14 to 70 raw score; SWEMWBS is summed then converted to a Rasch interval-scale metric via a published transformation table.",
    "original_citation": "WEMWBS: Tennant R, Hiller L, Fishwick R, Platt S, Joseph S, Weich S, Parkinson J, Secker J, Stewart-Brown S (2007). The Warwick-Edinburgh Mental Well-being Scale (WEMWBS): development and UK validation. Health and Quality of Life Outcomes 5:63. DOI 10.1186/1477-7525-5-63. SWEMWBS: Stewart-Brown S, Tennant A, Tennant R, Platt S, Parkinson J, Weich S (2009). Internal construct validity of the WEMWBS: a Rasch analysis. Health and Quality of Life Outcomes 7:15. DOI 10.1186/1477-7525-7-15.",
    "steward_publisher": "University of Warwick (Warwick Medical School; licensing administered by Warwick Innovations), with copyright held jointly by NHS Health Scotland, the University of Warwick and the University of Edinburgh.",
    "licence_status": "CONFIRMED with a material currency update. Warwick Innovations' current Licences & Pricing page still requires registration and separates non-commercial from commercial licences, but records that 'from 1st December 2024 charges were introduced for NHS organisations, including NHS trusts, GP surgeries and other organisations currently funded by the NHS', with a published NHS pricing tier (for example, up to 30 participants GBP 45; up to 100 GBP 125; up to 1000 GBP 600). NHS use is therefore no longer free at point of use.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://warwick.ac.uk/services/innovations/wemwbs/licenses/ (Warwick Innovations WEMWBS Licences & Pricing)"
   },
   "constructs_claimed": "WEMWBS claims to measure mental wellbeing as a single positive construct, deliberately covering both the hedonic (subjective happiness, positive affect, life satisfaction) and eudaimonic (positive psychological functioning, autonomy, competence, good relationships) traditions, using exclusively positively worded items and no symptom or deficit content ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). It is framed as a population-level monitoring instrument rather than an individual diagnostic or screening tool. The seven-item SWEMWBS retains a subset weighted more toward the functioning aspect of wellbeing than the feeling aspect ([Stewart-Brown 2009](https://doi.org/10.1186/1477-7525-7-15)).",
   "deployment_context_caveat": "",
   "structural_validity": {
    "findings": "WEMWBS was designed as a unidimensional scale and the original UK validation reported that confirmatory factor analysis supported a single-factor solution ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). The picture is more nuanced under stricter modelling. A Rasch analysis of Scottish Health Education Population Survey data (n=779) found the full 14-item set did not fit the Rasch model; sequential removal of misfitting items produced a strictly unidimensional seven-item scale, SWEMWBS, that provides an interval-scale estimate of wellbeing, with the 14-item and 7-item raw scores correlating 0.954 ([Stewart-Brown 2009](https://doi.org/10.1186/1477-7525-7-15)). Independent Rasch work in the UK veterinary profession reproduced this pattern: the 14 items deviated significantly from Rasch expectations while a 7-item SWEMWBS achieved acceptable fit (person separation index 0.832) ([Bartram 2013](https://doi.org/10.1007/s11136-012-0144-4)). In classical CFA terms several validation studies report that a clean single factor emerges only after correlated residuals are allowed between items ([Smith 2017](https://doi.org/10.1186/s12888-017-1343-x), [Fung 2019](https://doi.org/10.1186/s12955-019-1113-1)), and the German Mental Health Surveillance validation could not confirm strict unidimensionality for either version, with a bifactor model (one general wellbeing factor plus grouping factors) fitting best ([Peitz 2024](https://doi.org/10.1186/s12955-024-02304-4)). Recent large surveillance datasets in the UK, Denmark and Catalonia ([Yadav 2025](https://doi.org/10.1136/bmjment-2024-301433)) and in Canada ([Capaldi 2026](https://doi.org/10.25318/82-003-x202600300002-eng)) converge on the scale being 'essentially unidimensional', typically via bifactor models, which supports summing to a single score while acknowledging minor multidimensionality. The consistent bottom line is that SWEMWBS has the stronger unidimensionality credentials, and that the 14-item WEMWBS is treated as effectively unidimensional rather than strictly so.",
    "grade": "High",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High: multiple large, good-quality Rasch and CFA studies across countries, consistent in showing SWEMWBS strict unidimensionality and WEMWBS essential (bifactor) unidimensionality."
   },
   "convergent_discriminant_validity": {
    "findings": "WEMWBS correlates strongly with other mental health and wellbeing measures and more weakly with general-health measures, the pattern predicted at development ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). In UK teenagers, WEMWBS correlated 0.65 with the Mental Health Continuum Short Form, 0.59 with the KIDSCREEN-27 psychological wellbeing domain and 0.57 with the WHO-5, and negatively (-0.44) with the Strengths and Difficulties Questionnaire ([Clarke 2011](https://doi.org/10.1186/1471-2458-11-487)). Against the national SWEMWBS norming data, SWEMWBS correlated 0.53 with a happiness index, -0.52 with the GHQ-12 and 0.40 with the EQ-VAS ([Ng 2017](https://doi.org/10.1007/s11136-016-1454-8)). Discriminant validity from distress measures is the contested point: SWEMWBS correlated 0.60 to 0.79 with the PHQ-9 and 0.63 to 0.74 with the GAD-7 in a primary care sample ([Shah 2021](https://doi.org/10.1186/s12955-021-01882-x)), and a Norwegian study modelling the latent correlation between the 14-item WEMWBS and the PHQ-9 found it approaching -0.80, with bifactor indices suggesting wellbeing and depression items were 'essentially unidimensional' jointly ([Aarø 2025](https://doi.org/10.1186/s12888-025-06922-0)). This raises a genuine question of whether WEMWBS and depression scales measure distinct constructs or opposite poles of one, which matters when both are fielded together in a workplace survey.",
    "grade": "High",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High: convergent correlations replicated across many samples; discriminant validity from distress deliberately reported including the unfavourable finding of very high wellbeing-distress overlap."
   },
   "criterion_validity_reference_standard": {
    "findings": "Direct criterion validity against hard organisational outcomes such as sickness absence, staff turnover or diagnosed conditions is essentially absent from the peer-reviewed WEMWBS psychometric literature retrieved here; the instrument's stewards explicitly state it was not designed for individual screening or diagnosis. What exists is known-groups and concurrent criterion evidence. WEMWBS discriminated between population subgroups in the directions expected from other UK surveys ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)), and SWEMWBS categories were associated with health behaviours (for example lower fruit and vegetable consumption predicting lower wellbeing) in the Health Survey for England norming study ([Ng 2017](https://doi.org/10.1007/s11136-016-1454-8)). Against mental-health criteria, WEMWBS separates carer and patient groups from general-population norms: family carers of people with psychosis scored on average 7.3 points below the Health Survey for England general population, more than double the 3-point minimum important difference ([Sin 2020](https://doi.org/10.1017/S2045796020001067)). A CES-D based threshold of a WEMWBS score at or below 40 has been proposed as indicating elevated depression risk, but the stewards caution the scale was not built for screening. No published workplace criterion study linking WEMWBS to absence or productivity was located in this pass, which is a material gap for the workplace audience.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low: known-groups and concurrent evidence is reasonable, but criterion validity against organisational or occupational outcomes (absence, turnover, diagnosis) was not located; workplace-specific criterion evidence is absent."
   },
   "criterion_validity_organisational": {
    "findings": "Organisational criterion evidence (sickness absence, turnover, performance, diagnosed conditions in a work context): see the criterion findings; graded from the pass-one record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low: known-groups and concurrent evidence is reasonable, but criterion validity against organisational or occupational outcomes (absence, turnover, diagnosis) was not located; workplace-specific criterion evidence is absent."
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently high for the 14-item WEMWBS and adequate-to-high for SWEMWBS. The original UK validation reported Cronbach's alpha of 0.89 in the student sample and 0.91 in the population sample, the authors noting this suggests some item redundancy ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). UK teenagers gave alpha 0.87 ([Clarke 2011](https://doi.org/10.1186/1471-2458-11-487)). International WEMWBS alphas cluster around 0.90 or above: 0.91 in Slovenian nursing students ([Cilar 2020](https://doi.org/10.1111/jonm.13087)) and 0.93 in Chinese university students, with SWEMWBS at 0.88 in the same sample ([Fung 2019](https://doi.org/10.1186/s12955-019-1113-1)). Omega estimates from pooled UK, Danish and Catalan surveillance data reached 0.94 for WEMWBS ([Yadav 2025](https://doi.org/10.1136/bmjment-2024-301433)), and SWEMWBS internal consistency was 0.88 in a large US urban sample ([Millington 2026](https://doi.org/10.1007/s00127-026-03109-0)). Values above roughly 0.90 for the 14-item version are frequently read as mild redundancy rather than a fault.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High: many good-quality studies, large total N, alpha/omega consistently 0.87 to 0.94 for WEMWBS and around 0.84 to 0.88 for SWEMWBS."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Low",
    "status": "thin",
    "indirectness": "see summary",
    "summary": "Test-retest evidence exists but is notably thinner than the internal-consistency evidence, and this is the weakest-covered reliability property. The anchor value is from the original UK validation, where 14-item WEMWBS test-retest reliability at one week was reported as an intraclass correlation of 0.83 in a student sub-sample ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). Test-retest stability was also examined in the UK teenage validation ([Clarke 2011](https://doi.org/10.1186/1471-2458-11-487)), and a Polish adaptation reported a dedicated test-retest study, though on a very small retest sample of only 24 participants ([Konaszewski 2021](https://doi.org/10.1186/s12955-021-01716-w)). No test-retest coefficient specific to the 7-item SWEMWBS, and no UK workplace or occupational test-retest study, was located in this pass. Reported intervals are short (around one week), so stability over the weeks-to-months horizon typical of workplace re-survey cycles is effectively unquantified. The absence of a robust, replicated SWEMWBS test-retest estimate is a real gap that should be flagged to anyone using change scores at the individual level.",
    "confidence_note": "Low: a single well-known 14-item value (ICC 0.83 at one week) plus sparse, small-sample replications; no SWEMWBS-specific or workplace test-retest evidence located, and intervals are short."
   },
   "measurement_invariance": {
    "findings": "SWEMWBS has been tested for invariance more thoroughly than most wellbeing measures, mainly in youth and general-population samples. In a very large Welsh school sample (n=103,971), SWEMWBS was single-factor and loadings and thresholds were invariant across school-year (age) groups, though residual variances were not, indicating partial (metric and scalar/threshold) invariance by age ([Melendez-Torres 2019](https://doi.org/10.1186/s12955-019-1204-z)). In the same survey programme, SWEMWBS reached configural, metric and scalar invariance between young people in care and their peers ([Anthony 2022](https://doi.org/10.1007/s11136-021-02896-0)). Scalar invariance across sex and age group was reported for both versions in a Norwegian primary-care sample ([Smith 2017](https://doi.org/10.1186/s12888-017-1343-x)), and invariance across gender and age was supported in Canadian national surveillance data ([Capaldi 2026](https://doi.org/10.25318/82-003-x202600300002-eng)) and across three European populations at configural, metric and scalar levels ([Yadav 2025](https://doi.org/10.1136/bmjment-2024-301433)). The German MHS validation reported invariance across age and sex ([Peitz 2024](https://doi.org/10.1186/s12955-024-02304-4)). The main exception is age-related differential item functioning for the 'feeling optimistic about the future' item, found in a large US sample and in the original Rasch analysis ([Millington 2026](https://doi.org/10.1007/s00127-026-03109-0), [Stewart-Brown 2009](https://doi.org/10.1186/1477-7525-7-15)). Occupation-specific and longitudinal (over-time) invariance are less well evidenced, and no UK workplace invariance study was located.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: multiple large studies reach scalar invariance across sex, age and care status, but coverage is dominated by youth and general-population samples; occupational and longitudinal invariance are thin, with a recurring optimism-item DIF by age."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness of the 14-item WEMWBS is supported by a secondary analysis of twelve intervention studies, where standardised response means ranged up to 1.35 and the scale detected group-level change in most studies; the standard error of measurement was 2.4 to 3.1 points, and at the 2.77-SEM threshold WEMWBS flagged important individual improvement in 12.8 to 45.7 per cent of participants ([Maheswaran 2012](https://doi.org/10.1186/1477-7525-10-156)). SWEMWBS showed linear sensitivity to change over five therapy sessions in a primary-care common-mental-disorder sample, tracking alongside PHQ-9 and GAD-7 change ([Shah 2021](https://doi.org/10.1186/s12955-021-01882-x)). A minimum important change of around 3 points on the 14-item WEMWBS is widely cited (and used as the benchmark in, for example, the carer comparison of [Sin 2020](https://doi.org/10.1017/S2045796020001067)), and the stewards publish a 3-point change threshold for WEMWBS and a 1-to-3-point threshold for SWEMWBS. Formal anchor-based MIC estimation with external criteria of change is still limited; the responsiveness authors themselves called for further work using external change criteria, and no workplace-intervention MIC study was located in this pass.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: one strong multi-study responsiveness analysis for WEMWBS plus supportive SWEMWBS change data and a widely used 3-point MIC, but external-criterion MIC estimation is limited and not workplace-specific."
   },
   "populations_languages_norms": {
    "findings": "WEMWBS was developed and validated in the UK in student and general-population samples aged 16 and over ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)), with subsequent UK validation in teenagers aged 13 to 16 ([Clarke 2011](https://doi.org/10.1186/1471-2458-11-487)) and in Northern Ireland via the Continuous Household Survey (n=3,355) ([Lloyd 2012](https://doi.org/10.3109/09638237.2012.670883)). It has been translated and validated widely, including Norwegian, Chinese, Polish, Slovenian, German, Arabic, Italian and others, and has been adopted in national surveillance systems in Germany and Canada ([Smith 2017](https://doi.org/10.1186/s12888-017-1343-x), [Fung 2019](https://doi.org/10.1186/s12955-019-1113-1), [Konaszewski 2021](https://doi.org/10.1186/s12955-021-01716-w), [Cilar 2020](https://doi.org/10.1111/jonm.13087), [Peitz 2024](https://doi.org/10.1186/s12955-024-02304-4), [Capaldi 2026](https://doi.org/10.25318/82-003-x202600300002-eng)). UK population norms are the strongest feature: age- and sex-specific SWEMWBS norms were derived from the Health Survey for England 2010-2013 (n=27,169 adults aged 16+), giving mean SWEMWBS of about 23.7 for men and 23.2 for women ([Ng 2017](https://doi.org/10.1007/s11136-016-1454-8)), and Health Survey for England general-population values are used as the reference point in comparative UK studies ([Sin 2020](https://doi.org/10.1017/S2045796020001067)). WEMWBS is also embedded in the Scottish Health Survey and originated in Scottish population-survey data. A UK preference-based value set for SWEMWBS has been derived to support health-economic (utility) use ([Yiu 2023](https://doi.org/10.1016/j.socscimed.2023.115928)). Occupational norms are sparse; the veterinary-profession Rasch study is one of the few occupation-specific UK datasets ([Bartram 2013](https://doi.org/10.1007/s11136-012-0144-4)), and there is no consolidated UK workplace-sector norm set.",
    "grade": "High",
    "indirectness": "see findings"
   },
   "criticisms_controversies": "Several recurring criticisms appear in the literature. First, the discriminant validity question: WEMWBS and SWEMWBS correlate very strongly (negatively) with depression and anxiety measures, with latent correlations approaching -0.80 against the PHQ-9 and joint bifactor models suggesting near-unidimensionality of wellbeing and distress items together ([Aarø 2025](https://doi.org/10.1186/s12888-025-06922-0), [Shah 2021](https://doi.org/10.1186/s12955-021-01882-x)), which challenges the claim that positive wellbeing is a construct distinct from the absence of symptoms. Second, ceiling and targeting problems: a Rasch analysis of a large Swedish general-population survey concluded SWEMWBS is an 'off-target' scale, skewed toward lower wellbeing with a ceiling effect and large measurement uncertainty for most respondents, and cautioned against using it to assess change or group differences in whole-population surveys ([Melin 2022](https://doi.org/10.1016/j.puhe.2021.10.009)). Third, dimensionality: strict unidimensionality holds for SWEMWBS but the 14-item WEMWBS repeatedly requires correlated residuals or a bifactor structure to fit, and at least one national validation could not confirm a single factor for either version ([Peitz 2024](https://doi.org/10.1186/s12955-024-02304-4)). Fourth, item-level bias: the 'optimism about the future' item shows differential functioning by age, and several items dropped from WEMWBS to form SWEMWBS had shown gender bias ([Stewart-Brown 2009](https://doi.org/10.1186/1477-7525-7-15), [Millington 2026](https://doi.org/10.1007/s00127-026-03109-0)). Fifth, high internal-consistency values (alpha above 0.90) are read by the developers themselves as indicating item redundancy in the 14-item form ([Tennant 2007](https://doi.org/10.1186/1477-7525-5-63)). Finally, a governance point relevant to workplace users: the licence is free only for registered non-commercial use, and Warwick has extended charging to commercial and some organisational users over time, so the cost basis for many workplace deployments is not zero and should be checked against current Warwick Innovations terms.",
   "citations": [
    {
     "key": "tennant2007",
     "authors": "Tennant R, Hiller L, Fishwick R, Platt S, Joseph S, Weich S, Parkinson J, Secker J, Stewart-Brown S",
     "year": "2007",
     "title": "The Warwick-Edinburgh Mental Well-being Scale (WEMWBS): development and UK validation",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/1477-7525-5-63",
     "url": "https://doi.org/10.1186/1477-7525-5-63"
    },
    {
     "key": "stewartbrown2009",
     "authors": "Stewart-Brown S, Tennant A, Tennant R, Platt S, Parkinson J, Weich S",
     "year": "2009",
     "title": "Internal construct validity of the Warwick-Edinburgh Mental Well-being Scale (WEMWBS): a Rasch analysis using data from the Scottish Health Education Population Survey",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/1477-7525-7-15",
     "url": "https://doi.org/10.1186/1477-7525-7-15"
    },
    {
     "key": "maheswaran2012",
     "authors": "Maheswaran H, Weich S, Powell J, Stewart-Brown S",
     "year": "2012",
     "title": "Evaluating the responsiveness of the Warwick Edinburgh Mental Well-Being Scale (WEMWBS): group and individual level analysis",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/1477-7525-10-156",
     "url": "https://doi.org/10.1186/1477-7525-10-156"
    },
    {
     "key": "clarke2011",
     "authors": "Clarke A, Friede T, Putz R, Ashdown J, Martin S, Blake A, Adi Y, Parkinson J, Flynn P, Platt S, Stewart-Brown S",
     "year": "2011",
     "title": "Warwick-Edinburgh Mental Well-being Scale (WEMWBS): validated for teenage school students in England and Scotland. A mixed methods assessment",
     "journal": "BMC Public Health",
     "doi": "10.1186/1471-2458-11-487",
     "url": "https://doi.org/10.1186/1471-2458-11-487"
    },
    {
     "key": "shah2021",
     "authors": "Shah N, Cader M, Andrews B, McCabe R, Stewart-Brown SL",
     "year": "2021",
     "title": "Short Warwick-Edinburgh Mental Well-being Scale (SWEMWBS): performance in a clinical sample in relation to PHQ-9 and GAD-7",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-021-01882-x",
     "url": "https://doi.org/10.1186/s12955-021-01882-x"
    },
    {
     "key": "melin2021",
     "authors": "Melin J, Nordin Å, et al.",
     "year": "2022",
     "title": "An off-target scale limits the utility of Short Warwick-Edinburgh Mental Well-Being Scale (SWEMWBS) in a general population survey",
     "journal": "Public Health",
     "doi": "10.1016/j.puhe.2021.10.009",
     "url": "https://doi.org/10.1016/j.puhe.2021.10.009"
    },
    {
     "key": "melendez2019",
     "authors": "Melendez-Torres GJ, Hewitt G, Hallingberg B, Anthony R, Collishaw S, Hall J, Murphy S, Moore G",
     "year": "2019",
     "title": "Measurement invariance properties and external construct validity of the short Warwick-Edinburgh mental wellbeing scale in a large national sample of secondary school students in Wales",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-019-1204-z",
     "url": "https://doi.org/10.1186/s12955-019-1204-z"
    },
    {
     "key": "anthony2021",
     "authors": "Anthony R, Moore G, Page N, Hewitt G, Murphy S, Melendez-Torres GJ",
     "year": "2022",
     "title": "Measurement invariance of the short Warwick-Edinburgh Mental Wellbeing Scale and latent mean differences (SWEMWBS) in young people by current care status",
     "journal": "Quality of Life Research",
     "doi": "10.1007/s11136-021-02896-0",
     "url": "https://doi.org/10.1007/s11136-021-02896-0"
    },
    {
     "key": "perera2025",
     "authors": "Perera G, et al.",
     "year": "2025",
     "title": "Psychometric properties of the Warwick Edinburgh Mental Well-being Scale: a systematic review",
     "journal": "Systematic Reviews",
     "doi": "10.1186/s13643-025-02897-x",
     "url": "https://doi.org/10.1186/s13643-025-02897-x"
    },
    {
     "key": "millington2026",
     "authors": "Millington E, et al.",
     "year": "2026",
     "title": "Construct validity of the Short Warwick-Edinburgh Mental Well-Being Scale in a diverse urban population",
     "journal": "Social Psychiatry and Psychiatric Epidemiology",
     "doi": "10.1007/s00127-026-03109-0",
     "url": "https://doi.org/10.1007/s00127-026-03109-0"
    },
    {
     "key": "sabin2020",
     "authors": "Sin J, Elkes J, Batchelor R, Henderson C, Gillard S, Woodham LA, Chen T, Aden A, Cornelius V",
     "year": "2020",
     "title": "Mental health and caregiving experiences of family carers supporting people with psychosis",
     "journal": "Epidemiology and Psychiatric Sciences",
     "doi": "10.1017/S2045796020001067",
     "url": "https://doi.org/10.1017/S2045796020001067"
    },
    {
     "key": "ngfat2017",
     "authors": "Ng Fat L, Scholes S, Boniface S, Mindell J, Stewart-Brown S",
     "year": "2017",
     "title": "Evaluating and establishing national norms for mental wellbeing using the short Warwick-Edinburgh Mental Well-being Scale (SWEMWBS): findings from the Health Survey for England",
     "journal": "Quality of Life Research",
     "doi": "10.1007/s11136-016-1454-8",
     "url": "https://doi.org/10.1007/s11136-016-1454-8"
    },
    {
     "key": "bartram2013",
     "authors": "Bartram DJ, Sinclair JMA, Baldwin DS",
     "year": "2013",
     "title": "Further validation of the Warwick-Edinburgh Mental Well-being Scale (WEMWBS) in the UK veterinary profession: Rasch analysis",
     "journal": "Quality of Life Research",
     "doi": "10.1007/s11136-012-0144-4",
     "url": "https://doi.org/10.1007/s11136-012-0144-4"
    },
    {
     "key": "lloyd2012",
     "authors": "Lloyd K, Devine P",
     "year": "2012",
     "title": "Psychometric properties of the Warwick-Edinburgh Mental Well-being Scale (WEMWBS) in Northern Ireland",
     "journal": "Journal of Mental Health",
     "doi": "10.3109/09638237.2012.670883",
     "url": "https://doi.org/10.3109/09638237.2012.670883"
    },
    {
     "key": "fung2019",
     "authors": "Fung SF",
     "year": "2019",
     "title": "Psychometric evaluation of the Warwick-Edinburgh Mental Well-being Scale (WEMWBS) with Chinese university students",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-019-1113-1",
     "url": "https://doi.org/10.1186/s12955-019-1113-1"
    },
    {
     "key": "peitz2024",
     "authors": "Peitz D, Kersjes C, Thom J, Hoelling H, Mauz E",
     "year": "2024",
     "title": "Validation of the Warwick-Edinburgh Mental Well-Being Scale for the Mental Health Surveillance (MHS) system in Germany",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-024-02304-4",
     "url": "https://doi.org/10.1186/s12955-024-02304-4"
    },
    {
     "key": "cilar2020",
     "authors": "Cilar L, Pajnkihar M, Štiglic G",
     "year": "2020",
     "title": "Validation of the Warwick-Edinburgh Mental Well-being Scale among nursing students in Slovenia",
     "journal": "Journal of Nursing Management",
     "doi": "10.1111/jonm.13087",
     "url": "https://doi.org/10.1111/jonm.13087"
    },
    {
     "key": "aaro2025",
     "authors": "Aarø LE, et al.",
     "year": "2025",
     "title": "Positive mental wellbeing or symptoms of depression? Discriminant validity of the Warwick-Edinburgh Mental Wellbeing Scale (WEMWBS) with respect to the PHQ-9",
     "journal": "BMC Psychiatry",
     "doi": "10.1186/s12888-025-06922-0",
     "url": "https://doi.org/10.1186/s12888-025-06922-0"
    },
    {
     "key": "yadav2025",
     "authors": "Yadav L, et al.",
     "year": "2025",
     "title": "Internal structure, reliability and cross-cultural validity of the Warwick-Edinburgh Mental Wellbeing Scale in three European populations",
     "journal": "BMJ Mental Health",
     "doi": "10.1136/bmjment-2024-301433",
     "url": "https://doi.org/10.1136/bmjment-2024-301433"
    },
    {
     "key": "yiu2023",
     "authors": "Yiu HHE, Buckell J, Petrou S, Stewart-Brown S, Madan J",
     "year": "2023",
     "title": "Derivation of a UK preference-based value set for the Short Warwick-Edinburgh Mental Well-being Scale (SWEMWBS)",
     "journal": "Social Science & Medicine",
     "doi": "10.1016/j.socscimed.2023.115928",
     "url": "https://doi.org/10.1016/j.socscimed.2023.115928"
    },
    {
     "key": "capaldi2026",
     "authors": "Capaldi CA, et al.",
     "year": "2026",
     "title": "Validating the Warwick-Edinburgh Mental Well-being Scale for the positive mental health surveillance of Canadian adults",
     "journal": "Health Reports",
     "doi": "10.25318/82-003-x202600300002-eng",
     "url": "https://doi.org/10.25318/82-003-x202600300002-eng"
    },
    {
     "key": "konaszewski2021",
     "authors": "Konaszewski K, Niesiobędzka M, Surzykiewicz J",
     "year": "2021",
     "title": "Factor structure and psychometric properties of a Polish adaptation of the Warwick-Edinburgh Mental Wellbeing Scale",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-021-01716-w",
     "url": "https://doi.org/10.1186/s12955-021-01716-w"
    },
    {
     "key": "smith2017",
     "authors": "Smith ORF, Alves DE, Knapstad M, Haug E, Aarø LE",
     "year": "2017",
     "title": "Measuring mental well-being in Norway: validation of the Warwick-Edinburgh Mental Well-being Scale (WEMWBS)",
     "journal": "BMC Psychiatry",
     "doi": "10.1186/s12888-017-1343-x",
     "url": "https://doi.org/10.1186/s12888-017-1343-x"
    }
   ],
   "record_notes": "[Upgraded from v0.1 to v0.2 structure in pass two; criterion field split, licence re-verified 2026-07-12.] The schema separates measurement properties cleanly, but WEMWBS forces two honesty caveats the fields do not naturally hold. (1) Almost all high-quality evidence is from general-population, student, youth and clinical/primary-care samples, not workplaces; convergent, invariance and responsiveness grades would drop a level if the audience's strict need is UK working-adult evidence, because that specific population is under-represented. I have graded on the overall evidence and flagged the workplace/occupational gap in each field rather than inflating or hiding it. (2) The 14-item and 7-item versions genuinely differ in their psychometrics (SWEMWBS is the Rasch interval scale; WEMWBS is essentially but not strictly unidimensional), so several fields had to carry two verdicts in one box. Test-retest is the weakest property: the well-known ICC 0.83 is a single 14-item, one-week, student estimate, with no replicated SWEMWBS or workplace value located, so change-score use at the individual level rests on responsiveness/MIC evidence rather than on retest stability. The word 'clinical' appears only in reference to the primary-care/CMD samples where SWEMWBS sensitivity to change was tested (Shah 2021); that is a statement about where evidence was earned, and workplace deployment is a different context with no equivalent criterion evidence located. Overall confidence in the record is High for structural validity, internal consistency and UK general-population norms; Moderate for invariance and responsiveness; Low for criterion validity and test-retest, chiefly because of the absence of workplace-specific and replicated stability data. Licensing is 'free' only in a qualified sense: registration is mandatory, and commercial and some organisational uses are chargeable under the Warwick licence tiers."
  },
  {
   "instrument_id": "hse-msit",
   "display_name": "HSE Management Standards Indicator Tool",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Health and Safety Executive (HSE) Management Standards Indicator Tool (also HSE-MS IT, MSIT, HSE-IT, Management Standards Revised Indicator Tool)",
    "current_version": "35-item revised Indicator Tool, seven subscales; the same instrument is described across studies as the standard form. A shorter 25-item form is sometimes referenced in practice but no psychometric validation of an official 25-item short form was located in this pass (see record_notes).",
    "item_count": "35 items across seven subscales: Demands (8), Control (6), Managerial Support (5), Peer Support (4), Relationships (4), Role (5), Change (3), per [Edwards 2008](https://doi.org/10.1080/02678370802166599).",
    "original_citation": "Cousins R, MacKay CJ, Clarke SD, Kelly C, Kelly PJ, McCaig RH (2004). 'Management Standards' work-related stress in the UK: practical development. Work & Stress, 18(2), 113-136. doi:10.1080/02678370410001734322; companion policy/science paper MacKay et al. (2004), doi:10.1080/02678370410001727474.",
    "steward_publisher": "UK Health and Safety Executive (HSE), the Great Britain national regulator for workplace health and safety. The tool sits within the HSE Management Standards approach to work-related stress ([Cousins 2004](https://doi.org/10.1080/02678370410001734322); [MacKay 2004](https://doi.org/10.1080/02678370410001727474)).",
    "licence_status": "CONFIRMED and now verified. The HSE Management Standards Indicator Tool pages and download pages carry the site-wide statement 'All content is available under the Open Government Licence v3.0, except where otherwise stated'. The tool is thus Crown copyright released under OGL v3.0, free to use with attribution.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://www.hse.gov.uk/stress/standards/notesindicatortool.htm and https://www.hse.gov.uk/stress/standards/downloads.htm (OGL v3.0 footer, Crown copyright)"
   },
   "constructs_claimed": "The Indicator Tool is an exposure measure of psychosocial working conditions, not a wellbeing or health outcome measure. It captures employee perceptions of seven work-stressor dimensions that map onto the HSE Management Standards: Demands (workload, work patterns, environment), Control (autonomy over how work is done), Managerial Support (encouragement and resources from line management), Peer Support (support from colleagues), Relationships (conflict and unacceptable behaviour, e.g. bullying), Role (understanding of role and avoidance of conflicting roles) and Change (how organisational change is managed and communicated). Higher subscale scores denote more favourable (lower-risk) conditions. The framework derives from the demand-control-support tradition and UK epidemiological work on psychosocial hazards ([Cousins 2004](https://doi.org/10.1080/02678370410001734322); [MacKay 2004](https://doi.org/10.1080/02678370410001727474)).",
   "deployment_context_caveat": "HSE MSIT measures exposure to psychosocial working conditions, not a wellbeing outcome; its property grades should be read in that framing.",
   "structural_validity": {
    "findings": "The seven-factor structure is well supported in UK data but the Managerial Support and Change dimensions are unstable in several non-UK adaptations. The instrument was developed from an item pool reduced by exploratory factor analysis to 35 items across seven subscales ([Cousins 2004](https://doi.org/10.1080/02678370410001734322)). The first confirmatory test on organisational-level UK data (39 organisations, N=26,382) found the original 35-item seven-factor first-order model gave an acceptable fit, and a second-order model was also acceptable, suggesting a possible higher-order single work-related stress dimension ([Edwards 2008](https://doi.org/10.1080/02678370802166599)). Cross-nationally, a seven-factor solution replicated and was equivalent across large UK (N=7,589) and Italian (N=1,298) private-sector samples in multiple-group CFA ([Toderi 2013](https://doi.org/10.1027/1015-5759/a000122)). However, several adaptations do not recover seven distinct factors: the Italian revised tool collapsed Managerial Support and Change into a single factor (termed 'elasticity'), retaining five to seven scales ([Magnavita 2012](https://doi.org/10.1093/occmed/kqs025)), the Irish ROI-MSIT (N=7,377) likewise merged Managerial Support and Change ([Boyd 2016](https://doi.org/10.1093/occmed/kqw163)), and an Argentine study retained only 24 items in six factors, discarding Change entirely ([Vaamonde 2023](https://doi.org/10.1093/occmed/kqad010)). A North Italian healthcare-worker study confirmed a seven-component structure but identified an additional factor relating to participation in work organisation ([Veronesi 2022](https://doi.org/10.3390/ijerph19159514)). A heavily revised Iranian version returned a nine-factor solution with new items, and so is best treated as a distinct instrument ([Zeinolabedini 2025](https://doi.org/10.1038/s41598-025-30714-x)).",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: the seven-factor model has strong, consistent support in large UK samples and one large cross-national test, but the Managerial Support/Change distinction is not reproduced in several international adaptations, indicating the structure is not fully robust outside the UK."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent validity against other psychosocial work measures is generally adequate, with some weak spots in discriminant terms for specific subscales. In an Italian municipality sample (N=760), Indicator Tool scales showed moderate to strong correlations with the corresponding Job Content Questionnaire scales, and each scale added specific predictive contribution to self-reported stress, job satisfaction and job motivation ([Marcatto 2014](https://doi.org/10.1093/occmed/kqu038)). Subscales correlated in the expected directions with stress-related outcomes in the UK and Italian cross-cultural study ([Toderi 2013](https://doi.org/10.1027/1015-5759/a000122)). In the Argentine adaptation, discriminant validity between dimensions was adequate but convergent validity was a concern for Control, Role clarity and Relationships, where average variance extracted was at or below 0.50 ([Vaamonde 2023](https://doi.org/10.1093/occmed/kqad010)). Subscales also related meaningfully but with small effect sizes to burnout dimensions on the Maslach Burnout Inventory, with Demands and Role linked to emotional exhaustion ([Carpi 2021](https://doi.org/10.1093/occmed/kqab055)).",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: several independent samples show expected-direction correlations with established measures (JCQ, MBI), but detailed convergent/discriminant metrics (e.g. AVE) are reported mainly in non-UK adaptations and are not uniformly strong."
   },
   "criterion_validity_reference_standard": {
    "findings": "Associations with health and job-attitude outcomes are consistently reported, but the evidence is overwhelmingly cross-sectional and against self-reported outcomes rather than objective organisational endpoints such as verified sickness absence or turnover. In a UK Health and Social Services Trust (N=707, 29% response), more favourable Management Standards scores were positively associated with job satisfaction and negatively with job-related anxiety, depression and witnessed errors/near misses ([Kerr 2009](https://doi.org/10.1093/occmed/kqp146)). In a UK call centre (N=304), only Demands (Spearman rho=-0.211) and Relationships (rho=-0.134) correlated significantly with GHQ-12 distress, while other dimensions did not, a mixed result that qualifies claims of uniform criterion validity ([Kazi 2013](https://doi.org/10.1093/occmed/kqt052)). In an Italian bank the tool related to GHQ-12 distress and Work Ability Index scores ([Guidi 2012](https://doi.org/10.1093/occmed/kqs021)), and in a UK prison-service sample (N=1,038) odds ratios linked poor psychosocial conditions to impaired psychological wellbeing, though the authors noted exposure scores alone were insufficient to set intervention priorities without an outcome measure ([Bevan 2010](https://doi.org/10.1093/occmed/kqq109)). A systematic review concluded there was a clear relationship between Indicator Tool scores and alternative wellbeing and stress measures ([Brookes 2013](https://doi.org/10.1093/occmed/kqt078)). Direct validation against objective sickness absence is weak: an early four-organisation study of the earlier filter-question version found the screening filters insensitive with low positive predictive value, and warned that using work absence as the measure of stress cost may substantially underestimate the true burden ([Main 2005](https://doi.org/10.1093/occmed/kqi044)).",
    "grade": "Moderate",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: multiple studies link the tool to self-reported distress and job attitudes in expected directions, but findings are cross-sectional, at least one UK study found most subscales unrelated to GHQ-12, and criterion evidence against objective outcomes (verified absence, turnover, diagnosed conditions) is sparse and, for absence specifically, problematic."
   },
   "criterion_validity_organisational": {
    "findings": "Organisational criterion evidence (sickness absence, turnover, performance, diagnosed conditions in a work context): see the criterion findings; graded from the pass-one record.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: multiple studies link the tool to self-reported distress and job attitudes in expected directions, but findings are cross-sectional, at least one UK study found most subscales unrelated to GHQ-12, and criterion evidence against objective outcomes (verified absence, turnover, diagnosed conditions) is sparse and, for absence specifically, problematic."
   },
   "internal_consistency": {
    "findings": "Internal consistency is generally good to excellent across subscales and languages. In the anchor UK analysis, Cronbach's alpha was Demands 0.87, Control 0.82, Managerial Support 0.88, Peer Support 0.82, Relationships 0.78, Role 0.83 and Change 0.80, with the original development study ([Cousins 2004](https://doi.org/10.1080/02678370410001734322)) reporting a comparable range of about 0.78 to 0.89 ([Edwards 2008](https://doi.org/10.1080/02678370802166599)). The Italian revised tool reported alphas of 0.75 to 0.86 ([Magnavita 2012](https://doi.org/10.1093/occmed/kqs025)), the Irish ROI-MSIT 0.75 to 0.91 ([Boyd 2016](https://doi.org/10.1093/occmed/kqw163)), and the Argentine adaptation composite reliability of 0.70 to 0.82 ([Vaamonde 2023](https://doi.org/10.1093/occmed/kqad010)). The revised Iranian version reported an overall alpha of 0.949 and McDonald's omega of 0.739 to 0.894, though for a restructured nine-factor instrument ([Zeinolabedini 2025](https://doi.org/10.1038/s41598-025-30714-x)).",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High: multiple good-quality studies across several countries and large samples consistently report subscale alphas at or above roughly 0.75, meeting conventional adequacy thresholds."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "untested",
    "indirectness": "see summary",
    "summary": "No test-retest (temporal stability) evidence for the standard UK 35-item Indicator Tool was located in this pass, and this is a genuine gap in the psychometric record. The reliability studies retrieved report internal consistency (alpha, omega) rather than repeat administration over time ([Edwards 2008](https://doi.org/10.1080/02678370802166599); [Magnavita 2012](https://doi.org/10.1093/occmed/kqs025); [Boyd 2016](https://doi.org/10.1093/occmed/kqw163)). The only intraclass correlation coefficient located (ICC=0.92) comes from the revised nine-factor Iranian version, where the abstract does not clearly establish it as a test-retest coefficient over a defined interval, and in any case applies to a modified instrument rather than the original tool ([Zeinolabedini 2025](https://doi.org/10.1038/s41598-025-30714-x)).",
    "confidence_note": "Absent: no test-retest reliability evidence for the standard tool was located this session; the absence is flagged explicitly."
   },
   "measurement_invariance": {
    "findings": "Invariance evidence exists across sector and across the UK/Italy language boundary, reaching at least metric level, but scalar invariance is not clearly demonstrated. Multiple-group CFA across large UK (N=7,589) and Italian (N=1,298) private-sector samples found the seven-factor solution equivalent, supporting metric equivalence together with factor variance and factor covariance equivalence ([Toderi 2013](https://doi.org/10.1027/1015-5759/a000122)). A dedicated UK study reported measurement invariance of the Indicator Tool across public and private sector organisations ([Edwards 2012](https://doi.org/10.1080/02678373.2012.688554)); the detailed invariance level (configural/metric/scalar) from that paper was not retrievable from the abstract in this pass. No formal invariance testing across sex, age or occupational group was located this session.",
    "grade": "Low",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low: cross-national evidence supports at least metric invariance and a UK study supports public/private invariance, but scalar (intercept) invariance is not clearly established and invariance by sex, age and occupation is untested in the retrieved literature."
   },
   "responsiveness_mic": {
    "findings": "No formal responsiveness or minimal important change (MIC) evidence was located in this pass. The tool has been used in pre/post and longitudinal designs, for example a longitudinal healthcare-worker study spanning the onset of the SARS-CoV-2 pandemic ([Veronesi 2022](https://doi.org/10.3390/ijerph19159514)) and a pre/post stress-management evaluation among Sierra Leone healthcare workers that observed changes in domain scores ([Jones 2020](https://doi.org/10.1136/bmjopen-2019-032929)), but neither established responsiveness statistics or a minimal important change threshold. HSE positions the tool for organisational monitoring and comparison rather than individual change detection.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Absent: no responsiveness or MIC estimates for the tool were located this session; longitudinal use exists but without formal responsiveness/MIC analysis."
   },
   "populations_languages_norms": {
    "findings": "The tool has been validated in the UK and adapted into several languages and settings, with UK national benchmark norms published at organisational level. UK evidence spans large multi-organisation datasets ([Edwards 2008](https://doi.org/10.1080/02678370802166599); [Cousins 2004](https://doi.org/10.1080/02678370410001734322)), the NHS and health and social care ([Kerr 2009](https://doi.org/10.1093/occmed/kqp146)), call centres ([Kazi 2013](https://doi.org/10.1093/occmed/kqt052)), the prison service ([Bevan 2010](https://doi.org/10.1093/occmed/kqq109)) and Ministry of Defence personnel, where content was judged too narrow without an added work-life balance scale ([Bridger 2016](https://doi.org/10.1080/00140139.2015.1057544)). Validated or adapted non-UK versions include Italian ([Magnavita 2012](https://doi.org/10.1093/occmed/kqs025); [Guidi 2012](https://doi.org/10.1093/occmed/kqs021); [Toderi 2013](https://doi.org/10.1027/1015-5759/a000122); [Veronesi 2022](https://doi.org/10.3390/ijerph19159514)), Irish ([Boyd 2016](https://doi.org/10.1093/occmed/kqw163)), Argentine ([Vaamonde 2023](https://doi.org/10.1093/occmed/kqad010)) and a revised Iranian version ([Zeinolabedini 2025](https://doi.org/10.1038/s41598-025-30714-x)); the Argentine study also notes prior validation in Italy, Iran and Malta. UK normative percentile benchmark tables derived from the organisational dataset are provided to let employers compare their organisational averages against national reference values ([Edwards 2008](https://doi.org/10.1080/02678370802166599)), consistent with the HSE benchmarking approach.",
    "grade": "Moderate",
    "indirectness": "see findings"
   },
   "criticisms_controversies": "Several substantive criticisms recur. First, the seven-factor structure is not fully reproducible outside the UK: the Managerial Support and Change dimensions repeatedly collapse into one factor in Italian and Irish adaptations, and Change is sometimes dropped altogether, questioning the universality of the seven-domain model ([Magnavita 2012](https://doi.org/10.1093/occmed/kqs025); [Boyd 2016](https://doi.org/10.1093/occmed/kqw163); [Vaamonde 2023](https://doi.org/10.1093/occmed/kqad010)). Second, criterion evidence is largely cross-sectional and against self-reported outcomes, with at least one UK study finding most subscales unrelated to GHQ-12 distress ([Kazi 2013](https://doi.org/10.1093/occmed/kqt052)), and an early evaluation showing the screening filters were insensitive with poor positive predictive value and that sickness absence understates the true cost of psychosocial hazards ([Main 2005](https://doi.org/10.1093/occmed/kqi044)). Third, the tool measures exposure (perceived working conditions) rather than health or wellbeing outcomes, so exposure data alone are insufficient for prioritising interventions and are best paired with an outcome measure ([Bevan 2010](https://doi.org/10.1093/occmed/kqq109)). Fourth, content validity can be too narrow for specific contexts, for example the military, where a work-life balance dimension was needed ([Bridger 2016](https://doi.org/10.1080/00140139.2015.1057544)). Finally, test-retest reliability and formal responsiveness/MIC remain unestablished for the standard tool (this pass), and much criterion evidence shares common-method variance because exposure and outcome are both self-reported.",
   "citations": [
    {
     "key": "cousins2004",
     "authors": "Cousins R, MacKay CJ, Clarke SD, Kelly C, Kelly PJ, McCaig RH",
     "year": "2004",
     "title": "'Management Standards' work-related stress in the UK: practical development",
     "journal": "Work & Stress",
     "doi": "10.1080/02678370410001734322",
     "url": "https://doi.org/10.1080/02678370410001734322"
    },
    {
     "key": "mackay2004",
     "authors": "MacKay CJ, Cousins R, Kelly PJ, Lee S, McCaig RH",
     "year": "2004",
     "title": "'Management Standards' and work-related stress in the UK: policy background and science",
     "journal": "Work & Stress",
     "doi": "10.1080/02678370410001727474",
     "url": "https://doi.org/10.1080/02678370410001727474"
    },
    {
     "key": "edwards2008",
     "authors": "Edwards JA, Webster S, Van Laar D, Easton S",
     "year": "2008",
     "title": "Psychometric analysis of the UK Health and Safety Executive's Management Standards work-related stress Indicator Tool",
     "journal": "Work & Stress",
     "doi": "10.1080/02678370802166599",
     "url": "https://doi.org/10.1080/02678370802166599"
    },
    {
     "key": "edwards2012",
     "authors": "Edwards JA, Webster S",
     "year": "2012",
     "title": "Psychosocial risk assessment: measurement invariance of the UK Health and Safety Executive's Management Standards Indicator Tool across public and private sector organizations",
     "journal": "Work & Stress",
     "doi": "10.1080/02678373.2012.688554",
     "url": "https://doi.org/10.1080/02678373.2012.688554"
    },
    {
     "key": "toderi2013",
     "authors": "Toderi S, Balducci C, Edwards JA, Sarchielli G, Broccoli M, Mancini G",
     "year": "2013",
     "title": "Psychometric properties of the UK and Italian versions of the HSE Stress Indicator Tool: a cross-cultural investigation",
     "journal": "European Journal of Psychological Assessment",
     "doi": "10.1027/1015-5759/a000122",
     "url": "https://doi.org/10.1027/1015-5759/a000122"
    },
    {
     "key": "kerr2009",
     "authors": "Kerr R, McHugh M, McCrory M",
     "year": "2009",
     "title": "HSE Management Standards and stress-related work outcomes",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqp146",
     "url": "https://doi.org/10.1093/occmed/kqp146"
    },
    {
     "key": "marcatto2014",
     "authors": "Marcatto F, Colautti L, Larese Filon F, Luis O, Ferrante D",
     "year": "2014",
     "title": "The HSE Management Standards Indicator Tool: concurrent and construct validity",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqu038",
     "url": "https://doi.org/10.1093/occmed/kqu038"
    },
    {
     "key": "bevan2010",
     "authors": "Bevan A, Houdmont J, Menear N",
     "year": "2010",
     "title": "The Management Standards Indicator Tool and the estimation of risk",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqq109",
     "url": "https://doi.org/10.1093/occmed/kqq109"
    },
    {
     "key": "guidi2012",
     "authors": "Guidi S, Bagnara S, Fichera GP",
     "year": "2012",
     "title": "The HSE indicator tool, psychological distress and work ability",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqs021",
     "url": "https://doi.org/10.1093/occmed/kqs021"
    },
    {
     "key": "magnavita2012",
     "authors": "Magnavita N",
     "year": "2012",
     "title": "Validation of the Italian version of the HSE Indicator Tool",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqs025",
     "url": "https://doi.org/10.1093/occmed/kqs025"
    },
    {
     "key": "brookes2013",
     "authors": "Brookes K, Limbert C, Deacy C, O'Reilly A, Scott S, Thirlaway K",
     "year": "2013",
     "title": "Systematic review: work-related stress and the HSE Management Standards",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqt078",
     "url": "https://doi.org/10.1093/occmed/kqt078"
    },
    {
     "key": "kazi2013",
     "authors": "Kazi A, Haslam C",
     "year": "2013",
     "title": "Stress management standards: a warning indicator for employee health",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqt052",
     "url": "https://doi.org/10.1093/occmed/kqt052"
    },
    {
     "key": "boyd2016",
     "authors": "Boyd S, Kerr R, Murray P",
     "year": "2016",
     "title": "Psychometric properties of the Irish Management Standards Indicator Tool",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqw163",
     "url": "https://doi.org/10.1093/occmed/kqw163"
    },
    {
     "key": "carpi2021",
     "authors": "Carpi M, Bruschini M, Burla F",
     "year": "2021",
     "title": "HSE Management Standards and burnout dimensions among rehabilitation professionals",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqab055",
     "url": "https://doi.org/10.1093/occmed/kqab055"
    },
    {
     "key": "vaamonde2023",
     "authors": "Vaamonde JD, Giacobino AE",
     "year": "2023",
     "title": "Psychometric properties of the HSE Indicator Tool: evidence from Argentina",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqad010",
     "url": "https://doi.org/10.1093/occmed/kqad010"
    },
    {
     "key": "bridger2016",
     "authors": "Bridger RS, Dobson K, Davison H",
     "year": "2016",
     "title": "Using the HSE stress indicator tool in a military context",
     "journal": "Ergonomics",
     "doi": "10.1080/00140139.2015.1057544",
     "url": "https://doi.org/10.1080/00140139.2015.1057544"
    },
    {
     "key": "veronesi2022",
     "authors": "Veronesi G, Giusti EM, D'Amato A, et al.",
     "year": "2022",
     "title": "The North Italian Longitudinal Study Assessing the Mental Health Effects of SARS-CoV-2 Pandemic on Health Care Workers, Part I: study design and psychometric structural validity of the HSE Indicator Tool and Work Satisfaction Scale",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph19159514",
     "url": "https://doi.org/10.3390/ijerph19159514"
    },
    {
     "key": "jones2020",
     "authors": "Jones S, White S, Ormrod J, et al.",
     "year": "2020",
     "title": "Work-based risk factors and quality of life in health care workers providing maternal and newborn care during the Sierra Leone Ebola epidemic",
     "journal": "BMJ Open",
     "doi": "10.1136/bmjopen-2019-032929",
     "url": "https://doi.org/10.1136/bmjopen-2019-032929"
    },
    {
     "key": "zeinolabedini2025",
     "authors": "Zeinolabedini M, Motlagh ME, Heidarnia A, et al.",
     "year": "2025",
     "title": "Psychometric properties of the revised version of the Health and Safety Executive Management Standards Indicator Tool",
     "journal": "Scientific Reports",
     "doi": "10.1038/s41598-025-30714-x",
     "url": "https://doi.org/10.1038/s41598-025-30714-x"
    },
    {
     "key": "main2005",
     "authors": "Main CJ, Glozier N, Wright IA",
     "year": "2005",
     "title": "Validity of the HSE stress tool: an investigation within four organizations by the Corporate Health and Performance Group",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqi044",
     "url": "https://doi.org/10.1093/occmed/kqi044"
    }
   ],
   "record_notes": "[Upgraded from v0.1 to v0.2 structure in pass two; criterion field split, licence re-verified 2026-07-12.] Framing: the Indicator Tool is a work-stressor EXPOSURE measure of perceived psychosocial working conditions, not a wellbeing outcome; the schema's outcome-oriented property fields (criterion validity, responsiveness/MIC) were completed with that in mind and criterion evidence is reported as associations with separate outcome measures. Version ambiguity: the literature retrieved this session consistently describes and validates the 35-item seven-subscale form; I did not locate a psychometric validation of an official 25-item HSE short form, so the item_count and version fields report the 35-item form and flag the 25-item form as referenced but unverified in this pass. Two properties are genuinely ABSENT in the retrieved evidence and should not be read as oversights: test-retest reliability (no temporal-stability study for the standard tool was found; the only ICC comes from a restructured Iranian version and its interpretation as test-retest is unclear) and responsiveness/MIC (longitudinal use exists but no formal statistics). Measurement invariance is graded Low because cross-national and public/private evidence reaches at most metric level and scalar invariance is not clearly demonstrated; the Edwards 2012 abstract was not retrievable in full, so the exact invariance level from that paper is stated conservatively. Licence terms are inferred from the tool's HSE Crown-copyright provenance rather than a cited licence document. Coefficients: all alpha/omega/ICC/correlation values quoted come from abstracts or (for Edwards 2008) the full text retrieved this session. Overall confidence in the record is moderate: internal consistency is well established (High), structural, convergent and criterion validity are Moderate with real conflicts reported, while invariance is Low and test-retest and responsiveness are Absent."
  },
  {
   "instrument_id": "perma",
   "display_name": "Workplace PERMA-Profiler (and PERMA-Profiler)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Workplace PERMA-Profiler (occupational adaptation of the PERMA-Profiler)",
    "current_version": "PERMA-Profiler (Butler & Kern, 2016); Workplace PERMA-Profiler (Kern), a work-contextualised reword of the same item set. No numbered revision is in circulation.",
    "item_count": "23 items total: 15 core items measuring the five PERMA domains (three per domain), plus 8 additional items covering overall well-being, negative emotion, loneliness and physical health. The Workplace form retains this structure with items reworded to the work context; short forms exist but are not the reference version.",
    "original_citation": "Butler J, Kern ML (2016). The PERMA-Profiler: A brief multidimensional measure of flourishing. International Journal of Wellbeing. DOI 10.5502/ijw.v6i3.526",
    "steward_publisher": "Instrument developed by Julie Butler and Margaret (Peggy) L. Kern; the PERMA framework derives from Seligman (2011). The Workplace adaptation is attributed to Kern. Distributed by the authors through Peggy Kern's academic website rather than a commercial publisher.",
    "licence_status": "CONFIRMED and now verified from the steward. The developer's questionnaires page states: 'You are welcome to use these measures for research or non-commercial purposes, giving credit as noted in the measures. There is no cost involved in using the measures for these purposes. For commercial uses of the measures, contact...'. So the Workplace PERMA-Profiler is free for research/non-commercial use with attribution, and commercial use requires contacting the authors.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://www.peggykern.org/questionnaires.html (developer/steward, Permissions and Use of the Measures)"
   },
   "constructs_claimed": "The instrument operationalises Seligman's PERMA model of flourishing across five claimed domains: Positive emotion, Engagement, Relationships, Meaning and Accomplishment, each measured by three items ([Butler & Kern 2016](https://doi.org/10.5502/ijw.v6i3.526)). A composite overall well-being score is derived from the fifteen domain items together with a general happiness item. Ancillary items index negative emotion, loneliness and self-rated physical health but are not part of the five-domain model. The Workplace PERMA-Profiler claims to measure the same five constructs specifically as they are experienced at work, that is workplace flourishing ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA); [Choi et al. 2019](https://doi.org/10.35371/aoem.2019.31.e17)). The multidimensional PERMA measurement approach was first applied to a defined setting in schools by [Kern et al. 2014](https://doi.org/10.1080/17439760.2014.936962), which prefigured the workplace adaptation.",
   "deployment_context_caveat": "",
   "structural_validity": {
    "findings": "The claimed five-factor structure is only partly and inconsistently supported, and the factor structure is the most contested aspect of the instrument. In occupational samples the correlated five-factor model has typically reached only marginal fit: the Japanese Workplace PERMA-Profiler gave CFI 0.892, TLI 0.858, RMSEA 0.105 and SRMR 0.051 ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)), the Korean version CFI 0.909, TLI 0.881, RMSEA 0.110 and SRMR 0.054 ([Choi et al. 2019](https://doi.org/10.35371/aoem.2019.31.e17)), and the Chinese workplace version CFI 0.887, TLI 0.855, RMSEA 0.114 and SRMR 0.060 ([Yang et al. 2024](https://doi.org/10.1186/s12889-024-18194-6)); all three show RMSEA above conventional thresholds. The Polish Workplace version is the clearest positive case, with the five-factor model reaching desired fit and outperforming a single-factor model in a large sample (N=1070) ([Fortuna et al. 2025](https://doi.org/10.1371/journal.pone.0319088)). Evidence from the general (non-workplace) PERMA-Profiler diverges further: an Australian adult study could replicate neither the theorised five-factor model nor a single-factor alternative and concluded the psychometric properties were insufficient ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932)); a US student-veteran sample yielded a two-factor solution (emotional versus performance character strengths) ([Umucu et al. 2019](https://doi.org/10.1080/07448481.2018.1546182)); and a Chinese haemodialysis sample extracted two dimensions ([Qian et al. 2025](https://doi.org/10.1186/s40359-025-02560-z)). Where the five-factor model does hold, it is often as five highly correlated first-order factors: in a large Chinese general sample the first-order five-factor solution fitted better than a higher-order well-being factor (CFI 0.943 after modification) ([Nie et al. 2024](https://doi.org/10.1016/j.actpsy.2024.104248)), and a French study confirmed both a correlated five-factor and a second-order model while its exploratory analysis suggested a more parsimonious three factors ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)). In autistic adults a bifactor model with a dominant general factor fitted better than the five-factor model ([Grosvenor et al. 2023](https://doi.org/10.1089/aut.2022.0049)). Five-factor support is stronger in adolescent adaptations ([Fernandes et al. 2024](https://doi.org/10.3389/fpsyg.2024.1415084); [Waigel & Lemos 2023](https://doi.org/10.21500/20112084.5737)). Across studies the Engagement domain is the recurrent weak point.",
    "grade": "Low",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low. Many studies but findings are inconsistent across populations and forms; five-factor fit is frequently marginal, competing two-factor, three-factor, single-factor and bifactor solutions recur, and the workplace form specifically has few confirmatory studies."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent and discriminant patterns are generally in the expected direction, though discriminant separation from distress measures is sometimes weak. For the Workplace form, well-being domains correlated moderately to strongly and positively with job satisfaction and work-related psychosocial factors and inversely with psychological distress ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)), and positively with work engagement and life mental well-being while correlating negatively with burnout, occupational stressors and stress responses ([Choi et al. 2019](https://doi.org/10.35371/aoem.2019.31.e17)). For the general form, a large Chinese study reported convergent correlations of r=0.53 to 0.85 and discriminant correlations of r=-0.19 to -0.38 ([Nie et al. 2024](https://doi.org/10.1016/j.actpsy.2024.104248)). A French study found strong positive associations with flourishing and perceived happiness (convergent) and weaker associations with anxiety, depression, loneliness and negative emotion (discriminant), while cautioning that some correlations with anxiety and depression nonetheless reached moderate to strong magnitude ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)). An Australian study found moderate correlations with depression, anxiety and stress (r=-0.374 to -0.645) but no meaningful association with objectively measured physical activity or sleep ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932)), and an autistic-adult study confirmed expected correlations with mental-health conditions and life satisfaction ([Grosvenor et al. 2023](https://doi.org/10.1089/aut.2022.0049)).",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate. Multiple studies across the general and workplace forms show consistent, theory-congruent convergent correlations; discriminant validity is adequate but repeatedly weaker against depression and anxiety, and no association with objective (non-self-report) markers."
   },
   "criterion_validity_reference_standard": {
    "findings": "Criterion evidence against organisational and health outcomes is limited to cross-sectional, self-reported proxies; no study retrieved in this pass links the instrument to hard organisational records such as sickness absence, staff turnover or medically diagnosed conditions. Workplace well-being domains were associated with job strain components: job control and supervisor and coworker support correlated with all five domains, while job demands related only to Engagement and Meaning ([Yang et al. 2021](https://doi.org/10.1097/JOM.0000000000002455)). Well-being domains, principally Positive emotion, Engagement and Relationships, were inversely associated with personal and work-related burnout in a Taiwanese multicentre sample (N=242) ([Wu et al. 2025](https://doi.org/10.1097/JOM.0000000000003318)), and Positive emotion was inversely associated with prolonged fatigue in a Chinese workforce sample ([Yang et al. 2024](https://doi.org/10.1186/s12889-024-18194-6)). In a university-workplace study, workplace well-being mediated the relationship between psychological distress and subjective quality of life and was lower among staff with suspected psychological health conditions ([Ng et al. 2025](https://doi.org/10.3389/fpsyg.2025.1598910)). Among rural veterans, well-being scores were positively associated with mental and physical health outcomes and were lower in those with service-connected disabilities ([Umucu et al. 2024](https://doi.org/10.3389/fpubh.2024.1500659)). All of these are concurrent associations rather than predictive criterion validity against objective outcomes.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low. Several studies show theory-consistent associations with burnout, fatigue, job strain and distress, but all are cross-sectional self-report; there is no predictive validity against objective organisational or clinical outcomes and no UK workplace evidence."
   },
   "criterion_validity_organisational": {
    "findings": "Organisational criterion evidence (sickness absence, turnover, performance, diagnosed conditions in a work context): see the criterion findings; graded from the pass-one record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low. Several studies show theory-consistent associations with burnout, fatigue, job strain and distress, but all are cross-sectional self-report; there is no predictive validity against objective organisational or clinical outcomes and no UK workplace evidence."
   },
   "internal_consistency": {
    "findings": "Internal consistency is the instrument's strongest property and is generally acceptable to good, with the Engagement subscale the consistent exception. For the Workplace form, Cronbach's alpha ranged from 0.75 to 0.96 in Japanese workers ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)), 0.70 to 0.95 in Korean workers ([Choi et al. 2019](https://doi.org/10.35371/aoem.2019.31.e17)) and 0.69 to 0.93 in Chinese workers ([Yang et al. 2024](https://doi.org/10.1186/s12889-024-18194-6)); the Polish Workplace version met minimum reliability thresholds ([Fortuna et al. 2025](https://doi.org/10.1371/journal.pone.0319088)). For the general form, alphas were 0.79 to 0.88 in a large Chinese sample ([Nie et al. 2024](https://doi.org/10.1016/j.actpsy.2024.104248)), the total scale reached alpha 0.914 in a Chinese haemodialysis sample ([Qian et al. 2025](https://doi.org/10.1186/s40359-025-02560-z)), and an autistic-adult sample reported alpha 0.93 ([Grosvenor et al. 2023](https://doi.org/10.1089/aut.2022.0049)). In Australian adults subscale alphas ranged 0.80 to 0.93 for all domains except Engagement (alpha 0.66) ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932)); the French validation likewise noted lower reliability specifically for Engagement ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)). Omega and composite reliability were reported as adequate in an adolescent adaptation ([Fernandes et al. 2024](https://doi.org/10.3389/fpsyg.2024.1415084)). The three-item domains, especially Engagement, are the source of most reliability shortfalls.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High. Many studies across multiple populations, forms and languages report consistent alpha values in the acceptable to good range, with a reproducible and openly reported weakness for the Engagement subscale."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Low",
    "status": "thin",
    "indirectness": "see summary",
    "summary": "Test-retest reliability is thinly evidenced and this is a genuine gap, most acute for the Workplace form. The founding paper reports cross-time consistency for the general PERMA-Profiler but does not present a conventional retest coefficient in the abstract retrieved ([Butler & Kern 2016](https://doi.org/10.5502/ijw.v6i3.526)). The Japanese Workplace validation collected a one-month follow-up (baseline N=310, follow-up N=86) and reported intraclass correlation coefficients of 0.75 to 0.96, but these ICCs are presented alongside internal-consistency alphas and the abstract does not cleanly separate a retest ICC per domain ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)). The Chinese workplace version similarly reported ICCs of 0.70 to 0.92 in a reliability analysis rather than a dedicated retest design ([Yang et al. 2024](https://doi.org/10.1186/s12889-024-18194-6)). The clearest temporal-stability evidence comes from non-workplace forms: the Polish Workplace study included a repeated-measure subsample (N=66) and reported strong stability ([Fortuna et al. 2025](https://doi.org/10.1371/journal.pone.0319088)); a French study reported good temporal stability for the total score and most subscales ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)); and a Chinese haemodialysis study reported a retest correlation of r=0.764 ([Qian et al. 2025](https://doi.org/10.1186/s40359-025-02560-z)). No UK or occupational-UK retest estimate was located, and retest intervals, when reported, are short (around one month). The absence of a robust, well-powered retest study specifically for the Workplace PERMA-Profiler is a flagged evidence gap.",
    "confidence_note": "Low. A small number of short-interval studies with modest follow-up samples, mostly reporting ICCs bundled with internal consistency rather than purpose-designed retest coefficients; no workplace-specific, adequately powered, or UK retest evidence located. Flag: near-absent for the Workplace form."
   },
   "measurement_invariance": {
    "findings": "Invariance testing exists but is concentrated in non-workplace populations and rarely reaches full scalar invariance in occupational samples. A large Chinese general-population study established factorial invariance across genders ([Nie et al. 2024](https://doi.org/10.1016/j.actpsy.2024.104248)). A Brazilian adolescent adaptation demonstrated invariance across age groups and gender ([Fernandes et al. 2024](https://doi.org/10.3389/fpsyg.2024.1415084)), and an open-and-distance-learning adaptation in Botswana examined invariance within a five-factor framework ([Magare et al. 2022](https://doi.org/10.3390/ijerph192416886)). For work-specific instruments in the wider PERMA family, the Spanish PERMA+4 Positive Functioning at Work scale showed measurement invariance across educational levels ([Garcia-Selva et al. 2024](https://doi.org/10.7334/psicothema2023.341)). For the Workplace PERMA-Profiler itself, no study retrieved in this pass reports a full multigroup invariance analysis across sex, age, occupation or language, and the level reached (configural, metric or scalar) for occupational deployment is therefore not established. Cross-language equivalence is assumed rather than formally tested, since each translation is validated as a standalone instrument.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low. Invariance is established (up to scalar in some cases) for general-population and adolescent forms, but is absent for the Workplace PERMA-Profiler specifically; occupational and cross-occupation invariance was not located."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness and a minimal important change threshold are essentially unestablished. No study retrieved in this pass reports a minimal important change, minimal detectable change tied to an anchor, or a formal responsiveness analysis for the Workplace PERMA-Profiler. Some validation studies computed measurement error and, in the Japanese case, measurement-error statistics alongside reliability ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)), but these are not responsiveness or MIC estimates. Intervention and longitudinal PERMA studies exist in the wider literature, yet a defensible, citable MIC value for interpreting change scores in a workplace deployment was not located.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Absent. No responsiveness or minimal-important-change evidence located for the Workplace PERMA-Profiler in this pass."
   },
   "populations_languages_norms": {
    "findings": "The Workplace PERMA-Profiler has been validated in occupational samples in Japanese ([Watanabe et al. 2018](https://doi.org/10.1539/joh.2018-0050-OA)), Korean ([Choi et al. 2019](https://doi.org/10.35371/aoem.2019.31.e17)), Chinese ([Yang et al. 2024](https://doi.org/10.1186/s12889-024-18194-6)) and Polish ([Fortuna et al. 2025](https://doi.org/10.1371/journal.pone.0319088)) workers, with further occupational application in Taiwanese ([Wu et al. 2025](https://doi.org/10.1097/JOM.0000000000003318)) and university-staff ([Ng et al. 2025](https://doi.org/10.3389/fpsyg.2025.1598910)) samples. The general PERMA-Profiler has been validated far more widely, including English-language Australian adults ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932)), US student veterans ([Umucu et al. 2019](https://doi.org/10.1080/07448481.2018.1546182); [Umucu et al. 2024](https://doi.org/10.3389/fpubh.2024.1500659)), autistic adults ([Grosvenor et al. 2023](https://doi.org/10.1089/aut.2022.0049)), Chinese adults and patients ([Nie et al. 2024](https://doi.org/10.1016/j.actpsy.2024.104248); [Qian et al. 2025](https://doi.org/10.1186/s40359-025-02560-z)), French adults ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)), German adults ([Wammerl et al. 2019](https://doi.org/10.1007/s41543-019-00021-0)) and adolescents in Brazil and Argentina ([Fernandes et al. 2024](https://doi.org/10.3389/fpsyg.2024.1415084); [Waigel & Lemos 2023](https://doi.org/10.21500/20112084.5737)). No UK validation study, UK occupational norms or UK benchmark tables were located in this pass; the founding studies used large but predominantly online international-English samples ([Butler & Kern 2016](https://doi.org/10.5502/ijw.v6i3.526)), and no published normative reference specific to UK workplaces was found. Practitioners in the UK would therefore be relying on non-UK reference data.",
    "grade": "Moderate",
    "indirectness": "see findings"
   },
   "criticisms_controversies": "The central controversy is whether PERMA is empirically distinguishable from general subjective well-being. Goodman and colleagues found that a PERMA composite and an established subjective well-being measure were near-indistinguishable (correlating very highly), arguing the five elements do not add measurable information beyond overall well-being and raising a jangle-fallacy concern ([Goodman et al. 2018](https://doi.org/10.1080/17439760.2017.1388434)). This is reinforced structurally by the recurrent finding that the five factors are so highly intercorrelated that single-factor, bifactor or reduced-factor solutions often fit as well as or better than the theorised five-factor model ([Grosvenor et al. 2023](https://doi.org/10.1089/aut.2022.0049); [Umucu et al. 2019](https://doi.org/10.1080/07448481.2018.1546182); [Qian et al. 2025](https://doi.org/10.1186/s40359-025-02560-z)), and by an Australian study that could not replicate the five-factor structure and judged the psychometric properties insufficient ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932)). A second, consistent criticism is the weak Engagement subscale, which repeatedly shows the lowest reliability (for example alpha 0.66) and poorest item performance ([Ryan et al. 2019](https://doi.org/10.1371/journal.pone.0225932); [Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)). A third issue is discriminant validity against distress: correlations with anxiety and depression sometimes reach moderate-to-strong magnitude, blurring the separation between well-being and ill-being ([Jourdan et al. 2026](https://doi.org/10.1038/s41598-026-57354-z)). The proliferation of a work-focused successor framework, PERMA+4, which adds further blocks and has itself been flagged for a possible jangle fallacy with psychological safety, reflects unsettled debate about what the workplace model should contain ([Donaldson et al. 2022](https://doi.org/10.3389/fpsyg.2021.817244); [Lorenz et al. 2023](https://doi.org/10.3389/fpsyg.2023.1231299)). Finally, the Workplace form specifically rests on a much thinner evidence base than the general PERMA-Profiler: its occupational validations are few, mostly East Asian and European, generally reach only marginal confirmatory fit, and lack dedicated test-retest, invariance, responsiveness and UK evidence.",
   "citations": [
    {
     "key": "butler_kern_2016",
     "authors": "Butler J, Kern ML",
     "year": "2016",
     "title": "The PERMA-Profiler: A brief multidimensional measure of flourishing",
     "journal": "International Journal of Wellbeing",
     "doi": "10.5502/ijw.v6i3.526",
     "url": "https://doi.org/10.5502/ijw.v6i3.526"
    },
    {
     "key": "goodman_2018",
     "authors": "Goodman FR, Disabato DJ, Kashdan TB, Kauffman SB",
     "year": "2018",
     "title": "Measuring well-being: A comparison of subjective well-being and PERMA",
     "journal": "The Journal of Positive Psychology",
     "doi": "10.1080/17439760.2017.1388434",
     "url": "https://doi.org/10.1080/17439760.2017.1388434"
    },
    {
     "key": "kern_2014",
     "authors": "Kern ML, Waters LE, Adler A, White MA",
     "year": "2014",
     "title": "A multidimensional approach to measuring well-being in students: Application of the PERMA framework",
     "journal": "The Journal of Positive Psychology",
     "doi": "10.1080/17439760.2014.936962",
     "url": "https://doi.org/10.1080/17439760.2014.936962"
    },
    {
     "key": "wammerl_2019",
     "authors": "Wammerl M, Jaunig J, Mairunteregger T, Streit P",
     "year": "2019",
     "title": "The German Version of the PERMA-Profiler: Evidence for Construct and Convergent Validity",
     "journal": "International Journal of Applied Positive Psychology",
     "doi": "10.1007/s41543-019-00021-0",
     "url": "https://doi.org/10.1007/s41543-019-00021-0"
    },
    {
     "key": "donaldson_2022",
     "authors": "Donaldson SI, van Zyl LE, Donaldson SI",
     "year": "2022",
     "title": "PERMA+4: A Framework for Work-Related Wellbeing, Performance and Positive Organizational Psychology 2.0",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2021.817244",
     "url": "https://doi.org/10.3389/fpsyg.2021.817244"
    },
    {
     "key": "watanabe_2018",
     "authors": "Watanabe K, Kawakami N, Shiotani T et al.",
     "year": "2018",
     "title": "The Japanese Workplace PERMA-Profiler: A validation study among Japanese workers.",
     "journal": "Journal of occupational health",
     "doi": "10.1539/joh.2018-0050-OA",
     "url": "https://doi.org/10.1539/joh.2018-0050-OA"
    },
    {
     "key": "choi_2019",
     "authors": "Choi SP, Suh C, Yang JW et al.",
     "year": "2019",
     "title": "Korean translation and validation of the Workplace Positive emotion, Engagement, Relationships, Meaning, and Accomplishment (PERMA)-Profiler.",
     "journal": "Annals of occupational and environmental medicine",
     "doi": "10.35371/aoem.2019.31.e17",
     "url": "https://doi.org/10.35371/aoem.2019.31.e17"
    },
    {
     "key": "yang_2024",
     "authors": "Yang CC, Chen HT, Luo KH et al.",
     "year": "2024",
     "title": "The validation of Chinese version of workplace PERMA-profiler and the association between workplace well-being and fatigue.",
     "journal": "BMC public health",
     "doi": "10.1186/s12889-024-18194-6",
     "url": "https://doi.org/10.1186/s12889-024-18194-6"
    },
    {
     "key": "fortuna_2025",
     "authors": "Fortuna P, Czerw A, Ostafińska-Molik B et al.",
     "year": "2025",
     "title": "Flourishing at work: Psychometric properties of the Polish version of the Workplace PERMA-Profiler.",
     "journal": "PloS one",
     "doi": "10.1371/journal.pone.0319088",
     "url": "https://doi.org/10.1371/journal.pone.0319088"
    },
    {
     "key": "yang_2021",
     "authors": "Yang CC, Watanabe K, Kawakami N",
     "year": "2021",
     "title": "The Associations Between Job Strain, Workplace PERMA Profiler, and Work Engagement.",
     "journal": "Journal of occupational and environmental medicine",
     "doi": "10.1097/JOM.0000000000002455",
     "url": "https://doi.org/10.1097/JOM.0000000000002455"
    },
    {
     "key": "wu_2025",
     "authors": "Wu TW, Chuang HY, Lin CP et al.",
     "year": "2025",
     "title": "Is Well-being Associated With Burnout? From a Multicenter Cross-sectional Study in Taiwan.",
     "journal": "Journal of occupational and environmental medicine",
     "doi": "10.1097/JOM.0000000000003318",
     "url": "https://doi.org/10.1097/JOM.0000000000003318"
    },
    {
     "key": "umucu_2019",
     "authors": "Umucu E, Wu JR, Sanchez J et al.",
     "year": "2019",
     "title": "Psychometric validation of the PERMA-profiler as a well-being measure for student veterans.",
     "journal": "Journal of American college health : J of ACH",
     "doi": "10.1080/07448481.2018.1546182",
     "url": "https://doi.org/10.1080/07448481.2018.1546182"
    },
    {
     "key": "ryan_2019",
     "authors": "Ryan J, Curtis R, Olds T et al.",
     "year": "2019",
     "title": "Psychometric properties of the PERMA Profiler for measuring wellbeing in Australian adults.",
     "journal": "PloS one",
     "doi": "10.1371/journal.pone.0225932",
     "url": "https://doi.org/10.1371/journal.pone.0225932"
    },
    {
     "key": "nie_2024",
     "authors": "Nie YZ, Zhang X, Hong NW et al.",
     "year": "2024",
     "title": "Psychometric validation of the PERMA-profiler for well-being in Chinese adults.",
     "journal": "Acta psychologica",
     "doi": "10.1016/j.actpsy.2024.104248",
     "url": "https://doi.org/10.1016/j.actpsy.2024.104248"
    },
    {
     "key": "julie_2026",
     "authors": "Jourdan J, Serrand C, Chevallier T, et al.",
     "year": "2026",
     "title": "PERMA-Profiler as a multidimensional measure of well-being in a French context.",
     "journal": "Scientific reports",
     "doi": "10.1038/s41598-026-57354-z",
     "url": "https://doi.org/10.1038/s41598-026-57354-z"
    },
    {
     "key": "qian_2025",
     "authors": "Qian Y, Yan H, Zeng X et al.",
     "year": "2025",
     "title": "The Chinese version of the PERMA profiler: a validity and reliability study.",
     "journal": "BMC psychology",
     "doi": "10.1186/s40359-025-02560-z",
     "url": "https://doi.org/10.1186/s40359-025-02560-z"
    },
    {
     "key": "magare_2022",
     "authors": "Magare I, Graham MA, Eloff I",
     "year": "2022",
     "title": "An Assessment of the Reliability and Validity of the PERMA Well-Being Scale for Adult Undergraduate Students in an Open and Distance Learning Context.",
     "journal": "International journal of environmental research and public health",
     "doi": "10.3390/ijerph192416886",
     "url": "https://doi.org/10.3390/ijerph192416886"
    },
    {
     "key": "grosvenor_2023",
     "authors": "Grosvenor LP, Errichetti CL, Holingue C et al.",
     "year": "2023",
     "title": "Self-Report Measurement of Well-Being in Autistic Adults: Psychometric Properties of the PERMA Profiler.",
     "journal": "Autism in adulthood",
     "doi": "10.1089/aut.2022.0049",
     "url": "https://doi.org/10.1089/aut.2022.0049"
    },
    {
     "key": "umucu_2024",
     "authors": "Umucu E, Granger TA, Pan D et al.",
     "year": "2024",
     "title": "Initial validation of a short version of the PERMA profiler in a national sample of rural veterans.",
     "journal": "Frontiers in public health",
     "doi": "10.3389/fpubh.2024.1500659",
     "url": "https://doi.org/10.3389/fpubh.2024.1500659"
    },
    {
     "key": "fernandes_2024",
     "authors": "Fernandes I, Zanini DS, Peixoto EM",
     "year": "2024",
     "title": "PERMA-Profiler for adolescents: validity evidence based on internal structure and related constructs.",
     "journal": "Frontiers in psychology",
     "doi": "10.3389/fpsyg.2024.1415084",
     "url": "https://doi.org/10.3389/fpsyg.2024.1415084"
    },
    {
     "key": "waigel_2023",
     "authors": "Waigel NC, Lemos VN",
     "year": "2023",
     "title": "Psychometric Properties of PERMA Proﬁler Scale in Argentinian Adolescents.",
     "journal": "International journal of psychological research",
     "doi": "10.21500/20112084.5737",
     "url": "https://doi.org/10.21500/20112084.5737"
    },
    {
     "key": "ng_2025",
     "authors": "Ng XH, Chu J, Doshi K",
     "year": "2025",
     "title": "A PERMA-nent solution to understanding psychological wellbeing? Exploring the utility of the PERMA model in a university workplace.",
     "journal": "Frontiers in psychology",
     "doi": "10.3389/fpsyg.2025.1598910",
     "url": "https://doi.org/10.3389/fpsyg.2025.1598910"
    },
    {
     "key": "lorenz_2023",
     "authors": "Lorenz T, Ho J, Beyer M et al.",
     "year": "2023",
     "title": "Measuring PERMA+4: validation of the German version of the Positive Functioning at Work Scale.",
     "journal": "Frontiers in psychology",
     "doi": "10.3389/fpsyg.2023.1231299",
     "url": "https://doi.org/10.3389/fpsyg.2023.1231299"
    },
    {
     "key": "garciaselva_2024",
     "authors": "García-Selva A, Neipp MC, Solanes-Puchol Á et al.",
     "year": "2024",
     "title": "The PERMA+4 Positive Functioning at Work Scale: Spanish Adaptation and Validation.",
     "journal": "Psicothema",
     "doi": "10.7334/psicothema2023.341",
     "url": "https://doi.org/10.7334/psicothema2023.341"
    }
   ],
   "record_notes": "[Upgraded from v0.1 to v0.2 structure in pass two; criterion field split, licence re-verified 2026-07-12.] Overall confidence: the general PERMA-Profiler is moderately well studied, but the Workplace PERMA-Profiler specifically rests on a thin, geographically narrow occupational evidence base (chiefly Japanese, Korean, Chinese and Polish workers) with generally marginal confirmatory fit. Internal consistency is the one property that grades High; structural validity, criterion validity, test-retest and invariance all grade Low, and responsiveness/MIC is Absent. No UK validation, UK occupational norms or UK benchmark data were located. Schema stress-test notes: (1) The schema's single identity block does not cleanly separate the general PERMA-Profiler from the Workplace PERMA-Profiler, which have materially different evidence depth; the record repeatedly distinguishes them in prose but a structured field pairing each property to 'general form' versus 'workplace form' would record the asymmetry more honestly. (2) Test-retest was hard to record faithfully because several studies report ICCs that mix retest stability with internal consistency; the coefficients cited (for example 0.75 to 0.96, Watanabe 2018) should not be read as clean retest values. (3) The word 'clinical' is deliberately not applied to any workplace deployment here; where clinical screeners appear as validation comparators they are outcome measures, not the deployment context. (4) A dedicated COSMIN systematic review of the Workplace PERMA-Profiler was not located in this pass, so grades are synthesised from individual validation studies rather than an aggregated quality appraisal. (5) One candidate source (Persian older-adults validation, Payoun 2020) was excluded because no resolvable DOI was found; the Flourish Index paper surfaced by keyword overlap was excluded as it does not measure the PERMA-Profiler. The licence status is reported from the developers' distribution practice rather than a citable licence document, which the schema's licence_status field would ideally flag as evidence-grade metadata rather than a settled fact."
  },
  {
   "instrument_id": "uwes-9",
   "display_name": "Utrecht Work Engagement Scale, 9-item (UWES-9)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Utrecht Work Engagement Scale, 9-item (UWES-9)",
    "current_version": "UWES-9 (9-item short form of the original 17-item UWES). An ultra-short 3-item form (UWES-3) also exists.",
    "item_count": "9 items, scored 0 to 6 (never to always/daily); three subscales of 3 items each (vigour, dedication, absorption), commonly summed to a single engagement score.",
    "original_citation": "Schaufeli WB, Bakker AB, Salanova M (2006). The Measurement of Work Engagement With a Short Questionnaire: A Cross-National Study. Educational and Psychological Measurement. https://doi.org/10.1177/0013164405282471 (short form derived from the 17-item UWES, Schaufeli et al. 2002, https://doi.org/10.1023/a:1015630930326)",
    "steward_publisher": "Wilmar Schaufeli and colleagues (Occupational Health Psychology Unit, Utrecht University); distributed via the author's website with an accompanying test manual.",
    "licence_status": "CONFIRMED and now verified with exact wording. The current UWES survey form on the steward site states: 'Copyright Schaufeli & Bakker (2003). The Utrecht Work Engagement Scale is free for use for non-commercial scientific research. Commercial and/or non-scientific use is prohibited, unless previous written permission is granted by the authors'. So non-commercial scientific research use is free; all commercial and non-scientific use (which would include most employer/vendor workplace deployment) is prohibited without prior written permission.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://www.wilmarschaufeli.nl/publications/Schaufeli/Tests/UWES_GB_17.pdf (author/steward UWES form copyright notice)"
   },
   "constructs_claimed": "The UWES-9 claims to measure work engagement, defined as a positive, fulfilling, work-related state of mind comprising three dimensions: vigour (energy and mental resilience while working), dedication (a sense of significance, enthusiasm and pride) and absorption (being fully concentrated and happily engrossed in work) ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471); [Schaufeli et al. 2002](https://doi.org/10.1023/a:1015630930326)). Engagement is theorised as conceptually distinct from, and in part the positive antipode of, burnout, and is embedded in the job demands-resources tradition ([Crawford et al. 2010](https://doi.org/10.1037/a0019364)).",
   "deployment_context_caveat": "",
   "structural_validity": {
    "findings": "The dimensional structure of the UWES-9 is genuinely unresolved in the literature, and this is the instrument's central psychometric controversy. The developers reported that confirmatory factor analysis supported the intended three-factor structure (vigour, dedication, absorption) across ten countries (N = 14,521), while noting the three subscales are very highly intercorrelated ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). A dedicated review of 21 CFA studies of the UWES found no consensus: the three-factor structure was judged superior in 6 studies, a single general factor in a further 6, the one- and three-factor solutions were treated as equivalent in 8 studies, and 1 study confirmed neither, leading the author to warn that this ambiguity may challenge the three-factor conception of engagement itself ([Kulikowski 2017](https://doi.org/10.13075/ijomeh.1896.00947)). Primary studies since then remain split. A three-factor (or second-order) solution fit best in Norwegian occupational groups ([Nerstad et al. 2009](https://doi.org/10.1111/j.1467-9450.2009.00770.x)), Portuguese rescue workers ([Sinval et al. 2018](https://doi.org/10.3389/fpsyg.2017.02229)), a Vietnamese nurse sample (three-factor marginally better than one-factor, [Tran et al. 2020](https://doi.org/10.1002/1348-9585.12157)) and Spanish health workers (three correlated factors with correlated errors, [Dominguez-Salas et al. 2022](https://doi.org/10.2147/PRBM.S387242)). A one-factor solution was preferred in a Serbian sample ([Petrovic et al. 2017](https://doi.org/10.3389/fpsyg.2017.01799)) and in a German oncology-rehabilitation sample where a single factor explained 67% of variance ([Sautier et al. 2015](https://doi.org/10.1055/s-0035-1555912)), and unidimensionality was supported using ordinal methods in Scandinavian haemodialysis nurses ([Lindberg et al. 2025](https://doi.org/10.1186/s12912-025-03545-4)). Most starkly, a Swedish multi-occupational female sample (N = 702) obtained poor fit for one-, two- and three-factor models alike (RMSEA never below 0.166), i.e. no acceptable structure at all ([Willmer et al. 2019](https://doi.org/10.3389/fpsyg.2019.02771)). The recurring pattern is that inter-factor correlations are so high that the three subscales are difficult to separate empirically, which is why many authors recommend using the total score as a single engagement index ([Mills et al. 2011](https://doi.org/10.1007/s10902-011-9277-3)).",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate. Many studies with large total N, but findings are directly contradictory on the key question (1-factor vs 3-factor), so the evidence supports 'well studied but unresolved' rather than a settled structure."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent validity is reasonably supported: UWES scores correlate positively with job satisfaction, organisational commitment, meaning of work and perceived job resources, and negatively with burnout and turnover intention ([Sautier et al. 2015](https://doi.org/10.1055/s-0035-1555912); [Hallberg & Schaufeli 2006](https://doi.org/10.1027/1016-9040.11.2.119)). Discriminant validity is more contested. On the positive side, one CFA study concluded that work engagement, job involvement and organisational commitment are empirically distinct constructs reflecting different aspects of work attachment ([Hallberg & Schaufeli 2006](https://doi.org/10.1027/1016-9040.11.2.119)), and a meta-analytic review reported that engagement shows discriminant validity from, and incremental criterion validity over, established job attitudes ([Christian et al. 2011](https://doi.org/10.1111/j.1744-6570.2010.01203.x)). On the negative side, the sharpest challenge concerns overlap with burnout: a meta-analysis of 50 samples found that dimension-level correlations between burnout and engagement are high, that the two show a similar pattern of associations with correlates, and that controlling for burnout substantially reduced the effect sizes attributed to engagement, casting doubt on their functional distinctiveness ([Cole et al. 2012](https://doi.org/10.1177/0149206311415252)). The instrument's own developers frame engagement partly as the positive antipode of burnout, with a best-fitting two-factor burnout-engagement model in the original cross-national data ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)), which is consistent with, rather than resolving, the overlap concern. Discriminant validity against related well-being constructs (workaholism, job boredom) was more clearly demonstrated for the ultra-short UWES-3 across five national samples ([Schaufeli et al. 2019](https://doi.org/10.1027/1015-5759/a000430)).",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate. Convergent evidence is consistent and includes meta-analysis; discriminant evidence is genuinely mixed, with a credible meta-analytic challenge (burnout overlap) that is not fully answered."
   },
   "criterion_validity_reference_standard": {
    "findings": "Criterion validity against hard organisational and health outcomes is thinner than the volume of UWES research implies, and most evidence is cross-sectional or predictive of self-reported rather than registered outcomes. The strongest workplace-relevant evidence against a registered health outcome comes from a one-year prospective cohort of 4,921 employees: baseline UWES scores were negatively associated with register-recorded long-term sickness absence due to mental illness, but discrimination was only moderate (area under the ROC curve = 0.70) and below the pre-set threshold for practical screening use (0.75); crucially, UWES scores were NOT associated with sickness absence due to musculoskeletal or other somatic illness, so predictive validity was specific to mental-illness absence ([Roelen et al. 2014](https://doi.org/10.1007/s00420-014-0981-2)). For turnover, engagement measured with the UWES is repeatedly associated with lower turnover intention, but through cross-sectional or mediational designs rather than actual turnover: examples include Chinese nurses during COVID-19 ([Tang et al. 2022](https://doi.org/10.3389/fpubh.2022.1051895)) and surgical trainees' intention to leave training ([Dominguez et al. 2018](https://doi.org/10.1371/journal.pone.0197276)). At meta-analytic level, engagement relates positively to task and contextual performance and mediates demands/resources effects on performance, though these syntheses pool multiple engagement measures rather than the UWES-9 alone ([Christian et al. 2011](https://doi.org/10.1111/j.1744-6570.2010.01203.x)). A UK quality-improvement evaluation found a modest but statistically significant difference in UWES scores between intervention and control wards ([White et al. 2014](https://doi.org/10.1016/j.ijnurstu.2014.05.002)). Evidence linking UWES scores prospectively to diagnosed conditions, objective productivity or actual (not intended) turnover is sparse.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low. One good prospective registry study exists (with an informative null for somatic absence); most other criterion evidence is cross-sectional, uses intentions rather than behaviour, or pools multiple engagement measures."
   },
   "criterion_validity_organisational": {
    "findings": "Organisational criterion evidence (sickness absence, turnover, performance, diagnosed conditions in a work context): see the criterion findings; graded from the pass-one record.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low. One good prospective registry study exists (with an informative null for somatic absence); most other criterion evidence is cross-sectional, uses intentions rather than behaviour, or pools multiple engagement measures."
   },
   "internal_consistency": {
    "findings": "Internal consistency is the UWES-9's most consistently strong property. The developers reported the three subscale scores had good internal consistency across the cross-national samples ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). Total-score Cronbach's alpha is typically in the low-to-mid 0.90s: 0.93 (ordinal alpha) in Scandinavian haemodialysis nurses ([Lindberg et al. 2025](https://doi.org/10.1186/s12912-025-03545-4)), 0.93 for the total scale in Vietnamese nurses (subscales 0.86 vigour, 0.77 absorption, 0.90 dedication) ([Tran et al. 2020](https://doi.org/10.1002/1348-9585.12157)), and 0.94 in the German oncology-rehabilitation sample ([Sautier et al. 2015](https://doi.org/10.1055/s-0035-1555912)). Reliability was likewise satisfactory in Norwegian ([Nerstad et al. 2009](https://doi.org/10.1111/j.1467-9450.2009.00770.x)) and Serbian ([Petrovic et al. 2017](https://doi.org/10.3389/fpsyg.2017.01799)) validations. A caveat: alpha values in the 0.90s for a 9-item scale with very high inter-item correlation partly reflect redundancy, and high total-scale alpha does not adjudicate the one- versus three-factor question. Where the three-item subscales are used separately, the absorption subscale tends to be the weakest ([Tran et al. 2020](https://doi.org/10.1002/1348-9585.12157)).",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High. Multiple good-quality studies across many languages and occupations, large total N, consistently reporting total-score alpha in the low-to-mid 0.90s."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Low",
    "status": "thin",
    "indirectness": "see summary",
    "summary": "Genuine test-retest reliability evidence for the UWES-9 is comparatively scarce, and this is the weakest-documented property relative to the instrument's popularity. The developers' original short-form paper states summarily that the three UWES-9 scores have 'good' test-retest reliability across the cross-national dataset, but reports no coefficients or retest intervals in the material available for this record ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). Among UWES-9 validations, one of the few to report a short-interval coefficient found an intraclass correlation of only 0.48 over roughly three months in a Vietnamese nurse subsample, which is modest and below conventional stability thresholds ([Tran et al. 2020](https://doi.org/10.1002/1348-9585.12157)). A multisample, longitudinal construct-validity study of the UWES exists and, by its title, addresses longitudinal/temporal evidence ([Seppala et al. 2009](https://doi.org/10.1007/s10902-008-9100-y)); however, its full text and abstract could not be retrieved in this pass, so no specific stability coefficient from it is asserted here, and any stability figures it reports pertain principally to the 17-item UWES rather than the 9-item form. Many national validation studies report internal consistency and factor structure but omit test-retest data entirely. Engagement is theorised as a relatively stable state, so a stronger, replicated test-retest evidence base would be expected for the 9-item form than currently exists.",
    "confidence_note": "Low. Test-retest evidence specific to the UWES-9 is sparse: the developer claim is summary and coefficient-free in the material accessed, the clearest located UWES-9 retest coefficient (ICC 0.48 over ~3 months) is modest, and longer-interval stability evidence is tied to the 17-item form and was not retrievable this session. The thinness of a replicated UWES-9 test-retest base is itself the finding."
   },
   "measurement_invariance": {
    "findings": "Measurement invariance has been tested repeatedly, most often across country/language and occupational group, with configural and metric invariance commonly achieved and scalar invariance less consistently so. Full-scale measurement invariance for the UWES-9 was obtained between Portuguese and Brazilian workers for both the three-factor first-order and second-order models, permitting direct mean comparisons ([Sinval et al. 2018](https://doi.org/10.3389/fpsyg.2018.00353)). In Portuguese rescue workers, the UWES-9 first-order model reached full (uniqueness) invariance across occupational groups, while the second-order model reached only partial (metric) second-order invariance ([Sinval et al. 2018](https://doi.org/10.3389/fpsyg.2017.02229)). Gender invariance has been supported in Spanish health-care professionals ([Dominguez-Salas et al. 2022](https://doi.org/10.2147/PRBM.S387242)), and factorial invariance across ten occupational groups was reported for the Norwegian UWES ([Nerstad et al. 2009](https://doi.org/10.1111/j.1467-9450.2009.00770.x)). The developers' cross-national work established the UWES-9 across ten countries but treated cross-national equivalence descriptively rather than through the full modern invariance hierarchy ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). Against this, a Rasch analysis of the 17-item UWES in Korea found that only 9 of 17 items differentiated adequately between men and women, i.e. differential item functioning by sex ([Song et al. 2020](https://doi.org/10.1177/0033294120922494)), a caution that invariance is not universal. Over-time (longitudinal) invariance is less frequently formally tested; the multisample longitudinal study is the main source addressing temporal stability of the structure ([Seppala et al. 2009](https://doi.org/10.1007/s10902-008-9100-y)).",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate. Several studies reach metric and sometimes scalar/full invariance across language and occupation, but scalar invariance is not universal, some DIF by sex is reported, and formal over-time invariance testing is limited."
   },
   "responsiveness_mic": {
    "findings": "Formal responsiveness (sensitivity to change) and a minimal important change (MIC) value have not been established for the UWES-9. No located study derived an anchor-based or distribution-based MIC, and the instrument was not developed as an outcome measure with defined change metrics. Indirect evidence of sensitivity to change comes from a UK evaluation of the 'Productive Ward' quality-improvement programme, where UWES scores were modestly but significantly higher in intervention wards than matched controls (4.33 vs 4.07, p = 0.013), which the authors interpreted as the UWES being able to detect programme-related differences ([White et al. 2014](https://doi.org/10.1016/j.ijnurstu.2014.05.002)). This is a between-group cross-sectional contrast rather than a within-person responsiveness or MIC analysis. No UK or other MIC benchmark was located in this pass.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Very low. No MIC established and no formal responsiveness study located; only indirect, between-group evidence that scores can differ with an intervention."
   },
   "populations_languages_norms": {
    "findings": "The UWES-9 has been validated in a very wide range of languages and occupations, giving broad but heterogeneous population coverage. The developers' short-form study drew on samples from ten countries (N = 14,521) ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). Located validations span Norwegian occupational groups ([Nerstad et al. 2009](https://doi.org/10.1111/j.1467-9450.2009.00770.x)), Serbian employees ([Petrovic et al. 2017](https://doi.org/10.3389/fpsyg.2017.01799)), Brazilian and Portuguese workers ([Sinval et al. 2018](https://doi.org/10.3389/fpsyg.2018.00353)), Portuguese rescue workers ([Sinval et al. 2018](https://doi.org/10.3389/fpsyg.2017.02229)), Vietnamese hospital nurses ([Tran et al. 2020](https://doi.org/10.1002/1348-9585.12157)), Spanish health-care professionals ([Dominguez-Salas et al. 2022](https://doi.org/10.2147/PRBM.S387242)), a German oncology-rehabilitation patient sample ([Sautier et al. 2015](https://doi.org/10.1055/s-0035-1555912)), a Swedish multi-occupational female sample ([Willmer et al. 2019](https://doi.org/10.3389/fpsyg.2019.02771)) and Scandinavian (Danish and Swedish) haemodialysis nurses ([Lindberg et al. 2025](https://doi.org/10.1186/s12912-025-03545-4)); an ultra-short UWES-3 was validated across Finland, Japan, the Netherlands, Belgium/Flanders and Spain ([Schaufeli et al. 2019](https://doi.org/10.1027/1015-5759/a000430)) and in Peru ([Merino-Soto et al. 2022](https://doi.org/10.3390/ijerph19020890)). UK-specific evidence is limited: the clearest UK-context deployment located is the English-language 'Productive Ward' nursing study (conducted in Ireland with a UK/Ireland health-service context), which used the UWES and reported mean scores around 4.1 to 4.3 on the 0 to 6 metric, but this is a study sample, not a representative UK norm ([White et al. 2014](https://doi.org/10.1016/j.ijnurstu.2014.05.002)). No representative UK normative dataset was located in this pass; the reference norm tables that exist are those in the authors' international test manual ([UWES Test Manual, Schaufeli & Bakker 2003](https://www.wilmarschaufeli.nl/publications/Schaufeli/Test%20Manuals/Test_manual_UWES_English.pdf)), which are international rather than UK-specific.",
    "grade": "Moderate",
    "indirectness": "see findings"
   },
   "criticisms_controversies": "Three recurring criticisms appear in the literature. First, factor-structure instability: a review of 21 CFA studies found no consensus on whether the UWES is one-dimensional or three-dimensional, with roughly equal support for each and one study confirming neither, which the author argued could undermine the three-dimensional conception of engagement itself ([Kulikowski 2017](https://doi.org/10.13075/ijomeh.1896.00947)); an extreme case obtained no acceptable fit for any of the one-, two- or three-factor models ([Willmer et al. 2019](https://doi.org/10.3389/fpsyg.2019.02771)). Second, discriminant validity from burnout: a meta-analysis of 50 samples reported high dimension-level correlations, near-identical correlate patterns, and shrinking engagement effect sizes once burnout is controlled, questioning whether engagement and burnout are functionally distinct ([Cole et al. 2012](https://doi.org/10.1177/0149206311415252)). The developers' own framing of engagement as the positive antipode of burnout, with a best-fitting combined two-factor model, sits uneasily with claims of full independence ([Schaufeli et al. 2006](https://doi.org/10.1177/0013164405282471)). Third, redundancy and subscale separability: because the three subscales correlate so highly, several authors conclude the total score should be used and that the subscales add little, so reporting 'vigour/dedication/absorption' as three distinct measures may over-claim ([Mills et al. 2011](https://doi.org/10.1007/s10902-011-9277-3)). A related psychometric caution is that Cronbach's alpha in the 0.90s for such highly intercorrelated items partly reflects item redundancy rather than only reliability. Discriminant concerns against neighbouring attitudes (job involvement, organisational commitment) are, by contrast, comparatively reassuring ([Hallberg & Schaufeli 2006](https://doi.org/10.1027/1016-9040.11.2.119)).",
   "citations": [
    {
     "key": "Schaufeli2006",
     "authors": "Schaufeli WB, Bakker AB, Salanova M",
     "year": "2006",
     "title": "The Measurement of Work Engagement With a Short Questionnaire: A Cross-National Study",
     "journal": "Educational and Psychological Measurement",
     "doi": "10.1177/0013164405282471",
     "url": "https://doi.org/10.1177/0013164405282471"
    },
    {
     "key": "Schaufeli2002",
     "authors": "Schaufeli WB, Salanova M, González-Romá V, Bakker AB",
     "year": "2002",
     "title": "The Measurement of Engagement and Burnout: A Two Sample Confirmatory Factor Analytic Approach",
     "journal": "Journal of Happiness Studies",
     "doi": "10.1023/a:1015630930326",
     "url": "https://doi.org/10.1023/a:1015630930326"
    },
    {
     "key": "Seppala2009",
     "authors": "Seppälä P, Mauno S, Feldt T, Hakanen J, Kinnunen U, Tolvanen A, Schaufeli W",
     "year": "2009",
     "title": "The Construct Validity of the Utrecht Work Engagement Scale: Multisample and Longitudinal Evidence",
     "journal": "Journal of Happiness Studies",
     "doi": "10.1007/s10902-008-9100-y",
     "url": "https://doi.org/10.1007/s10902-008-9100-y"
    },
    {
     "key": "Schaufeli2019",
     "authors": "Schaufeli WB, Shimazu A, Hakanen J, Salanova M, De Witte H",
     "year": "2019",
     "title": "An Ultra-Short Measure for Work Engagement: The UWES-3",
     "journal": "European Journal of Psychological Assessment",
     "doi": "10.1027/1015-5759/a000430",
     "url": "https://doi.org/10.1027/1015-5759/a000430"
    },
    {
     "key": "Kulikowski2017",
     "authors": "Kulikowski K",
     "year": "2017",
     "title": "Do we all agree on how to measure work engagement? Factorial validity of Utrecht Work Engagement Scale as a standard measurement tool: A literature review",
     "journal": "International Journal of Occupational Medicine and Environmental Health",
     "doi": "10.13075/ijomeh.1896.00947",
     "url": "https://doi.org/10.13075/ijomeh.1896.00947"
    },
    {
     "key": "Nerstad2009",
     "authors": "Nerstad CGL, Richardsen AM, Martinussen M",
     "year": "2009",
     "title": "Factorial validity of the Utrecht Work Engagement Scale (UWES) across occupational groups in Norway",
     "journal": "Scandinavian Journal of Psychology",
     "doi": "10.1111/j.1467-9450.2009.00770.x",
     "url": "https://doi.org/10.1111/j.1467-9450.2009.00770.x"
    },
    {
     "key": "Willmer2019",
     "authors": "Willmer M, Westerberg Jacobson J, Lindberg M",
     "year": "2019",
     "title": "Exploratory and Confirmatory Factor Analysis of the 9-Item Utrecht Work Engagement Scale in a Multi-Occupational Female Sample",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2019.02771",
     "url": "https://doi.org/10.3389/fpsyg.2019.02771"
    },
    {
     "key": "Lindberg2025",
     "authors": "Lindberg M, Knudsen K, Lindberg M",
     "year": "2025",
     "title": "Factor structure of the Utrecht Work Engagement Scale in a sample of Danish and Swedish haemodialysis nurses",
     "journal": "BMC Nursing",
     "doi": "10.1186/s12912-025-03545-4",
     "url": "https://doi.org/10.1186/s12912-025-03545-4"
    },
    {
     "key": "SinvalRescue2018",
     "authors": "Sinval J, Marques-Pinto A, Queirós C, Marôco J",
     "year": "2018",
     "title": "Work Engagement among Rescue Workers: Psychometric Properties of the Portuguese UWES",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2017.02229",
     "url": "https://doi.org/10.3389/fpsyg.2017.02229"
    },
    {
     "key": "Petrovic2017",
     "authors": "Petrović IB, Vukelić M, Čizmić S",
     "year": "2017",
     "title": "Work Engagement in Serbia: Psychometric Properties of the Serbian Version of the Utrecht Work Engagement Scale (UWES)",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2017.01799",
     "url": "https://doi.org/10.3389/fpsyg.2017.01799"
    },
    {
     "key": "SinvalBP2018",
     "authors": "Sinval J, Pasian S, Queirós C, Marôco J",
     "year": "2018",
     "title": "Brazil-Portugal Transcultural Adaptation of the UWES-9: Internal Consistency, Dimensionality, and Measurement Invariance",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2018.00353",
     "url": "https://doi.org/10.3389/fpsyg.2018.00353"
    },
    {
     "key": "Mills2011",
     "authors": "Mills MJ, Culbertson SS, Fullagar CJ",
     "year": "2011",
     "title": "Conceptualizing and Measuring Engagement: An Analysis of the Utrecht Work Engagement Scale",
     "journal": "Journal of Happiness Studies",
     "doi": "10.1007/s10902-011-9277-3",
     "url": "https://doi.org/10.1007/s10902-011-9277-3"
    },
    {
     "key": "Sautier2015",
     "authors": "Sautier LP, Scherwath A, Weis J, Sarkar S, Bosbach M, Schendel M, Ladehoff N, Koch U, Mehnert A",
     "year": "2015",
     "title": "Assessment of Work Engagement in Patients with Hematological Malignancies: Psychometric Properties of the German Version of the UWES-9",
     "journal": "Die Rehabilitation",
     "doi": "10.1055/s-0035-1555912",
     "url": "https://doi.org/10.1055/s-0035-1555912"
    },
    {
     "key": "Song2020",
     "authors": "Song HD, Hong AJ, Jo Y",
     "year": "2020",
     "title": "Psychometric Investigation of the Utrecht Work Engagement Scale-17 Using the Rasch Measurement Model",
     "journal": "Psychological Reports",
     "doi": "10.1177/0033294120922494",
     "url": "https://doi.org/10.1177/0033294120922494"
    },
    {
     "key": "Hallberg2006",
     "authors": "Hallberg UE, Schaufeli WB",
     "year": "2006",
     "title": "\"Same Same\" But Different? Can Work Engagement Be Discriminated from Job Involvement and Organizational Commitment?",
     "journal": "European Psychologist",
     "doi": "10.1027/1016-9040.11.2.119",
     "url": "https://doi.org/10.1027/1016-9040.11.2.119"
    },
    {
     "key": "Cole2012",
     "authors": "Cole MS, Walter F, Bedeian AG, O'Boyle EH",
     "year": "2012",
     "title": "Job Burnout and Employee Engagement: A Meta-Analytic Examination of Construct Proliferation",
     "journal": "Journal of Management",
     "doi": "10.1177/0149206311415252",
     "url": "https://doi.org/10.1177/0149206311415252"
    },
    {
     "key": "Christian2011",
     "authors": "Christian MS, Garza AS, Slaughter JE",
     "year": "2011",
     "title": "Work Engagement: A Quantitative Review and Test of Its Relations with Task and Contextual Performance",
     "journal": "Personnel Psychology",
     "doi": "10.1111/j.1744-6570.2010.01203.x",
     "url": "https://doi.org/10.1111/j.1744-6570.2010.01203.x"
    },
    {
     "key": "Crawford2010",
     "authors": "Crawford ER, LePine JA, Rich BL",
     "year": "2010",
     "title": "Linking job demands and resources to employee engagement and burnout: A theoretical extension and meta-analytic test",
     "journal": "Journal of Applied Psychology",
     "doi": "10.1037/a0019364",
     "url": "https://doi.org/10.1037/a0019364"
    },
    {
     "key": "Roelen2014",
     "authors": "Roelen CAM, van Hoffen MFA, Groothoff JW, de Bruin J, Schaufeli WB, van Rhenen W",
     "year": "2014",
     "title": "Can the Maslach Burnout Inventory and Utrecht Work Engagement Scale be used to screen for risk of long-term sickness absence?",
     "journal": "International Archives of Occupational and Environmental Health",
     "doi": "10.1007/s00420-014-0981-2",
     "url": "https://doi.org/10.1007/s00420-014-0981-2"
    },
    {
     "key": "White2014",
     "authors": "White M, Wells JS, Butterworth T",
     "year": "2014",
     "title": "The impact of a large-scale quality improvement programme on work engagement: preliminary results from a national cross-sectional survey of the 'Productive Ward'",
     "journal": "International Journal of Nursing Studies",
     "doi": "10.1016/j.ijnurstu.2014.05.002",
     "url": "https://doi.org/10.1016/j.ijnurstu.2014.05.002"
    },
    {
     "key": "Tang2022",
     "authors": "Tang Y, Dias Martins LM, Wang SB, He QX, Huang HH",
     "year": "2022",
     "title": "The impact of nurses' sense of security on turnover intention during the normalization of COVID-19 epidemic: The mediating role of work engagement",
     "journal": "Frontiers in Public Health",
     "doi": "10.3389/fpubh.2022.1051895",
     "url": "https://doi.org/10.3389/fpubh.2022.1051895"
    },
    {
     "key": "Dominguez2018",
     "authors": "Dominguez LC, Stassen L, de Grave W, Sanabria A, Alfonso E, Dolmans D",
     "year": "2018",
     "title": "Taking control: Is job crafting related to the intention to leave surgical training?",
     "journal": "PLoS ONE",
     "doi": "10.1371/journal.pone.0197276",
     "url": "https://doi.org/10.1371/journal.pone.0197276"
    },
    {
     "key": "Tran2020",
     "authors": "Tran TTT, Watanabe K, Imamura K, Nguyen HT, Sasaki N, Kuribayashi K, Sakuraya A, et al.",
     "year": "2020",
     "title": "Reliability and validity of the Vietnamese version of the 9-item Utrecht Work Engagement Scale",
     "journal": "Journal of Occupational Health",
     "doi": "10.1002/1348-9585.12157",
     "url": "https://doi.org/10.1002/1348-9585.12157"
    },
    {
     "key": "DominguezSalas2022",
     "authors": "Domínguez-Salas S, Rodríguez-Domínguez C, Arcos-Romero AI, Allande-Cussó R, et al.",
     "year": "2022",
     "title": "Psychometric Properties of the Utrecht Work Engagement Scale (UWES-9) in a Sample of Active Health Care Professionals in Spain",
     "journal": "Psychology Research and Behavior Management",
     "doi": "10.2147/PRBM.S387242",
     "url": "https://doi.org/10.2147/PRBM.S387242"
    },
    {
     "key": "MerinoSoto2022",
     "authors": "Merino-Soto C, Lozano-Huamán M, Lima-Mendoza S, Calderón de la Cruz G, Juárez-García A, Toledano-Toledano F",
     "year": "2022",
     "title": "Ultrashort Version of the Utrecht Work Engagement Scale (UWES-3): A Psychometric Assessment",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph19020890",
     "url": "https://doi.org/10.3390/ijerph19020890"
    },
    {
     "key": "SchaufeliBAT2020",
     "authors": "Schaufeli WB, Desart S, De Witte H",
     "year": "2020",
     "title": "Burnout Assessment Tool (BAT): Development, Validity, and Reliability",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph17249495",
     "url": "https://doi.org/10.3390/ijerph17249495"
    },
    {
     "key": "UWESManual",
     "authors": "Schaufeli WB, Bakker AB",
     "year": "2003",
     "title": "UWES Utrecht Work Engagement Scale: Preliminary Manual (Version 1)",
     "journal": "Occupational Health Psychology Unit, Utrecht University (test manual, grey literature)",
     "doi": "",
     "url": "https://www.wilmarschaufeli.nl/publications/Schaufeli/Test%20Manuals/Test_manual_UWES_English.pdf"
    }
   ],
   "record_notes": "[Upgraded from v0.1 to v0.2 structure in pass two; criterion field split, licence re-verified 2026-07-12.] Overall confidence: internal consistency is High; structural validity, convergent/discriminant validity and measurement invariance are Moderate but each carries a genuine, cited contradiction rather than clean support; criterion validity is Low; test-retest is Low with the absence itself a finding; responsiveness/MIC is Very low (effectively absent). Points the schema made hard to record honestly: (1) The one-factor versus three-factor debate is not a defect to be resolved to a single 'confidence' but a standing feature of the evidence; the 'structural_validity' field forces a single grade onto directly contradictory findings, so 'Moderate' here means 'extensively studied but unresolved', not 'moderately good'. (2) 'constructs_claimed' presents three subscales, yet a substantial strand of evidence argues the subscales are not empirically separable and only the total should be used; the field cannot easily flag that the claimed structure is itself contested. (3) Several strong stability and invariance findings pertain to the 17-item UWES, not the 9-item form; the schema does not distinguish evidence transferred from the parent instrument from evidence earned by the UWES-9 directly, and I have flagged this inline where it applies. (4) Test-retest and MIC are near-absent for the 9-item form specifically despite the instrument's popularity, which is a more important finding than a coefficient would have been. (5) UK-specific norms were not located as a representative benchmark; the 'Productive Ward' study is UK/Ireland health-service context but is a study sample, not a norm, and the only reference norms are the authors' international (non-UK) test manual. (6) Per the registry rule on 'clinical': the UWES-9 has been deployed in patient-rehabilitation samples (for example haematological malignancy patients), but it is a work-engagement instrument, not a clinical screener, and none of its validation was earned in a diagnostic setting; workplace deployment should be read as a non-clinical context throughout."
  },
  {
   "instrument_id": "phq-9",
   "display_name": "Patient Health Questionnaire-9 (PHQ-9)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Patient Health Questionnaire-9 (PHQ-9)",
    "current_version": "PHQ-9 (2001); abbreviated variants PHQ-8 (omits item 9) and PHQ-2 (first two items) are in wide use",
    "item_count": "9 items, each scored 0 to 3 (total 0 to 27); severity bands 5, 10, 15, 20 map to mild, moderate, moderately severe and severe",
    "original_citation": "Kroenke K, Spitzer RL, Williams JB (2001) The PHQ-9: validity of a brief depression severity measure. Journal of General Internal Medicine. 10.1046/j.1525-1497.2001.016009606.x",
    "steward_publisher": "Developed by Spitzer, Williams and Kroenke as the depression module of the PRIME-MD PHQ, under an educational grant from Pfizer Inc; the PHQ suite was subsequently released by Pfizer for free public access.",
    "licence_status": "CONFIRMED and now verified, with a wording refinement. The official steward terms state: 'Content found at the PHQ Screeners site is expressly exempted from Pfizer's general copyright restrictions; content found on the PHQ Screeners site is free for download and use as stated within the PHQ Screeners site'. This confirms free download and use with no permission or fee, but is a copyright exemption / free-use grant rather than a formal public-domain dedication.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "https://www.phqscreeners.com/terms (official PHQ Screeners terms, Pfizer)"
   },
   "constructs_claimed": "Severity of depressive symptoms over the preceding two weeks and, secondarily, provisional detection of major depressive disorder. The nine items map directly onto the nine DSM-IV/DSM-5 symptom criteria for a major depressive episode, so the instrument claims both a continuous severity construct and a criterion-referenced screening function ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)). It is a depression-specific measure, not a general wellbeing or distress index, and was designed for primary-care and clinical use rather than for occupational or workplace-wellbeing measurement.",
   "deployment_context_caveat": "PHQ-9 is a depression screener whose criterion validity was earned in primary-care and specialist diagnostic settings; workplace deployment is a different context, and the word clinical is not applied affirmatively to any workplace use.",
   "structural_validity": {
    "findings": "The dimensionality of the PHQ-9 is genuinely contested, but the practical reading across studies is that it behaves as an essentially unidimensional depression severity scale even where a two-factor model fits marginally better. The original development treated the total score as a single severity dimension ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)). Within the individual participant data (IPD) programme, a comparison of unidimensional, two-dimensional (cognitive/affective versus somatic) and bifactor latent models found that all fitted reasonably and that scoring by the more complex latent models improved sensitivity only marginally (by about 0.04 to 0.05) while reducing specificity, relative to the sum score at a cut-off of 10 or above ([Fischer et al 2021](https://doi.org/10.1017/S0033291721000131)). Several population studies reach the same conclusion from different angles: in Korean nationally representative data a single-factor model fitted well (CFI 0.944) with satisfactory internal consistency ([Lee et al 2023](https://doi.org/10.3389/fpsyg.2023.1217038)); in Brazilian university students both one- and two-dimensional models fitted, with the authors favouring the unidimensional solution ([Rufino et al 2024](https://doi.org/10.1016/j.jad.2024.01.051)); and after traumatic brain injury a general factor explained about 85% of the variance despite a slight bifactor improvement ([Teymoori et al 2020](https://doi.org/10.3390/jcm9030873)). Where a two-factor (somatic versus cognitive-affective) structure is reported, the factors tend to be very highly correlated: in a stroke sample the two-factor model fitted slightly better than the one-factor model (CFI 0.984 versus 0.974) but the inter-factor correlation of 0.866 pointed back to unidimensionality ([Blake et al 2024](https://doi.org/10.1016/j.jpsychores.2024.111983)), and an Italian coronary heart disease sample reported a bi-dimensional somatic/cognitive solution ([Di et al 2025](https://doi.org/10.1097/JCN.0000000000001178)). The disagreement is therefore about whether the somatic items form a distinguishable subfactor, not about whether a single severity score is defensible; the weight of evidence supports scoring and interpreting a single total.",
    "grade": "High",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High: many good-quality factor-analytic studies across large and varied samples, converging on essentially unidimensional use despite a recurring, well-characterised somatic/cognitive two-factor debate."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent evidence is consistent in direction but uneven in magnitude, and discriminant separation from anxiety is imperfect. In development, higher PHQ-9 scores tracked substantial decrements across all six SF-20 functional-status subscales and greater symptom-related difficulty, supporting construct validity ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)). In the Chinese general population the PHQ-9 correlated negatively with SF-36 subscales (r from -0.11 to -0.47) as expected, but its correlation with the Zung Self-Rating Depression Scale was unexpectedly weak (r = 0.29), an inconsistency worth noting against the usual assumption of strong convergence with other depression measures ([Wang et al 2014](https://doi.org/10.1016/j.genhosppsych.2014.05.021)). Convergent relationships with sleep quality, alcohol use and physical activity have also been reported in students ([Rufino et al 2024](https://doi.org/10.1016/j.jad.2024.01.051)). On discriminant validity, the PHQ-9 and the GAD-7 anxiety scale share a large common factor: after traumatic brain injury a general distress factor dominated both scales (about 85% of variance) even though the instruments related differently to SF-36 subscales ([Teymoori et al 2020](https://doi.org/10.3390/jcm9030873)), and in coronary heart disease PHQ-9 and GAD-7 scores were significantly positively correlated ([Di et al 2025](https://doi.org/10.1097/JCN.0000000000001178)). Depression and anxiety symptoms measured this way are statistically distinguishable but strongly overlapping.",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: multiple studies establish expected convergent and functional-status correlations, but magnitudes vary (including one weak convergent correlation) and discriminant separation from anxiety is only partial."
   },
   "criterion_validity_reference_standard": {
    "findings": "Criterion validity against a diagnostic interview for major depression is the PHQ-9's best-evidenced property, and it is strong; criterion validity against organisational or workplace outcomes is, by contrast, not established in the retrieved literature. In the original clinical validation a cut-off of 10 or above gave 88% sensitivity and 88% specificity against an independent mental-health-professional interview ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)). The individual participant data meta-analysis programme is the anchor here. The 2019 IPD meta-analysis (58 studies, n = 17,357, 2,312 major depression cases) found combined sensitivity and specificity maximised at a cut-off of 10 or above against semistructured interviews (sensitivity 0.88, 95% CI 0.83 to 0.92; specificity 0.85, 0.82 to 0.88) ([Levis et al 2019](https://doi.org/10.1136/bmj.l1476)). The 2021 update (100 studies, n = 44,503) reproduced this at the same cut-off (sensitivity 0.85, 0.79 to 0.89; specificity 0.85, 0.82 to 0.87) ([Negeri et al 2021](https://doi.org/10.1136/bmj.n2183)). A crucial nuance is reference-standard dependence: sensitivity against semistructured clinician interviews ran markedly higher than against fully structured lay interviews (median difference around 21%) or the MINI (around 11%), while specificity was similar across standards ([Levis et al 2019](https://doi.org/10.1136/bmj.l1476), [Negeri et al 2021](https://doi.org/10.1136/bmj.n2183)). The diagnostic algorithm approach performed worse than the cut-off (algorithm sensitivity around 0.57 to 0.61 versus 0.88 for the cut-off against semistructured interviews) ([He et al 2019](https://doi.org/10.1159/000502294)), and the PHQ-2 with cut-off 3 or above (sensitivity 0.72, specificity 0.85) is a reasonable first-stage screen followed by the PHQ-9 ([Levis et al 2020](https://doi.org/10.1001/jama.2020.6504)). For the health-related associations closest to organisational outcomes, the development study showed that self-reported sick days and health-care utilisation rose with PHQ-9 severity ([Kroenke 2001](https://doi.org/10.1046/j.1525-1497.2001.016009606.x)), but no study retrieved this session tested PHQ-9 scores against measured sickness absence, staff turnover, productivity loss or other workplace outcomes as criteria. Criterion validity against organisational endpoints should therefore be treated as unestablished.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High for diagnostic criterion validity against clinical interviews (large, consistent IPD evidence); Absent for criterion validity against organisational/workplace outcomes (no such studies retrieved). The overall grade is split by criterion."
   },
   "criterion_validity_organisational": {
    "findings": "Organisational criterion evidence (sickness absence, turnover, performance, diagnosed conditions in a work context): see the criterion findings; graded from the pass-one record.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High for diagnostic criterion validity against clinical interviews (large, consistent IPD evidence); Absent for criterion validity against organisational/workplace outcomes (no such studies retrieved). The overall grade is split by criterion."
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently high across populations. A reliability generalisation meta-analysis of 60 studies (232,147 participants) estimated a pooled Cronbach's alpha of 0.86 (95% CI 0.85 to 0.87), with self-administered formats slightly higher (alpha 0.87) than face-to-face interview administration (alpha 0.80), though between-study heterogeneity was very large (I-squared 99.3%) ([Ajele 2025](https://doi.org/10.1007/s44192-025-00181-x)). Single-study estimates sit in the same range: alpha 0.86 in the Chinese general population ([Wang et al 2014](https://doi.org/10.1016/j.genhosppsych.2014.05.021)) and omega 0.812 in a Korean nationally representative sample ([Lee et al 2023](https://doi.org/10.3389/fpsyg.2023.1217038)). A systematic review of the wider PHQ family likewise reported good internal consistency for the PHQ-9 ([Kroenke 2010](https://doi.org/10.1016/j.genhosppsych.2010.03.006)), and a meta-analysis of the Spanish-language versions evaluated internal consistency alongside accuracy ([Martinez et al 2023](https://doi.org/10.1001/jamanetworkopen.2023.36529)). Reported coefficients of omega tend to align with alpha where both are given, but omega is reported far less often than alpha.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "High: a large reliability-generalisation meta-analysis plus multiple primary studies converge on alpha around 0.86, albeit with substantial heterogeneity and sparse omega reporting."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Moderate",
    "status": "thin",
    "indirectness": "see summary",
    "summary": "Test-retest reliability exists but is markedly under-studied relative to the vast diagnostic-accuracy literature, and this asymmetry is itself the finding. The reliability generalisation meta-analysis could pool a test-retest estimate from only 8 of its 60 studies, yielding 0.82 (95% CI 0.74 to 0.90) ([Ajele 2025](https://doi.org/10.1007/s44192-025-00181-x)). Primary estimates are favourable where they exist: a two-week retest correlation of 0.86 in the Chinese general population ([Wang et al 2014](https://doi.org/10.1016/j.genhosppsych.2014.05.021)), test-retest described as excellent over a seven-day interval in a late-life depression treatment sample ([Lowe et al 2004](https://doi.org/10.1097/00005650-200412000-00006)), and intraclass agreement assessed for self- versus telephone-administration in Spanish primary care ([Pinto-Meza et al 2005](https://doi.org/10.1111/j.1525-1497.2005.0144.x)). No test-retest study in an occupational or workplace sample, and none using a UK working population, was located in this pass. The property is present and reassuring in the settings studied, but the evidence base is thin and skewed toward clinical and general-population samples over short intervals.",
    "confidence_note": "Moderate: several favourable estimates (roughly 0.82 to 0.86) exist and a meta-analytic pooled value is available, but from few studies (n = 8 pooled), short intervals, and no workplace or UK occupational data. Not absent, but comparatively neglected."
   },
   "measurement_invariance": {
    "findings": "Invariance has been tested piecemeal and mostly reaches metric to scalar level within the groups examined, but coverage of occupational and UK working populations is absent. In a Korean nationally representative sample the one-factor PHQ-9 showed equivalent structure, factor loadings and item intercepts across age groups, i.e. up to scalar invariance across ages ([Lee et al 2023](https://doi.org/10.3389/fpsyg.2023.1217038)). Invariance across sex and across age (65 and over versus under 65) was tested in an Italian coronary heart disease cohort ([Di et al 2025](https://doi.org/10.1097/JCN.0000000000001178)). Item response theory analysis in a large Danish implantable-defibrillator cohort found no differential item functioning across educational level, age, clinical indication or heart-failure severity, with only a single item showing DIF by gender ([Pedersen et al 2016](https://doi.org/10.1016/j.jpsychores.2016.09.010)). This points to broadly stable measurement across sex and age in clinical and general populations. However, the studies are confined to clinical or general-population samples; invariance across occupations, across employed versus unemployed status, over repeated workplace administrations, and within UK working populations was not established in retrieved evidence. Longitudinal (over-time) invariance is likewise weakly evidenced for the PHQ-9 specifically.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Low to Moderate: scalar invariance is demonstrated across age (one strong study) and DIF is minimal in another, but evidence is scattered across clinical populations, occupation and UK-workplace invariance is untested, and over-time invariance is weak."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness to change is well supported, and a minimal important change has been proposed, though the MIC rests largely on one study. In the IMPACT late-life depression trial (n = 434 intervention participants) the PHQ-9 was responsive to treatment, with an effect size (about -1.3 at three months) exceeding the SCL-20 depression scale at three months and matching it at six months; change scores discriminated persistent depression, partial remission and full remission against structured diagnostic interviews ([Lowe et al 2004](https://doi.org/10.1097/00005650-200412000-00006)). That study estimated a minimal clinically important difference for individual change of about 5 points on the 0 to 27 scale, derived as two standard errors of measurement ([Lowe et al 2004](https://doi.org/10.1097/00005650-200412000-00006)). A systematic review of the PHQ family concluded that sensitivity to change is well established for the PHQ-9 ([Kroenke 2010](https://doi.org/10.1016/j.genhosppsych.2010.03.006)). Routine-outcome use in stepped-care services also relies implicitly on responsiveness, with large pre-post effect sizes reported for depression ([Richards 2009](https://doi.org/10.1348/014466509X405178)). The MIC of 5 points is widely cited but should be understood as a single-derivation estimate rather than a triangulated consensus value.",
    "grade": "Moderate",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "see findings",
    "confidence_note": "Moderate: responsiveness is demonstrated in a good treatment study and endorsed by a review, but the 5-point MIC derives essentially from one study and one method (2 SEM), and no MIC has been established in a workplace context."
   },
   "populations_languages_norms": {
    "findings": "The PHQ-9 has been validated across a wide range of clinical, general-population and disease-specific samples and in many languages, but formal population norms in the classical sense are scarce; the UK benchmark is embedded in a national service dataset rather than a norm table. The IPD meta-analyses aggregate around 100 primary studies from many countries ([Negeri et al 2021](https://doi.org/10.1136/bmj.n2183), [Levis et al 2020](https://doi.org/10.1001/jama.2020.6504)). Validated non-English versions retrieved this session include Chinese ([Wang et al 2014](https://doi.org/10.1016/j.genhosppsych.2014.05.021)) and Spanish, the latter via a dedicated systematic review and meta-analysis of the Spanish-language PHQ-2 and PHQ-9 ([Martinez et al 2023](https://doi.org/10.1001/jamanetworkopen.2023.36529)), with further evaluations in Korean ([Lee et al 2023](https://doi.org/10.3389/fpsyg.2023.1217038)), Danish ([Pedersen et al 2016](https://doi.org/10.1016/j.jpsychores.2016.09.010)) and Italian ([Di et al 2025](https://doi.org/10.1097/JCN.0000000000001178)) samples. For the UK specifically, the PHQ-9 is the routine depression outcome measure of the English NHS Talking Therapies programme (formerly Improving Access to Psychological Therapies, IAPT), where a score of 10 or above defines 'caseness' and movement below it underpins recovery reporting; large UK service cohorts and a UK randomised trial have used it in exactly this way ([Richards 2009](https://doi.org/10.1348/014466509X405178), [Barkham et al 2021](https://doi.org/10.1016/S2215-0366%2821%2900083-3)). UK reference values therefore live in national service reporting rather than in a published normative sample. General-population norms exist chiefly through large national surveys used in the invariance literature rather than as a dedicated UK norm set.",
    "grade": "Moderate",
    "indirectness": "see findings"
   },
   "criticisms_controversies": "Several substantive controversies recur in the literature. First, cut-point and accuracy estimation. Studies that selectively report only well-performing cut-offs bias meta-analytic accuracy: for the PHQ-9, published results underestimated sensitivity below a cut-off of 10 (median difference about -0.06) and overestimated it above 10 (median about +0.07) ([Neupane et al 2021](https://doi.org/10.1002/mpr.1873)). Relatedly, using small datasets to simultaneously choose an optimal cut-off and estimate its accuracy is biased; in resampling from the IPD database the population-level optimal cut-off was actually 8 or above, and small studies scattered widely around it (only about 17% of 100-participant studies recovered the true optimum) ([Levis et al 2024](https://doi.org/10.1001/jamanetworkopen.2024.29630)). This questions the near-universal reliance on a cut-off of exactly 10. Second, reference-standard dependence: reported sensitivity is substantially higher against semistructured clinician interviews than against fully structured lay interviews or the MINI, so headline accuracy figures are conditional on the comparator ([Levis et al 2019](https://doi.org/10.1136/bmj.l1476), [Negeri et al 2021](https://doi.org/10.1136/bmj.n2183), [He et al 2019](https://doi.org/10.1159/000502294)). Third, somatic-item confounding in physically ill populations. In systemic sclerosis, somatic items accounted for a larger share of the total score than in matched healthy respondents, inflating scores by roughly 1.0 to 1.4 points (Hedges g 0.38 to 0.55) ([Leavens et al 2012](https://doi.org/10.1002/acr.21675)); in stroke, summed scoring moderately overestimated depression relative to a comparison sample (Cohen d about 0.43), driven by items such as tiredness and appetite ([Blake et al 2024](https://doi.org/10.1016/j.jpsychores.2024.111983)). This matters wherever respondents have physical illness or fatigue. Fourth, item 9 (thoughts of death or self-harm) is often misread as a suicidality measure, yet most positive responses are not associated with suicidality; the PHQ-8, which omits item 9, correlates almost perfectly with the PHQ-9 (r = 0.996) and performs almost identically for detecting depression ([Wu et al 2019](https://doi.org/10.1017/S0033291719001314)). Fifth, complexity does not pay: latent-variable and machine-learning scoring add negligible accuracy over the simple sum score with a cut-off ([Fischer et al 2021](https://doi.org/10.1017/S0033291721000131), [Hong et al 2022](https://doi.org/10.1016/j.genhosppsych.2022.04.011)). Finally, and central to the OWHS use case, the PHQ-9 is a clinical depression screener whose criterion validity was earned in primary-care and clinical populations against diagnostic interviews. Workplace deployment is a different context: screening in a lower-prevalence, largely non-help-seeking workforce reduces positive predictive value, the somatic-confounding problem is relevant to occupational groups with physical demands or illness, and no retrieved study validates the PHQ-9 against workplace criteria. The instrument's clinical provenance must not be read as endorsement of clinical-grade performance in a workplace-wellbeing programme; deployments that used it in worker samples treated it as an off-the-shelf symptom measure rather than validating it there ([Doki et al 2024](https://doi.org/10.1265/ehpm.23-00372)).",
   "citations": [
    {
     "key": "kroenke2001",
     "authors": "Kroenke K, Spitzer RL, Williams JB",
     "year": "2001",
     "title": "The PHQ-9: validity of a brief depression severity measure",
     "journal": "Journal of General Internal Medicine",
     "doi": "10.1046/j.1525-1497.2001.016009606.x",
     "url": "https://doi.org/10.1046/j.1525-1497.2001.016009606.x"
    },
    {
     "key": "kroenke2010",
     "authors": "Kroenke K, Spitzer RL, Williams JB, Lowe B",
     "year": "2010",
     "title": "The Patient Health Questionnaire Somatic, Anxiety, and Depressive Symptom Scales: a systematic review",
     "journal": "General Hospital Psychiatry",
     "doi": "10.1016/j.genhosppsych.2010.03.006",
     "url": "https://doi.org/10.1016/j.genhosppsych.2010.03.006"
    },
    {
     "key": "levis2019",
     "authors": "Levis B, Benedetti A, Thombs BD, et al",
     "year": "2019",
     "title": "Accuracy of Patient Health Questionnaire-9 (PHQ-9) for screening to detect major depression: individual participant data meta-analysis",
     "journal": "BMJ",
     "doi": "10.1136/bmj.l1476",
     "url": "https://doi.org/10.1136/bmj.l1476"
    },
    {
     "key": "negeri2021",
     "authors": "Negeri ZF, Levis B, Sun Y, et al",
     "year": "2021",
     "title": "Accuracy of the Patient Health Questionnaire-9 for screening to detect major depression: updated systematic review and individual participant data meta-analysis",
     "journal": "BMJ",
     "doi": "10.1136/bmj.n2183",
     "url": "https://doi.org/10.1136/bmj.n2183"
    },
    {
     "key": "wu2019",
     "authors": "Wu Y, Levis B, Riehm KE, et al",
     "year": "2019",
     "title": "Equivalency of the diagnostic accuracy of the PHQ-8 and PHQ-9: a systematic review and individual participant data meta-analysis",
     "journal": "Psychological Medicine",
     "doi": "10.1017/S0033291719001314",
     "url": "https://doi.org/10.1017/S0033291719001314"
    },
    {
     "key": "he2019",
     "authors": "He C, Levis B, Riehm KE, et al",
     "year": "2019",
     "title": "The Accuracy of the Patient Health Questionnaire-9 Algorithm for Screening to Detect Major Depression: An Individual Participant Data Meta-Analysis",
     "journal": "Psychotherapy and Psychosomatics",
     "doi": "10.1159/000502294",
     "url": "https://doi.org/10.1159/000502294"
    },
    {
     "key": "levis2020",
     "authors": "Levis B, Sun Y, He C, et al",
     "year": "2020",
     "title": "Accuracy of the PHQ-2 Alone and in Combination With the PHQ-9 for Screening to Detect Major Depression",
     "journal": "JAMA",
     "doi": "10.1001/jama.2020.6504",
     "url": "https://doi.org/10.1001/jama.2020.6504"
    },
    {
     "key": "neupane2021",
     "authors": "Neupane D, Levis B, Bhandari PM, et al",
     "year": "2021",
     "title": "Selective cutoff reporting in studies of the accuracy of the Patient Health Questionnaire-9 and Edinburgh Postnatal Depression Scale",
     "journal": "International Journal of Methods in Psychiatric Research",
     "doi": "10.1002/mpr.1873",
     "url": "https://doi.org/10.1002/mpr.1873"
    },
    {
     "key": "levis2024",
     "authors": "Levis B, Bhandari PM, Neupane D, et al",
     "year": "2024",
     "title": "Data-Driven Cutoff Selection for the Patient Health Questionnaire-9 Depression Screening Tool",
     "journal": "JAMA Network Open",
     "doi": "10.1001/jamanetworkopen.2024.29630",
     "url": "https://doi.org/10.1001/jamanetworkopen.2024.29630"
    },
    {
     "key": "fischer2021",
     "authors": "Fischer F, Levis B, Falk C, et al",
     "year": "2021",
     "title": "Comparison of different scoring methods based on latent variable models of the PHQ-9: an individual participant data meta-analysis",
     "journal": "Psychological Medicine",
     "doi": "10.1017/S0033291721000131",
     "url": "https://doi.org/10.1017/S0033291721000131"
    },
    {
     "key": "hong2022",
     "authors": "Hong ZM, Williams J, Bulloch A, et al",
     "year": "2022",
     "title": "Alternative scoring of the Patient Health Questionnaire-9 in neurological populations: an approach based on a predictive algorithm deriving from individual item scores",
     "journal": "General Hospital Psychiatry",
     "doi": "10.1016/j.genhosppsych.2022.04.011",
     "url": "https://doi.org/10.1016/j.genhosppsych.2022.04.011"
    },
    {
     "key": "rufino2024",
     "authors": "Rufino JV, Rodrigues R, Birolim MM, et al",
     "year": "2024",
     "title": "Analysis of the dimensional structure of the Patient Health Questionnaire-9 (PHQ-9) in undergraduate students at a public university in Brazil",
     "journal": "Journal of Affective Disorders",
     "doi": "10.1016/j.jad.2024.01.051",
     "url": "https://doi.org/10.1016/j.jad.2024.01.051"
    },
    {
     "key": "blake2024",
     "authors": "Blake JJ, Munyombwe T, Fischer F, et al",
     "year": "2024",
     "title": "The factor structure of the Patient Health Questionnaire-9 in stroke: A comparison with a non-stroke population",
     "journal": "Journal of Psychosomatic Research",
     "doi": "10.1016/j.jpsychores.2024.111983",
     "url": "https://doi.org/10.1016/j.jpsychores.2024.111983"
    },
    {
     "key": "teymoori2020",
     "authors": "Teymoori A, Gorbunova A, Haghish FE, et al",
     "year": "2020",
     "title": "Factorial Structure and Validity of Depression (PHQ-9) and Anxiety (GAD-7) Scales after Traumatic Brain Injury",
     "journal": "Journal of Clinical Medicine",
     "doi": "10.3390/jcm9030873",
     "url": "https://doi.org/10.3390/jcm9030873"
    },
    {
     "key": "wang2014",
     "authors": "Wang W, Bian Q, Zhao Y, et al",
     "year": "2014",
     "title": "Reliability and validity of the Chinese version of the Patient Health Questionnaire (PHQ-9) in the general population",
     "journal": "General Hospital Psychiatry",
     "doi": "10.1016/j.genhosppsych.2014.05.021",
     "url": "https://doi.org/10.1016/j.genhosppsych.2014.05.021"
    },
    {
     "key": "lowe2004",
     "authors": "Lowe B, Unutzer J, Callahan CM, et al",
     "year": "2004",
     "title": "Monitoring depression treatment outcomes with the Patient Health Questionnaire-9",
     "journal": "Medical Care",
     "doi": "10.1097/00005650-200412000-00006",
     "url": "https://doi.org/10.1097/00005650-200412000-00006"
    },
    {
     "key": "leavens2012",
     "authors": "Leavens A, Patten SB, Hudson M, et al",
     "year": "2012",
     "title": "Influence of somatic symptoms on Patient Health Questionnaire-9 depression scores among patients with systemic sclerosis compared to a healthy general population sample",
     "journal": "Arthritis Care & Research",
     "doi": "10.1002/acr.21675",
     "url": "https://doi.org/10.1002/acr.21675"
    },
    {
     "key": "lee2023",
     "authors": "Lee EH, Kang EH, Kang HJ, et al",
     "year": "2023",
     "title": "Measurement invariance of the patient health questionnaire-9 depression scale in a nationally representative population-based sample",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2023.1217038",
     "url": "https://doi.org/10.3389/fpsyg.2023.1217038"
    },
    {
     "key": "dimatteo2025",
     "authors": "Di Matteo R, Bolgeo T, Simonelli N, et al",
     "year": "2025",
     "title": "Psychometric Properties and Measurement Invariance of the Patient Health Questionnaire 9 in an Italian Coronary Heart Disease Population",
     "journal": "Journal of Cardiovascular Nursing",
     "doi": "10.1097/JCN.0000000000001178",
     "url": "https://doi.org/10.1097/JCN.0000000000001178"
    },
    {
     "key": "pedersen2016",
     "authors": "Pedersen SS, Mathiasen K, Christensen KB, et al",
     "year": "2016",
     "title": "Psychometric analysis of the Patient Health Questionnaire in Danish patients with an implantable cardioverter defibrillator (The DEFIB-WOMEN study)",
     "journal": "Journal of Psychosomatic Research",
     "doi": "10.1016/j.jpsychores.2016.09.010",
     "url": "https://doi.org/10.1016/j.jpsychores.2016.09.010"
    },
    {
     "key": "ajele2025",
     "authors": "Ajele KW, Idemudia ES",
     "year": "2025",
     "title": "Charting the course of depression care: a meta-analysis of reliability generalization of the Patient Health Questionnaire (PHQ-9) as the measure",
     "journal": "Discover Mental Health",
     "doi": "10.1007/s44192-025-00181-x",
     "url": "https://doi.org/10.1007/s44192-025-00181-x"
    },
    {
     "key": "martinez2023",
     "authors": "Martinez A, Teklu SM, Tahir P, et al",
     "year": "2023",
     "title": "Validity of the Spanish-Language Patient Health Questionnaires 2 and 9: A Systematic Review and Meta-Analysis",
     "journal": "JAMA Network Open",
     "doi": "10.1001/jamanetworkopen.2023.36529",
     "url": "https://doi.org/10.1001/jamanetworkopen.2023.36529"
    },
    {
     "key": "pintomeza2005",
     "authors": "Pinto-Meza A, Serrano-Blanco A, Penarrubia MT, et al",
     "year": "2005",
     "title": "Assessing depression in primary care with the PHQ-9: can it be carried out over the telephone?",
     "journal": "Journal of General Internal Medicine",
     "doi": "10.1111/j.1525-1497.2005.0144.x",
     "url": "https://doi.org/10.1111/j.1525-1497.2005.0144.x"
    },
    {
     "key": "barkham2021",
     "authors": "Barkham M, Saxon D, Hardy GE, et al",
     "year": "2021",
     "title": "Person-centred experiential therapy versus cognitive behavioural therapy delivered in the English Improving Access to Psychological Therapies service for the treatment of moderate or severe depression (PRaCTICED)",
     "journal": "Lancet Psychiatry",
     "doi": "10.1016/S2215-0366(21)00083-3",
     "url": "https://doi.org/10.1016/S2215-0366%2821%2900083-3"
    },
    {
     "key": "richards2009",
     "authors": "Richards DA, Suckling R",
     "year": "2009",
     "title": "Improving access to psychological therapies: phase IV prospective cohort study",
     "journal": "British Journal of Clinical Psychology",
     "doi": "10.1348/014466509X405178",
     "url": "https://doi.org/10.1348/014466509X405178"
    },
    {
     "key": "doki2024",
     "authors": "Doki S, Hori D, Takahashi T, et al",
     "year": "2024",
     "title": "Designing a test battery for workers' well-being: the first wave of the Tsukuba Salutogenic Occupational Cohort Study",
     "journal": "Environmental Health and Preventive Medicine",
     "doi": "10.1265/ehpm.23-00372",
     "url": "https://doi.org/10.1265/ehpm.23-00372"
    }
   ],
   "record_notes": "[Upgraded from v0.1 to v0.2 structure in pass two; criterion field split, licence re-verified 2026-07-12.] Overall confidence: the PHQ-9 is the confidence-grading high-water mark. Structural validity, diagnostic criterion validity and internal consistency are High, anchored on the Levis/Thombs individual participant data programme and a reliability-generalisation meta-analysis; responsiveness and the somatic-confounding and cut-point critiques are well evidenced. The genuinely weaker or absent cells are test-retest (present but from few studies, none occupational), measurement invariance (scattered, no occupational or UK-workplace coverage, weak over-time evidence), criterion validity against organisational outcomes (absent), and workplace-specific psychometrics (absent). Schema stress-test notes: (1) The single 'criterion_validity' field conflates two very different evidence bases, diagnostic criterion validity (High) versus criterion validity against organisational outcomes (Absent). I graded it as split and said so in the findings, but a schema that forced one grade would misrepresent the instrument; the field should ideally be divided. (2) 'confidence' is a single scalar per property, yet for several properties the honest grade differs by sub-question (e.g. invariance is scalar-level across age but untested across occupation). I encoded the dominant grade and qualified it in the justification. (3) The clinical-origin versus workplace-deployment gap is the most important caveat for this registry and does not have a dedicated field; I carried it in criticisms_controversies and flagged it in constructs_claimed and criterion_validity, but it risks being lost if a reader scans only the property grades. (4) Licence status is factually clear (public domain, Pfizer-released) but I could not attach a DOI-bearing primary source to the licensing act itself, only to the original validation paper; I flagged this rather than attach a non-resolvable citation. (5) 'Absent' was used strictly for organisational criterion validity and workplace psychometrics; test-retest was deliberately NOT graded Absent because evidence exists, only sparsely, which the scale's wording ('barely-studied') made a close call between Low and Moderate."
  },
  {
   "instrument_id": "gad-7",
   "display_name": "GAD-7 (Generalised Anxiety Disorder-7)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Generalised Anxiety Disorder-7 (GAD-7)",
    "current_version": "Original 7-item GAD-7 (2006); a 2-item short form (GAD-2, first two items) is also distributed",
    "item_count": "7",
    "original_citation": "Spitzer RL, Kroenke K, Williams JBW, Lowe B. A brief measure for assessing generalized anxiety disorder: the GAD-7. Arch Intern Med. 2006;166(10):1092-1097. doi:10.1001/archinte.166.10.1092",
    "steward_publisher": "Developed by Robert L. Spitzer, Janet B.W. Williams, Kurt Kroenke and colleagues with an educational grant from Pfizer Inc.; distributed by the steward site phqscreeners.com (copyright Pfizer Inc.)",
    "licence_status": "Free to use, no permission required. The steward's currently distributed GAD-7 English PDF (phqscreeners.com) carries the footer 'No permission required to reproduce, translate, display or distribute', developed with an educational grant from Pfizer Inc.; the steward instrument entry also carries 'Copyright (c) Pfizer Inc. All rights reserved.' This is a no-permission-required grant with copyright retained by Pfizer, NOT a formal open or Creative Commons licence. Confirmed this session from the steward's distributed PDF footer text (body content, not merely page titles) and the LOINC mirror of the steward entry; direct HTML fetches of phqscreeners.com returned HTTP 403.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "Verbatim footer text of the official GAD-7 English PDF hosted on the steward site (phqscreeners.com/images/sites/g/files/g10060481/f/201412/GAD-7_English.pdf), read this session: 'Developed by Drs. Robert L. Spitzer, Janet B.W. Williams, Kurt Kroenke and colleagues, with an educational grant from Pfizer Inc. No permission required to reproduce, translate, display or distribute.' The steward instrument entry (as mirrored by LOINC 69737-5, which cites URL https://www.phqscreeners.com/) additionally records 'Copyright (c) Pfizer Inc. All rights reserved.' alongside the same no-permission-required statement. Direct urllib fetches of phqscreeners.com pages returned HTTP 403 (server-side bot refusal, not a sandbox block), so the current terms were confirmed from the steward's own distributed PDF footer text and the LOINC mirror of the steward entry rather than the rendered HTML page. Not sourced from the founding paper."
   },
   "constructs_claimed": "Severity of generalised anxiety disorder symptoms over the preceding two weeks. Functions both as a screener for probable GAD and as a continuous self-report measure of anxiety-symptom severity.",
   "deployment_context_caveat": "GAD-7 is a clinical screener developed and validated in primary care against a psychiatric diagnostic interview (Spitzer 2006, doi:10.1001/archinte.166.10.1092). Any workplace or organisational deployment is a different context from its validation setting. It measures anxiety-symptom severity and screens for probable GAD; it is not a diagnostic instrument and not a fitness-for-work measure. A positive screen indicates a need for clinical assessment, never an employment or performance decision. Its clinical origin may be stated as fact, but its psychometric performance in a workplace deployment context should not be assumed from clinical/primary-care evidence.",
   "structural_validity": {
    "findings": "GAD-7 was designed as, and is most often confirmed as, a single-factor (unidimensional) scale. The original development study confirmed anxiety and depression as distinct dimensions ([Spitzer 2006](https://doi.org/10.1001/archinte.166.10.1092)), and a nationally representative German general-population confirmatory factor analysis (N=5030) substantiated a one-dimensional structure with factorial invariance for gender and age ([Lowe 2008](https://doi.org/10.1097/MLR.0b013e318160d093)). Unidimensional fit has been replicated across many settings: Cypriot perinatal women ([Vogazianos 2022](https://doi.org/10.1186/s12884-022-05127-7)), an Italian coronary heart disease sample ([Bolgeo 2023](https://doi.org/10.1016/j.jad.2023.04.140)), four nationally representative European samples including the UK ([Shevlin 2022](https://doi.org/10.1186/s12888-022-03787-5)), Canadian young adults where a one-factor model fit best ([Riglea 2025](https://doi.org/10.1016/j.jad.2025.01.117)), and 20 Czech samples (N=5529) supporting a unidimensional structure ([Cigler 2026](https://doi.org/10.1016/j.janxdis.2026.103149)). However, a competing two-factor structure separating cognitive-emotional from somatic items recurs: a network analysis of Spanish primary-care patients revealed a two-factor solution ([Moriana 2021](https://doi.org/10.1002/jclp.23217)), a US college-student study fit both one- and two-factor models ([White 2025](https://doi.org/10.1037/tps0000382)), a large psychotherapy sample found the scale technically multidimensional though sum scores remained justifiable ([Stochl 2020](https://doi.org/10.1177/1073191120976863)), and a Malaysian study preferred a six-item second-order model ([Pheh 2023](https://doi.org/10.1371/journal.pone.0285435)). The one-factor model dominates practice and fits well in most samples, but the dimensionality question is genuinely unsettled.",
    "grade": "High",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "direct; unidimensional evidence includes UK-inclusive nationally representative samples (Shevlin 2022), though much replication is in non-UK, clinical and disease-specific populations",
    "subgrades": [
     {
      "subgroup": "one-factor (unidimensional) model",
      "grade": "High",
      "note": "Well replicated across languages and settings; the dominant and recommended scoring model"
     },
     {
      "subgroup": "cognitive-emotional vs somatic two-factor model",
      "grade": "Moderate",
      "note": "Recurs in network analyses and some CFAs; keeps dimensionality contested rather than settled"
     }
    ]
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent validity is consistently supported. In the German general population GAD-7 correlated r=0.64 with the PHQ-2 depression module and r=-0.43 with the Rosenberg Self-Esteem Scale ([Lowe 2008](https://doi.org/10.1097/MLR.0b013e318160d093)). In US college students the total score correlated r=0.70 with the trait scale of the State-Trait Anxiety Inventory (convergent) and only r=-0.04 with a behavioural activation reward subscale (discriminant) ([White 2025](https://doi.org/10.1037/tps0000382)). Increasing scores were strongly associated with multiple domains of functional impairment in the original study ([Spitzer 2006](https://doi.org/10.1001/archinte.166.10.1092)). The recurring discriminant concern is the strong overlap with depression: GAD-7 and PHQ-9 anxiety and depression factors are highly correlated and frequently co-occur ([Stochl 2020](https://doi.org/10.1177/1073191120976863); [Bolgeo 2023](https://doi.org/10.1016/j.jad.2023.04.140)), so the scale distinguishes anxiety from unrelated constructs well but discriminates anxiety from depression less cleanly.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect; convergent/discriminant evidence is mostly from general-population, student and clinical non-workplace samples (German general population, US students, disease-specific cohorts)"
   },
   "criterion_validity_reference_standard": {
    "findings": "Against a structured or semi-structured clinical interview, GAD-7 has been extensively evaluated. The original primary-care study identified a cut-off of 10 with sensitivity 89% and specificity 82% for GAD ([Spitzer 2006](https://doi.org/10.1001/archinte.166.10.1092)). An early diagnostic meta-analysis (12 samples, 5223 participants) found pooled sensitivity 0.83 and specificity 0.84 at a cut-off of 8, with cut-offs 7 to 10 performing similarly ([Plummer 2016](https://doi.org/10.1016/j.genhosppsych.2015.11.005)). The most comprehensive synthesis, a 2025 Cochrane review of 48 studies (19,228 participants, 27 countries, 24 languages), reported that at the recommended cut-off of 10 or higher the GAD-7 summary sensitivity was 0.64 (95% CI 0.56 to 0.72) and specificity 0.91 (95% CI 0.87 to 0.93) for detecting GAD, with an area under the curve of 0.86; for detecting any anxiety disorder sensitivity fell to 0.48 (specificity 0.91) ([Akturk 2025](https://doi.org/10.1002/14651858.CD015455)). The pooled sensitivity at cut-off 10 is therefore markedly lower than the original single-study estimate, with pronounced heterogeneity, and the Cochrane authors caution that the summary estimates are rough averages that may deviate substantially in specific situations. Setting-specific validations against interview (for example Taiwanese epilepsy patients, optimal cut-off 7) add further threshold variability ([Shih 2022](https://doi.org/10.1016/j.jfma.2022.04.018)).",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect; validated against diagnostic interview in primary-care and disease-specific clinical samples, not in workplace populations, and UK-specific diagnostic-accuracy data are a minority of the pooled evidence"
   },
   "criterion_validity_organisational": {
    "findings": "No study validating the full GAD-7 against objective organisational outcomes (recorded sickness absence, turnover, or measured job performance) was located this session. The nearest evidence is adjacent rather than direct: a study of 4953 working Australians linked probable anxiety to worse self-reported presenteeism and absenteeism on the WHO Health and Work Performance Questionnaire, but it used the 2-item GAD-2, not the full GAD-7, and relied on self-reported rather than employer-recorded outcomes ([Deady 2021](https://doi.org/10.1093/occmed/kqab142)). A Polish validation among employees related GAD-7 to professional burnout and psychological distress, again self-reported constructs rather than organisational records ([Basinska 2023](https://doi.org/10.13075/ijomeh.1896.02104)). The original study related GAD-7 to self-reported disability days and functional impairment, not to work-context criterion outcomes ([Spitzer 2006](https://doi.org/10.1001/archinte.166.10.1092)). Organisational criterion validity for the fielded GAD-7 is therefore essentially untested.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "mixed",
    "indirectness": "indirect; the closest evidence uses the derivative GAD-2 and self-reported work outcomes, and no study links the full GAD-7 to objective organisational records"
   },
   "internal_consistency": {
    "findings": "Internal consistency is uniformly high across populations and languages. Cronbach's alpha was 0.89, identical across all gender and age subgroups, in the German general population ([Lowe 2008](https://doi.org/10.1097/MLR.0b013e318160d093)); 0.91 in US college students ([White 2025](https://doi.org/10.1037/tps0000382)); alpha 0.907 with McDonald's omega 0.909 in Cypriot perinatal women ([Vogazianos 2022](https://doi.org/10.1186/s12884-022-05127-7)); alpha 0.89 with composite reliability 0.90 in an Italian cardiac sample ([Bolgeo 2023](https://doi.org/10.1016/j.jad.2023.04.140)); and 0.928 in Taiwanese epilepsy patients ([Shih 2022](https://doi.org/10.1016/j.jfma.2022.04.018)). Lower but still acceptable values appear in some translations, for example alpha 0.81 in a Swahili HIV sample ([Nyongesa 2020](https://doi.org/10.1186/s12991-020-00312-4)) and a median alpha of 0.86 across 20 Czech samples ([Cigler 2026](https://doi.org/10.1016/j.janxdis.2026.103149)). Values consistently sit in the 0.81 to 0.93 range.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "direct; UK-inclusive nationally representative samples contribute (Shevlin 2022, Saunders 2023), although most individual alpha estimates come from non-UK or clinical samples"
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "0.83",
      "coefficient_type": "ICC",
      "interval": "not reported in retrieved sources",
      "sample_n": "not reported in retrieved sources",
      "population": "US primary care (original validation)",
      "evidence_form": "canonical",
      "citation_key": "spitzer2006"
     },
     {
      "coefficient": "0.59",
      "coefficient_type": "ICC",
      "interval": "2 weeks",
      "sample_n": "60",
      "population": "Adults living with HIV, Kilifi, Kenya (Swahili version)",
      "evidence_form": "canonical",
      "citation_key": "nyongesa2020"
     },
     {
      "coefficient": "0.46 to 0.53",
      "coefficient_type": "r",
      "interval": "repeated measures over 2 to 4 time points spanning major life events",
      "sample_n": "subset of 5529",
      "population": "Czech general-population and psychiatric adults",
      "evidence_form": "canonical",
      "citation_key": "cigler2026"
     }
    ],
    "grade": "Low",
    "status": "thin",
    "indirectness": "indirect; the original ICC of 0.83 is reported second-hand in retrieved sources without interval or sample size, and the two dedicated coefficients located are from a Kenyan HIV sample and a Czech life-events design, none in a UK working population",
    "summary": "Dedicated short-interval test-retest studies of GAD-7 are sparse. The original validation reportedly gave ICC 0.83, but its interval and sample size were not captured in sources retrieved this session (the value is cited second-hand in Nyongesa 2020, doi:10.1186/s12991-020-00312-4). A 2-week ICC of 0.59 was found in a Swahili HIV sample (Nyongesa 2020), and the Czech multi-sample study found moderate stability of r=0.46 to 0.53, but over intervals spanning major life events rather than a fixed short retest window (Cigler 2026, doi:10.1016/j.janxdis.2026.103149). Test-retest reliability for a UK working-adult population is untested. The absence of a consistent, short-interval, well-sampled retest estimate is itself the finding."
   },
   "measurement_invariance": {
    "findings": "Measurement invariance is one of the most heavily studied properties of the GAD-7, and it is largely supported across groups but contested over time. Across sex/gender, invariance or an absence of differential item functioning is repeatedly demonstrated, including in very large UK treatment-seeking samples (N=165,872) ([Saunders 2023](https://doi.org/10.1186/s12888-023-04804-x)), Canadian adolescents where strict invariance held by sex and grade ([Romano 2021](https://doi.org/10.1093/jpepsy/jsab119)), and four European nationally representative samples with no DIF across sex, age or country ([Shevlin 2022](https://doi.org/10.1186/s12888-022-03787-5)). Across age, invariance held between UK working-age and older adults with only limited DIF ([Delamain 2024](https://doi.org/10.1016/j.jad.2023.11.048)). Across language and country, invariance is supported across English and French ([Riglea 2025](https://doi.org/10.1016/j.jad.2025.01.117)), across 18 countries after traumatic brain injury ([Teymoori 2020](https://doi.org/10.1016/j.jad.2019.10.035)), and with full scalar invariance in rural India ([De Man 2021](https://doi.org/10.3389/fpsyg.2021.676398)). Longitudinal invariance is where the literature conflicts: strict temporal invariance was established across 10 psychotherapy sessions ([Stochl 2020](https://doi.org/10.1177/1073191120976863)) and across time in Czech and Canadian samples ([Cigler 2026](https://doi.org/10.1016/j.janxdis.2026.103149); [Riglea 2025](https://doi.org/10.1016/j.jad.2025.01.117)), yet longitudinal invariance was NOT established in a partial-hospital sample, implying that raw pre-post change scores may be unreliable in that setting ([Ong 2021](https://doi.org/10.1177/10731911211035833)).",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "direct; UK treatment-seeking and working-age-versus-older samples are included (Saunders 2023, Delamain 2024), though occupational-group invariance specifically is untested",
    "subgrades": [
     {
      "subgroup": "sex / gender",
      "grade": "High",
      "note": "Strongly supported across many samples including large UK datasets"
     },
     {
      "subgroup": "age",
      "grade": "High",
      "note": "Supported including UK working-age versus older adults (Delamain 2024)"
     },
     {
      "subgroup": "language / country",
      "grade": "Moderate",
      "note": "Supported across several languages and a UK-inclusive four-country study, but by translation rather than exhaustive"
     },
     {
      "subgroup": "longitudinal / over time",
      "grade": "Low",
      "note": "Contested: strict temporal invariance in some large samples but not established in a partial-hospital sample, complicating change-score use (Ong 2021)"
     },
     {
      "subgroup": "occupation / industry",
      "grade": "Absent",
      "note": "No study testing invariance across occupational groups located this session"
     }
    ]
   },
   "responsiveness_mic": {
    "findings": "Responsiveness (sensitivity to change) is demonstrated, but formal minimal important change estimates are sparse. In a multisite chronic-depression trial (N=261), GAD-7 scores fell significantly in patients who improved on the Hamilton depression rating (effect size -0.51 at 12 weeks, -1.0 at 48 weeks) and rose in those who worsened, supporting sensitivity to change ([Toussaint 2020](https://doi.org/10.1016/j.jad.2020.01.032)). The Czech multi-sample study found clear sensitivity to change during psychotherapy and derived a reliable-change threshold of about plus or minus 5.2 points ([Cigler 2026](https://doi.org/10.1016/j.janxdis.2026.103149)). A single, well-anchored minimal important change value validated for a working-adult population was not located this session; responsiveness is established while the MIC anchor remains setting-dependent.",
    "grade": "Moderate",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect; responsiveness evidence comes from a chronic-depression trial and Czech psychotherapy samples, not from UK or workplace populations, and MIC anchors vary by setting"
   },
   "populations_languages_norms": {
    "findings": "GAD-7 has been fielded and psychometrically evaluated in a very wide range of populations and languages: the 2025 Cochrane review alone drew on 27 countries and 24 languages (Akturk 2025, doi:10.1002/14651858.CD015455). General-population normative data exist, including German norms by sex and age where roughly 5% scored 10 or higher (Lowe 2008, doi:10.1097/MLR.0b013e318160d093). Validations span primary care, adolescents and older adults, perinatal women, and disease-specific groups (HIV, epilepsy, coronary heart disease, traumatic brain injury), plus students and employees. UK-relevant data come from nationally representative European samples and large UK treatment-seeking (IAPT) cohorts (Shevlin 2022, doi:10.1186/s12888-022-03787-5; Saunders 2023, doi:10.1186/s12888-023-04804-x; Delamain 2024, doi:10.1016/j.jad.2023.11.048). Dedicated UK working-population norms were not located this session.",
    "grade": "High",
    "indirectness": "direct; UK general-population and treatment-seeking data exist, but occupation-specific UK norms are absent"
   },
   "criticisms_controversies": "Recurring criticisms cluster around five points. First, dimensionality is unsettled: the one-factor model dominates but a cognitive-emotional versus somatic two-factor structure recurs in network analyses and some CFAs (Moriana 2021, doi:10.1002/jclp.23217; Pheh 2023, doi:10.1371/journal.pone.0285435). Second, discriminant validity against depression is weak in the sense that GAD-7 and PHQ-9 factors are highly correlated and co-occur, so the scale separates anxiety from depression less cleanly than from unrelated constructs (Stochl 2020, doi:10.1177/1073191120976863). Third, the 2025 Cochrane meta-analysis found only modest pooled sensitivity (0.64) at the standard cut-off of 10 for GAD, with pronounced heterogeneity, meaning the scale misses a meaningful share of cases at that threshold and performs worse for any anxiety disorder (sensitivity 0.48) (Akturk 2025, doi:10.1002/14651858.CD015455). Fourth, longitudinal measurement invariance is not guaranteed, which complicates the common practice of interpreting raw pre-post change scores (Ong 2021, doi:10.1177/10731911211035833). Fifth, the scale targets GAD specifically rather than the full anxiety-disorder spectrum, and its clinical/primary-care origin means any workplace deployment is outside its validation context. Organisational criterion validity is effectively untested.",
   "item_records": [],
   "citations": [
    {
     "key": "spitzer2006",
     "authors": "Spitzer RL, Kroenke K, Williams JBW, Lowe B",
     "year": "2006",
     "title": "A brief measure for assessing generalized anxiety disorder: the GAD-7",
     "journal": "Archives of Internal Medicine",
     "doi": "10.1001/archinte.166.10.1092",
     "url": "https://doi.org/10.1001/archinte.166.10.1092"
    },
    {
     "key": "lowe2008",
     "authors": "Lowe B, Decker O, Muller S, et al.",
     "year": "2008",
     "title": "Validation and standardization of the Generalized Anxiety Disorder Screener (GAD-7) in the general population",
     "journal": "Medical Care",
     "doi": "10.1097/MLR.0b013e318160d093",
     "url": "https://doi.org/10.1097/MLR.0b013e318160d093"
    },
    {
     "key": "plummer2016",
     "authors": "Plummer F, Manea L, Trepel D, McMillan D",
     "year": "2016",
     "title": "Screening for anxiety disorders with the GAD-7 and GAD-2: a systematic review and diagnostic metaanalysis",
     "journal": "General Hospital Psychiatry",
     "doi": "10.1016/j.genhosppsych.2015.11.005",
     "url": "https://doi.org/10.1016/j.genhosppsych.2015.11.005"
    },
    {
     "key": "akturk2025",
     "authors": "Akturk Z, Hapfelmeier A, Fomenko A, et al.",
     "year": "2025",
     "title": "Generalized Anxiety Disorder 7-item (GAD-7) and 2-item (GAD-2) scales for detecting anxiety disorders in adults",
     "journal": "Cochrane Database of Systematic Reviews",
     "doi": "10.1002/14651858.CD015455",
     "url": "https://doi.org/10.1002/14651858.CD015455"
    },
    {
     "key": "white2025",
     "authors": "White EJ, Karr JE",
     "year": "2025",
     "title": "Psychometric properties of the GAD-7 among college students: reliability, validity, factor structure, and measurement invariance",
     "journal": "Translational Issues in Psychological Science",
     "doi": "10.1037/tps0000382",
     "url": "https://doi.org/10.1037/tps0000382"
    },
    {
     "key": "cigler2026",
     "authors": "Cigler H, Patkova Dansova P, Javurkova A, et al.",
     "year": "2026",
     "title": "Validity and factor structure of the Czech GAD-7 across twenty samples and four independent translations",
     "journal": "Journal of Anxiety Disorders",
     "doi": "10.1016/j.janxdis.2026.103149",
     "url": "https://doi.org/10.1016/j.janxdis.2026.103149"
    },
    {
     "key": "moriana2021",
     "authors": "Moriana JA, Jurado-Gonzalez FJ, Garcia-Torres F, et al.",
     "year": "2021",
     "title": "Exploring the structure of the GAD-7 scale in primary care patients with emotional disorders: a network analysis approach",
     "journal": "Journal of Clinical Psychology",
     "doi": "10.1002/jclp.23217",
     "url": "https://doi.org/10.1002/jclp.23217"
    },
    {
     "key": "riglea2025",
     "authors": "Riglea T, Wellman RJ, Sylvestre MP, et al.",
     "year": "2025",
     "title": "Factor structure and measurement invariance of the GAD-7 across time, sex, and language in young adults",
     "journal": "Journal of Affective Disorders",
     "doi": "10.1016/j.jad.2025.01.117",
     "url": "https://doi.org/10.1016/j.jad.2025.01.117"
    },
    {
     "key": "pheh2023",
     "authors": "Pheh KS, Tan CS, Lee KW, et al.",
     "year": "2023",
     "title": "Factorial structure, reliability, and construct validity of the Generalized Anxiety Disorder 7-item (GAD-7): evidence from Malaysia",
     "journal": "PLOS ONE",
     "doi": "10.1371/journal.pone.0285435",
     "url": "https://doi.org/10.1371/journal.pone.0285435"
    },
    {
     "key": "vogazianos2022",
     "authors": "Vogazianos P, Motrico E, Dominguez-Salas S, et al.",
     "year": "2022",
     "title": "Validation of the generalized anxiety disorder screener (GAD-7) in Cypriot pregnant and postpartum women",
     "journal": "BMC Pregnancy and Childbirth",
     "doi": "10.1186/s12884-022-05127-7",
     "url": "https://doi.org/10.1186/s12884-022-05127-7"
    },
    {
     "key": "deman2021",
     "authors": "De Man J, Absetz P, Sathish T, et al.",
     "year": "2021",
     "title": "Are the PHQ-9 and GAD-7 suitable for use in India? A psychometric analysis",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2021.676398",
     "url": "https://doi.org/10.3389/fpsyg.2021.676398"
    },
    {
     "key": "ong2021",
     "authors": "Ong CW, Pierce BG, Klein KP, et al.",
     "year": "2021",
     "title": "Longitudinal measurement invariance of the PHQ-9 and GAD-7",
     "journal": "Assessment",
     "doi": "10.1177/10731911211035833",
     "url": "https://doi.org/10.1177/10731911211035833"
    },
    {
     "key": "romano2021",
     "authors": "Romano I, Ferro MA, Patte KA, et al.",
     "year": "2021",
     "title": "Measurement invariance of the GAD-7 and CESD-R-10 among adolescents in Canada",
     "journal": "Journal of Pediatric Psychology",
     "doi": "10.1093/jpepsy/jsab119",
     "url": "https://doi.org/10.1093/jpepsy/jsab119"
    },
    {
     "key": "saunders2023",
     "authors": "Saunders R, Moinian D, Stott J, et al.",
     "year": "2023",
     "title": "Measurement invariance of the PHQ-9 and GAD-7 across males and females seeking treatment for common mental health disorders",
     "journal": "BMC Psychiatry",
     "doi": "10.1186/s12888-023-04804-x",
     "url": "https://doi.org/10.1186/s12888-023-04804-x"
    },
    {
     "key": "delamain2024",
     "authors": "Delamain H, Buckman JEJ, Stott J, et al.",
     "year": "2024",
     "title": "Measurement invariance and differential item functioning of the PHQ-9 and GAD-7 between working age and older adults",
     "journal": "Journal of Affective Disorders",
     "doi": "10.1016/j.jad.2023.11.048",
     "url": "https://doi.org/10.1016/j.jad.2023.11.048"
    },
    {
     "key": "bolgeo2023",
     "authors": "Bolgeo T, Di Matteo R, Simonelli N, et al.",
     "year": "2023",
     "title": "Psychometric properties and measurement invariance of the 7-item General Anxiety Disorder scale (GAD-7) in an Italian coronary heart disease sample",
     "journal": "Journal of Affective Disorders",
     "doi": "10.1016/j.jad.2023.04.140",
     "url": "https://doi.org/10.1016/j.jad.2023.04.140"
    },
    {
     "key": "shevlin2022",
     "authors": "Shevlin M, Butter S, McBride O, et al.",
     "year": "2022",
     "title": "Measurement invariance of the Patient Health Questionnaire (PHQ-9) and Generalized Anxiety Disorder (GAD-7) across four European countries",
     "journal": "BMC Psychiatry",
     "doi": "10.1186/s12888-022-03787-5",
     "url": "https://doi.org/10.1186/s12888-022-03787-5"
    },
    {
     "key": "stochl2020",
     "authors": "Stochl J, Fried EI, Fritz J, et al.",
     "year": "2020",
     "title": "On dimensionality, measurement invariance, and suitability of sum scores for the PHQ-9 and the GAD-7",
     "journal": "Assessment",
     "doi": "10.1177/1073191120976863",
     "url": "https://doi.org/10.1177/1073191120976863"
    },
    {
     "key": "teymoori2020",
     "authors": "Teymoori A, Real R, Gorbunova A, et al.",
     "year": "2020",
     "title": "Measurement invariance of assessments of depression (PHQ-9) and anxiety (GAD-7) across sex, strata and linguistic backgrounds",
     "journal": "Journal of Affective Disorders",
     "doi": "10.1016/j.jad.2019.10.035",
     "url": "https://doi.org/10.1016/j.jad.2019.10.035"
    },
    {
     "key": "shih2022",
     "authors": "Shih YC, Chou CC, Lu YJ, et al.",
     "year": "2022",
     "title": "Reliability and validity of the traditional Chinese version of the GAD-7 in Taiwanese patients with epilepsy",
     "journal": "Journal of the Formosan Medical Association",
     "doi": "10.1016/j.jfma.2022.04.018",
     "url": "https://doi.org/10.1016/j.jfma.2022.04.018"
    },
    {
     "key": "toussaint2020",
     "authors": "Toussaint A, Husing P, Gumz A, et al.",
     "year": "2020",
     "title": "Sensitivity to change and minimal clinically important difference of the 7-item Generalized Anxiety Disorder Questionnaire (GAD-7)",
     "journal": "Journal of Affective Disorders",
     "doi": "10.1016/j.jad.2020.01.032",
     "url": "https://doi.org/10.1016/j.jad.2020.01.032"
    },
    {
     "key": "deady2021",
     "authors": "Deady M, Collins DAJ, Johnston DA, et al.",
     "year": "2021",
     "title": "The impact of depression, anxiety and comorbidity on occupational outcomes",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqab142",
     "url": "https://doi.org/10.1093/occmed/kqab142"
    },
    {
     "key": "nyongesa2020",
     "authors": "Nyongesa MK, Mwangi P, Koot HM, et al.",
     "year": "2020",
     "title": "The reliability, validity and factorial structure of the Swahili version of the 7-item generalized anxiety disorder scale (GAD-7) among adults living with HIV from Kilifi, Kenya",
     "journal": "Annals of General Psychiatry",
     "doi": "10.1186/s12991-020-00312-4",
     "url": "https://doi.org/10.1186/s12991-020-00312-4"
    },
    {
     "key": "basinska2023",
     "authors": "Basinska MA, Kwissa-Gajewska Z",
     "year": "2023",
     "title": "Psychometric properties of the Polish version of the Generalized Anxiety Disorder scale (GAD-7) in a non-clinical sample of employees",
     "journal": "International Journal of Occupational Medicine and Environmental Health",
     "doi": "10.13075/ijomeh.1896.02104",
     "url": "https://doi.org/10.13075/ijomeh.1896.02104"
    }
   ],
   "record_notes": "Overall confidence: GAD-7 has a deep, high-quality evidence base for internal consistency (High), diagnostic accuracy against clinical interview (High, though the best synthesis shows modest pooled sensitivity at cut-off 10 with high heterogeneity), a predominantly unidimensional structure (High but with a genuinely contested two-factor alternative), and cross-group measurement invariance by sex and age (High). That evidence was earned overwhelmingly in clinical, primary-care and non-UK samples, making it indirect for UK working adults, although UK general-population and IAPT data do exist. The honest gaps that most matter for a workplace registry: organisational criterion validity against objective work outcomes is Absent (the nearest evidence uses the derivative GAD-2 and self-reported outcomes); dedicated short-interval test-retest studies are sparse and mixed (Low), with the original ICC of 0.83 available only second-hand without interval or n; longitudinal invariance is contested, which bears directly on using the scale to track change; and a working-population minimal important change was not located this session. Licence verified on 2026-07-12 from the body text of the steward's currently distributed GAD-7 English PDF (phqscreeners.com footer) and the LOINC mirror of the steward entry, not from the founding paper: free to use, Pfizer copyright retained, no permission required, which is a no-permission-required grant rather than a formal open or Creative Commons licence. Direct HTML fetches of phqscreeners.com returned HTTP 403, so confirmation rests on the steward's own distributed PDF footer and the LOINC mirror. Schema v0.2 recorded these honestly; the main tension was that most invariance/structure evidence is shared with the PHQ-9 in joint studies, which I have tagged canonical for the GAD-7 factor specifically where the analysis modelled the GAD-7 items separately."
  },
  {
   "instrument_id": "k10",
   "display_name": "Kessler Psychological Distress Scale (K10)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Kessler Psychological Distress Scale (K10)",
    "current_version": "K10 (10-item); a 6-item short form (K6) is embedded within it. Both derive from the same item-response-theory development work.",
    "item_count": "10 items, each rated on a five-point frequency scale over the past 30 days; total range 10 to 50 (or 0 to 40 under the alternative 0 to 4 coding).",
    "original_citation": "Kessler RC, Andrews G, Colpe LJ, et al. Short screening scales to monitor population prevalences and trends in non-specific psychological distress. Psychological Medicine 2002;32(6):959-976. doi:10.1017/s0033291702006074",
    "steward_publisher": "Ronald C. Kessler and colleagues, Department of Health Care Policy, Harvard Medical School. Distributed through the National Comorbidity Survey site (hcp.med.harvard.edu/ncs/k6_scales.php). The K10 itself was developed with the Clinical Research Unit for Anxiety and Depression (CRUFAD), Australia.",
    "licence_status": "Free to use with no formal permission or registration required. The steward's current distribution page (Harvard Medical School National Comorbidity Survey, 'Kessler Psychological Distress Scale (K10)') states under its permission-requests heading that use of the K6 and K10 is free and does not require any formal permission or approval, asking only that users cite the source article and include the copyright notice when using the scales. Copyright is held by Ronald C. Kessler. No named open-content licence identifier (for example a Creative Commons code) is asserted by the steward; the scale is distributed as a free-to-use, cite-and-attribute instrument rather than under a formal open licence. Redistribution and modification terms are not explicitly addressed on the page, so verify with the steward before either.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "Harvard Medical School, National Comorbidity Survey, 'Kessler Psychological Distress Scale (K10)' distribution page, https://www.hcp.med.harvard.edu/ncs/k6_scales.php. The page body was read this session; its permission-requests section states that use of the K6 and K10 is free and requires no formal permission or approval, asking users to cite the article and include the copyright. Not sourced from any founding paper or review."
   },
   "constructs_claimed": "Non-specific psychological distress over the preceding 30 days, indexed through symptoms of anxiety and depression (nervousness, agitation, fatigue, hopelessness, negative affect). It is a dimensional screen for the likely presence of a common mental disorder, not a diagnostic instrument and not a measure of any single named disorder.",
   "deployment_context_caveat": "The K10 is an epidemiological and clinical screening instrument developed for population surveys and validated against structured psychiatric diagnosis. It is not a workplace-designed measure and carries no evidence base against workplace outcomes. Any organisational deployment is off-label relative to its validation: scores index general psychological distress, not work-related distress, and no occupational cut-offs or work-outcome linkages have been established. 'Clinical' or 'diagnostic' language is appropriate only to its origin and reference-standard validation, never to a workplace-screening use.",
   "structural_validity": {
    "findings": "The dimensionality of the K10 is genuinely contested. The developers treated it as essentially unidimensional (a single distress continuum) from item-response-theory modelling ([Kessler 2002](https://doi.org/10.1017/s0033291702006074)). A widely cited community-sample analysis instead found four first-order factors (labelled nervous, negative affect, fatigue and agitation) resolving into two second-order factors interpreted as depression and anxiety, replicated across two survey waves ([Brooks 2006](https://doi.org/10.1037/1040-3590.18.1.62)). Subsequent confirmatory work has not settled on one model: a large Chinese healthcare-professional sample favoured a two-factor (depression and anxiety) oblique model over a one-factor model ([Wang 2025](https://doi.org/10.1016/j.genhosppsych.2025.02.017)), a two-factor solution also emerged among children of Chinese migrant workers ([Ren 2021](https://doi.org/10.1002/brb3.2417)) and in a Portuguese adult sample the original one-dimensional structure was not confirmed in favour of correlated anxiety and depression factors ([Pereira 2019](https://doi.org/10.1590/1413-81232018243.06322017)). Several sub-Saharan validations converged instead on a unidimensional model with correlated error terms as the best-fitting and most parsimonious solution, with four-factor solutions judged to be overfitted or artefactual ([Milkias 2022](https://doi.org/10.1016/j.jad.2022.02.013); [Naisanga 2022](https://doi.org/10.1016/j.jad.2022.05.022); [Hoffman 2022](https://doi.org/10.1186/s40359-022-00883-9)). Bifactor models (a general distress factor plus specific anxiety and depression factors) have fitted well in Brazilian samples ([Perrelli 2024](https://doi.org/10.1590/1518-8345.7073.4254); [Peixoto 2021](https://doi.org/10.1186/s41155-021-00186-9)). Critically, neither the single-factor nor the multifactor model fitted in a treatment-seeking clinical sample, which the authors read as a warning against assuming the epidemiological structure holds in clinical settings ([Berle 2010](https://doi.org/10.1097/NMD.0b013e3181ef1f16)). The pattern is best summarised as a robust general distress dimension overlaid by a separable anxiety and depression split, with the winning statistical model depending heavily on sample, estimator and whether correlated errors are permitted.",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "indirect. Structural evidence is abundant but earned mainly in Australian, sub-Saharan African, Chinese and South American general or clinical samples; no UK working-adult factor-analytic study was located this session."
   },
   "convergent_discriminant_validity": {
    "findings": "The K10 correlates strongly with other distress and internalising measures. It showed a strong correlation with the Self-Reporting Questionnaire (r = 0.81) in Brazilian higher-education students ([Perrelli 2024](https://doi.org/10.1590/1518-8345.7073.4254)) and a moderate correlation (r = 0.63) with the emotional-symptoms subscale of the Strengths and Difficulties Questionnaire in Australian adolescents ([Blake 2023](https://doi.org/10.1177/00048674231216601)). Latent-variable work places the K10 firmly on a single internalising-psychopathology dimension shared with DSM-IV depression and anxiety diagnoses ([Sunderland 2013](https://doi.org/10.1037/a0034562)). Higher K10 scores track strongly and monotonically with suicidal ideation, those in the very-high band being an order of magnitude more likely to report ideation than those in the low band ([Chamberlain/Goldney 2009](https://doi.org/10.1027/0227-5910.30.1.39)). Discriminant separation from unrelated constructs is less formally documented; most validation studies emphasise convergent rather than discriminant coefficients.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect. Convergent evidence comes from Australian, Canadian, Brazilian and other non-UK general, student and adolescent samples rather than UK workers."
   },
   "criterion_validity_reference_standard": {
    "findings": "This is the K10's strongest evidence base. Against structured diagnostic interviews (CIDI or SCID for DSM-IV disorders) the scale discriminates cases from non-cases well. In the original development and clinical reappraisal, areas under the ROC curve were 0.87 to 0.88 for disorders of moderate severity and 0.95 to 0.96 for severe disorders ([Kessler 2002](https://doi.org/10.1017/s0033291702006074)). In the nationally representative Australian National Survey of Mental Health and Well-Being the K10 achieved an AUC of 0.90 (95% CI 0.89 to 0.91) for CIDI/DSM-IV mood and anxiety disorders, outperforming the GHQ-12 (AUC 0.80) ([Furukawa 2003](https://doi.org/10.1017/s0033291702006700)). Against serious mental illness defined by SCID plus functional impairment, the K10 AUC was 0.85 ([Kessler 2003](https://doi.org/10.1001/archpsyc.60.2.184)). Performance is more modest outside high-income settings: in the South African Stress and Health study the K10 showed only moderate discrimination (AUC 0.73 for depression, 0.72 for anxiety) and failed the authors' joint sensitivity and positive-predictive-value criteria, with poorer discrimination in the Black subgroup ([Andersen 2011](https://doi.org/10.1002/mpr.351)). In Canadian military personnel the AUC against four past-month disorders was 0.92, with a screening cut-off of 10 or greater giving 86% sensitivity and 83% specificity ([Sampasa-Kanyinga 2018](https://doi.org/10.1371/journal.pone.0196562)). A Swiss community study found much poorer agreement with MINI diagnoses than the classic Australian work, with low Cohen's kappa at every cut-off, cautioning against community screening use ([Osman 2022](https://doi.org/10.1111/eip.13296)). Cut-offs are population-dependent and not universal, and the reference-standard time frame (12-month diagnosis) often mismatches the K10's 30-day window, a recognised limitation ([Andersen 2011](https://doi.org/10.1002/mpr.351)).",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect. The strongest diagnostic-accuracy evidence is Australian, US and Canadian general-population and military samples; no UK working-adult criterion study against a diagnostic standard was located, and accuracy varies by setting."
   },
   "criterion_validity_organisational": {
    "findings": "No study located this session validated the K10 against an organisational or work outcome (sickness absence, turnover, job performance, or work-recorded diagnosis) as a criterion. The K10 and its K6 short form appear widely in occupational and workplace surveys, but as an exposure or prevalence measure rather than as a screen validated against a work-outcome standard. Examples of such descriptive use include an Australian industry comparison linking very-high distress to self-reported productivity loss and work-cutback days ([Burns 2023](https://doi.org/10.1002/1348-9585.12428)), a thirty-seven-year US panel study relating occupation and job tenure to new distress cases measured with the K6 ([Laditka 2023](https://doi.org/10.1186/s40359-023-01119-0)), and a Japanese occupational cohort using a K6 cut-off to define distress caseness at baseline ([workplace social support, cited in record notes]). These establish that K10/K6 distress associates with work-relevant variables, but none provides criterion validity in the COSMIN sense: distress is the predictor or outcome of interest, not a test being validated against an independent work-outcome gold standard, and no work-based cut-off has been derived or calibrated. Organisational criterion validity is therefore Absent as a formal psychometric property.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect. Even the associational workplace literature is largely Australian, US and Japanese; no UK workplace criterion evidence was located and none is criterion-validation in design."
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently high and this is the K10's best-replicated reliability property. A reliability-generalisation meta-analysis of 48 studies (2002 to 2024) estimated a pooled Cronbach's alpha of 0.90 (95% CI 0.88 to 0.91) for the K10, with variation across populations from about 0.78 (a Tanzanian sample) to 0.97 (Australian samples), and highest values among adolescents (0.93) and carers (0.91) ([Wojujutari 2024](https://doi.org/10.1155/2024/3801950)). Primary studies corroborate this: alpha 0.88 in Canadian military personnel ([Sampasa-Kanyinga 2018](https://doi.org/10.1371/journal.pone.0196562)), 0.95 in Chinese healthcare professionals ([Wang 2025](https://doi.org/10.1016/j.genhosppsych.2025.02.017)), 0.83 to 0.86 across Ethiopian, Ugandan and South African general and medical samples ([Milkias 2022](https://doi.org/10.1016/j.jad.2022.02.013); [Naisanga 2022](https://doi.org/10.1016/j.jad.2022.05.022); [Hoffman 2022](https://doi.org/10.1186/s40359-022-00883-9), the last also reporting McDonald's omega total of 0.88), and 0.91 in Portuguese adults ([Pereira 2019](https://doi.org/10.1590/1413-81232018243.06322017)). Lower values appear in some community samples, for example alpha 0.81 in a Swiss young-adult sample ([Osman 2022](https://doi.org/10.1111/eip.13296)). Values in the 0.80s to low 0.90s are the norm.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect. Alpha is high almost everywhere it has been measured, but the pooled evidence is dominated by non-UK general, clinical, student and military samples; no UK working-adult alpha was isolated this session, though a UK adult study using the K6 short form reported adequate fit ([Lantos 2023](https://doi.org/10.1016/j.jad.2023.06.033))."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "thin",
    "indirectness": "indirect. No canonical K10 test-retest coefficient was located this session; the absence itself is population-general.",
    "summary": "No test-retest or temporal-stability coefficient for the canonical K10 (or K6) was located in this session's searches. The K10 was developed as a cross-sectional epidemiological screen with a 30-day recall window, and its reliability evidence base is overwhelmingly internal-consistency (alpha) based; the large reliability-generalisation meta-analysis synthesised alpha only, not retest reliability ([Wojujutari 2024](https://doi.org/10.1155/2024/3801950)). This is a genuine and important gap: high internal consistency does not establish temporal stability, and the two must not be conflated. For an instrument used to track change (including any longitudinal workplace use), the absence of established test-retest reliability is a material limitation and is recorded here as the finding it is, not averaged away."
   },
   "measurement_invariance": {
    "findings": "Invariance evidence is moderate and mostly supportive across sex and age, thinner across culture and untested across occupation. Strict measurement invariance held across two Australian survey administrations ten years apart and across age bands for a latent internalising model incorporating K10 items ([Sunderland 2013](https://doi.org/10.1037/a0034562)). Full measurement invariance across gender was supported among children of Chinese migrant workers ([Ren 2021](https://doi.org/10.1002/brb3.2417)), and Rasch analysis in older Australians found scale invariance across sex, age and education after minor model modification ([Calkin 2023](https://doi.org/10.1016/j.jad.2023.02.116)). A Brazilian adaptation reported multiple-group invariance by gender and age range ([Peixoto 2021](https://doi.org/10.1186/s41155-021-00186-9)). A UK adult study reported support for measurement invariance across birth sex and age for the K6 short form ([Lantos 2023](https://doi.org/10.1016/j.jad.2023.06.033)). Against this, cross-national and cross-ethnic comparability is questioned: the South African study found significantly lower discrimination in the Black subgroup, implying non-equivalence by race/ethnicity ([Andersen 2011](https://doi.org/10.1002/mpr.351)). No study located this session tested invariance across occupational groups or between working and non-working adults, which is the comparison most relevant to workplace deployment.",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect. Sex and age invariance is reasonably supported but in Australian, Chinese and Brazilian samples; cross-cultural equivalence is contested and occupational invariance is untested.",
    "subgrades": [
     {
      "subgroup": "across sex",
      "grade": "Moderate",
      "note": "Full or strict invariance supported in several samples (Ren 2021; Calkin 2023; Sunderland 2013)."
     },
     {
      "subgroup": "across age",
      "grade": "Moderate",
      "note": "Invariance supported across age bands and over a ten-year interval in Australian data (Sunderland 2013; Calkin 2023)."
     },
     {
      "subgroup": "across culture/ethnicity",
      "grade": "Low",
      "note": "Differential performance by race/ethnicity in South Africa suggests non-equivalence (Andersen 2011)."
     },
     {
      "subgroup": "across occupation",
      "grade": "Absent",
      "note": "No test of invariance across occupational groups or working vs non-working adults located this session."
     }
    ]
   },
   "responsiveness_mic": {
    "findings": "No formal responsiveness study (ability to detect within-person change) or minimal important change (MIC) value for the K10 was located this session. Related psychometric work is severity-oriented rather than change-oriented: severity banding into no, mild, moderate and severe distress is well established for cross-sectional interpretation ([Ul Husnain 2024](https://doi.org/10.1007/s40258-024-00879-z), using Australian HILDA data), and Rasch work has produced ordinal-to-interval conversion tables intended to improve measurement precision in older adults ([Calkin 2023](https://doi.org/10.1016/j.jad.2023.02.116)), but neither reports a minimal important change or an anchor-based responsiveness estimate. The K10's design and evidence base are cross-sectional; responsiveness and MIC are effectively unestablished.",
    "grade": "Absent",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect. No responsiveness or MIC evidence located in any population, UK or otherwise."
   },
   "populations_languages_norms": {
    "findings": "Norms and translations are extensive. Australian normative data are the reference standard: interpretive norms from the 1997 survey ([Andrews 2001](https://doi.org/10.1111/j.1467-842x.2001.tb00310.x)) and full normative tables by sex, age and disorder status from the 2007 National Survey of Mental Health and Wellbeing (n = 8841), with stratum-specific likelihood ratios for estimating disorder probability ([Slade 2011](https://doi.org/10.3109/00048674.2010.543653)). Australian adolescent norms (ages 11 to 17, n = 2964) are available with the caveat of low specificity for major depressive disorder ([Blake 2023](https://doi.org/10.1177/00048674231216601)). Multi-country European general-population norms across seven countries (n = 7087) show mean total scores varying from about 6.9 (Netherlands) to 9.9 (Spain) on the 0 to 40 metric, women and younger adults scoring higher ([Lehmann 2023](https://doi.org/10.1038/s41598-023-45124-0)). Validated translations located this session include Arabic ([Easton 2017, cited in record notes]), Brazilian Portuguese ([Perrelli 2024](https://doi.org/10.1590/1518-8345.7073.4254); [Peixoto 2021](https://doi.org/10.1186/s41155-021-00186-9)), European Portuguese ([Pereira 2019](https://doi.org/10.1590/1413-81232018243.06322017)), and multiple sub-Saharan African adaptations ([Milkias 2022](https://doi.org/10.1016/j.jad.2022.02.013); [Naisanga 2022](https://doi.org/10.1016/j.jad.2022.05.022); [Hoffman 2022](https://doi.org/10.1186/s40359-022-00883-9)). UK usage is real: the K10 and K6 appear in UK survey and social-science research, and a UK adult sample analysed the K6 short form ([Lantos 2023](https://doi.org/10.1016/j.jad.2023.06.033)); however, no UK working-adult normative table for the full K10 was located this session. For a UK working-adult audience, the closest directly usable norms are the seven-country European set, which does not include the UK.",
    "grade": "High",
    "indirectness": "indirect. Norms are rich for Australia and continental Europe; UK-specific working-adult norms for the full K10 were not located this session."
   },
   "criticisms_controversies": "Three issues recur. First, dimensionality is unresolved: the developers' unidimensional treatment coexists with well-supported two-factor (anxiety and depression) and bifactor models, and the best-fitting model is sample- and method-dependent ([Brooks 2006](https://doi.org/10.1037/1040-3590.18.1.62); [Wang 2025](https://doi.org/10.1016/j.genhosppsych.2025.02.017); [Milkias 2022](https://doi.org/10.1016/j.jad.2022.02.013)). Second, the epidemiological factor structure and cut-offs do not transfer cleanly to clinical or community-screening settings: neither standard model fitted a treatment-seeking sample ([Berle 2010](https://doi.org/10.1097/NMD.0b013e3181ef1f16)), and a Swiss community study found agreement with diagnostic interviews too poor to recommend community screening ([Osman 2022](https://doi.org/10.1111/eip.13296)). Third, cut-offs are not universal, vary by country, age and sex, and diagnostic accuracy drops in some non-Western and minority-ethnic samples ([Andersen 2011](https://doi.org/10.1002/mpr.351); [Blake 2023](https://doi.org/10.1177/00048674231216601)). Under-recognised gaps are the near-absence of published test-retest reliability and of responsiveness or minimal-important-change evidence, both material for any use that tracks change over time. For workplace use specifically, the instrument has no work-outcome validation and no occupational norms or invariance testing.",
   "item_records": [],
   "citations": [
    {
     "key": "kessler2002",
     "authors": "Kessler R C, Andrews G, Colpe L J et al.",
     "year": "2002",
     "title": "Short screening scales to monitor population prevalences and trends in non-specific psychological distress.",
     "journal": "Psychological medicine",
     "doi": "10.1017/s0033291702006074",
     "url": "https://doi.org/10.1017/s0033291702006074"
    },
    {
     "key": "andrews2001",
     "authors": "Andrews G, Slade T",
     "year": "2001",
     "title": "Interpreting scores on the Kessler Psychological Distress Scale (K10).",
     "journal": "Australian and New Zealand journal of public health",
     "doi": "10.1111/j.1467-842x.2001.tb00310.x",
     "url": "https://doi.org/10.1111/j.1467-842x.2001.tb00310.x"
    },
    {
     "key": "kessler2003smi",
     "authors": "Kessler Ronald C, Barker Peggy R, Colpe Lisa J et al.",
     "year": "2003",
     "title": "Screening for serious mental illness in the general population.",
     "journal": "Archives of general psychiatry",
     "doi": "10.1001/archpsyc.60.2.184",
     "url": "https://doi.org/10.1001/archpsyc.60.2.184"
    },
    {
     "key": "furukawa2003",
     "authors": "Furukawa T A, Kessler R C, Slade T et al.",
     "year": "2003",
     "title": "The performance of the K6 and K10 screening scales for psychological distress in the Australian National Survey of Mental Health and Well-Being.",
     "journal": "Psychological medicine",
     "doi": "10.1017/s0033291702006700",
     "url": "https://doi.org/10.1017/s0033291702006700"
    },
    {
     "key": "brooks2006",
     "authors": "Brooks Robert T, Beard John, Steel Zachary",
     "year": "2006",
     "title": "Factor structure and interpretation of the K10.",
     "journal": "Psychological assessment",
     "doi": "10.1037/1040-3590.18.1.62",
     "url": "https://doi.org/10.1037/1040-3590.18.1.62"
    },
    {
     "key": "berle2010",
     "authors": "Berle David, Starcevic Vladan, Milicevic Denise et al.",
     "year": "2010",
     "title": "The factor structure of the Kessler-10 questionnaire in a treatment-seeking sample.",
     "journal": "The Journal of nervous and mental disease",
     "doi": "10.1097/NMD.0b013e3181ef1f16",
     "url": "https://doi.org/10.1097/NMD.0b013e3181ef1f16"
    },
    {
     "key": "andersen2011",
     "authors": "Andersen L S, Grimsrud A, Myer L et al.",
     "year": "2011",
     "title": "The psychometric properties of the K10 and K6 scales in screening for mood and anxiety disorders in the South African Stress and Health study.",
     "journal": "International journal of methods in psychiatric research",
     "doi": "10.1002/mpr.351",
     "url": "https://doi.org/10.1002/mpr.351"
    },
    {
     "key": "slade2011",
     "authors": "Slade Tim, Grove Rachel, Burgess Philip",
     "year": "2011",
     "title": "Kessler Psychological Distress Scale: normative data from the 2007 Australian National Survey of Mental Health and Wellbeing.",
     "journal": "The Australian and New Zealand journal of psychiatry",
     "doi": "10.3109/00048674.2010.543653",
     "url": "https://doi.org/10.3109/00048674.2010.543653"
    },
    {
     "key": "sunderland2014",
     "authors": "Sunderland Matthew, Slade Tim, Carragher Natacha et al.",
     "year": "2013",
     "title": "Age-related differences in internalizing psychopathology amongst the Australian general population.",
     "journal": "Journal of abnormal psychology",
     "doi": "10.1037/a0034562",
     "url": "https://doi.org/10.1037/a0034562"
    },
    {
     "key": "ren2021",
     "authors": "Ren Qiang, Li Yong, Chen Ding-Geng",
     "year": "2021",
     "title": "Measurement invariance of the Kessler Psychological Distress Scale (K10) among children of Chinese rural-to-urban migrant workers.",
     "journal": "Brain and behavior",
     "doi": "10.1002/brb3.2417",
     "url": "https://doi.org/10.1002/brb3.2417"
    },
    {
     "key": "lehmann2023",
     "authors": "Lehmann J, Pilz M J, Holzner B et al.",
     "year": "2023",
     "title": "General population normative data from seven European countries for the K10 and K6 scales for psychological distress.",
     "journal": "Scientific reports",
     "doi": "10.1038/s41598-023-45124-0",
     "url": "https://doi.org/10.1038/s41598-023-45124-0"
    },
    {
     "key": "blake2023",
     "authors": "Blake Julie A, Farugia Taya L, Andrew Brooke et al.",
     "year": "2023",
     "title": "The Kessler Psychological Distress Scale in Australian adolescents: Analysis of the second Australian Child and Adolescent Survey of Mental Health and Wellbeing.",
     "journal": "The Australian and New Zealand journal of psychiatry",
     "doi": "10.1177/00048674231216601",
     "url": "https://doi.org/10.1177/00048674231216601"
    },
    {
     "key": "calkin2023",
     "authors": "Calkin Cailen J, Numbers Katya, Brodaty Henry et al.",
     "year": "2023",
     "title": "Measuring distress in older population: Rasch analysis of the Kessler Psychological Distress Scale.",
     "journal": "Journal of affective disorders",
     "doi": "10.1016/j.jad.2023.02.116",
     "url": "https://doi.org/10.1016/j.jad.2023.02.116"
    },
    {
     "key": "wang2025",
     "authors": "Wang Ye, Zeng Zheng, Huang Changqun et al.",
     "year": "2025",
     "title": "Large-scale validation of the Kessler-10 Scale's psychometric properties among healthcare professionals in China.",
     "journal": "General hospital psychiatry",
     "doi": "10.1016/j.genhosppsych.2025.02.017",
     "url": "https://doi.org/10.1016/j.genhosppsych.2025.02.017"
    },
    {
     "key": "wojujutari2024",
     "authors": "Wojujutari Ajele Kenni, Idemudia Erhabor Sunday",
     "year": "2024",
     "title": "Consistency as the Currency in Psychological Measures: A Reliability Generalization Meta-Analysis of Kessler Psychological Distress Scale (K-10 and K-6).",
     "journal": "Depression and anxiety",
     "doi": "10.1155/2024/3801950",
     "url": "https://doi.org/10.1155/2024/3801950"
    },
    {
     "key": "sampasa2018",
     "authors": "Sampasa-Kanyinga Hugues, Zamorski Mark A, Colman Ian",
     "year": "2018",
     "title": "The psychometric properties of the 10-item Kessler Psychological Distress Scale (K10) in Canadian military personnel.",
     "journal": "PloS one",
     "doi": "10.1371/journal.pone.0196562",
     "url": "https://doi.org/10.1371/journal.pone.0196562"
    },
    {
     "key": "milkias2022",
     "authors": "Milkias Barkot, Ametaj Amantia, Alemayehu Melkam et al.",
     "year": "2022",
     "title": "Psychometric properties and factor structure of the Kessler-10 among Ethiopian adults.",
     "journal": "Journal of affective disorders",
     "doi": "10.1016/j.jad.2022.02.013",
     "url": "https://doi.org/10.1016/j.jad.2022.02.013"
    },
    {
     "key": "naisanga2022",
     "authors": "Naisanga Molly, Ametaj Amantia, Kim Hannah H et al.",
     "year": "2022",
     "title": "Construct validity and factor structure of the K-10 among Ugandan adults.",
     "journal": "Journal of affective disorders",
     "doi": "10.1016/j.jad.2022.05.022",
     "url": "https://doi.org/10.1016/j.jad.2022.05.022"
    },
    {
     "key": "hoffman2022",
     "authors": "Hoffman Jacob, Cossie Qhama, Ametaj Amantia A et al.",
     "year": "2022",
     "title": "Construct validity and factor structure of the Kessler-10 in South Africa.",
     "journal": "BMC psychology",
     "doi": "10.1186/s40359-022-00883-9",
     "url": "https://doi.org/10.1186/s40359-022-00883-9"
    },
    {
     "key": "prade2022swiss",
     "authors": "Osman Naweed, Chow Winnie S, Michel Chantal et al.",
     "year": "2022",
     "title": "Psychometric properties of the Kessler psychological scales in a Swiss young-adult community sample indicate poor suitability for community screening for mental disorders.",
     "journal": "Early intervention in psychiatry",
     "doi": "10.1111/eip.13296",
     "url": "https://doi.org/10.1111/eip.13296"
    },
    {
     "key": "davies2023uk",
     "authors": "Lantos Dorottya, Moreno-Agostino Darío, Harris Lasana T et al.",
     "year": "2023",
     "title": "The performance of long vs. short questionnaire-based measures of depression, anxiety, and psychological distress among UK adults: A comparison of the patient health questionnaires, generalized anxiety disorder scales, malaise inventory, and Kessler scales.",
     "journal": "Journal of affective disorders",
     "doi": "10.1016/j.jad.2023.06.033",
     "url": "https://doi.org/10.1016/j.jad.2023.06.033"
    },
    {
     "key": "goldney2009",
     "authors": "Chamberlain Peter, Goldney Robert, Delfabbro Paul et al.",
     "year": "2009",
     "title": "Suicidal ideation. The clinical utility of the K10.",
     "journal": "Crisis",
     "doi": "10.1027/0227-5910.30.1.39",
     "url": "https://doi.org/10.1027/0227-5910.30.1.39"
    },
    {
     "key": "dollard2023industry",
     "authors": "Burns Kristy, Schroeder Elizabeth-Ann, Fung Thomas et al.",
     "year": "2023",
     "title": "Industry differences in psychological distress and distress-related productivity loss: A cross-sectional study of Australian workers.",
     "journal": "Journal of occupational health",
     "doi": "10.1002/1348-9585.12428",
     "url": "https://doi.org/10.1002/1348-9585.12428"
    },
    {
     "key": "kim2023occup",
     "authors": "Laditka James N, Laditka Sarah B, Arif Ahmed A et al.",
     "year": "2023",
     "title": "Psychological distress is more common in some occupations and increases with job tenure: a thirty-seven year panel study in the United States.",
     "journal": "BMC psychology",
     "doi": "10.1186/s40359-023-01119-0",
     "url": "https://doi.org/10.1186/s40359-023-01119-0"
    },
    {
     "key": "engel2024hilda",
     "authors": "Ul Husnain Muhammad Iftikhar, Hajizadeh Mohammad, Ahmad Hasnat et al.",
     "year": "2024",
     "title": "The Hidden Toll of Psychological Distress in Australian Adults and Its Impact on Health-Related Quality of Life Measured as Health State Utilities.",
     "journal": "Applied health economics and health policy",
     "doi": "10.1007/s40258-024-00879-z",
     "url": "https://doi.org/10.1007/s40258-024-00879-z"
    },
    {
     "key": "monteiro2019pt",
     "authors": "Pereira Anabela, Oliveira Carla Andreia, Bártolo Ana et al.",
     "year": "2019",
     "title": "Reliability and Factor Structure of the 10-item Kessler Psychological Distress Scale (K10) among Portuguese adults.",
     "journal": "Ciencia & saude coletiva",
     "doi": "10.1590/1413-81232018243.06322017",
     "url": "https://doi.org/10.1590/1413-81232018243.06322017"
    },
    {
     "key": "easton2017arabic",
     "authors": "Easton Scott D, Safadi Nadia S, Wang Yao",
     "year": "2017",
     "title": "The Kessler psychological distress scale: translation and validation of an Arabic version",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-017-0783-9",
     "url": "https://doi.org/10.1186/s12955-017-0783-9"
    },
    {
     "key": "damasio2021brazil",
     "authors": "Peixoto Evandro Morais, Zanini Daniela Sacramento, de Andrade Josemberg Moura",
     "year": "2021",
     "title": "Cross-cultural adaptation and psychometric properties of the Kessler Distress Scale (K10): an application of the rating scale model",
     "journal": "Psicologia: Reflexao e Critica",
     "doi": "10.1186/s41155-021-00186-9",
     "url": "https://doi.org/10.1186/s41155-021-00186-9"
    },
    {
     "key": "japan_social_support2022",
     "authors": "Inoue Yosuke, Hikichi Hiroyuki, Inoue Mariko et al.",
     "year": "2022",
     "title": "Workplace Social Support and Reduced Psychological Distress: A 1-Year Occupational Cohort Study",
     "journal": "Journal of Occupational and Environmental Medicine",
     "doi": "10.1097/JOM.0000000000002675",
     "url": "https://doi.org/10.1097/JOM.0000000000002675"
    },
    {
     "key": "perrelli2024brazil",
     "authors": "Perrelli Jaqueline Galdino Albuquerque, Vasconcelos Gabriel Vinicius Souza de, Correia e Sa Jessica Rodrigues et al.",
     "year": "2024",
     "title": "Validity of the Kessler Psychological Distress scale in Brazilian higher education students",
     "journal": "Revista Latino-Americana de Enfermagem",
     "doi": "10.1590/1518-8345.7073.4254",
     "url": "https://doi.org/10.1590/1518-8345.7073.4254"
    }
   ],
   "record_notes": "Overall confidence: the K10 is a strong screener for the likely presence of common mental disorders against a diagnostic reference standard (High, well-established, principally Australian, US and Canadian samples), with high internal consistency (High, well-established) and extensive norms (High, but not UK-specific for working adults). Its factor structure is genuinely contested (recorded as contested, not averaged), and three properties are honestly Absent or thin: test-retest reliability (no coefficient located this session), responsiveness/MIC (none located), and organisational criterion validity (no work-outcome validation exists; the workplace literature uses the K10/K6 as an exposure or prevalence measure, not as a validated work-outcome screen). For the UK working-adult audience every graded property is indirect: the evidence was earned in non-UK, non-workplace, general-population, student, clinical or military samples, and the instrument is clinical/epidemiological in origin, making any workplace deployment off-label (see deployment_context_caveat). Licence verified against the Harvard Medical School NCS steward page this session (2026-07-12): free to use, no permission or fee required, cite the source; no named open-content licence code is asserted by the steward. A few supporting sources are named in prose but not separately DOI-listed where they duplicate an already-cited finding (Arabic validation Easton 2017 doi:10.1186/s12955-017-0783-9; Brazilian rating-scale-model adaptation, Peixoto 2021 doi:10.1186/s41155-021-00186-9; Japanese workplace social-support cohort using K6 doi:10.1097/JOM.0000000000002675); these resolve but were retrieved as context rather than as primary graded evidence. Schema v0.2 friction: the criterion split worked well and made the organisational Absent finding explicit; the single evidence_form tag per property is a simplification where evidence is genuinely mixed (canonical K10 plus K6 short-form and parent-item evidence), flagged as 'mixed' where that applies."
  },
  {
   "instrument_id": "ghq-12",
   "display_name": "General Health Questionnaire-12 (GHQ-12)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "General Health Questionnaire-12 (GHQ-12)",
    "current_version": "GHQ-12 (12-item short form of Goldberg's General Health Questionnaire; the family comprises GHQ-60, GHQ-30, GHQ-28 and GHQ-12)",
    "item_count": "12",
    "original_citation": "Goldberg DP (1972) The Detection of Psychiatric Illness by Questionnaire, Maudsley Monograph No. 21, Oxford University Press; the 12-item short form was later derived from this work. Validity of the two short versions summarised by [Goldberg et al. 1997](https://doi.org/10.1017/s0033291796004242).",
    "steward_publisher": "GL Assessment (part of the GL Education Group), United Kingdom, copyright holder; translated versions distributed internationally by the Mapi Research Trust via ePROVIDE. Part of the permissions payment is paid as a royalty to the Institute of Psychiatry.",
    "licence_status": "Proprietary and commercial (licensed, paid). The GHQ is not open access or public domain. The steward's current copyright statement reads that the General Health Questionnaire is protected worldwide by international copyright laws in all languages, with all rights reserved to GL Assessment, UK, and must not be used without permission; permission for licensees in the United Kingdom, Republic of Ireland and Channel Islands is obtained from GL Assessment (permissions@gl-assessment.co.uk) and, for other countries, from the Mapi Research Trust via ePROVIDE (https://eprovide.mapi-trust.org). GL Assessment's current permissions policy states that all GL Assessment products are protected by copyright and may not usually be reproduced in hard copy or electronic form, or translated, without permission. GL Assessment's product page markets the GHQ range (GHQ-12, GHQ-28, GHQ-30, GHQ-60) as a purchasable assessment requiring professional-eligibility registration to order. The steward's GHQ support FAQ states that photocopying a record form without abiding by the permission conditions is regarded as theft and a criminal offence, that part of the permissions payment is paid as a royalty to the Institute of Psychiatry, and that translated versions are distributed by the Mapi Research Trust but have not been validated by GL Assessment. Verified this session against the steward's current pages, not founding papers.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "GL Assessment GHQ product page (https://www.gl-assessment.co.uk/products/general-health-questionnaire/, page current May 2026); GL Assessment permissions policy (https://www.gl-assessment.co.uk/policies/permissions/); GL Assessment / GL Education GHQ support FAQ page (https://support.gl-assessment.co.uk/knowledge-base/assessments/general-health-questionnaire-support/about-the-general-health-questionnaire/faqs), which carries the 'photocopying a record form is regarded as theft and a criminal offence', Institute of Psychiatry royalty, and Mapi-distributed-unvalidated-translations statements; and the GL Assessment copyright/permission notice ('all rights reserved to GL Assessment, UK; do not use without permission; UK/Ireland/Channel Islands contact permissions@gl-assessment.co.uk, other countries contact the Mapi Research Trust'). All read this session via web search of the steward's live pages."
   },
   "constructs_claimed": "Non-specific psychiatric morbidity / common mental disorder, that is current psychological distress. The GHQ detects breaks in normal healthy functioning and the appearance of new distressing symptoms over the recent past (roughly the last few weeks), rather than long-standing or enduring conditions. Marketed and used as a screener for minor psychiatric disorder in community, primary-care and occupational-health settings.",
   "deployment_context_caveat": "The GHQ-12 is a psychiatric-morbidity screener developed and validated for case detection in general medical, primary-care and community settings; its origin and validation are stated here as fact, but workplace deployment is a different context and any occupational use must be treated as such. It measures short-term states (symptoms of less than roughly two weeks are captured, and the instrument does not detect chronic or long-standing conditions), so it is a distress screener rather than a diagnostic or wellbeing measure. It is also a proprietary, licensed instrument requiring paid permission, which contrasts with the free instruments in this registry and constrains organisational deployment (cost, permission, and prohibition on uncontrolled reproduction).",
   "structural_validity": {
    "findings": "The internal structure is the central and long-running controversy for this instrument, and it is driven by scoring method rather than by any true multidimensionality. Although Goldberg designed the GHQ-12 as a unidimensional measure, factor-analytic studies have reported one-, two- and three-factor solutions, the best known being Graetz's three factors (anxiety/depression, social dysfunction, loss of confidence) and various two-factor solutions splitting positively and negatively worded items. [Hankins 2008](https://doi.org/10.1186/1471-2458-8-355) showed in Health Survey for England data (n=3705) that the best fitting model is one-dimensional with a response-bias factor on the negatively phrased items, concluding that earlier multifactor structures were artefacts of the analysis method and of the mixing of positively and negatively worded items across scoring schemes (Likert, GHQ 0-0-1-1 and C-GHQ). [Rey et al. 2014](https://doi.org/10.1037/a0036468) replicated this in a large Spanish sample (n=27,674), finding that spurious multidimensionality appears only under corrected and Likert scoring because of ambiguous response categories in the negative items, and recommending standard GHQ scoring with a single global score. [Romppel et al. 2013](https://doi.org/10.1016/j.comppsych.2012.10.010) reached the same conclusion in a German population sample (N=2041), with subscale correlations against external criteria (BDI, PHQ-2, SF-36) not differing substantially from one another. The strongest evidence comes from [Gnambs and Staufenbiel 2018](https://doi.org/10.1080/17437199.2018.1426484), two meta-analyses (summary data from 38 studies, total N=76,473; and individual responses from 84 samples, N=410,640): confirmatory and bifactor modelling showed that although two wording-based factors are recoverable, almost all common variance loads on a general factor, so the GHQ-12 is essentially unidimensional and subscale scores should not be interpreted. [Hystad and Johnsen 2020](https://doi.org/10.3389/fpsyg.2020.01300) corroborated the bifactor structure (general factor plus positive- and negative-wording method factors) in military samples. The modern consensus is therefore essential unidimensionality with method effects from item wording; the historical multifactor literature is treated here as contested and largely superseded.",
    "grade": "High",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "direct - large representative general-population and survey samples (England, Spain, Germany) plus meta-analytic aggregation; the definitive evidence is not workplace-specific but the structural conclusion is not population-dependent."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent evidence is consistent though less systematically studied than structure or criterion validity. The GHQ-12 total correlates moderately with other distress and wellbeing measures: r=0.58 with the Subjective Well-being Inventory in older Indian adults ([Qin et al. 2018](https://doi.org/10.4103/psychiatry.IndianJPsychiatry_112_17)). In head-to-head comparisons against diagnostic interviews the GHQ-12 discriminates common mental disorders comparably to the SRQ, K10, K6 and PHQ ([Patel et al. 2007](https://doi.org/10.1017/S0033291707002334); [Gill et al. 2007](https://doi.org/10.1016/j.psychres.2006.11.005)), although [Gill et al. 2007](https://doi.org/10.1016/j.psychres.2006.11.005) found the SF-12 mental component and the K6/K10 outperformed the GHQ-12 for detecting diagnosed depression. In UK cancer patients the GHQ-12 tracked the Distress Thermometer and HADS over time ([Gessler et al. 2008](https://doi.org/10.1002/pon.1273)). Discriminant separation from distinct constructs is less directly documented in the sources retrieved.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect - convergent coefficients come from non-UK, non-workplace samples (India, Australia) and clinical cancer cohorts; no UK working-adult convergent study was located this session."
   },
   "criterion_validity_reference_standard": {
    "findings": "Validity against a diagnostic reference standard is the GHQ-12's best-evidenced property. In the WHO multi-centre study of mental illness in general health care (5438 patients, 15 centres, CIDI primary-care interview), the GHQ-12 achieved a mean area under the ROC curve of 0.88 (range 0.83 to 0.95), performing as well as the longer GHQ-28, with complex scoring offering no advantage and no significant effect of gender, age or education on validity ([Goldberg et al. 1997](https://doi.org/10.1017/s0033291796004242)). Against the CIS-R in Indian primary care the GHQ-12 was among the best of five screeners, though positive predictive value at optimal cut-offs was modest across all five screeners compared, ranging from 51% to 77% ([Patel et al. 2007](https://doi.org/10.1017/S0033291707002334)). Against the CIDI in an Australian general-population survey (N=10,504) the GHQ-12 discriminated depression and anxiety, albeit less well than the K6/K10 for depression ([Gill et al. 2007](https://doi.org/10.1016/j.psychres.2006.11.005)). In adolescents against a SCID DSM-IV-TR interview the AUC was 0.781, with sex-specific optimal thresholds ([Baksheev et al. 2011](https://doi.org/10.1016/j.psychres.2010.10.010)). Case-detection performance is thus robust and reproducible across languages and settings, but is anchored in primary-care, community and clinical samples rather than workplaces.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect - reference-standard validation is extensive but drawn from general medical, primary-care, community and international samples, and (for Baksheev) adolescents; no criterion validation against a diagnostic standard in a UK working-adult sample was located this session."
   },
   "criterion_validity_organisational": {
    "findings": "No study validating the GHQ-12 against organisational outcomes (sickness absence, staff turnover, job performance, or diagnosed conditions recorded in a work context) as a criterion was located this session. The GHQ-12 is very widely used within occupational and workforce samples, for example among South African healthcare workers where its reliability and dimensionality were examined ([Kufe et al. 2024](https://doi.org/10.4102/ajopa.v6i0.144)) and in UK primary-care detection studies ([Plummer et al. 2000](https://doi.org/10.1017/s0033291799002597)), but in those studies the GHQ is deployed as the distress measure itself, not validated against a separate work-outcome standard. The absence of organisational criterion evidence is reported here as the finding it is: the instrument's predictive relationship to absence, turnover or performance in the workplace was not established in the retrieved literature.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct - the absence applies squarely to the UK working-adult deployment question; no evidence located to grade."
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently high. Cronbach's alpha of about 0.90 has been reported under Likert and GHQ scoring in Health Survey for England data, falling to 0.75 under C-GHQ scoring ([Hankins 2008](https://doi.org/10.1186/1471-2458-8-355)); alpha of 0.9 in older Indian adults ([Qin et al. 2018](https://doi.org/10.4103/psychiatry.IndianJPsychiatry_112_17)); and a lower KR-20 of 0.70 under dichotomous scoring in Colombian students ([Simancas-Pallares et al. 2017](https://doi.org/10.7705/biomedica.v37i3.3240)). Values across the wider literature typically sit in the 0.80 to 0.90 range. An important caveat is that [Hankins 2008](https://doi.org/10.1186/1471-2458-8-355) showed Cronbach's alpha overestimates reliability once the negative-item response bias is modelled, so the true measurement precision is somewhat lower than the headline alpha values suggest.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect - alpha estimates come from England (general population), India and international student/community samples rather than UK working adults specifically, though the property is stable across populations."
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "greater than 0.70",
      "coefficient_type": "ICC",
      "interval": "2 weeks",
      "sample_n": "137 analysed (154 tested at baseline)",
      "population": "young adults, Japan (non-clinical)",
      "evidence_form": "canonical",
      "citation_key": "ohno2017"
     }
    ],
    "grade": "Low",
    "status": "thin",
    "indirectness": "indirect - the single located estimate is from Japanese young adults over a two-week interval, not UK working adults; it is also a two-way random-effects ICC for agreement rather than a workplace test-retest.",
    "summary": "Genuine test-retest (stability) evidence for the GHQ-12 is thin. Only one clean estimate was located this session: an intraclass correlation exceeding 0.70 over a two-week interval in 137 young adults, reported alongside a standard error of measurement of 1.47 (bimodal scoring) and 2.44 (Likert scoring) and corresponding smallest detectable change values ([Ohno et al. 2017](https://doi.org/10.1111/jep.12795)). Because the GHQ measures a current, changeable state rather than a trait, high test-retest coefficients are not necessarily expected, and the scarcity of stability data is itself a graded finding. Reported alpha values from other studies must not be mistaken for retest reliability."
   },
   "measurement_invariance": {
    "findings": "Invariance evidence is partial. [Goldberg et al. 1997](https://doi.org/10.1017/s0033291796004242) found no significant effect of gender, age or educational level on the validity of the GHQ across the WHO multi-centre sample, and validity coefficients held across ten translated languages, supporting broad cross-population comparability of case detection. [Hystad and Johnsen 2020](https://doi.org/10.3389/fpsyg.2020.01300) demonstrated that a bifactor structure was invariant across two independent military samples and, in a multi-group analysis, across time. The meta-analytic confirmation of a common structure across 84 samples ([Gnambs and Staufenbiel 2018](https://doi.org/10.1080/17437199.2018.1426484)) is consistent with structural stability across populations. However, formal configural/metric/scalar invariance testing across sex, age band and, in particular, occupation is not comprehensively established in the retrieved literature, and translated versions are noted by the steward as not all validated by the publisher.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "mixed",
    "indirectness": "indirect - invariance evidence comes from military, international and cross-language samples; occupation-level and UK working-adult invariance were not established this session.",
    "subgrades": [
     {
      "subgroup": "sex / age / education",
      "grade": "Moderate",
      "note": "no effect on validity across the WHO multi-centre study (Goldberg et al. 1997); reasonably supported."
     },
     {
      "subgroup": "across time / repeated samples",
      "grade": "Low",
      "note": "bifactor structure invariant across two military samples and over time (Hystad and Johnsen 2020); limited to one military cohort."
     },
     {
      "subgroup": "occupation",
      "grade": "Absent",
      "note": "no formal invariance test across occupational groups located this session."
     }
    ]
   },
   "responsiveness_mic": {
    "findings": "Formal responsiveness and a minimal important change value are not established. [Gessler et al. 2008](https://doi.org/10.1002/pon.1273) found that GHQ-12 scores changed over four and eight weeks in the same direction as the HADS and Distress Thermometer in UK cancer outpatients, providing indirect support for sensitivity to change. [Ohno et al. 2017](https://doi.org/10.1111/jep.12795) derived smallest detectable change values from the standard error of measurement (for example a smallest detectable change of about 4.06 points at the individual level under bimodal scoring), which bounds interpretable change but is a distribution-based statistic, not an anchor-based minimal important change. No validated minimal important change for the GHQ-12 was located this session.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect - the change evidence comes from a UK cancer cohort and Japanese young adults, not UK working adults; no anchor-based minimal important change exists."
   },
   "populations_languages_norms": {
    "findings": "The GHQ-12 is one of the most widely translated and used distress screeners internationally, with validated or examined versions across, among many others, English general-population and survey samples ([Hankins 2008](https://doi.org/10.1186/1471-2458-8-355)), German ([Romppel et al. 2013](https://doi.org/10.1016/j.comppsych.2012.10.010)), Spanish ([Rey et al. 2014](https://doi.org/10.1037/a0036468)), Chinese adolescents ([Li et al. 2009](https://doi.org/10.1111/j.1365-2702.2009.02905.x)), Ukrainian refugees ([Benoni et al. 2024](https://doi.org/10.1186/s12955-024-02226-1)), older Indian adults ([Qin et al. 2018](https://doi.org/10.4103/psychiatry.IndianJPsychiatry_112_17)) and South African healthcare workers ([Kufe et al. 2024](https://doi.org/10.4102/ajopa.v6i0.144)). UK normative data are provided in the publisher's GHQ user guide (a commercial document, not verified this session); the steward notes that translated versions are distributed by the Mapi Research Trust but have not all been validated by GL Assessment. Optimal cut-off scores vary by population, scoring method and sex, so a single universal threshold should not be assumed.",
    "grade": "Moderate",
    "indirectness": "indirect - extensive cross-national coverage but the material anchoring UK working-adult norms sits in a proprietary user guide not accessible this session."
   },
   "criticisms_controversies": "Three related criticisms dominate. First, the scoring-method controversy: the choice between binary GHQ (0-0-1-1), Likert (0-1-2-3) and C-GHQ scoring materially changes the apparent factor structure and reliability, and much of the historical multi-factor literature is now regarded as an artefact of scoring and of the mix of positively and negatively worded items ([Hankins 2008](https://doi.org/10.1186/1471-2458-8-355); [Rey et al. 2014](https://doi.org/10.1037/a0036468); [Gnambs and Staufenbiel 2018](https://doi.org/10.1080/17437199.2018.1426484)). Practitioners should pre-specify scoring and treat the GHQ-12 as a single total score, not interpret subscales. Second, response bias on the negatively phrased items inflates Cronbach's alpha and can create spurious dimensions, so reported reliabilities overstate true precision ([Hankins 2008](https://doi.org/10.1186/1471-2458-8-355)). Third, as a screener the GHQ-12 has modest positive predictive value at realistic prevalence (the five screeners compared, including the GHQ, gave PPVs of 51 to 77% in primary care) and detects only recent-onset distress, so it is not a diagnostic instrument and misses long-standing conditions ([Patel et al. 2007](https://doi.org/10.1017/S0033291707002334)). For this registry's audience two further points matter: it is a licensed commercial product (unlike the free instruments here), and its criterion evidence is clinical/primary-care in origin with no validation against organisational outcomes.",
   "item_records": [],
   "citations": [
    {
     "key": "gnambs2018",
     "authors": "Gnambs T; Staufenbiel T",
     "year": "2018",
     "title": "The structure of the General Health Questionnaire (GHQ-12): two meta-analytic factor analyses",
     "journal": "Health Psychology Review",
     "doi": "10.1080/17437199.2018.1426484",
     "url": "https://doi.org/10.1080/17437199.2018.1426484"
    },
    {
     "key": "hankins2008",
     "authors": "Hankins M",
     "year": "2008",
     "title": "The reliability of the twelve-item general health questionnaire (GHQ-12) under realistic assumptions",
     "journal": "BMC Public Health",
     "doi": "10.1186/1471-2458-8-355",
     "url": "https://doi.org/10.1186/1471-2458-8-355"
    },
    {
     "key": "rey2014",
     "authors": "Rey JJ; Abad FJ; Barrada JR; Garrido LE; Ponsoda V",
     "year": "2014",
     "title": "The impact of ambiguous response categories on the factor structure of the GHQ-12",
     "journal": "Psychological Assessment",
     "doi": "10.1037/a0036468",
     "url": "https://doi.org/10.1037/a0036468"
    },
    {
     "key": "romppel2013",
     "authors": "Romppel M; Braehler E; Roth M; Glaesmer H",
     "year": "2013",
     "title": "What is the General Health Questionnaire-12 assessing? Dimensionality and psychometric properties of the GHQ-12 in a large scale German population sample",
     "journal": "Comprehensive Psychiatry",
     "doi": "10.1016/j.comppsych.2012.10.010",
     "url": "https://doi.org/10.1016/j.comppsych.2012.10.010"
    },
    {
     "key": "goldberg1997",
     "authors": "Goldberg DP; Gater R; Sartorius N; Ustun TB; Piccinelli M; Gureje O; Rutter C",
     "year": "1997",
     "title": "The validity of two versions of the GHQ in the WHO study of mental illness in general health care",
     "journal": "Psychological Medicine",
     "doi": "10.1017/s0033291796004242",
     "url": "https://doi.org/10.1017/s0033291796004242"
    },
    {
     "key": "hystad2020",
     "authors": "Hystad SW; Johnsen BH",
     "year": "2020",
     "title": "The Dimensionality of the 12-Item General Health Questionnaire (GHQ-12): Comparisons of Factor Structures and Invariance Across Samples and Time",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2020.01300",
     "url": "https://doi.org/10.3389/fpsyg.2020.01300"
    },
    {
     "key": "baksheev2011",
     "authors": "Baksheev GN; Robinson J; Cosgrave EM; Baker K; Yung AR",
     "year": "2011",
     "title": "Validity of the 12-item General Health Questionnaire (GHQ-12) in detecting depressive and anxiety disorders among high school students",
     "journal": "Psychiatry Research",
     "doi": "10.1016/j.psychres.2010.10.010",
     "url": "https://doi.org/10.1016/j.psychres.2010.10.010"
    },
    {
     "key": "simancas2017",
     "authors": "Simancas-Pallares MA; Arrieta Vergara KM; Arevalo Tovar L",
     "year": "2017",
     "title": "Construct validity and internal consistency of three factor structures and two scoring methods of the 12-item General Health Questionnaire",
     "journal": "Biomedica",
     "doi": "10.7705/biomedica.v37i3.3240",
     "url": "https://doi.org/10.7705/biomedica.v37i3.3240"
    },
    {
     "key": "qin2018",
     "authors": "Qin T; Vlachantoni A; Evandrou M; Falkingham J",
     "year": "2018",
     "title": "General Health Questionnaire-12 reliability, factor structure, and external validity among older adults in India",
     "journal": "Indian Journal of Psychiatry",
     "doi": "10.4103/psychiatry.IndianJPsychiatry_112_17",
     "url": "https://doi.org/10.4103/psychiatry.IndianJPsychiatry_112_17"
    },
    {
     "key": "patel2007",
     "authors": "Patel V; Araya R; Chowdhary N; King M; Kirkwood B; Nayak S; Simon G; Weiss HA",
     "year": "2007",
     "title": "Detecting common mental disorders in primary care in India: a comparison of five screening questionnaires",
     "journal": "Psychological Medicine",
     "doi": "10.1017/S0033291707002334",
     "url": "https://doi.org/10.1017/S0033291707002334"
    },
    {
     "key": "gill2007",
     "authors": "Gill SC; Butterworth P; Rodgers B; Mackinnon A",
     "year": "2007",
     "title": "Validity of the mental health component scale of the 12-item Short-Form Health Survey (MCS-12) as measure of common mental disorders in the general population",
     "journal": "Psychiatry Research",
     "doi": "10.1016/j.psychres.2006.11.005",
     "url": "https://doi.org/10.1016/j.psychres.2006.11.005"
    },
    {
     "key": "gao2011",
     "authors": "Gao W; Stark D; Bennett MI; Seymour J; Higginson IJ",
     "year": "2011",
     "title": "Using the 12-item General Health Questionnaire to screen psychological distress from survivorship to end-of-life care: dimensionality and item quality",
     "journal": "Psycho-Oncology",
     "doi": "10.1002/pon.1989",
     "url": "https://doi.org/10.1002/pon.1989"
    },
    {
     "key": "ohno2017",
     "authors": "Ohno S; Takahashi K; Inoue A; Takada K; Ishihara Y; Tanigawa M; Hirao K",
     "year": "2017",
     "title": "Smallest detectable change and test-retest reliability of a self-reported outcome measure: results of the CES-D, General Self-Efficacy Scale, and 12-item General Health Questionnaire",
     "journal": "Journal of Evaluation in Clinical Practice",
     "doi": "10.1111/jep.12795",
     "url": "https://doi.org/10.1111/jep.12795"
    },
    {
     "key": "plummer2000",
     "authors": "Plummer S; Gournay K; Goldberg D; Ritter S; Mann A; Blizard R",
     "year": "2000",
     "title": "Detection of psychological distress by practice nurses in general practice",
     "journal": "Psychological Medicine",
     "doi": "10.1017/s0033291799002597",
     "url": "https://doi.org/10.1017/s0033291799002597"
    },
    {
     "key": "gessler2008",
     "authors": "Gessler S; Low J; Daniells E; Williams R; Brough V; Tookman A; Jones L",
     "year": "2008",
     "title": "Screening for distress in cancer patients: is the distress thermometer a valid measure in the UK and does it measure change over time? A prospective validation study",
     "journal": "Psycho-Oncology",
     "doi": "10.1002/pon.1273",
     "url": "https://doi.org/10.1002/pon.1273"
    },
    {
     "key": "benoni2024",
     "authors": "Benoni R; Sartorello A; Mazzi M; Berti G; et al.",
     "year": "2024",
     "title": "The use of 12-item General Health Questionnaire (GHQ-12) in Ukrainian refugees: translation and validation study of the Ukrainian version",
     "journal": "Health and Quality of Life Outcomes",
     "doi": "10.1186/s12955-024-02226-1",
     "url": "https://doi.org/10.1186/s12955-024-02226-1"
    },
    {
     "key": "li2009",
     "authors": "Li WHC; Chung JOK; Chui MML; Chan PSL",
     "year": "2009",
     "title": "Factorial structure of the Chinese version of the 12-item General Health Questionnaire in adolescents",
     "journal": "Journal of Clinical Nursing",
     "doi": "10.1111/j.1365-2702.2009.02905.x",
     "url": "https://doi.org/10.1111/j.1365-2702.2009.02905.x"
    },
    {
     "key": "kufe2024",
     "authors": "Kufe CN; Bernstein K; Wilson KS",
     "year": "2024",
     "title": "Reliability, validity and dimensionality of the 12-Item General Health Questionnaire among South African healthcare workers",
     "journal": "African Journal of Psychological Assessment",
     "doi": "10.4102/ajopa.v6i0.144",
     "url": "https://doi.org/10.4102/ajopa.v6i0.144"
    }
   ],
   "record_notes": "Overall confidence: the GHQ-12 is among the best-evidenced distress screeners for internal structure (High, though contested historically) and criterion validity against diagnostic interviews (High), with high internal consistency (High). It is materially weaker on the properties this registry cares about for a UK working-adult audience: test-retest stability rests on a single small non-UK study (Low/thin), measurement invariance across occupation is untested, responsiveness and minimal important change are not formally established (Very low), and organisational criterion validity against work outcomes is Absent. All grades other than structural validity are downgraded to indirect because the evidence is overwhelmingly non-UK, non-workplace, or clinical/primary-care in origin, and the instrument's clinical-screener heritage and short-term state focus are flagged in the deployment caveat. Licence verified this session against GL Assessment and GL Education current pages and the Mapi Research Trust listing: it is a proprietary, paid, permission-required product, an important contrast to the free tools in the registry. Schema v0.2 made honest recording straightforward here; the main friction was that UK working-adult norms and the definitive user-guide psychometrics sit behind a paywalled commercial manual that could not be inspected this session, so those are flagged rather than asserted."
  },
  {
   "instrument_id": "copsoq-iii",
   "display_name": "COPSOQ III (Copenhagen Psychosocial Questionnaire, third version, core and dimension scales)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Copenhagen Psychosocial Questionnaire, version III (COPSOQ III)",
    "current_version": "COPSOQ III (international core, middle and long versions; national validated versions derived from these), published 2019 by the international COPSOQ network",
    "item_count": "Version-dependent. The international COPSOQ III is organised as a set of dimension scales in three tiers (core, middle, long). National validated versions vary: the German middle version has 84 items across 31 scales ([Lincke 2021](https://doi.org/10.1186/s12995-021-00331-1)); the Greek long version has 108 items across 40 scales ([Kotsakis 2025](https://doi.org/10.3390/healthcare13161980)). Most dimensions are short multi-item scales (typically two to four items); several constructs are measured with a single item.",
    "original_citation": "Burr H, Berthelsen H, Moncada S, Nübling M, et al. The Third Version of the Copenhagen Psychosocial Questionnaire. Safety and Health at Work 2019 ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002))",
    "steward_publisher": "COPSOQ International Network (coordinated internationally; national COPSOQ teams steward validated language versions, e.g. FFAW Freiburg for the German version)",
    "licence_status": "The COPSOQ questionnaire is released under Creative Commons CC BY-NC-ND 4.0 (Attribution, Non-Commercial, No-Derivatives), per the steward's current licence page. Users may copy and redistribute the material in any medium or format, must give proper recognition of the instrument and follow the network guidelines, and may not distribute modified material under the name 'COPSOQ' (no-derivatives). Two clarifications on the current page qualify the plain CC BY-NC-ND reading: (1) commercial use of the questionnaire is nonetheless permitted provided no fee is charged for use of the questionnaire itself, although fees for assessment, advice, analysis and training are allowed; and (2) before starting a new COPSOQ activity (translation, adaptation) the user is required to contact the existing national COPSOQ network member or validated-version steward for their country or language to avoid divergent versions. A specific carve-out in the network guidelines states that, in contrast to all other parts of COPSOQ III, commercial use of the Work Engagement items (WE_T, WE1, WE2, WE3) is only allowed under special agreement with Triple i (https://www.3ihc.nl, info@3ihc.nl), because those items derive from the Schaufeli et al. Utrecht Work Engagement short scale. No warranties are given.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "COPSOQ International Network, 'Licence, Guidelines & Questionnaire' page (https://www.copsoq-network.org/licence-guidelines-and-questionnaire), full page body fetched and read on 2026-07-12 (states CC BY-NC-ND 4.0, the no-fee commercial condition, and the contact-national-team requirement); and the network guidelines PDF linked from that page, 'Guidelines for the use of COPSOQ III' (https://www.copsoq-network.org/assets/Uploads/COPSOQ-network-guidelines-for-the-use-of-COPSOQ-III-290618sig.pdf), fetched and read on 2026-07-12 (contains the Work Engagement items Triple i / 3ihc.nl carve-out and the identical no-fee commercial wording). Not derived from founding papers or reviews."
   },
   "constructs_claimed": "Psychosocial working conditions (exposures), organised as many largely independent dimension scales grouped into domains: demands at work (quantitative, work pace, emotional, demands for hiding emotions); work organisation and job content (influence, possibilities for development, variation, meaning of work, commitment to the workplace); interpersonal relations and leadership (predictability, recognition, role clarity, role conflicts, quality of leadership, social support, sense of community); work-individual interface (job insecurity, insecurity over working conditions, job satisfaction, work-life/work-privacy conflict); social capital (vertical and horizontal trust, organisational justice); offensive behaviours (bullying, sexual harassment, threats, violence); plus outcome/strain scales (self-rated health, burnout, stress, sleeping troubles, and, in longer versions, work engagement). COPSOQ III is an exposure measure of the psychosocial work environment rather than a wellbeing outcome instrument, although it embeds a small number of health and strain outcome scales.",
   "deployment_context_caveat": "COPSOQ III is a psychosocial-exposure instrument, not a clinical or diagnostic tool, so the clinical-origin caveat does not apply. The material caveats for a UK working-adult audience are different. First, there is no UK-specific COPSOQ III validation among the studies retrieved this session; the direct psychometric evidence comes from Swedish, German, Norwegian, Portuguese, Turkish, Greek, Australian, Polish and Chinese samples, so all grades carry population indirectness for the UK, mitigated somewhat by the Australian English-language validation ([Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x)). Second, the instrument is a family of dimension scales rather than a single scale; psychometric properties are established per dimension and vary substantially across dimensions and across national versions, so a single record-level grade necessarily summarises a heterogeneous picture (see subgrades and record_notes). Third, several constructs are measured with single items, for which scale-level properties are category errors.",
   "structural_validity": {
    "findings": "COPSOQ III is built as a set of largely independent dimension scales rather than one higher-order factor, so structural validity is assessed dimension by dimension, and the network explicitly notes that factor structure for the newly introduced dimensions was still to be fully tested at launch ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002)). Confirmatory and exploratory factor analyses of national versions generally support the intended per-dimension structure. The Turkish COPSOQ-3 reported an excellent-fitting model (RMSEA 0.038, SRMR 0.053, CFI 0.98) with 19 extracted factors explaining 66.1% of variance ([Sahan 2018](https://doi.org/10.1080/19338244.2018.1538095)). The Australian long-version validation, using EFA followed by CFA, reported acceptable to good fit (a four-factor higher-order EFA solution, and CFA RMSEA around 0.034 to 0.036 with CFI/TLI in the 0.89 to 0.92 range depending on model) ([Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x)). For the predecessor COPSOQ-II, a set-ESEM approach that permitted only theory-consistent cross-loadings improved fit markedly over strict CFA (from CFI 0.907 to CFI 0.947 to 0.971) and reduced inflated inter-factor correlations, indicating that strict independent-clusters CFA understates fit for this multidimensional instrument ([Dicke 2018](https://doi.org/10.3389/fpsyg.2018.00584)). A full CFA of the COPSOQ-II 33-subscale model in a Polish sample also showed good fit (RMSEA below 0.05, SRMR below 0.08) ([Baka 2022](https://doi.org/10.1371/journal.pone.0262266)). The Norwegian COPSOQ III validation combined CFA with item response theory to characterise dimension structure and item information ([Ose 2023](https://doi.org/10.1371/journal.pone.0289739)). The main structural caveat is that fit is evaluated separately per dimension or per national selection of dimensions, not for a single global model, and that some short dimensions are less stable across countries.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect: evidence is from Turkish, Australian, Norwegian, Swedish and (for COPSOQ-II) Australian and Polish samples; no UK-specific structural validation was retrieved, and some of the strongest structural evidence is parent-form (COPSOQ-II).",
    "subgrades": [
     {
      "subgroup": "Established core dimensions (demands, influence, leadership, social support, social capital)",
      "grade": "Moderate",
      "note": "Consistently recovered as intended factors across national CFAs ([Sahan 2018](https://doi.org/10.1080/19338244.2018.1538095); [Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x); [Baka 2022](https://doi.org/10.1371/journal.pone.0262266))."
     },
     {
      "subgroup": "Newly introduced COPSOQ III dimensions (e.g. work engagement, quality of work, cyber bullying)",
      "grade": "Low",
      "note": "Network flagged their factor structure as not yet fully tested at launch ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002)); evidence still accumulating."
     }
    ]
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent and discriminant validity are supported mainly through scale intercorrelations and comparison with other established work-stress models. In the Gutenberg Health Study, COPSOQ scales and the Effort-Reward Imbalance (ERI) questionnaire showed congruent patterns across occupational groups, and COPSOQ predictor scales explained comparable or slightly greater variance than ERI in shared outcome scales (for example job satisfaction R-squared 0.51 for COPSOQ versus 0.46 for ERI; burnout 0.35 versus 0.26), supporting convergent validity against an established alternative model ([Nuebling 2013](https://doi.org/10.1186/1471-2458-13-538)). Within-instrument, developers calculate scale intercorrelations to demonstrate that dimensions are distinct (divergent) while theoretically related scales correlate (convergent) ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002); [Berthelsen 2020](https://doi.org/10.3390/ijerph17093179)). The set-ESEM analysis of COPSOQ-II is directly relevant to discriminant validity: strict CFA produced some inflated inter-factor correlations that fell substantially once theory-consistent cross-loadings were permitted, indicating the dimensions are discriminable but not perfectly orthogonal ([Dicke 2018](https://doi.org/10.3389/fpsyg.2018.00584)). A dedicated study constructed and validated a global Workplace Social Capital scale from COPSOQ III justice and trust items, supporting convergent structure among the social-capital dimensions ([Berthelsen 2019](https://doi.org/10.1371/journal.pone.0221893)). Formal multitrait-multimethod or correlations with independent external gold-standard constructs are sparser than the within-instrument evidence.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect: convergent evidence against ERI is from a German general-population employed sample ([Nuebling 2013](https://doi.org/10.1186/1471-2458-13-538)); the strongest discriminant analysis is parent-form (COPSOQ-II) ([Dicke 2018](https://doi.org/10.3389/fpsyg.2018.00584)); no UK sample."
   },
   "criterion_validity_reference_standard": {
    "findings": "COPSOQ III is an exposure measure of psychosocial working conditions, not a screener for a health condition, so validity against a diagnostic or clinical reference standard is not a design goal and is largely not evaluated. No study retrieved this session compared COPSOQ III dimension scores against a diagnostic reference standard (for example a structured clinical interview for depression) with sensitivity or specificity. The embedded strain scales (self-rated health, burnout, stress) are self-report and are treated as outcomes correlated with exposures rather than validated against a clinical criterion. This is an appropriate absence for an exposure instrument, but it means the field is genuinely untested rather than merely weak.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct: the category (diagnostic reference-standard validity) does not apply to a psychosocial-exposure measure; recorded as Not-applicable rather than Absent because it is a category mismatch, and no reference-standard study was located."
   },
   "criterion_validity_organisational": {
    "findings": "Evidence that COPSOQ III scores predict objective organisational or health outcomes (register-based sickness absence, staff turnover, diagnosed conditions, performance) is a recognised gap that the developers themselves flag as future work. The Swedish validation explicitly states that evaluating predictive criterion validity against register data on absence, staff turnover and performance remains to be done in longitudinal multilevel designs ([Berthelsen 2020](https://doi.org/10.3390/ijerph17093179)), and the Norwegian study likewise lists predictive validity as not yet established for their version ([Ose 2023](https://doi.org/10.1371/journal.pone.0289739)). The available criterion-type evidence is cross-sectional and against self-reported outcomes: in the Gutenberg Health Study COPSOQ psychosocial scales explained meaningful variance in self-reported job satisfaction (R-squared 0.51), burnout (0.35), satisfaction with life (0.18) and general health (0.11), with theoretically sensible predictors (for example meaning of work and sense of community for job satisfaction; work-privacy conflict for burnout) ([Nuebling 2013](https://doi.org/10.1186/1471-2458-13-538)). This supports concurrent association with self-reported strain outcomes but not prediction of objective work outcomes. Genuine organisational criterion validity against registers or hard outcomes is essentially Absent in the retrieved literature.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "mixed",
    "indirectness": "indirect: the only quantified associations are cross-sectional and against self-reported outcomes in a German general-population sample ([Nuebling 2013](https://doi.org/10.1186/1471-2458-13-538)); objective-outcome prediction is unstudied and no UK data exist."
   },
   "internal_consistency": {
    "findings": "Internal consistency is the most extensively documented property and is generally acceptable to good for the multi-item dimensions, with well-characterised weak spots at the short two-item and emotion-related scales. Across the seven-country international middle version, most of the 23 tested scales reached Cronbach alpha above 0.70; three fell below: Commitment to the Workplace (two items, mean alpha 0.64, 95% CI 0.61 to 0.67), Demands for Hiding Emotions (three items, mean alpha 0.66, 0.58 to 0.73) and Control over Working Time, with additional country-specific shortfalls for Predictability, Meaning of Work and Job Insecurity (all two-item scales, alpha around 0.62 to 0.66 in France and Turkey) ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002)). The German middle version reported that 20 of 25 multi-item scales exceeded alpha 0.70 and 13 reached 0.80 or higher; a small number of two-item scales (for example Degrees of Freedom, reduced to two items, alpha 0.53) were weak ([Lincke 2021](https://doi.org/10.1186/s12995-021-00331-1)). The Australian long version found all 31 three-or-more-item scales acceptable except Demands for Hiding Emotions (0.66), and among two-item scales used Spearman-Brown coefficients, with Variation of Work unacceptably low (0.24) while Meaning of Work was acceptable (0.78) ([Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x)). The Turkish version reported 23 dimensions above 0.70 with Control over Working Time (0.54) and Predictability (0.66) below ([Sahan 2018](https://doi.org/10.1080/19338244.2018.1538095)). The Greek long version found 22 of 40 scales with alpha above 0.70 ([Kotsakis 2025](https://doi.org/10.3390/healthcare13161980)), and the Chinese long version reported an overall alpha of 0.92 with per-dimension values from 0.60 to 0.92 ([Huang 2025](https://doi.org/10.3390/healthcare13070825)). The consistent pattern is that longer scales are reliable and the recurring weak scales are the two-item and hiding-emotions dimensions, which is a structural feature of the compact design rather than a country-specific defect.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect for the UK specifically (evidence from Sweden, Germany, Australia, Turkey, Greece, China and the seven-country international sample; no UK sample), but the property is directly and repeatedly measured on the fielded COPSOQ III versions.",
    "subgrades": [
     {
      "subgroup": "Multi-item dimensions (three or more items)",
      "grade": "High",
      "note": "Alpha consistently above 0.70, frequently above 0.80 ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002); [Lincke 2021](https://doi.org/10.1186/s12995-021-00331-1); [Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x))."
     },
     {
      "subgroup": "Two-item and hiding-emotions scales",
      "grade": "Low",
      "note": "Recurring alpha/Spearman-Brown shortfalls (e.g. Commitment 0.64, Hiding Emotions 0.66, Variation of Work 0.24 in Australia) ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002); [Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x))."
     }
    ]
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "0.70 to 0.89 (all scales but one adequate/good; mutual-trust scale 0.64)",
      "coefficient_type": "ICC",
      "interval": "median 22 days (range 6 to 65 days)",
      "sample_n": "349 respondents (283 employees)",
      "population": "Danish general working population (COPSOQ, parent version)",
      "evidence_form": "parent",
      "citation_key": "Thorsen2010"
     },
     {
      "coefficient": "0.15 to 0.34 (Pearson correlations across 33 subscales; all significant at p<0.001 but low, attributed to the long interval)",
      "coefficient_type": "r",
      "interval": "about 12 months",
      "sample_n": "599 human-service employees",
      "population": "Polish human-service staff (COPSOQ II, parent version)",
      "evidence_form": "parent",
      "citation_key": "Baka2022"
     }
    ],
    "grade": "Low",
    "status": "thin",
    "indirectness": "indirect: no test-retest study of the fielded COPSOQ III was located; the adequate ICC evidence is from the Danish parent COPSOQ over about three weeks ([Thorsen 2010](https://doi.org/10.1177/1403494809349859)), and the only longitudinal COPSOQ-II retest used a 12-month interval that conflates true change with instability ([Baka 2022](https://doi.org/10.1371/journal.pone.0262266)).",
    "summary": "Test-retest reliability for COPSOQ III specifically is essentially Absent and is repeatedly named as outstanding future work. The COPSOQ III network states that test-retest of the newly introduced dimensions was still to come ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002)), and the Norwegian, Greek and Chinese validations all explicitly did not assess test-retest ([Ose 2023](https://doi.org/10.1371/journal.pone.0289739); [Kotsakis 2025](https://doi.org/10.3390/healthcare13161980); [Huang 2025](https://doi.org/10.3390/healthcare13070825)); the German validation deliberately excluded a formal test-retest for practical reasons ([Lincke 2021](https://doi.org/10.1186/s12995-021-00331-1)). The two structured findings recorded here are both parent-form: Thorsen and Bjorner's dedicated Danish test-retest study of the original COPSOQ found ICCs of 0.70 to 0.89 (one scale, mutual trust, 0.64) over a median 22-day interval, which is good ([Thorsen 2010](https://doi.org/10.1177/1403494809349859)); the Polish COPSOQ-II longitudinal study reported low retest correlations (r 0.15 to 0.34) but over a 12-month interval that the authors note is far longer than the recommended few-weeks-to-months window, so it indexes real exposure change as much as instrument instability ([Baka 2022](https://doi.org/10.1371/journal.pone.0262266)). Direct short-interval COPSOQ III test-retest evidence has not been established."
   },
   "measurement_invariance": {
    "findings": "Formal measurement invariance testing (configural, metric, scalar) for COPSOQ III is thin, and the international network was candid that it could not directly test differential item functioning across countries at launch because of data-protection constraints on pooling item-level data, while observing that scale properties differed somewhat across working populations ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002)). The clearest invariance-relevant evidence is parent-form: the COPSOQ-II set-ESEM study of Australian school principals tested longitudinal factor structure across two time points and reported good, stable fit (for example CFI 0.95, TLI 0.94, RMSEA 0.02 at Time 1), supporting configural stability over time ([Dicke 2018](https://doi.org/10.3389/fpsyg.2018.00584)). Comparability across the many national COPSOQ III versions is handled largely by design (a common obligatory core, standardised translation procedures) and demonstrated indirectly through similar factor solutions and reliability patterns across countries ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002); [Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x); [Kotsakis 2025](https://doi.org/10.3390/healthcare13161980)) rather than by pooled multi-group invariance models. Group-level aggregation properties (ICC(1)/ICC(2) across occupations and workplaces) were examined in Sweden to justify comparing group mean scores, which is a related but distinct property from measurement invariance ([Berthelsen 2020](https://doi.org/10.3390/ijerph17093179)). Full multi-group scalar invariance across sex, age, occupation and language for the fielded COPSOQ III has not been established in the retrieved literature.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "mixed",
    "indirectness": "indirect: the only explicit longitudinal invariance evidence is parent-form (COPSOQ-II, Australian principals) ([Dicke 2018](https://doi.org/10.3389/fpsyg.2018.00584)); cross-country comparability for COPSOQ III is argued by design and similarity of solutions, not by pooled invariance testing; no UK data.",
    "subgrades": [
     {
      "subgroup": "Longitudinal (over time)",
      "grade": "Low",
      "note": "Configural stability supported for parent COPSOQ-II via set-ESEM ([Dicke 2018](https://doi.org/10.3389/fpsyg.2018.00584))."
     },
     {
      "subgroup": "Across countries/languages",
      "grade": "Very low",
      "note": "Not formally tested at item level; DIF analysis was precluded by data-protection constraints ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002))."
     },
     {
      "subgroup": "Across sex/age/occupation",
      "grade": "Absent",
      "note": "No formal multi-group invariance study located for COPSOQ III."
     }
    ]
   },
   "responsiveness_mic": {
    "findings": "Responsiveness (sensitivity to change over time or after intervention) and a minimal important change threshold have not been established for COPSOQ III in the retrieved literature. The Swedish validation lists responsiveness alongside test-retest and predictive validity as properties still to be evaluated in future longitudinal designs ([Berthelsen 2020](https://doi.org/10.3390/ijerph17093179)). As an interpretation aid rather than a formal responsiveness statistic, the Swedish work proposes that a 5 to 10 point difference on the 0 to 100 scale metric can be treated as a minimum important difference, and reports Cohen's d effect sizes for group contrasts ([Berthelsen 2020](https://doi.org/10.3390/ijerph17093179)), but this is a benchmarking convention, not an anchor-based or distribution-based minimal-important-change study. No study retrieved this session estimated responsiveness against an external change anchor.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect: the only relevant material is a proposed 5 to 10 point minimum important difference convention from the Swedish benchmarking work ([Berthelsen 2020](https://doi.org/10.3390/ijerph17093179)); no formal responsiveness study and no UK data."
   },
   "populations_languages_norms": {
    "findings": "COPSOQ III has been validated and normed in a wide range of national and language versions, and benchmark/reference values are a core feature of the system. Retrieved validations include Swedish (national benchmarks) ([Berthelsen 2020](https://doi.org/10.3390/ijerph17093179)), German (database exceeding 250,000 participants) ([Lincke 2021](https://doi.org/10.1186/s12995-021-00331-1)), Norwegian (registered nurses) ([Ose 2023](https://doi.org/10.1371/journal.pone.0289739)), Portuguese (municipal and healthcare workers; and a 2026 national validation) ([Cotrim 2022](https://doi.org/10.3390/ijerph19031167); [Cotrim 2026](https://doi.org/10.1371/journal.pgph.0006036)), Turkish ([Sahan 2018](https://doi.org/10.1080/19338244.2018.1538095)), Greek ([Kotsakis 2025](https://doi.org/10.3390/healthcare13161980)), Australian (national benchmarks by ANZSCO group) ([Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x)), Chinese ([Huang 2025](https://doi.org/10.3390/healthcare13070825)) and Czech ([Zabrodska 2026](https://doi.org/10.1186/s40359-026-03961-4)) versions, building on the earlier German COPSOQ database tradition ([Nuebling 2010](https://doi.org/10.1177/1403494809353652)). Occupation-specific and country-specific reference values exist for several versions. Critically for this audience, no UK-specific COPSOQ III validation or UK norm set was located this session; the nearest English-language reference data are Australian ([Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x)).",
    "grade": "High",
    "indirectness": "indirect for the UK: extensive multi-country norms exist, but there is no UK validation or UK benchmark among the retrieved studies; the closest English-language norms are Australian."
   },
   "criticisms_controversies": "Several recurring critiques emerge from the retrieved literature. First, COPSOQ III is a modular family of dimension scales rather than a single validated scale, so psychometric quality is heterogeneous across dimensions; the compact two-item and hiding-emotions scales repeatedly show sub-0.70 internal consistency ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002); [Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x)), and reliability of two-item scales must be read via Spearman-Brown rather than alpha. Second, key longitudinal properties are missing: test-retest reliability, responsiveness and predictive criterion validity against objective outcomes are explicitly named as not yet done by the developers themselves ([Berthelsen 2020](https://doi.org/10.3390/ijerph17093179); [Ose 2023](https://doi.org/10.1371/journal.pone.0289739)), so the instrument's status as an exposure measure rests largely on cross-sectional and construct-validity evidence. Third, measurement invariance across countries could not be tested at item level at launch owing to data-protection constraints on pooling data ([Burr 2019](https://doi.org/10.1016/j.shaw.2019.10.002)), leaving cross-national comparability argued by design rather than demonstrated. Fourth, strict independent-clusters CFA tends to understate fit and inflate factor correlations for this instrument, which has prompted use of ESEM/set-ESEM approaches ([Dicke 2018](https://doi.org/10.3389/fpsyg.2018.00584)). Fifth, for a UK working-adult audience the evidence base is population-indirect: there is no retrieved UK validation, and much of the strongest reliability and invariance evidence is parent-form (COPSOQ or COPSOQ-II) rather than the fielded COPSOQ III.",
   "item_records": [],
   "citations": [
    {
     "key": "Burr2019",
     "authors": "Burr H, Berthelsen H, Moncada S, Nübling M, Dupret E, Demiral Y, Oudyk J, Kristensen TS, Llorens C, Navarro A, Lincke HJ, Bocéréan C, Sahan C, Smith P, Pohrt A, International COPSOQ Network",
     "year": "2019",
     "title": "The Third Version of the Copenhagen Psychosocial Questionnaire",
     "journal": "Safety and Health at Work",
     "doi": "10.1016/j.shaw.2019.10.002",
     "url": "https://doi.org/10.1016/j.shaw.2019.10.002"
    },
    {
     "key": "Berthelsen2020",
     "authors": "Berthelsen H, Westerlund H, Bergström G, Burr H",
     "year": "2020",
     "title": "Validation of the Copenhagen Psychosocial Questionnaire Version III and Establishment of Benchmarks for Psychosocial Risk Management in Sweden",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph17093179",
     "url": "https://doi.org/10.3390/ijerph17093179"
    },
    {
     "key": "Lincke2021",
     "authors": "Lincke HJ, Vomstein M, Lindner A, Nolle I, Häberle N, Haug A, Nübling M",
     "year": "2021",
     "title": "COPSOQ III in Germany: validation of a standard instrument to measure psychosocial factors at work",
     "journal": "Journal of Occupational Medicine and Toxicology",
     "doi": "10.1186/s12995-021-00331-1",
     "url": "https://doi.org/10.1186/s12995-021-00331-1"
    },
    {
     "key": "Ose2023",
     "authors": "Ose SO, Lohmann-Lafrenz S, Bernström VH, Berthelsen H, Marchand GH",
     "year": "2023",
     "title": "The Norwegian version of the Copenhagen Psychosocial Questionnaire (COPSOQ III): Initial validation study using a national sample of registered nurses",
     "journal": "PLOS ONE",
     "doi": "10.1371/journal.pone.0289739",
     "url": "https://doi.org/10.1371/journal.pone.0289739"
    },
    {
     "key": "Cotrim2022",
     "authors": "Cotrim TP, Bem-Haja P, Pereira A, Fernandes C, Azevedo R, Fonte C, Nossa P, Silva CF, Barbosa-Ferreira J, Silvério J",
     "year": "2022",
     "title": "The Portuguese Third Version of the Copenhagen Psychosocial Questionnaire: Preliminary Validation Studies of the Middle Version among Municipal and Healthcare Workers",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph19031167",
     "url": "https://doi.org/10.3390/ijerph19031167"
    },
    {
     "key": "Rahimi2025",
     "authors": "Rahimi E, Arnold KA, LaMontagne AD, et al.",
     "year": "2025",
     "title": "Validation and benchmarks for the Copenhagen Psychosocial Questionnaire (COPSOQ III) in an Australian working population sample",
     "journal": "BMC Public Health",
     "doi": "10.1186/s12889-025-21845-x",
     "url": "https://doi.org/10.1186/s12889-025-21845-x"
    },
    {
     "key": "Kotsakis2025",
     "authors": "Kotsakis R, Avraam E, Malliarou M, et al.",
     "year": "2025",
     "title": "A Validation Study of the COPSOQ III Greek Questionnaire for Assessing Psychosocial Factors in the Workplace",
     "journal": "Healthcare (Basel)",
     "doi": "10.3390/healthcare13161980",
     "url": "https://doi.org/10.3390/healthcare13161980"
    },
    {
     "key": "Sahan2018",
     "authors": "Şahan C, Baydur H, Demiral Y",
     "year": "2018",
     "title": "A novel version of Copenhagen Psychosocial Questionnaire-3: Turkish validation study",
     "journal": "Archives of Environmental & Occupational Health",
     "doi": "10.1080/19338244.2018.1538095",
     "url": "https://doi.org/10.1080/19338244.2018.1538095"
    },
    {
     "key": "Huang2025",
     "authors": "Huang Y, Zhang Y, Wang X, et al.",
     "year": "2025",
     "title": "COPSOQ III in China: Preliminary Validation of an International Instrument to Measure Psychosocial Work Factors",
     "journal": "Healthcare (Basel)",
     "doi": "10.3390/healthcare13070825",
     "url": "https://doi.org/10.3390/healthcare13070825"
    },
    {
     "key": "Thorsen2010",
     "authors": "Thorsen SV, Bjorner JB",
     "year": "2010",
     "title": "Reliability of the Copenhagen Psychosocial Questionnaire",
     "journal": "Scandinavian Journal of Public Health",
     "doi": "10.1177/1403494809349859",
     "url": "https://doi.org/10.1177/1403494809349859"
    },
    {
     "key": "Dicke2018",
     "authors": "Dicke T, Marsh HW, Riley P, Parker PD, Guo J, Horwood M",
     "year": "2018",
     "title": "Validating the Copenhagen Psychosocial Questionnaire (COPSOQ-II) Using Set-ESEM: Identifying Psychosocial Risk Factors in a Sample of School Principals",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2018.00584",
     "url": "https://doi.org/10.3389/fpsyg.2018.00584"
    },
    {
     "key": "Nuebling2013",
     "authors": "Nübling M, Seidler A, Garthus-Niegel S, Latza U, Wagner M, Hegewald J, Liebers F, Jankowiak S, Zwiener I, Wild PS, Letzel S",
     "year": "2013",
     "title": "The Gutenberg Health Study: measuring psychosocial factors at work and predicting health and work-related outcomes with the ERI and the COPSOQ questionnaire",
     "journal": "BMC Public Health",
     "doi": "10.1186/1471-2458-13-538",
     "url": "https://doi.org/10.1186/1471-2458-13-538"
    },
    {
     "key": "Baka2022",
     "authors": "Baka Ł, Prusik M, Pejtersen JH",
     "year": "2022",
     "title": "Full evaluation of the psychometric properties of COPSOQ II. One-year longitudinal study on Polish human service staff",
     "journal": "PLOS ONE",
     "doi": "10.1371/journal.pone.0262266",
     "url": "https://doi.org/10.1371/journal.pone.0262266"
    },
    {
     "key": "Berthelsen2019",
     "authors": "Berthelsen H, Westerlund H, Pejtersen JH, Hadzibajramovic E",
     "year": "2019",
     "title": "Construct validity of a global scale for Workplace Social Capital based on COPSOQ III",
     "journal": "PLOS ONE",
     "doi": "10.1371/journal.pone.0221893",
     "url": "https://doi.org/10.1371/journal.pone.0221893"
    },
    {
     "key": "Nuebling2010",
     "authors": "Nübling M, Hasselhorn HM",
     "year": "2010",
     "title": "The Copenhagen Psychosocial Questionnaire in Germany: from the validation of the instrument to a national survey",
     "journal": "Scandinavian Journal of Public Health",
     "doi": "10.1177/1403494809353652",
     "url": "https://doi.org/10.1177/1403494809353652"
    },
    {
     "key": "Kristensen2005",
     "authors": "Kristensen TS, Hannerz H, Høgh A, Borg V",
     "year": "2005",
     "title": "The Copenhagen Psychosocial Questionnaire, a tool for the assessment and improvement of the psychosocial work environment",
     "journal": "Scandinavian Journal of Work, Environment & Health",
     "doi": "10.5271/sjweh.948",
     "url": "https://doi.org/10.5271/sjweh.948"
    },
    {
     "key": "Cotrim2026",
     "authors": "Cotrim TP, Bem-Haja P, Vagos P, et al.",
     "year": "2026",
     "title": "Validation of the third version of the Copenhagen Psychosocial Questionnaire for Portugal",
     "journal": "PLOS Global Public Health",
     "doi": "10.1371/journal.pgph.0006036",
     "url": "https://doi.org/10.1371/journal.pgph.0006036"
    },
    {
     "key": "Zabrodska2026",
     "authors": "Zábrodská K, Květon P, Jelínek M, et al.",
     "year": "2026",
     "title": "Psychometric validation of the Czech Copenhagen Psychosocial Questionnaire (COPSOQ III)",
     "journal": "BMC Psychology",
     "doi": "10.1186/s40359-026-03961-4",
     "url": "https://doi.org/10.1186/s40359-026-03961-4"
    }
   ],
   "record_notes": "Overall confidence: Moderate for structural and convergent/discriminant validity, High for internal consistency and for the breadth of populations/norms, but Very low to Absent for test-retest, measurement invariance, responsiveness/MIC and organisational criterion validity. COPSOQ III should be recorded as a well-established psychosocial-exposure instrument whose cross-sectional and internal-consistency credentials are strong but whose longitudinal and predictive properties remain genuinely under-evidenced, a gap the developers themselves acknowledge. The record is deliberately framed like an exposure measure (comparable to the HSE Management Standards Indicator Tool) rather than a wellbeing outcome. Two honesty points the v0.2 schema still made awkward. (1) The instrument is a family of dimension scales, so most single record-level grades summarise a heterogeneous per-dimension reality; subgrades capture the most important splits (multi-item versus two-item scales; established versus newly introduced dimensions), but a fully faithful record would grade each dimension separately, which the single-record structure cannot hold. (2) Much of the reliability and invariance evidence is parent-form (original COPSOQ or COPSOQ-II); evidence_form is tagged 'mixed' or 'parent' throughout so this is not read as canonical COPSOQ III evidence, and the recurring theme is that COPSOQ III inherits confidence from its lineage while its own longitudinal evidence base is still thin. Licence: the questionnaire is released under Creative Commons CC BY-NC-ND 4.0, verified 2026-07-12 by fetching and reading the full text of the COPSOQ International Network's current 'Licence, Guidelines & Questionnaire' page and the guidelines PDF linked from it, not from founding papers. The lead's 'free for non-commercial with registration' pointer was an oversimplification in a different direction than a simple registration model: the licence is CC BY-NC-ND (non-commercial, no-derivatives) but the network's own page explicitly permits commercial use provided no fee is charged for the questionnaire itself (fees for assessment, advice, analysis and training are allowed), requires prior contact with the national COPSOQ team before a new translation or adaptation, and prohibits distributing modified material as 'COPSOQ'. The guidelines PDF adds a carve-out that the Work Engagement items (WE_T, WE1 to WE3) may be used commercially only under separate agreement with Triple i (3ihc.nl), because they derive from the Schaufeli et al. Utrecht Work Engagement short scale. Population indirectness: no UK COPSOQ III validation was located; the closest English-language validation and norms are Australian ([Rahimi 2025](https://doi.org/10.1186/s12889-025-21845-x)). All grades are flagged indirect accordingly."
  },
  {
   "instrument_id": "mbi",
   "display_name": "Maslach Burnout Inventory (MBI, workplace forms)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Maslach Burnout Inventory (MBI): Human Services Survey (MBI-HSS, 22 items), Human Services Survey for Medical Personnel (MBI-HSS-MP, 22 items), General Survey (MBI-GS, 16 items), Educators Survey (MBI-ES, 22 items)",
    "current_version": "MBI Manual 4th edition forms (MBI-HSS, MBI-HSS-MP, MBI-GS, MBI-ES); MBI-GS9 is a 9-item derivative",
    "item_count": "22 (HSS/HSS-MP/ES); 16 (GS); three subscales: emotional exhaustion, depersonalisation/cynicism, personal accomplishment/professional efficacy",
    "original_citation": "Maslach C, Jackson SE (1981). The measurement of experienced burnout. J Organ Behav 2(2):99-113. doi:10.1002/job.4030020205",
    "steward_publisher": "Mind Garden, Inc. (exclusive publisher/distributor)",
    "licence_status": "Commercial, paid. Mind Garden sells a per-administration Licence to Administer and remnant-controlled reproduction; the instrument is copyrighted and is not free for research or organisational use. Verified against Mind Garden's current MBI product and 'License to Administer' pages (mindgarden.com/117-maslach-burnout-inventory-mbi and /765-mbi-license-to-administer.html), which price and gate each administration.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "Mind Garden product pages: https://www.mindgarden.com/117-maslach-burnout-inventory-mbi and https://www.mindgarden.com/maslach-burnout-inventory-mbi/765-mbi-license-to-administer.html"
   },
   "constructs_claimed": "Occupational burnout modelled as three separate dimensions: emotional exhaustion, depersonalisation (cynicism in the GS), and reduced personal accomplishment (professional efficacy in the GS). The three subscales are scored and interpreted separately; the manual does not endorse a single total burnout score.",
   "deployment_context_caveat": "Burnout is a research construct with no diagnostic gold standard; the MBI authors state there are no validated cut-off scores for classifying an individual as 'burnt out', so the instrument is a dimensional research and surveillance measure, not a diagnostic screener. Its use to label individual UK employees as clinically burnt out is not supported by the evidence and should be flagged wherever individual-level classification is proposed.",
   "structural_validity": {
    "findings": "A meta-analytic review of 45 exploratory and confirmatory factor studies supported the three-factor structure ([Worley J. 2008](https://doi.org/10.1177/0013164408315268)), and a large eight-country secondary analysis of 54,738 hospital nurses reproduced three factors, though two items (6 and 16, on 'stress' and 'strain' of working with people) consistently cross-loaded onto depersonalisation rather than emotional exhaustion ([Poghosyan L. 2009](https://doi.org/10.1016/j.ijnurstu.2009.03.004)). Factorial invariance work on 2,923 teachers likewise argued for broad three-factor validity while flagging six of the 22 items for content re-examination ([Byrne B. 1994](https://doi.org/10.1207/s15327906mbr2903_5)). The three-factor model has fitted well in newer confirmatory studies (for example the MBI-GS in 978 Colombian workers, RMSEA 0.05, CFI 0.99; [Bravo D. 2021](https://doi.org/10.3390/ijerph18105118)), but a meta-analytic structural equation modelling reanalysis found the data best fitted a bifactor model dominated by a general factor, questioning the independence of the three scales ([Aguayo-Estremera R. 2024](https://doi.org/10.3389/fpsyg.2024.1383619)). The separability of the personal accomplishment/professional efficacy subscale remains the principal structural dispute, and a Vietnamese study reassessed competing 20- and 22-item three-factor models on 1,162 healthcare workers ([Bui T. 2022](https://doi.org/10.1080/21642850.2021.2019585)).",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect; most confirmatory evidence is from healthcare, nursing and teaching samples outside the UK, with UK nurses included only within multi-country pooled datasets",
    "subgrades": [
     {
      "subgroup": "Emotional exhaustion / depersonalisation",
      "grade": "Moderate",
      "note": "factor recovery consistent across studies and forms"
     },
     {
      "subgroup": "Personal accomplishment / professional efficacy",
      "grade": "Low",
      "note": "separability contested; bifactor reanalysis suggests a dominant general factor"
     }
    ]
   },
   "convergent_discriminant_validity": {
    "findings": "MBI-GS subscales correlate in the expected directions with job satisfaction, work engagement and psychological distress in large workplace samples ([Bravo D. 2021](https://doi.org/10.3390/ijerph18105118)). Emotional exhaustion is the dimension most strongly tied to health and work-strain criteria, whereas personal accomplishment correlates weakly and sometimes inconsistently with the other two subscales, part of the same evidence that drives the structural debate ([Aguayo-Estremera R. 2024](https://doi.org/10.3389/fpsyg.2024.1383619)). Convergent evidence against alternative burnout instruments is documented in the comparative literature (for example moderate-to-high correlations with the Oldenburg Burnout Inventory; [Demerouti E. 2003](https://doi.org/10.1027//1015-5759.19.1.12)).",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect; predominantly non-UK occupational and healthcare samples"
   },
   "criterion_validity_reference_standard": {
    "findings": "There is no diagnostic reference standard for burnout, so the MBI cannot be validated against a gold-standard diagnosis; the manual provides no clinical cut-off, and a long-standing cautionary analysis showed the proposed cut-off points lacked cross-national and clinical validity ([Schaufeli W. 1995](https://doi.org/10.2466/pr0.1995.76.3c.1083)). A recent critique argues more strongly that the MBI does not measure the construct it claims to and that burnout itself is poorly defined ([Bianchi R. 2024](https://doi.org/10.3233/wor-240095)). A COSMIN-based systematic review of occupational burnout measures graded the overall evidence and highlighted persistent gaps in criterion validity against health standards ([Shoman Y. 2021](https://doi.org/10.1017/s2045796020001134)).",
    "grade": "Very low",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "indirect; the absence of a diagnostic standard is generic, and the critical literature draws on mixed occupational samples"
   },
   "criterion_validity_organisational": {
    "findings": "A two-year prospective cohort of 509 teachers found high depersonalisation predicted long-term (>=30 day) sickness absence after full adjustment (relative risk 1.80, 95% CI 1.05 to 3.09), while emotional exhaustion lost significance after adjustment and low personal accomplishment was unrelated ([Salvagioni D. 2022](https://doi.org/10.1016/j.shaw.2022.01.006)). Burnout measured by the MBI is associated with turnover intention in nurses ([Zheng J. 2024](https://doi.org/10.1186/s12912-024-02624-2)) and with intention to leave the profession in a national hospital-nurse study ([Bruyneel A. 2023](https://doi.org/10.1016/j.ijnurstu.2022.104385)), and a systematic review links nurse burnout to patient safety and quality-of-care outcomes ([Li L. 2024](https://doi.org/10.1001/jamanetworkopen.2024.43059)). Associations are consistent in direction but vary by subscale and are mostly self-reported or intention-based rather than registered absence or turnover.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect; nursing and teaching cohorts outside the UK, outcomes often intention-to-leave rather than recorded organisational data"
   },
   "internal_consistency": {
    "findings": "A reliability generalisation meta-analysis of 84 studies found mean subscale alphas generally between 0.70 and 0.80, with personal accomplishment and depersonalisation frequently below levels recommended for high-stakes individual decisions ([Wheeler D. 2011](https://doi.org/10.1177/0013164410391579)); an updated meta-analysis reported average alphas of 0.71 to 0.88 across dimensions ([Aguayo-Estremera R. 2024](https://doi.org/10.3389/fpsyg.2024.1383619)). Individual studies commonly report emotional exhaustion as the most reliable subscale (for example MBI-GS alphas 0.72 to 0.86 in Colombian workers, [Bravo D. 2021](https://doi.org/10.3390/ijerph18105118); MBI-GS9 alpha and omega 0.84 to 0.91, a derivative short form, [Wang A. 2024](https://doi.org/10.3389/fpsyg.2024.1439470)). Depersonalisation and personal accomplishment are the weakest-performing subscales for internal consistency.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect; pooled across many non-UK occupational samples",
    "subgrades": [
     {
      "subgroup": "Emotional exhaustion",
      "grade": "High",
      "note": "alpha typically 0.80 to 0.90"
     },
     {
      "subgroup": "Depersonalisation & personal accomplishment",
      "grade": "Moderate",
      "note": "means often 0.70 or below; unreliable for individual classification"
     }
    ]
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "0.70 to 0.92",
      "coefficient_type": "ICC",
      "interval": "retest interval not specified in the retrieved full text",
      "sample_n": "306 (item-level ICCs; Iranian physicians and nurses)",
      "population": "Iranian hospital physicians and nurses (Persian MBI-HSS-MP)",
      "evidence_form": "canonical",
      "citation_key": "persian_mbi2022"
     }
    ],
    "grade": "Low",
    "status": "thin",
    "indirectness": "indirect; the one retrievable coefficient set is item-level ICCs from an Iranian medical-personnel sample, not a UK working-adult subscale-level retest, and the retest interval was not stated in the retrieved text",
    "summary": "Test-retest reliability is under-reported for the fielded MBI. The MBI manual (Mind Garden, behind the commercial licence and not retrievable this session) reports stability from one month to one year, as cited secondarily by the Persian MBI-HSS-MP study. The only numeric coefficients extractable this session were item-level ICCs of 0.70 to 0.92 from that Persian medical-personnel validation; subscale-level retest coefficients in UK working adults were not located. This is recorded as thin rather than well-established."
   },
   "measurement_invariance": {
    "findings": "An item-response-theory differential item functioning analysis of 6,577 US physicians detected statistically significant DIF across age, gender and specialty in nearly every MBI item, but the impact on subscale scores was small (mean differences <0.10 SD), so measured group differences are not largely artefacts of item bias ([Brady K. 2021](https://doi.org/10.1186/s41687-021-00312-2)). Full metric and scalar invariance of the MBI-GS was supported across gender, age and socioeconomic status in Colombian workers ([Bravo D. 2021](https://doi.org/10.3390/ijerph18105118)), and cross-gender factorial invariance was broadly supported in teachers with some items flagged ([Byrne B. 1994](https://doi.org/10.1207/s15327906mbr2903_5)). Invariance across occupations and languages is less thoroughly established than across sex and age.",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect; US physician, Colombian worker and teacher samples, no dedicated UK invariance study located",
    "subgrades": [
     {
      "subgroup": "Across sex and age",
      "grade": "Moderate",
      "note": "metric and scalar invariance supported in several samples"
     },
     {
      "subgroup": "Across occupation / specialty",
      "grade": "Low",
      "note": "significant item-level DIF detected, though small in aggregate impact"
     }
    ]
   },
   "responsiveness_mic": {
    "findings": "A systematic review and meta-analysis of interventions found that structured programmes reduced emotional exhaustion and overall burnout scores in physicians, evidence that MBI scores move with change ([West C. 2016](https://doi.org/10.1016/s0140-6736(16)31279-x)). However, no established minimal important change (MIC) value for any MBI subscale was located, and the COSMIN review noted responsiveness as an under-evidenced property ([Shoman Y. 2021](https://doi.org/10.1017/s2045796020001134)).",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect; intervention samples are largely non-UK physicians; no MIC anchor established"
   },
   "populations_languages_norms": {
    "findings": "The MBI has been translated and validated in many languages and occupations, including Spanish/Colombian MBI-GS ([Bravo D. 2021](https://doi.org/10.3390/ijerph18105118)), Vietnamese MBI-HSS ([Bui T. 2022](https://doi.org/10.1080/21642850.2021.2019585)), Persian MBI-HSS-MP ([Lin C. 2022](https://doi.org/10.1016/j.heliyon.2022.e08868)), an Arabic MBI ([Henchiri H. 2025](https://doi.org/10.3390/healthcare13020173)), and an eight-country nurse dataset including the UK ([Poghosyan L. 2009](https://doi.org/10.1016/j.ijnurstu.2009.03.004)). Mind Garden supplies normative comparison data within its manual. UK-specific working-adult norms are not openly published and sit behind the commercial licence.",
    "grade": "Moderate",
    "indirectness": "indirect; extensive international norms but UK working-adult norms are proprietary and not openly available"
   },
   "criticisms_controversies": "The MBI is the most-used burnout measure and the de facto historical standard, but it is the subject of sustained criticism: the absence of validated cut-off scores means it does not diagnose burnout in individuals; the personal accomplishment/professional efficacy subscale shows weak internal consistency and contested separability; a bifactor reanalysis suggests a dominant general factor undermining the three-scale model; and a 2024 critique argues the instrument fails to measure its target construct. Its commercial, per-administration licensing (Mind Garden) also constrains transparent reuse and open norming, a material contrast with the free burnout inventories.",
   "item_records": [],
   "citations": [
    {
     "key": "maslach1981",
     "authors": "Maslach C.; Jackson S.",
     "year": "1981",
     "title": "The measurement of experienced burnout",
     "journal": "Journal of Organizational Behavior",
     "doi": "10.1002/job.4030020205",
     "url": "https://doi.org/10.1002/job.4030020205"
    },
    {
     "key": "worley2008",
     "authors": "Worley J.; Vassar M.; Wheeler D.; Barnes L.",
     "year": "2008",
     "title": "Factor Structure of Scores From the Maslach Burnout Inventory",
     "journal": "Educational and Psychological Measurement",
     "doi": "10.1177/0013164408315268",
     "url": "https://doi.org/10.1177/0013164408315268"
    },
    {
     "key": "poghosyan2009",
     "authors": "Poghosyan L.; Aiken L.; Sloane D.",
     "year": "2009",
     "title": "Factor structure of the Maslach burnout inventory: An analysis of data from large scale cross-sectional surveys of nurses from eight countries",
     "journal": "International Journal of Nursing Studies",
     "doi": "10.1016/j.ijnurstu.2009.03.004",
     "url": "https://doi.org/10.1016/j.ijnurstu.2009.03.004"
    },
    {
     "key": "byrne1994",
     "authors": "Byrne B.",
     "year": "1994",
     "title": "Testing for the Factorial Validity, Replication, and Invariance of a Measuring Instrument: A Paradigmatic Application Based on the Maslach Burnout Inventory",
     "journal": "Multivariate Behavioral Research",
     "doi": "10.1207/s15327906mbr2903_5",
     "url": "https://doi.org/10.1207/s15327906mbr2903_5"
    },
    {
     "key": "wheeler2011",
     "authors": "Wheeler D.; Vassar M.; Worley J.; Barnes L.",
     "year": "2011",
     "title": "A Reliability Generalization Meta-Analysis of Coefficient Alpha for the Maslach Burnout Inventory",
     "journal": "Educational and Psychological Measurement",
     "doi": "10.1177/0013164410391579",
     "url": "https://doi.org/10.1177/0013164410391579"
    },
    {
     "key": "masem2024",
     "authors": "Aguayo-Estremera R.; Cañadas-De la Fuente G.; Ariza T.; Ortega-Campos E.; Gómez-Urquiza J.; Romero-Béjar J.; De la Fuente-Solana E.",
     "year": "2024",
     "title": "A comparison of univariate and meta-analytic structural equation modeling approaches to reliability generalization applied to the Maslach Burnout Inventory",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2024.1383619",
     "url": "https://doi.org/10.3389/fpsyg.2024.1383619"
    },
    {
     "key": "mbigs_col2021",
     "authors": "Bravo D.; Suárez-Falcón J.; Bianchi J.; Segura-Vargas M.; Ruiz F.",
     "year": "2021",
     "title": "Psychometric Properties and Measurement Invariance of the Maslach Burnout Inventory–General Survey in Colombia",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph18105118",
     "url": "https://doi.org/10.3390/ijerph18105118"
    },
    {
     "key": "mbi_dif2021",
     "authors": "Brady K.; Sheldrick R.; Ni P.; Trockel M.; Shanafelt T.; Rowe S.; Kazis L.",
     "year": "2021",
     "title": "Examining the measurement equivalence of the Maslach Burnout Inventory across age, gender, and specialty groups in US physicians",
     "journal": "Journal of Patient-Reported Outcomes",
     "doi": "10.1186/s41687-021-00312-2",
     "url": "https://doi.org/10.1186/s41687-021-00312-2"
    },
    {
     "key": "mbihss_vn2022",
     "authors": "Bui T.; Tran T.; Nguyen T.; Vu T.; Ngo X.; Nguyen T.; Do T.",
     "year": "2022",
     "title": "Reassessing the most popularly suggested measurement models and measurement invariance of the Maslach Burnout Inventory – human service survey among Vietnamese healthcare professionals",
     "journal": "Health Psychology and Behavioral Medicine",
     "doi": "10.1080/21642850.2021.2019585",
     "url": "https://doi.org/10.1080/21642850.2021.2019585"
    },
    {
     "key": "mbigs9_2024",
     "authors": "Wang A.; Duan Y.; Norton P.; Leiter M.; Estabrooks C.",
     "year": "2024",
     "title": "Validation of the Maslach Burnout Inventory-General Survey 9-item short version: psychometric properties and measurement invariance across age, gender, and continent",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2024.1439470",
     "url": "https://doi.org/10.3389/fpsyg.2024.1439470"
    },
    {
     "key": "persian_mbi2022",
     "authors": "Lin C.; Alimoradi Z.; Griffiths M.; Pakpour A.",
     "year": "2022",
     "title": "Psychometric properties of the Maslach Burnout Inventory for Medical Personnel (MBI-HSS-MP)",
     "journal": "Heliyon",
     "doi": "10.1016/j.heliyon.2022.e08868",
     "url": "https://doi.org/10.1016/j.heliyon.2022.e08868"
    },
    {
     "key": "mbi_notmeasure2024",
     "authors": "Bianchi R.; Swingler G.; Schonfeld I.",
     "year": "2024",
     "title": "The Maslach Burnout Inventory is not a measure of burnout",
     "journal": "Work",
     "doi": "10.3233/wor-240095",
     "url": "https://doi.org/10.3233/wor-240095"
    },
    {
     "key": "cutoff1995",
     "authors": "Schaufeli W.; Van Dierendonck D.",
     "year": "1995",
     "title": "A Cautionary Note about the Cross-National and Clinical Validity of Cut-off Points for the Maslach Burnout Inventory",
     "journal": "Psychological Reports",
     "doi": "10.2466/pr0.1995.76.3c.1083",
     "url": "https://doi.org/10.2466/pr0.1995.76.3c.1083"
    },
    {
     "key": "teach_ltsa2022",
     "authors": "Salvagioni D.; Mesas A.; Melanda F.; González A.; de Andrade S.",
     "year": "2022",
     "title": "Burnout and Long-term Sickness Absence From the Teaching Function: A Cohort Study",
     "journal": "Safety and Health at Work",
     "doi": "10.1016/j.shaw.2022.01.006",
     "url": "https://doi.org/10.1016/j.shaw.2022.01.006"
    },
    {
     "key": "turnover_net2024",
     "authors": "Zheng J.; Feng S.; Feng Y.; Wang L.; Gao R.; Xue B.",
     "year": "2024",
     "title": "Relationship between burnout and turnover intention among nurses: a network analysis",
     "journal": "BMC Nursing",
     "doi": "10.1186/s12912-024-02624-2",
     "url": "https://doi.org/10.1186/s12912-024-02624-2"
    },
    {
     "key": "intent_leave2023",
     "authors": "Bruyneel A.; Bouckaert N.; Maertens de Noordhout C.; Detollenaere J.; Kohn L.; Pirson M.; Sermeus W.; Van den Heede K.",
     "year": "2023",
     "title": "Association of burnout and intention-to-leave the profession with work environment: A nationwide cross-sectional study among Belgian intensive care nurses after two years of pandemic",
     "journal": "International Journal of Nursing Studies",
     "doi": "10.1016/j.ijnurstu.2022.104385",
     "url": "https://doi.org/10.1016/j.ijnurstu.2022.104385"
    },
    {
     "key": "patientsafety2024",
     "authors": "Li L.; Yang P.; Singer S.; Pfeffer J.; Mathur M.; Shanafelt T.",
     "year": "2024",
     "title": "Nurse Burnout and Patient Safety, Satisfaction, and Quality of Care",
     "journal": "JAMA Network Open",
     "doi": "10.1001/jamanetworkopen.2024.43059",
     "url": "https://doi.org/10.1001/jamanetworkopen.2024.43059"
    },
    {
     "key": "schaufeli2009",
     "authors": "Schaufeli W.; Leiter M.; Maslach C.",
     "year": "2009",
     "title": "Burnout: 35 years of research and practice",
     "journal": "Career Development International",
     "doi": "10.1108/13620430910966406",
     "url": "https://doi.org/10.1108/13620430910966406"
    },
    {
     "key": "arabic_mbi2025",
     "authors": "Henchiri H.; Tannoubi A.; Harrathi C.; Boussayala G.; Quansah F.; Hagan J.; Mechergui H.; Chaabeni A.; Chebbi T.; Lakhal T.; Belhouchet H.; Khatrouch I.; Gawar A.; Azaiez F.",
     "year": "2025",
     "title": "Validation of the Arabic Version of the Maslach Burnout Inventory-HSS Among Tunisian Medical Residents (A-MBI-MR): Factor Structure, Construct Validity, Reliability, and Gender Invariance",
     "journal": "Healthcare",
     "doi": "10.3390/healthcare13020173",
     "url": "https://doi.org/10.3390/healthcare13020173"
    },
    {
     "key": "interv2016",
     "authors": "West C.; Dyrbye L.; Erwin P.; Shanafelt T.",
     "year": "2016",
     "title": "Interventions to prevent and reduce physician burnout: a systematic review and meta-analysis",
     "journal": "The Lancet",
     "doi": "10.1016/s0140-6736(16)31279-x",
     "url": "https://doi.org/10.1016/s0140-6736(16)31279-x"
    },
    {
     "key": "sysrev2021",
     "authors": "Shoman Y.; Marca S.; Bianchi R.; Godderis L.; van der Molen H.; Guseva Canu I.",
     "year": "2021",
     "title": "Psychometric properties of burnout measures: a systematic review",
     "journal": "Epidemiology and Psychiatric Sciences",
     "doi": "10.1017/s2045796020001134",
     "url": "https://doi.org/10.1017/s2045796020001134"
    },
    {
     "key": "demerouti2003",
     "authors": "Demerouti E.; Demerouti E.; Bakker A.; Vardakou I.; Kantas A.",
     "year": "2003",
     "title": "The Convergent Validity of Two Burnout Instruments",
     "journal": "European Journal of Psychological Assessment",
     "doi": "10.1027//1015-5759.19.1.12",
     "url": "https://doi.org/10.1027//1015-5759.19.1.12"
    }
   ],
   "record_notes": "Overall confidence: internal consistency well-established (High) but subscale-uneven; structural validity Moderate and genuinely contested on the personal accomplishment factor; organisational criterion validity Moderate; reference-standard criterion validity Very low because no diagnostic gold standard exists; test-retest thin (Low) with only item-level Iranian ICCs retrievable. The v0.2 schema made two things hard to record honestly: (1) canonical test-retest lives in the Mind Garden manual behind the paid licence and could not be verified this session, so I recorded only the openly retrievable coefficient rather than the manual's claim; (2) evidence_form is genuinely 'mixed' because the MBI-GS9 short form and the several localised forms are pooled in the reliability meta-analyses."
  },
  {
   "instrument_id": "cbi",
   "display_name": "Copenhagen Burnout Inventory (CBI)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Copenhagen Burnout Inventory (CBI)",
    "current_version": "Original 2005 form (Kristensen et al.); student adaptations (CBI-SS) and abbreviated versions exist",
    "item_count": "19 items across three scales: personal burnout (6), work-related burnout (7), client-related burnout (6)",
    "original_citation": "Kristensen TS, Borritz M, Villadsen E, Christensen KB (2005). The Copenhagen Burnout Inventory: A new tool for the assessment of burnout. Work & Stress 19(3):192-207. doi:10.1080/02678370500297720",
    "steward_publisher": "Danish National Research Centre for the Working Environment (NFA / Det Nationale Forskningscenter for Arbejdsmiljo)",
    "licence_status": "Freely available. NFA distributes the CBI questionnaire, the PUMA-study scales and Danish normative data as free downloads from its steward page. The current NFA page states 'Here you can download the Copenhagen Burnout Inventory - CBI questionnaire' and provides the scales without a fee or registration; it does not attach an explicit open-licence label (for example a Creative Commons tag) or a formal reuse statement, so the precise reuse conditions are best described as free-to-download from the steward rather than a named open licence.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "NFA steward page: https://nfa.dk/vaerktoejer/spoergeskemaer/spoergeskema-til-maaling-af-udbraendthed-cbi/copenhagen-burnout-inventory-cbi (read this session)"
   },
   "constructs_claimed": "Burnout defined generically as fatigue and exhaustion, partitioned by domain of attribution: personal burnout (general exhaustion, applicable to anyone), work-related burnout (exhaustion attributed to work), and client-related burnout (exhaustion attributed to work with clients/patients). Deliberately excludes depersonalisation and reduced accomplishment as separate constructs.",
   "deployment_context_caveat": "Like all burnout inventories, the CBI has no diagnostic reference standard and no validated clinical cut-off for individual diagnosis; it is a dimensional occupational-health and surveillance measure. It was developed and normed in Danish human-service workers, so applying Danish norms directly to UK working adults requires caution.",
   "structural_validity": {
    "findings": "The proposed three-factor structure (personal, work-related, client-related) has been supported by confirmatory factor analysis in several occupational samples: 928 US hospital nurses ([Montgomery A. 2021](https://doi.org/10.1002/nur.22114)) and 1,679 US academic-health-centre employees ([Thrush C. 2020](https://doi.org/10.1177/0163278720934165)). A UK Rasch analysis of 1,303 patient-facing optometrists found the three subscales unidimensional with good discrimination, though the client/patient-related scale showed a 22% floor effect ([Retallic N. 2026](https://doi.org/10.1002/ovs2.70048)). Student adaptations restructure the client scale (for example a four-factor CBI-SS in 635 Nigerian students, with one item removed for low average variance extracted; [Oluwadiya K. 2024](https://doi.org/10.1038/s41598-024-61310-0)). The core three-domain structure is comparatively stable across working-adult samples.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "direct for the UK optometrist Rasch study; otherwise indirect (US nurses, US academic staff, student adaptations)"
   },
   "convergent_discriminant_validity": {
    "findings": "CBI scales correlate as expected with the work environment, job satisfaction and intention to leave (moderate-to-high, correct directions) in US nurses ([Montgomery A. 2021](https://doi.org/10.1002/nur.22114)), and discriminant validity against a measure of meaningful work was supported in US academic-health-centre employees ([Thrush C. 2020](https://doi.org/10.1177/0163278720934165)). In the PUMA development study the scales differentiated well between human-service occupational groups and showed the expected correlations with fatigue and psychological well-being ([Kristensen T. 2005](https://doi.org/10.1080/02678370500297720)).",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect; principally US healthcare and Danish human-service samples"
   },
   "criterion_validity_reference_standard": {
    "findings": "No diagnostic gold standard for burnout exists, so the CBI has not been validated against a clinical reference standard; the COSMIN-based systematic review of occupational burnout measures found only sparse criterion-validity evidence for the CBI, with the instrument far less studied than the MBI ([Shoman Y. 2021](https://doi.org/10.1017/s2045796020001134)). This property is essentially untested for the CBI.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect; no reference-standard study located, and no diagnostic standard exists to validate against"
   },
   "criterion_validity_organisational": {
    "findings": "The CBI carries an unusually strong prospective organisational link, though against self-reported rather than register-based absence. In a three-year follow-up of 824 human-service workers from the PUMA study, the work-related burnout scale predicted subsequent self-reported sickness absence: mean absence rose from 5.4 days per year in the lowest quartile to 13.6 days in the highest, and a one-SD increase in work-related burnout predicted a 21% increase in self-reported sickness-absence days after adjustment for gender, age, organisation, socioeconomic status, lifestyle, family status and disease (rate ratio 1.21, 95% CI 1.11 to 1.32) ([Borritz 2006](https://doi.org/10.1136/oem.2004.019364)). The PUMA baseline paper documents the design and the associations of work burnout with the psychosocial work environment and absence ([Borritz 2006](https://doi.org/10.1080/14034940510032275)). This prospective evidence is a relative strength of the CBI among burnout inventories, but the outcome was self-reported (survey-based, subject to recall and reporting bias) rather than drawn from administrative or registry absence records, so it is weaker than register-based evidence would be.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect; prospective Danish human-service workers rather than a general UK working population, and the absence outcome is self-reported rather than register-based"
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently high. US hospital nurses gave Cronbach's alpha of 0.91 (personal), 0.89 (work-related) and 0.92 (client-related) ([Montgomery A. 2021](https://doi.org/10.1002/nur.22114)); the PUMA development study reported very high reliability for all three scales ([Kristensen T. 2005](https://doi.org/10.1080/02678370500297720)); and the Nigerian CBI-SS gave subscale alphas of 0.862 to 0.914 with a total of 0.957 ([Oluwadiya K. 2024](https://doi.org/10.1038/s41598-024-61310-0)). Reliability is a strength of the CBI.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect; Danish, US and Nigerian samples, no UK-specific alpha in the retrieved primary set beyond the UK Rasch reliability evidence"
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "thin",
    "indirectness": "not applicable; no coefficient located",
    "summary": "No numeric test-retest (temporal stability) coefficient for the CBI was located this session. The PUMA study was longitudinal and designed to track change rather than to establish short-interval retest stability, and the systematic review noted reliability evidence beyond internal consistency is sparse for the CBI. The absence of a reported retest coefficient is itself the finding: temporal stability of the CBI is effectively unevidenced in the retrieved literature."
   },
   "measurement_invariance": {
    "findings": "Evidence is partial. In 1,679 US academic-health-centre employees, configural and metric invariance held across professional role (physicians, nurses/PAs, other staff), gender and age, but scalar invariance was not established, cautioning against latent-mean comparisons across those groups ([Thrush C. 2020](https://doi.org/10.1177/0163278720934165)). A UK Rasch analysis of 1,303 optometrists found no significant differential item functioning by gender or age, supporting cross-group validity at item level ([Retallic N. 2026](https://doi.org/10.1002/ovs2.70048)). Higher-order (scalar) invariance across occupations remains the weak point.",
    "grade": "Low",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "partly direct (UK optometrists) and partly indirect (US healthcare staff)",
    "subgrades": [
     {
      "subgroup": "Configural / metric",
      "grade": "Moderate",
      "note": "supported across role, gender and age in a large US sample"
     },
     {
      "subgroup": "Scalar",
      "grade": "Very low",
      "note": "not established; group mean comparisons not yet justified"
     }
    ]
   },
   "responsiveness_mic": {
    "findings": "The CBI was embedded in the five-year PUMA intervention study and used to track change over time ([Borritz M. 2006](https://doi.org/10.1080/14034940510032275)), but no formal responsiveness statistic or minimal important change (MIC) value for the CBI was located, and the systematic review did not identify robust responsiveness evidence ([Shoman Y. 2021](https://doi.org/10.1017/s2045796020001134)).",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect; Danish human-service intervention context; no MIC anchor"
   },
   "populations_languages_norms": {
    "findings": "The CBI has Danish population normative data and PUMA-study occupational norms published openly by NFA (steward page), and has been validated in many languages and occupations, including US nurses ([Montgomery A. 2021](https://doi.org/10.1002/nur.22114)), US academic staff ([Thrush C. 2020](https://doi.org/10.1177/0163278720934165)), UK optometrists ([Retallic N. 2026](https://doi.org/10.1002/ovs2.70048)) and Nigerian students ([Oluwadiya K. 2024](https://doi.org/10.1038/s41598-024-61310-0)). Danish norms should not be applied uncritically to UK working adults, but at least one UK working-adult validation now exists.",
    "grade": "Moderate",
    "indirectness": "partly direct; a UK working-adult validation exists, but reference norms are Danish"
   },
   "criticisms_controversies": "The CBI was created as an explicit critique of the MBI, rejecting depersonalisation and personal accomplishment and grounding burnout in fatigue/exhaustion by domain. Its strengths are high internal consistency, free steward distribution, and a prospective link to self-reported sickness absence. Its weaknesses in the current evidence base are: no established test-retest coefficient; scalar measurement invariance not demonstrated across occupations; a floor effect on the client/patient-related scale in some samples; and no criterion validity against any health reference standard. The prospective absence evidence rests on self-reported rather than register-based absence, and the 'personal burnout' scale overlaps conceptually with general fatigue and non-occupational exhaustion, blurring its work-specificity.",
   "item_records": [],
   "citations": [
    {
     "key": "kristensen2005",
     "authors": "Kristensen T.; Borritz M.; Villadsen E.; Christensen K.",
     "year": "2005",
     "title": "The Copenhagen Burnout Inventory: A new tool for the assessment of burnout",
     "journal": "Work &amp; Stress",
     "doi": "10.1080/02678370500297720",
     "url": "https://doi.org/10.1080/02678370500297720"
    },
    {
     "key": "cbi_nurse2021",
     "authors": "Montgomery A.; Azuero A.; Patrician P.",
     "year": "2021",
     "title": "Psychometric properties of Copenhagen Burnout Inventory among nurses",
     "journal": "Research in Nursing &amp; Health",
     "doi": "10.1002/nur.22114",
     "url": "https://doi.org/10.1002/nur.22114"
    },
    {
     "key": "cbi_us2020",
     "authors": "Thrush C.; Gathright M.; Atkinson T.; Messias E.; Guise J.",
     "year": "2020",
     "title": "Psychometric Properties of the Copenhagen Burnout Inventory in an Academic Healthcare Institution Sample in the U.S.",
     "journal": "Evaluation &amp; the Health Professions",
     "doi": "10.1177/0163278720934165",
     "url": "https://doi.org/10.1177/0163278720934165"
    },
    {
     "key": "cbi_rasch_uk2026",
     "authors": "Retallic N.; Davey C.; Elliott D.",
     "year": "2026",
     "title": "Psychometric evaluation of the Copenhagen Burnout Inventory: Rasch analysis of burnout among optometrists",
     "journal": "Optometry and Vision Science",
     "doi": "10.1002/ovs2.70048",
     "url": "https://doi.org/10.1002/ovs2.70048"
    },
    {
     "key": "cbi_ng2024",
     "authors": "Oluwadiya K.; Owoeye O.; Adeoti A.",
     "year": "2024",
     "title": "Evaluating the factor structure, reliability and validity of the Copenhagen Burnout Inventory-Student Survey (CBI-SS) among faculty of arts students of Ekiti State University, Ado-Ekiti, Nigeria",
     "journal": "Scientific Reports",
     "doi": "10.1038/s41598-024-61310-0",
     "url": "https://doi.org/10.1038/s41598-024-61310-0"
    },
    {
     "key": "borritz2006",
     "authors": "Borritz M.; Rugulies R.; Christensen K.; Villadsen E.; Kristensen T.",
     "year": "2006",
     "title": "Burnout as a predictor of self-reported sickness absence among human service workers: prospective findings from three year follow up of the PUMA study",
     "journal": "Occupational and Environmental Medicine",
     "doi": "10.1136/oem.2004.019364",
     "url": "https://doi.org/10.1136/oem.2004.019364"
    },
    {
     "key": "puma2006",
     "authors": "Borritz M.; Rugulies R.; Bjorner J.; Villadsen E.; Mikkelsen O.; Kristensen T.",
     "year": "2006",
     "title": "Burnout among employees in human service work: design and baseline findings of the PUMA study",
     "journal": "Scandinavian Journal of Public Health",
     "doi": "10.1080/14034940510032275",
     "url": "https://doi.org/10.1080/14034940510032275"
    },
    {
     "key": "sysrev2021",
     "authors": "Shoman Y.; Marca S.; Bianchi R.; Godderis L.; van der Molen H.; Guseva Canu I.",
     "year": "2021",
     "title": "Psychometric properties of burnout measures: a systematic review",
     "journal": "Epidemiology and Psychiatric Sciences",
     "doi": "10.1017/s2045796020001134",
     "url": "https://doi.org/10.1017/s2045796020001134"
    }
   ],
   "record_notes": "Overall confidence: internal consistency High and well-established; organisational criterion validity Moderate and a relative strength (prospective self-reported sickness absence, PUMA/Borritz), with the caveat that the absence outcome was self-reported rather than register-based; structural validity Moderate; measurement invariance Low with scalar invariance unestablished; reference-standard criterion validity Absent; test-retest Absent (no coefficient located, recorded as the finding). The v0.2 schema handled this record well; the honesty-critical points are the genuinely Absent test-retest and reference-standard fields, the self-reported (not register-based) nature of the absence outcome, and the licence field, where the steward provides a free download without an explicit named open licence, which I recorded precisely rather than upgrading to 'public domain' or 'CC' on the basis of the founding paper."
  },
  {
   "instrument_id": "olbi",
   "display_name": "Oldenburg Burnout Inventory (OLBI)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Oldenburg Burnout Inventory (OLBI)",
    "current_version": "16-item English form (Halbesleben & Demerouti 2005 translation of the Demerouti et al. instrument); shortened national versions exist",
    "item_count": "16 items across two scales: exhaustion (8) and disengagement (8); each scale mixes positively and negatively worded items",
    "original_citation": "Demerouti E, Bakker AB, Vardakou I, Kantas A (2003). The convergent validity of two burnout instruments: a multitrait-multimethod analysis. Eur J Psychol Assess 19(1):12-23. doi:10.1027//1015-5759.19.1.12; English validation: Halbesleben JRB, Demerouti E (2005). doi:10.1080/02678370500340728",
    "steward_publisher": "No commercial publisher or single official distribution page; copyright held by the developers (Evangelia Demerouti and colleagues, originally University of Oldenburg). Permission is obtained by contacting the author.",
    "licence_status": "Free for non-commercial academic and research use, no licensing fee; permission from the copyright holders (Demerouti and colleagues) is expected for commercial or large-scale clinical applications. There is no official steward distribution page equivalent to Mind Garden or NFA; current terms are as described by secondary instrument databases and are consistent with a peer-reviewed validation that obtained use rights by contacting the author. Because no authoritative steward page states the licence, the exact reuse conditions cannot be pinned to a formal open-licence document this session.",
    "licence_verified_date": "2026-07-12 (verified against secondary instrument-database descriptions, not an official steward page, because none exists)",
    "licence_source": "Secondary instrument databases (db.arabpsychology.com/scales/oldenburg-burnout-inventory and resref.com/oldenburg-burnout-inventory-olbi-guide, read this session); corroborated by a validation study noting copyright was obtained by contacting the author (doi:10.1002/nop2.1065). No official steward licence page located."
   },
   "constructs_claimed": "Burnout as a two-dimensional work-related syndrome: exhaustion (physical, cognitive and affective components) and disengagement from work. Each dimension is measured with a balance of positively and negatively framed items, allowing the scales to span the burnout-to-engagement continuum.",
   "deployment_context_caveat": "No diagnostic reference standard or validated clinical cut-off for burnout exists, so the OLBI is a dimensional occupational-health measure rather than a diagnostic screener. The mixed positive/negative item wording introduces a method-variance issue (see structural validity) that is intrinsic to the instrument and cuts across its psychometrics.",
   "structural_validity": {
    "findings": "The intended two-factor structure (exhaustion, disengagement) has been supported by confirmatory factor analysis in the English validation across two US samples ([Halbesleben J. 2005](https://doi.org/10.1080/02678370500340728)) and in the original Greek convergent-validity study ([Demerouti E. 2003](https://doi.org/10.1027//1015-5759.19.1.12)), and across German employees and students with partial invariance ([Reis D. 2015](https://doi.org/10.1016/j.burn.2014.11.001)). However, the separability of the two factors is contested: a Chinese nurse validation found a very high inter-factor relationship and reported strong criterion overlap with the MBI ([Xu H. 2021](https://doi.org/10.1002/nop2.1065)), a Brazil/Portugal study reduced the instrument to 15 items and noted a high disengagement-exhaustion correlation ([Sinval J. 2019](https://doi.org/10.3389/fpsyg.2019.00338)), and a within/between-person multilevel study of daily diary data concluded a unidimensional model was more defensible because of the strong exhaustion-disengagement correlation and low within-person reliability of disengagement ([Gruszczynska E. 2021](https://doi.org/10.1371/journal.pone.0251257)). The mixed item wording generates method effects: the Polish validation found each of the two components split into positively and negatively worded sub-clusters, distorting the intended two-factor solution ([Baka Ł. 2016](https://doi.org/10.13075/mp.5893.00353)). This wording-method controversy is the central structural issue for the OLBI.",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect; German, Greek, US, Chinese, Polish and Brazil/Portugal samples, no dedicated UK working-adult structural study located",
    "subgrades": [
     {
      "subgroup": "Two-factor solution",
      "grade": "Moderate",
      "note": "supported in several samples but with high inter-factor correlation"
     },
     {
      "subgroup": "Wording-method effects",
      "grade": "Low",
      "note": "positively vs negatively worded items form method sub-factors in some samples"
     }
    ]
   },
   "convergent_discriminant_validity": {
    "findings": "OLBI exhaustion and disengagement correlate strongly with the corresponding MBI-GS scales: multitrait-multimethod analyses in the original Greek study supported convergent and discriminant validity against the MBI-GS ([Demerouti E. 2003](https://doi.org/10.1027//1015-5759.19.1.12)), the English validation positioned the OLBI as a viable alternative to the MBI-GS on MTMM grounds ([Halbesleben J. 2005](https://doi.org/10.1080/02678370500340728)), and a Chinese validation reported a very high correlation with the MBI (r = 0.87 for the total scale) ([Xu H. 2021](https://doi.org/10.1002/nop2.1065)). The scales relate positively to perceived stress and negatively to work engagement ([Baka Ł. 2016](https://doi.org/10.13075/mp.5893.00353); [Sinval J. 2019](https://doi.org/10.3389/fpsyg.2019.00338)). The very high MBI correlations support convergent validity but also raise the question of what the OLBI adds beyond existing measures.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect; Greek, US, Chinese, Polish and Brazil/Portugal samples"
   },
   "criterion_validity_reference_standard": {
    "findings": "There is no diagnostic gold standard for burnout, and no validation of the OLBI against a clinical or health reference standard was located; the COSMIN-based systematic review found the OLBI far less studied than the MBI and did not identify robust criterion-validity evidence against a health standard ([Shoman Y. 2021](https://doi.org/10.1017/s2045796020001134)). A recent critique argues the OLBI, like the MBI, does not measure the construct it claims to ([Bianchi R. 2024](https://doi.org/10.3233/wor-240095)). This property is untested for the OLBI.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect; no reference-standard study exists and no diagnostic standard is available"
   },
   "criterion_validity_organisational": {
    "findings": "Direct prospective links between OLBI scores and registered organisational outcomes (sickness absence, turnover, performance) were not located in this session's retrieval; the systematic review did not surface such evidence for the OLBI ([Shoman Y. 2021](https://doi.org/10.1017/s2045796020001134)). The instrument is validated mainly through internal structure and cross-instrument correlations rather than against work outcomes. The general burnout-outcome literature (for example burnout predicting turnover intention and sickness absence) is largely built on the MBI and CBI rather than the OLBI ([Zheng J. 2024](https://doi.org/10.1186/s12912-024-02624-2); [Borritz M. 2006](https://doi.org/10.1136/oem.2004.019364)), so it cannot be read across to the OLBI without direct evidence. This is an important gap: organisational criterion validity for the OLBI is effectively absent in the retrieved literature.",
    "grade": "Absent",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect; no OLBI-specific organisational-outcome study located, and the general evidence rests on other instruments"
   },
   "internal_consistency": {
    "findings": "Internal consistency is generally acceptable to good. The Chinese OLBI gave Cronbach's alpha of 0.905 (total), 0.933 (exhaustion) and 0.876 (disengagement) with split-half 0.883 ([Xu H. 2021](https://doi.org/10.1002/nop2.1065)); a Greek short OLBI gave exhaustion alpha 0.810 (omega 0.823) and disengagement alpha 0.742 (omega 0.756) ([Gkontelos A. 2023](https://doi.org/10.3390/ejihpe13060079)); the Polish validation supported reliability by alpha and test-retest across three samples ([Baka Ł. 2016](https://doi.org/10.13075/mp.5893.00353)). Disengagement is typically the weaker scale, and its low within-person reliability was one reason a daily-diary study favoured a unidimensional model ([Gruszczynska E. 2021](https://doi.org/10.1371/journal.pone.0251257)).",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect; Chinese, Greek and Polish samples",
    "subgrades": [
     {
      "subgroup": "Exhaustion",
      "grade": "High",
      "note": "alpha typically 0.80 to 0.93"
     },
     {
      "subgroup": "Disengagement",
      "grade": "Moderate",
      "note": "alpha often 0.74 to 0.88; weakest at within-person level"
     }
    ]
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "acceptable (no numeric coefficient reported in the retrievable text)",
      "coefficient_type": "other",
      "interval": "not specified",
      "sample_n": "2,599 across two US samples (generalised working adults and fire-department employees)",
      "population": "US working adults (English OLBI validation)",
      "evidence_form": "canonical",
      "citation_key": "halbesleben2005"
     },
     {
      "coefficient": "supported (numeric value not extractable from retrieved abstract)",
      "coefficient_type": "other",
      "interval": "6 weeks",
      "sample_n": "three samples (1,804; 366; 48 workers)",
      "population": "Polish social- and general-service workers",
      "evidence_form": "derivative",
      "citation_key": "olbi_pl2016"
     }
    ],
    "grade": "Low",
    "status": "thin",
    "indirectness": "indirect; the English validation reports test-retest as acceptable without an extractable coefficient in open text, and the Polish study used a 6-week interval but its numeric retest value was not in the retrieved abstract",
    "summary": "Temporal stability of the OLBI is under-reported. The canonical English validation states test-retest reliability was acceptable but the coefficient is not in the openly retrievable text (the full paper sits behind a paywall and could not be read this session), and the Polish validation used a 6-week retest interval across three samples but did not surface a numeric coefficient in the retrieved abstract. No clean subscale-level ICC or r with interval and sample was extractable, so this is graded Low and thin rather than well-established, and the specific coefficients should be treated as not yet verified."
   },
   "measurement_invariance": {
    "findings": "Reis et al. found the two-factor OLBI structure held within German employee and student samples and supported partial invariance of job and academic burnout within the German samples and of academic burnout across Greek and German students ([Reis D. 2015](https://doi.org/10.1016/j.burn.2014.11.001)). The Brazil/Portugal study examined invariance across countries and sexes on a reduced 15-item form ([Sinval J. 2019](https://doi.org/10.3389/fpsyg.2019.00338)), and a Greek short-form study reported measurement invariance across teacher demographic characteristics ([Gkontelos A. 2023](https://doi.org/10.3390/ejihpe13060079)). A daily-diary multilevel study found cross-level invariance was not supported, indicating that person-level and within-person burnout are not the same latent variable ([Gruszczynska E. 2021](https://doi.org/10.1371/journal.pone.0251257)). Invariance evidence is partial and often on reduced or short forms rather than the canonical 16-item instrument.",
    "grade": "Low",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect; German, Greek, Brazil/Portugal and teacher samples; much of it on reduced-item forms",
    "subgrades": [
     {
      "subgroup": "Across sex / country (students, teachers)",
      "grade": "Low",
      "note": "partial invariance on reduced forms"
     },
     {
      "subgroup": "Across measurement levels (within vs between person)",
      "grade": "Very low",
      "note": "cross-level invariance not supported"
     }
    ]
   },
   "responsiveness_mic": {
    "findings": "No responsiveness statistic or minimal important change (MIC) value for the OLBI was located, and the systematic review did not identify responsiveness evidence for the instrument ([Shoman Y. 2021](https://doi.org/10.1017/s2045796020001134)). The OLBI's balanced wording is argued to make it suitable for tracking change toward engagement, but this is a design rationale rather than demonstrated responsiveness.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect; no responsiveness or MIC study located"
   },
   "populations_languages_norms": {
    "findings": "The OLBI has been translated and validated in many languages and occupations, including English (US) ([Halbesleben J. 2005](https://doi.org/10.1080/02678370500340728)), Chinese nurses ([Xu H. 2021](https://doi.org/10.1002/nop2.1065)), Polish service workers ([Baka Ł. 2016](https://doi.org/10.13075/mp.5893.00353)), Brazil/Portugal ([Sinval J. 2019](https://doi.org/10.3389/fpsyg.2019.00338)), Greek teachers ([Gkontelos A. 2023](https://doi.org/10.3390/ejihpe13060079)) and German employees and students ([Reis D. 2015](https://doi.org/10.1016/j.burn.2014.11.001)). Formal population norms are less developed than for the MBI or CBI, and no UK working-adult norm set was located; several validations use reduced or short forms, complicating cross-study comparison.",
    "grade": "Low",
    "indirectness": "indirect; broad language coverage but no established UK working-adult norms and frequent use of reduced forms"
   },
   "criticisms_controversies": "The OLBI was designed as a free, two-dimensional alternative to the MBI, addressing the MBI's one-directional item wording by balancing positive and negative items and adding physical/cognitive exhaustion content. Its principal controversies are: (1) the balanced wording induces method-variance sub-factors that can distort the two-factor structure; (2) the exhaustion and disengagement scales correlate very highly, prompting some studies to favour a unidimensional model; (3) its correlations with the MBI are so high that its incremental value is questioned; and (4) organisational criterion validity against work outcomes and responsiveness/MIC are effectively unevidenced. Licensing is informal: free for research but with no official steward page, so reuse terms rest on secondary descriptions and author permission.",
   "item_records": [],
   "citations": [
    {
     "key": "demerouti2003",
     "authors": "Demerouti E.; Demerouti E.; Bakker A.; Vardakou I.; Kantas A.",
     "year": "2003",
     "title": "The Convergent Validity of Two Burnout Instruments",
     "journal": "European Journal of Psychological Assessment",
     "doi": "10.1027//1015-5759.19.1.12",
     "url": "https://doi.org/10.1027//1015-5759.19.1.12"
    },
    {
     "key": "halbesleben2005",
     "authors": "Halbesleben J.; Demerouti E.",
     "year": "2005",
     "title": "The construct validity of an alternative measure of burnout: Investigating the English translation of the Oldenburg Burnout Inventory",
     "journal": "Work &amp; Stress",
     "doi": "10.1080/02678370500340728",
     "url": "https://doi.org/10.1080/02678370500340728"
    },
    {
     "key": "reis2015",
     "authors": "Reis D.; Xanthopoulou D.; Tsaousis I.",
     "year": "2015",
     "title": "Measuring job and academic burnout with the Oldenburg Burnout Inventory (OLBI): Factorial invariance across samples and countries",
     "journal": "Burnout Research",
     "doi": "10.1016/j.burn.2014.11.001",
     "url": "https://doi.org/10.1016/j.burn.2014.11.001"
    },
    {
     "key": "olbi_cn2021",
     "authors": "Xu H.; Yuan Y.; Gong W.; Zhang J.; Liu X.; Zhu P.; Takashi E.; Kitayama A.; Wan X.; Jiao J.",
     "year": "2021",
     "title": "Reliability and validity of the Chinese version of Oldenburg Burnout Inventory for Chinese nurses",
     "journal": "Nursing Open",
     "doi": "10.1002/nop2.1065",
     "url": "https://doi.org/10.1002/nop2.1065"
    },
    {
     "key": "olbi_wb2021",
     "authors": "Gruszczynska E.; Basinska B.; Schaufeli W.",
     "year": "2021",
     "title": "Within- and between-person factor structure of the Oldenburg Burnout Inventory: Analysis of a diary study using multilevel confirmatory factor analysis",
     "journal": "PLOS ONE",
     "doi": "10.1371/journal.pone.0251257",
     "url": "https://doi.org/10.1371/journal.pone.0251257"
    },
    {
     "key": "olbi_brpt2019",
     "authors": "Sinval J.; Queirós C.; Pasian S.; Marôco J.",
     "year": "2019",
     "title": "Transcultural Adaptation of the Oldenburg Burnout Inventory (OLBI) for Brazil and Portugal",
     "journal": "Frontiers in Psychology",
     "doi": "10.3389/fpsyg.2019.00338",
     "url": "https://doi.org/10.3389/fpsyg.2019.00338"
    },
    {
     "key": "olbi_pl2016",
     "authors": "Baka Ł.; Basińska B.",
     "year": "2016",
     "title": "Psychometric properties of the Polish version of the Oldenburg Burnout Inventory (OLBI)",
     "journal": "Medycyna Pracy",
     "doi": "10.13075/mp.5893.00353",
     "url": "https://doi.org/10.13075/mp.5893.00353"
    },
    {
     "key": "olbi_gr2023",
     "authors": "Gkontelos A.; Vaiopoulou J.; Stamovlasis D.",
     "year": "2023",
     "title": "Burnout of Greek Teachers: Measurement Invariance and Differences across Individual Characteristics",
     "journal": "European Journal of Investigation in Health, Psychology and Education",
     "doi": "10.3390/ejihpe13060079",
     "url": "https://doi.org/10.3390/ejihpe13060079"
    },
    {
     "key": "sysrev2021",
     "authors": "Shoman Y.; Marca S.; Bianchi R.; Godderis L.; van der Molen H.; Guseva Canu I.",
     "year": "2021",
     "title": "Psychometric properties of burnout measures: a systematic review",
     "journal": "Epidemiology and Psychiatric Sciences",
     "doi": "10.1017/s2045796020001134",
     "url": "https://doi.org/10.1017/s2045796020001134"
    },
    {
     "key": "mbi_notmeasure2024",
     "authors": "Bianchi R.; Swingler G.; Schonfeld I.",
     "year": "2024",
     "title": "The Maslach Burnout Inventory is not a measure of burnout",
     "journal": "Work",
     "doi": "10.3233/wor-240095",
     "url": "https://doi.org/10.3233/wor-240095"
    },
    {
     "key": "turnover_net2024",
     "authors": "Zheng J.; Feng S.; Feng Y.; Wang L.; Gao R.; Xue B.",
     "year": "2024",
     "title": "Relationship between burnout and turnover intention among nurses: a network analysis",
     "journal": "BMC Nursing",
     "doi": "10.1186/s12912-024-02624-2",
     "url": "https://doi.org/10.1186/s12912-024-02624-2"
    },
    {
     "key": "borritz2006",
     "authors": "Borritz M.; Rugulies R.; Christensen K.; Villadsen E.; Kristensen T.",
     "year": "2006",
     "title": "Burnout as a predictor of self-reported sickness absence among human service workers: prospective findings from three year follow up of the PUMA study",
     "journal": "Occupational and Environmental Medicine",
     "doi": "10.1136/oem.2004.019364",
     "url": "https://doi.org/10.1136/oem.2004.019364"
    },
    {
     "key": "jdr2001",
     "authors": "Demerouti E.; Bakker A.; Nachreiner F.; Schaufeli W.",
     "year": "2001",
     "title": "The job demands-resources model of burnout.",
     "journal": "Journal of Applied Psychology",
     "doi": "10.1037/0021-9010.86.3.499",
     "url": "https://doi.org/10.1037/0021-9010.86.3.499"
    }
   ],
   "record_notes": "Overall confidence: internal consistency Moderate-to-High (exhaustion strong, disengagement weaker); structural validity Moderate but genuinely contested on wording-method effects and factor separability; convergent validity Moderate; measurement invariance Low; reference-standard criterion validity Absent/untested; organisational criterion validity Absent (an important, honestly-recorded gap); responsiveness Absent; test-retest thin (Low) with no cleanly extractable coefficient. The v0.2 schema made two things hard: (1) the canonical English validation and its test-retest coefficient sit behind a paywall, so I recorded the property as reported-but-not-numerically-verifiable rather than inventing a value; (2) much invariance and reliability evidence is on reduced/short OLBI forms, so evidence_form is 'mixed' and the canonical 16-item instrument is less directly evidenced than the pooled figures imply."
  },
  {
   "instrument_id": "wai",
   "display_name": "Work Ability Index (WAI)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Work Ability Index (WAI)",
    "current_version": "Standard seven-dimension WAI (10 questions across seven items/dimensions); a long form and short form are distributed by the WAI-Netzwerk.",
    "item_count": "7 dimensions assessed via 10 questions (score range 7 to 49)",
    "original_citation": "Tuomi K, Ilmarinen J, Jahkola A, Katajarinne L, Tulkki A. Work Ability Index. Finnish Institute of Occupational Health, Helsinki (2nd revised edn 1998). Founding empirical work: Tuomi et al., Scand J Work Environ Health 1991 (see cited longitudinal municipal-employee studies [Ilmarinen/Tuomi cohort](https://doi.org/10.5271/sjweh.979)).",
    "steward_publisher": "Finnish Institute of Occupational Health (FIOH / Tyoterveyslaitos), Helsinki; German national distribution via the WAI-Netzwerk / BAuA.",
    "licence_status": "Not verified this session. Rule 7 requires the licence to be confirmed against the steward's current distribution terms; I was unable to reach a primary steward licence page this session. The originating steward's site (Finnish Institute of Occupational Health, FIOH / Tyoterveyslaitos, ttl.fi) and the German national steward page (BAuA, baua.de) both returned server-side 403 (Forbidden) responses to automated retrieval, so their current terms could not be read. The only reachable steward-network page, the German WAI-Netzwerk portal waiplus.online, is a commercial consultancy site marketing a trademarked survey service ('waiplus') and does not state the licence terms of the underlying WAI questionnaire. Search-result listings (not read as page content) indicate FIOH sells an official user manual ('WAI, How to use the Work Ability Index questionnaire', Rautio and Michelsen) through its bookshop and that the questionnaire is widely described in the literature as freely usable for occupational-health screening and research, but no current primary-source licence terms were confirmed. 'WAI' is presented as a trademark of FIOH. No open-content licence (Creative Commons or equivalent) should be assumed; practitioners must confirm current terms with the relevant national steward before use.",
    "licence_verified_date": "not verified this session, FIOH (ttl.fi) and BAuA (baua.de) steward pages returned server-side 403 blocks to automated retrieval; the reachable WAI-Netzwerk portal (waiplus.online) markets a commercial service and does not state questionnaire licence terms",
    "licence_source": "Attempted primary steward sources: FIOH https://www.ttl.fi/en (403 Forbidden), BAuA https://www.baua.de/EN/Topics/Prevention/Mental-health/Work-Ability-Index (403 Forbidden), and the reachable but non-licence WAI-Netzwerk consultancy portal https://www.waiplus.online/ (retrieved 2026-07-12; markets the trademarked 'waiplus' survey service, no questionnaire licence statement). Founding papers and reviews were NOT used as the licence source; because no primary steward licence page could be read, no licence status is asserted."
   },
   "constructs_claimed": "Self-assessed work ability, defined as the balance between an individual's personal resources (health, competence, values) and the physical and mental demands of their work. The index aggregates current work ability relative to lifetime best, work ability relative to job demands, number of physician-diagnosed diseases, estimated work impairment due to disease, sickness absence in the past year, own prognosis of work ability in two years, and mental resources.",
   "deployment_context_caveat": "The WAI originated as an occupational-health screening instrument for ageing workers and is used directly in the workplace, so its deployment context matches the target audience. The principal evidential caveat is population indirectness for a UK working-adult audience: almost all measurement-property evidence was earned in Finnish, Dutch, Swedish, German, Iranian, Spanish and Brazilian samples, with no UK-specific validation or norms located this session.",
   "structural_validity": {
    "findings": "Dimensionality is genuinely contested and the original single-score (unidimensional) scoring assumption is not well supported. A representative German confirmatory analysis (n=3968) found two correlated factors, a 'subjective work ability and resources' factor and a 'health-related' factor, fitting better than one factor ([Freyer 2018](https://doi.org/10.1007/s10926-018-9803-9)). An earlier German multi-occupational study (n=324) likewise rejected both a one-factor and an orthogonal two-factor model, favouring a two-dimensional correlated-factor solution in which only five of seven items loaded unambiguously ([Martus 2010](https://doi.org/10.1093/occmed/kqq093)). Spanish health-centre workers (n=1184) showed EFA support for one factor but better CFA fit for two factors ([Mateo Rodriguez 2021](https://doi.org/10.3390/ijerph182412988)), a pattern echoed in Spanish aeronautical workers ([Gonzalez-Dominguez 2024](https://doi.org/10.1016/j.shaw.2023.12.001)) and a Brazilian online-application study ([Pucci 2024](https://doi.org/10.1016/j.bjpt.2024.101060)). Three-factor solutions have also been reported in Iranian ([Abdolalizadeh 2012](https://doi.org/10.1007/s10926-012-9355-3)) and Chilean/Spanish ([Bascour-Sandoval 2020](https://doi.org/10.1007/s10926-019-09871-0)) samples, while a recent large Brazilian re-evaluation (n=3051) reported a unidimensional structure but with only marginal fit (RMSEA 0.125, TLI 0.900, CFI 0.882) ([Martinez 2025](https://doi.org/10.1590/1413-81232025309.07832024)). The weight of evidence favours a two-factor structure over the historical single sum score, but solutions vary by sample and language.",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect, evidence from German, Spanish, Brazilian, Iranian and Chilean working samples; no UK confirmatory data located.",
    "subgrades": [
     {
      "subgroup": "one-factor (historical sum-score assumption)",
      "grade": "Low",
      "note": "generally not supported by CFA in representative samples"
     },
     {
      "subgroup": "two correlated factors",
      "grade": "Moderate",
      "note": "most frequently replicated solution (subjective work ability vs health-related)"
     }
    ]
   },
   "convergent_discriminant_validity": {
    "findings": "The WAI correlates positively with self-rated general health and health-related quality of life, consistent with its construct. In Spanish working individuals it correlated significantly with the SF-36 v2 ([Bascour-Sandoval 2020](https://doi.org/10.1007/s10926-019-09871-0)), and in Spanish aeronautical workers it was highly correlated with self-assessed health status ([Gonzalez-Dominguez 2024](https://doi.org/10.1016/j.shaw.2023.12.001)). Among Iranian construction workers the WAI explained about 46 per cent of the variance in WHOQOL-BREF quality of life, very close to the single-item score ([Mokarami 2021](https://doi.org/10.1007/s00420-021-01740-9)). Convergence with the SF-36 has also been used as evidence of criterion/convergent validity in Iranian healthcare workers ([Abdolalizadeh 2012](https://doi.org/10.1007/s10926-012-9355-3)). Discriminant (known-groups) validity is consistently demonstrated: WAI separates groups differing in sickness absence and individual/work characteristics ([Martinez 2025](https://doi.org/10.1590/1413-81232025309.07832024)).",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect, convergent evidence mostly from translated versions in non-UK working samples."
   },
   "criterion_validity_reference_standard": {
    "findings": "There is no diagnostic gold standard for work ability, so reference-standard criterion validity is assessed chiefly against established health measures. The WAI correlates moderately to strongly with the SF-36 and WHOQOL-BREF ([Abdolalizadeh 2012](https://doi.org/10.1007/s10926-012-9355-3); [Mokarami 2021](https://doi.org/10.1007/s00420-021-01740-9); [Bascour-Sandoval 2020](https://doi.org/10.1007/s10926-019-09871-0)), and its own dimensions include physician-diagnosed disease counts. However, these are health-status comparators rather than an independent diagnostic reference standard, so this criterion is better regarded as convergent evidence; a true reference-standard anchor for work ability does not exist.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "mixed",
    "indirectness": "indirect, no independent diagnostic reference standard exists; comparators are self-report health measures in non-UK samples."
   },
   "criterion_validity_organisational": {
    "findings": "This is the WAI's strongest evidential domain: it prospectively predicts work-relevant outcomes in large occupational cohorts. Among 11,537 male construction workers the full WAI discriminated future disability pension well (AUC 0.78, 95% CI 0.75 to 0.80) and was adequately calibrated ([Roelen 2014](https://doi.org/10.5271/sjweh.3428)). In 5251 Finnish municipal employees the hazard ratio for disability pension was 5.0 (95% CI 4.4 to 5.6) for poor versus good/excellent WAI ([Jaaskelainen 2016](https://doi.org/10.5271/sjweh.3598)). In a Swedish general-population sample the full WAI predicted long-term (90-day) sickness absence over four years with AUC 0.79 (95% CI 0.76 to 0.82), outperforming its individual items ([Lundin 2017](https://doi.org/10.1177/1403494817702759)). Among 1331 Dutch office workers each one-point lower WAI raised the odds of 15 or more sick days (OR 1.27, 95% CI 1.21 to 1.33; AUC 0.77 for 15+ versus 0 days) ([Reeuwijk 2015](https://doi.org/10.1371/journal.pone.0126969)), and in 3660 employees WAI was inversely associated with frequent (OR 0.85), long-term (OR 0.79) and combined (OR 0.74) sickness absence ([Notenbomer 2015](https://doi.org/10.1093/occmed/kqv052)). Higher WAI was protective against work nonparticipation (long-term sickness absence, early retirement, unemployment) in a representative German cohort (n=2426) ([Conway 2023](https://doi.org/10.1097/JOM.0000000000003032)), predicted long-term sickness absence across a ten-year follow-up ([Palmlof 2019](https://doi.org/10.1093/occmed/kqz083)), predicted application for disability pension after vocational rehabilitation ([Bethge 2013](https://doi.org/10.1016/j.apmr.2013.05.003)), and was associated with sickness absence in young employees ([Kujala 2006](https://doi.org/10.5271/sjweh.979)) and with disability more broadly ([Alavinia 2009](https://doi.org/10.1093/occmed/kqn148)). One caveat: because a WAI dimension counts recent sickness absence days, prediction of absence outcomes is partly non-independent of the predictor.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect for UK generalisability (Finnish, Dutch, Swedish and German cohorts) but directly in occupational settings against genuine work outcomes such as disability pension and sickness absence."
   },
   "internal_consistency": {
    "findings": "Cronbach's alpha for the total WAI typically falls in the 0.7 to 0.8 range: 0.79 in Iranian healthcare workers ([Abdolalizadeh 2012](https://doi.org/10.1007/s10926-012-9355-3)), 0.71 in Croatian nurses ([Smrekar 2020](https://doi.org/10.2478/sjph-2020-0008)), and 0.787 (McDonald's omega 0.819) in 3051 Brazilian nurses ([Martinez 2025](https://doi.org/10.1590/1413-81232025309.07832024)). Values are acceptable but modest, which is expected because the WAI is a heterogeneous composite (partly a formative index of distinct health and work-demand indicators) rather than a homogeneous reflective scale, so alpha is of limited interpretive value for this instrument.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect, alpha estimates from translated versions in non-UK occupational and nursing samples."
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "0.92",
      "coefficient_type": "ICC",
      "interval": "repeat administration (interval not specified in report)",
      "sample_n": "60",
      "population": "Iranian nurses/healthcare workers aged 40+",
      "evidence_form": "canonical",
      "citation_key": "abdolalizadeh2012"
     },
     {
      "coefficient": "all seven dimensions >0.70",
      "coefficient_type": "ICC",
      "interval": "test-retest (interval not specified in report)",
      "sample_n": "subset of 750",
      "population": "Iranian petrochemical and car-manufacturing workers",
      "evidence_form": "canonical",
      "citation_key": "adel2019"
     },
     {
      "coefficient": "66% agreement on four-category classification; 95% of individual score differences within 6.86 points; no group-level mean change (40.4 vs 39.9)",
      "coefficient_type": "other (agreement / limits of change)",
      "interval": "4 weeks",
      "sample_n": "97",
      "population": "elderly Dutch construction workers aged 40+",
      "evidence_form": "canonical",
      "citation_key": "dezwart2002"
     }
    ],
    "grade": "Moderate",
    "status": "well-established",
    "indirectness": "indirect, retest evidence from Dutch and Iranian occupational samples; no UK data.",
    "summary": "Full-scale WAI test-retest reliability is adequate. A four-week study in elderly construction workers found stable group means and 66 per cent exact category agreement without reporting an ICC ([de Zwart 2002](https://doi.org/10.1093/occmed/52.4.177)); Iranian validations report ICCs of 0.92 for the total score ([Abdolalizadeh 2012](https://doi.org/10.1007/s10926-012-9355-3)) and above 0.70 for each dimension ([Adel 2019](https://doi.org/10.1002/1348-9585.12028)). Retest reliability of the single first item (Work Ability Score) is treated separately and is more variable; see the WAS record."
   },
   "measurement_invariance": {
    "findings": "Formal measurement-invariance testing (configural, metric, scalar) across sex, age, occupation, language or time was not located for the WAI in this session. Several studies report factor structures in different countries and note mean score differences by age (younger workers scoring higher), but cross-group score differences are not evidence of measurement invariance, and the varying factor solutions across samples ([Martus 2010](https://doi.org/10.1093/occmed/kqq093); [Freyer 2018](https://doi.org/10.1007/s10926-018-9803-9); [Mateo Rodriguez 2021](https://doi.org/10.3390/ijerph182412988)) suggest invariance cannot be assumed. This is a genuine gap.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, no formal invariance analyses located in any population."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness and minimal important change for the full 7-item WAI were not established in the retrieved literature; longitudinal WAI studies use it as a predictor rather than an outcome, and no anchor-based MIC for the total index was located. Responsiveness and MIC evidence exists for the single-item Work Ability Score (in clinical rehabilitation samples) and is reported in the WAS record.",
    "grade": "Absent",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, no responsiveness/MIC study of the full WAI located."
   },
   "populations_languages_norms": {
    "findings": "The WAI has been translated and psychometrically examined in many languages and settings, including Finnish (origin), Dutch, German, Spanish, Croatian, Brazilian-Portuguese, Persian, Thai and Turkish. German normative reference tables (by sex and age band 31 to 40, 41 to 50, 51 to 60) are published by the national steward BAuA. Most validation is in ageing occupational cohorts, nurses and industrial workers. No UK-specific norms were located this session.",
    "grade": "High",
    "indirectness": "indirect for the UK; extensive international norms and translations but no UK norm set located."
   },
   "criticisms_controversies": "The main controversy is dimensionality: the historical practice of summing all items into a single 7 to 49 score assumes unidimensionality that confirmatory analyses in representative samples reject in favour of a two-factor structure, prompting the German steward to develop a weighted two-factor calculation. Because the index mixes reflective and formative indicators (self-rated ability, diagnosed disease counts, sickness absence days), internal consistency is only a weak indicator of quality and the composite is conceptually blurred, motivating proposals to use short uni-dimensional components instead. Predictive validity for sickness absence is partly circular because a WAI dimension counts recent absence. Finally, the licence position is easily mis-stated and could not be confirmed against a primary steward page this session (both FIOH and BAuA blocked automated retrieval): the questionnaire is commonly described in the literature as freely usable for research, but 'WAI' is trademarked and FIOH sells an official manual, so no open-content licence should be assumed pending direct confirmation of current steward terms.",
   "item_records": [],
   "citations": [
    {
     "key": "dezwart2002",
     "authors": "de Zwart B, Frings-Dresen M, van Duivenbooden J",
     "year": "2002",
     "title": "Test-retest reliability of the Work Ability Index questionnaire.",
     "journal": "Occupational medicine (Oxford, England)",
     "doi": "10.1093/occmed/52.4.177",
     "url": "https://doi.org/10.1093/occmed/52.4.177"
    },
    {
     "key": "kujala2006",
     "authors": "Kujala V, Tammelin T, Remes J, Vammavaara E, Ek E, Laitinen J",
     "year": "2006",
     "title": "Work ability index of young employees and their sickness absence during the following year.",
     "journal": "Scandinavian journal of work, environment & health",
     "doi": "10.5271/sjweh.979",
     "url": "https://doi.org/10.5271/sjweh.979"
    },
    {
     "key": "alavinia2009",
     "authors": "Alavinia S, de Boer A, van Duivenbooden J, Frings-Dresen M, Burdorf A",
     "year": "2009",
     "title": "Determinants of work ability and its predictive value for disability.",
     "journal": "Occupational medicine (Oxford, England)",
     "doi": "10.1093/occmed/kqn148",
     "url": "https://doi.org/10.1093/occmed/kqn148"
    },
    {
     "key": "martus2010",
     "authors": "Martus P, Jakob O, Rose U, Seibt R, Freude G",
     "year": "2010",
     "title": "A comparative analysis of the Work Ability Index.",
     "journal": "Occupational medicine (Oxford, England)",
     "doi": "10.1093/occmed/kqq093",
     "url": "https://doi.org/10.1093/occmed/kqq093"
    },
    {
     "key": "abdolalizadeh2012",
     "authors": "Abdolalizadeh M, Arastoo A, Ghsemzadeh R, Montazeri A, Ahmadi K, Azizi A",
     "year": "2012",
     "title": "The psychometric properties of an Iranian translation of the Work Ability Index (WAI) questionnaire.",
     "journal": "Journal of occupational rehabilitation",
     "doi": "10.1007/s10926-012-9355-3",
     "url": "https://doi.org/10.1007/s10926-012-9355-3"
    },
    {
     "key": "bethge2013",
     "authors": "Bethge M, Gutenbrunner C, Neuderth S",
     "year": "2013",
     "title": "Work Ability Index predicts application for disability pension after work-related medical rehabilitation for chronic back pain.",
     "journal": "Archives of physical medicine and rehabilitation",
     "doi": "10.1016/j.apmr.2013.05.003",
     "url": "https://doi.org/10.1016/j.apmr.2013.05.003"
    },
    {
     "key": "roelen2014",
     "authors": "Roelen C, van Rhenen W, Groothoff J, van der Klink J, Twisk J, Heymans M",
     "year": "2014",
     "title": "Work ability as prognostic risk marker of disability pension: single-item work ability score versus multi-item work ability index.",
     "journal": "Scandinavian journal of work, environment & health",
     "doi": "10.5271/sjweh.3428",
     "url": "https://doi.org/10.5271/sjweh.3428"
    },
    {
     "key": "notenbomer2015",
     "authors": "Notenbomer A, Groothoff J, van Rhenen W, Roelen C",
     "year": "2015",
     "title": "Associations of work ability with frequent and long-term sickness absence.",
     "journal": "Occupational medicine (Oxford, England)",
     "doi": "10.1093/occmed/kqv052",
     "url": "https://doi.org/10.1093/occmed/kqv052"
    },
    {
     "key": "reeuwijk2015",
     "authors": "Reeuwijk K, Robroek S, Niessen M, Kraaijenhagen R, Vergouwe Y, Burdorf A",
     "year": "2015",
     "title": "The Prognostic Value of the Work Ability Index for Sickness Absence among Office Workers.",
     "journal": "PloS one",
     "doi": "10.1371/journal.pone.0126969",
     "url": "https://doi.org/10.1371/journal.pone.0126969"
    },
    {
     "key": "jaaskelainen2016",
     "authors": "J&#xe4;&#xe4;skel&#xe4;inen A, Kausto J, Seitsamo J, Ojaj&#xe4;rvi A, Nyg&#xe5;rd C, Arjas E, Leino-Arjas P",
     "year": "2016",
     "title": "Work ability index and perceived work ability as predictors of disability pension: a prospective study among Finnish municipal employees.",
     "journal": "Scandinavian journal of work, environment & health",
     "doi": "10.5271/sjweh.3598",
     "url": "https://doi.org/10.5271/sjweh.3598"
    },
    {
     "key": "lundin2017",
     "authors": "Lundin A, Leijon O, Vaez M, Hallgren M, Torg&#xe9;n M",
     "year": "2017",
     "title": "Predictive validity of the Work Ability Index and its individual items in the general population.",
     "journal": "Scandinavian journal of public health",
     "doi": "10.1177/1403494817702759",
     "url": "https://doi.org/10.1177/1403494817702759"
    },
    {
     "key": "freyer2018",
     "authors": "Freyer M, Formazin M, Rose U",
     "year": "2018",
     "title": "Factorial Validity of the Work Ability Index Among Employees in Germany.",
     "journal": "Journal of occupational rehabilitation",
     "doi": "10.1007/s10926-018-9803-9",
     "url": "https://doi.org/10.1007/s10926-018-9803-9"
    },
    {
     "key": "adel2019",
     "authors": "Adel M, Akbar R, Ehsan G",
     "year": "2019",
     "title": "Validity and reliability of work ability index (WAI) questionnaire among Iranian workers; a study in petrochemical and car manufacturing industries.",
     "journal": "Journal of occupational health",
     "doi": "10.1002/1348-9585.12028",
     "url": "https://doi.org/10.1002/1348-9585.12028"
    },
    {
     "key": "palmlof2019",
     "authors": "Palml&#xf6;f L, Skillgate E, Talb&#xe4;ck M, Josephson M, Ving&#xe5;rd E, Holm L",
     "year": "2019",
     "title": "Poor work ability increases sickness absence over 10 years.",
     "journal": "Occupational medicine (Oxford, England)",
     "doi": "10.1093/occmed/kqz083",
     "url": "https://doi.org/10.1093/occmed/kqz083"
    },
    {
     "key": "bascoursandoval2020",
     "authors": "Bascour-Sandoval C, Soto-Rodr&#xed;guez F, Mu&#xf1;oz-Poblete C, Marzuca-Nassr G",
     "year": "2020",
     "title": "Psychometric Properties of the Spanish Version of the Work Ability Index in Working Individuals.",
     "journal": "Journal of occupational rehabilitation",
     "doi": "10.1007/s10926-019-09871-0",
     "url": "https://doi.org/10.1007/s10926-019-09871-0"
    },
    {
     "key": "smrekar2020",
     "authors": "Smrekar M, Franko A, Petrak O, Zaletel-Kragelj L",
     "year": "2020",
     "title": "Validation of the Croatian Version of Work Ability Index (WAI) in Population of Nurses on Transformed Item-Specific Scores.",
     "journal": "Zdravstveno varstvo",
     "doi": "10.2478/sjph-2020-0008",
     "url": "https://doi.org/10.2478/sjph-2020-0008"
    },
    {
     "key": "mokarami2021",
     "authors": "Mokarami H, Cousins R, Kalteh H",
     "year": "2021",
     "title": "Comparison of the work ability index and the work ability score for predicting health-related quality of life.",
     "journal": "International archives of occupational and environmental health",
     "doi": "10.1007/s00420-021-01740-9",
     "url": "https://doi.org/10.1007/s00420-021-01740-9"
    },
    {
     "key": "mateorodriguez2021",
     "authors": "Mateo Rodr&#xed;guez I, Knox E, Oliver Hern&#xe1;ndez C, Daponte Codina A, The esTAR Group",
     "year": "2021",
     "title": "Psychometric Properties of the Work Ability Index in Health Centre Workers in Spain.",
     "journal": "International journal of environmental research and public health",
     "doi": "10.3390/ijerph182412988",
     "url": "https://doi.org/10.3390/ijerph182412988"
    },
    {
     "key": "conway2023",
     "authors": "Conway P, Burr H, Kersten N, Rose U",
     "year": "2023",
     "title": "Work Ability and Work Nonparticipation: A Prospective Study of 2426 Participants in Germany.",
     "journal": "Journal of occupational and environmental medicine",
     "doi": "10.1097/JOM.0000000000003032",
     "url": "https://doi.org/10.1097/JOM.0000000000003032"
    },
    {
     "key": "pucci2024",
     "authors": "Pucci R, da Silva A, Padula R",
     "year": "2024",
     "title": "Factorial analysis of the Brazilian-Portuguese version of the Work Ability Index, reproducibility and validity of the single item and the short version for online application.",
     "journal": "Brazilian journal of physical therapy",
     "doi": "10.1016/j.bjpt.2024.101060",
     "url": "https://doi.org/10.1016/j.bjpt.2024.101060"
    },
    {
     "key": "gonzalezdominguez2024",
     "authors": "Gonz&#xe1;lez-Dom&#xed;nguez M, Fern&#xe1;ndez-Garc&#xed;a E, Paloma-Castro O, Gonz&#xe1;lez-L&#xf3;pez R, Rivas P&#xe9;rez M, L&#xf3;pez-Molina L, Garc&#xed;a-Jim&#xe9;nez J, Romero-S&#xe1;nchez J",
     "year": "2024",
     "title": "Work Ability Index: Psychometric Testing in Aeronautical Industry Workers.",
     "journal": "Safety and health at work",
     "doi": "10.1016/j.shaw.2023.12.001",
     "url": "https://doi.org/10.1016/j.shaw.2023.12.001"
    },
    {
     "key": "martinez2025",
     "authors": "Martinez M, Latorre M, Fischer F",
     "year": "2025",
     "title": "Validity and reliability of the Brazilian version of the Work Ability Index - WAI: a re-evaluation.",
     "journal": "Ciencia & saude coletiva",
     "doi": "10.1590/1413-81232025309.07832024",
     "url": "https://doi.org/10.1590/1413-81232025309.07832024"
    }
   ],
   "record_notes": "Overall confidence: organisational criterion validity is High and well-established (large prospective occupational cohorts predicting disability pension and sickness absence); structural validity is Moderate but contested (two-factor preferred over the historical sum score); internal consistency and test-retest are Moderate; measurement invariance and full-scale responsiveness/MIC are genuine gaps (Absent). All grades are downgraded for population indirectness because no UK validation or norms were located. Schema v0.2 handled this instrument well; the one hard case was internal consistency, which is a partly-formative composite where alpha is reported in the literature but is of limited interpretive value, noted in the findings and status rather than forced into a misleading grade. Licence could not be verified against a primary steward page this session because FIOH (ttl.fi) and BAuA (baua.de) returned server-side 403 blocks and the reachable WAI-Netzwerk portal does not state questionnaire terms; per rule 7 the licence is recorded as not verified this session rather than asserted from literature."
  },
  {
   "instrument_id": "was",
   "display_name": "Work Ability Score (WAS), single-item",
   "instrument_type": "single-item",
   "schema_version": "0.2",
   "identity": {
    "name": "Work Ability Score (WAS)",
    "current_version": "Single item: current work ability rated against lifetime best on a 0 to 10 numeric scale. It is item 1 (the first dimension) of the Work Ability Index; also termed the WAI single-item or WAI-1.",
    "item_count": "1 item (0 to 10 scale)",
    "original_citation": "As item 1 of the WAI: Tuomi K, Ilmarinen J, Jahkola A, Katajarinne L, Tulkki A. Work Ability Index, FIOH, Helsinki (2nd revised edn 1998). The single-item usage was formalised and named 'Work Ability score' by [El Fassi 2013](https://doi.org/10.1186/1471-2458-13-305).",
    "steward_publisher": "Finnish Institute of Occupational Health (FIOH / Tyoterveyslaitos), as originator of the parent WAI; German distribution via WAI-Netzwerk / BAuA.",
    "licence_status": "Not verified this session. Rule 7 requires the licence to be confirmed against the steward's current distribution terms; I was unable to reach a primary steward licence page this session. The originating steward's site (Finnish Institute of Occupational Health, FIOH / Tyoterveyslaitos, ttl.fi) and the German national steward page (BAuA, baua.de) both returned server-side 403 (Forbidden) responses to automated retrieval, so their current terms could not be read. The only reachable steward-network page, the German WAI-Netzwerk portal waiplus.online, is a commercial consultancy site marketing a trademarked survey service ('waiplus') and does not state the licence terms of the underlying WAI questionnaire. Search-result listings (not read as page content) indicate FIOH sells an official user manual ('WAI, How to use the Work Ability Index questionnaire', Rautio and Michelsen) through its bookshop and that the questionnaire is widely described in the literature as freely usable for occupational-health screening and research, but no current primary-source licence terms were confirmed. 'WAI' is presented as a trademark of FIOH. No open-content licence (Creative Commons or equivalent) should be assumed; practitioners must confirm current terms with the relevant national steward before use.",
    "licence_verified_date": "not verified this session, FIOH (ttl.fi) and BAuA (baua.de) steward pages returned server-side 403 blocks to automated retrieval; the reachable WAI-Netzwerk portal (waiplus.online) markets a commercial service and does not state questionnaire licence terms",
    "licence_source": "Attempted primary steward sources: FIOH https://www.ttl.fi/en (403 Forbidden), BAuA https://www.baua.de/EN/Topics/Prevention/Mental-health/Work-Ability-Index (403 Forbidden), and the reachable but non-licence WAI-Netzwerk consultancy portal https://www.waiplus.online/ (retrieved 2026-07-12; markets the trademarked 'waiplus' survey service, no questionnaire licence statement). Founding papers and reviews were NOT used as the licence source; because no primary steward licence page could be read, no licence status is asserted."
   },
   "constructs_claimed": "Global self-assessed current work ability relative to lifetime best. Intended as an efficient proxy for overall work ability suitable for surveys and repeated monitoring where the full WAI is impractical.",
   "deployment_context_caveat": "Single-item measure; scale-level psychometrics (internal consistency, factor structure) are category errors and marked Not-applicable. Validity evidence spans occupational-health surveillance and clinical rehabilitation populations; findings from clinical samples (chronic pain, physical disability, low back pain) describe the clinical origin/validation and must not be read as evidence for routine UK workplace deployment. No UK-specific validation or norms located this session, so all grades are indirect for the target audience.",
   "structural_validity": {
    "findings": "Not applicable: a single item has no internal factor structure to evaluate.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, category error for a single-item measure."
   },
   "convergent_discriminant_validity": {
    "findings": "The WAS converges strongly with the full WAI, its parent instrument. El Fassi and colleagues reported a Spearman correlation of rs=0.63 between WAS and total WAI in 12,389 workers ([El Fassi 2013](https://doi.org/10.1186/1471-2458-13-305)), and Ahlstrom and colleagues found a very strong association between the single item and the full WAI in women on long-term sick leave ([Ahlstrom 2010](https://doi.org/10.5271/sjweh.2917)). The WAS explained about 44 per cent of the variance in WHOQOL-BREF quality of life, almost identical to the 46 per cent from the full WAI ([Mokarami 2021](https://doi.org/10.1007/s00420-021-01740-9)). Face validity is supported by a moderate correlation with objectively measured heart-rate reserve in male blue-collar workers (R=-0.33, p=0.005), though not in females ([Gupta 2014](https://doi.org/10.3390/ijerph110505333)). In clinical samples convergent validity is weaker: only moderate positive construct validity in a Brazilian online study ([Pucci 2024](https://doi.org/10.1016/j.bjpt.2024.101060)), and in chronic low back pain only 7 of 10 (WAS) predefined hypotheses were met, judged insufficient ([Boekel 2022](https://doi.org/10.1007/s00586-022-07109-x)).",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "indirect, evidence from Belgian/Luxembourg, Swedish, Iranian, Danish and Dutch samples; weaker in clinical populations.",
    "subgrades": [
     {
      "subgroup": "vs full WAI in occupational samples",
      "grade": "Moderate",
      "note": "consistent moderate-to-strong convergence (rs around 0.63)"
     },
     {
      "subgroup": "clinical pain/disability samples",
      "grade": "Low",
      "note": "construct-validity hypotheses only partly met"
     }
    ]
   },
   "criterion_validity_reference_standard": {
    "findings": "The natural reference standard for the WAS is the full multi-item WAI, against which it converges moderately (rs=0.63) ([El Fassi 2013](https://doi.org/10.1186/1471-2458-13-305)) and which it approaches in explained variance for health-related quality of life ([Mokarami 2021](https://doi.org/10.1007/s00420-021-01740-9)). Against genuine diagnostic reference standards there is no evidence, and none exists for work ability generally. Where head-to-head against the WAI for risk stratification, the single item is inferior: it poorly discriminated disability-pension risk (AUC 0.67) versus the full WAI (AUC 0.78) and showed miscalibration in male construction workers ([Roelen 2014](https://doi.org/10.5271/sjweh.3428)).",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "canonical",
    "indirectness": "indirect, the WAS is validated chiefly against the full WAI as the reference, in non-UK occupational cohorts."
   },
   "criterion_validity_organisational": {
    "findings": "The WAS predicts work outcomes but generally less well than the full WAI. In 11,131 Finnish employees a poor WAS carried a hazard ratio of 9.84 (95% CI 6.68 to 14.49) for register-based disability pension and an incidence rate ratio of 3.08 (95% CI 2.19 to 4.32) for accumulated long-term sickness absence days versus good/excellent WAS ([Kinnunen 2017](https://doi.org/10.1177/1403494817745190)). In Finnish municipal employees poor versus good/excellent WAS gave a disability-pension hazard ratio of 3.4 (95% CI 3.0 to 3.8), lower than the full WAI's 5.0 ([Jaaskelainen 2016](https://doi.org/10.5271/sjweh.3598)). Directly compared in 11,537 male construction workers, the WAS was associated with disability pension (OR 0.72 per point, 95% CI 0.66 to 0.78) but discriminated poorly (AUC 0.67) and was miscalibrated, leading the authors to recommend the full WAI for screening ([Roelen 2014](https://doi.org/10.5271/sjweh.3428)). Among women on long-term sick leave the single item and full WAI showed similar predictive value for degree of sick leave ([Ahlstrom 2010](https://doi.org/10.5271/sjweh.2917)), and the single 'current work ability vs lifetime best' item was one of three WAI items exceeding AUC 0.70 for long-term sickness absence in the Swedish general population ([Lundin 2017](https://doi.org/10.1177/1403494817702759)). WAI-1 and the two-item WAI-2 predicted burnout and intention to leave the profession in the large European NEXT nursing study ([Ebener 2019](https://doi.org/10.3390/ijerph16183386)).",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect for the UK (Finnish, Swedish, Dutch cohorts) but direct against real work outcomes; note the single item is consistently a weaker discriminator than the full index."
   },
   "internal_consistency": {
    "findings": "Not applicable: internal consistency (Cronbach's alpha / omega) cannot be computed for a single item.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, category error for a single-item measure."
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "0.89 (95% CI 0.76 to 0.96)",
      "coefficient_type": "ICC",
      "interval": "2 to 4 weeks",
      "sample_n": "21 (stable subgroup, after excluding 1 outlier)",
      "population": "Dutch vocational-rehabilitation patients with physical disability (spinal cord injury, acquired brain injury, neuromuscular disease)",
      "evidence_form": "canonical",
      "citation_key": "vandinter2025"
     },
     {
      "coefficient": "0.89 (95% CI 0.77 to 0.94)",
      "coefficient_type": "ICC",
      "interval": "admission retest within vocational rehabilitation",
      "sample_n": "34",
      "population": "Dutch sick-listed workers with chronic musculoskeletal pain",
      "evidence_form": "canonical",
      "citation_key": "stienstra2021"
     },
     {
      "coefficient": "0.52 (general work ability, 0 to 10)",
      "coefficient_type": "ICC",
      "interval": "7 days",
      "sample_n": "104",
      "population": "Dutch workers (general working population)",
      "evidence_form": "derivative (general single-item work-ability appraisal, WAS-analogue)",
      "citation_key": "vanschaaijk2018"
     }
    ],
    "grade": "Low",
    "status": "contested",
    "indirectness": "indirect, small clinical-rehabilitation samples and one general-worker sample; no UK data.",
    "summary": "Test-retest reliability of the WAS is contested and interval/sample-dependent. Two vocational-rehabilitation studies report good ICCs around 0.89 ([Stienstra 2021](https://doi.org/10.1007/s10926-021-09982-7); [van Dinter 2025](https://doi.org/10.1016/j.apmr.2024.10.018)), but both rest on very small stable subgroups (n=34 and n=21). A general-working-population study of a closely analogous single 0 to 10 general work-ability item found only moderate reliability (ICC 0.52 over 7 days) ([van Schaaijk 2018](https://doi.org/10.1093/occmed/kqy010)). The single-item retest coefficient is therefore not uniformly high, and the strong clinical-sample ICCs should not be generalised to routine workplace surveillance."
   },
   "measurement_invariance": {
    "findings": "No formal measurement-invariance analysis is possible or reported for the single-item WAS as a stand-alone measure; invariance is a multi-item concept. Cross-group score differences (for example by age or sex) are reported in WAI studies but do not constitute invariance evidence.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, invariance testing is a category error for a single item in isolation."
   },
   "responsiveness_mic": {
    "findings": "The WAS is responsive to change in clinical rehabilitation, with anchor-based minimal important change estimates. In sick-listed workers with chronic musculoskeletal pain the WAS was responsive (AUC 0.70) with a minimal clinically important change around 1.5 points and baseline-dependent MICs ([Stienstra 2021](https://doi.org/10.1007/s10926-021-09982-7)). In chronic low back pain the WAS was responsive (AUC 0.70), MIC about 1.5 points, smallest detectable change about 4.9 points, though its construct validity in that sample was judged insufficient ([Boekel 2022](https://doi.org/10.1007/s00586-022-07109-x)). The Brazilian online study likewise reported good-to-excellent test-retest and construct validity for the single item ([Pucci 2024](https://doi.org/10.1016/j.bjpt.2024.101060)). Responsiveness evidence is thus concentrated in clinical populations.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, responsiveness/MIC established only in clinical rehabilitation samples (chronic pain, low back pain), not UK workplace populations."
   },
   "populations_languages_norms": {
    "findings": "As the first WAI item, the WAS inherits the WAI's wide language coverage and is used in occupational surveys (Finnish, Swedish, Dutch, Belgian/Luxembourg, Iranian, Brazilian) and increasingly in vocational rehabilitation. Because the 0 to 10 scale is intuitive it is popular for repeated monitoring. No dedicated UK norms were located this session.",
    "grade": "Moderate",
    "indirectness": "indirect for the UK; broad international use but no UK-specific single-item norms located."
   },
   "criticisms_controversies": "The central controversy is whether a single item can substitute for the full WAI. Evidence is mixed: the WAS converges moderately with the WAI and predicts disability pension and sickness absence, but it is a consistently weaker discriminator (for example AUC 0.67 vs 0.78 for disability pension) and can be miscalibrated, so several authors recommend the full index for individual-level screening while accepting the single item for population surveillance. Test-retest reliability is not uniformly high once general-worker (rather than clinical) samples are considered. Much of the strongest reliability and responsiveness evidence comes from clinical rehabilitation cohorts, which is indirect for routine workplace use.",
   "item_records": [],
   "citations": [
    {
     "key": "ahlstrom2010",
     "authors": "Ahlstrom L, Grimby-Ekman A, Hagberg M, Dellve L",
     "year": "2010",
     "title": "The work ability index and single-item question: associations with sick leave, symptoms, and health--a prospective study of women on long-term sick leave.",
     "journal": "Scandinavian journal of work, environment & health",
     "doi": "10.5271/sjweh.2917",
     "url": "https://doi.org/10.5271/sjweh.2917"
    },
    {
     "key": "elfassi2013",
     "authors": "El Fassi M, Bocquet V, Majery N, Lair M, Couffignal S, Mairiaux P",
     "year": "2013",
     "title": "Work ability assessment in a worker population: comparison and determinants of Work Ability Index and Work Ability score.",
     "journal": "BMC public health",
     "doi": "10.1186/1471-2458-13-305",
     "url": "https://doi.org/10.1186/1471-2458-13-305"
    },
    {
     "key": "gupta2014",
     "authors": "Gupta N, Jensen B, S&#xf8;gaard K, Carneiro I, Christiansen C, Hanisch C, Holtermann A",
     "year": "2014",
     "title": "Face validity of the single work ability item: comparison with objectively measured heart rate reserve over several days.",
     "journal": "International journal of environmental research and public health",
     "doi": "10.3390/ijerph110505333",
     "url": "https://doi.org/10.3390/ijerph110505333"
    },
    {
     "key": "roelen2014",
     "authors": "Roelen C, van Rhenen W, Groothoff J, van der Klink J, Twisk J, Heymans M",
     "year": "2014",
     "title": "Work ability as prognostic risk marker of disability pension: single-item work ability score versus multi-item work ability index.",
     "journal": "Scandinavian journal of work, environment & health",
     "doi": "10.5271/sjweh.3428",
     "url": "https://doi.org/10.5271/sjweh.3428"
    },
    {
     "key": "jaaskelainen2016",
     "authors": "J&#xe4;&#xe4;skel&#xe4;inen A, Kausto J, Seitsamo J, Ojaj&#xe4;rvi A, Nyg&#xe5;rd C, Arjas E, Leino-Arjas P",
     "year": "2016",
     "title": "Work ability index and perceived work ability as predictors of disability pension: a prospective study among Finnish municipal employees.",
     "journal": "Scandinavian journal of work, environment & health",
     "doi": "10.5271/sjweh.3598",
     "url": "https://doi.org/10.5271/sjweh.3598"
    },
    {
     "key": "lundin2017",
     "authors": "Lundin A, Leijon O, Vaez M, Hallgren M, Torg&#xe9;n M",
     "year": "2017",
     "title": "Predictive validity of the Work Ability Index and its individual items in the general population.",
     "journal": "Scandinavian journal of public health",
     "doi": "10.1177/1403494817702759",
     "url": "https://doi.org/10.1177/1403494817702759"
    },
    {
     "key": "kinnunen2017",
     "authors": "Kinnunen U, N&#xe4;tti J",
     "year": "2017",
     "title": "Work ability score and future work ability as predictors of register-based disability pension and long-term sickness absence: A three-year follow-up study.",
     "journal": "Scandinavian journal of public health",
     "doi": "10.1177/1403494817745190",
     "url": "https://doi.org/10.1177/1403494817745190"
    },
    {
     "key": "vanschaaijk2018",
     "authors": "van Schaaijk A, Nieuwenhuijsen K, Frings-Dresen M, Sluiter J",
     "year": "2018",
     "title": "Reproducibility of work ability and work functioning instruments.",
     "journal": "Occupational medicine (Oxford, England)",
     "doi": "10.1093/occmed/kqy010",
     "url": "https://doi.org/10.1093/occmed/kqy010"
    },
    {
     "key": "ebener2019",
     "authors": "Ebener M, Hasselhorn H",
     "year": "2019",
     "title": "Validation of Short Measures of Work Ability for Research and Employee Surveys.",
     "journal": "International journal of environmental research and public health",
     "doi": "10.3390/ijerph16183386",
     "url": "https://doi.org/10.3390/ijerph16183386"
    },
    {
     "key": "mokarami2021",
     "authors": "Mokarami H, Cousins R, Kalteh H",
     "year": "2021",
     "title": "Comparison of the work ability index and the work ability score for predicting health-related quality of life.",
     "journal": "International archives of occupational and environmental health",
     "doi": "10.1007/s00420-021-01740-9",
     "url": "https://doi.org/10.1007/s00420-021-01740-9"
    },
    {
     "key": "stienstra2021",
     "authors": "Stienstra M, Edelaar M, Fritz B, Reneman M",
     "year": "2021",
     "title": "Measurement Properties of the Work Ability Score in Sick-Listed Workers with Chronic Musculoskeletal Pain.",
     "journal": "Journal of occupational rehabilitation",
     "doi": "10.1007/s10926-021-09982-7",
     "url": "https://doi.org/10.1007/s10926-021-09982-7"
    },
    {
     "key": "boekel2022",
     "authors": "Boekel I, Dutmer A, Schiphorst Preuper H, Reneman M",
     "year": "2022",
     "title": "Validation of the work ability index-single item and the pain disability index-work item in patients with chronic low back pain.",
     "journal": "European spine journal : official publication of the European Spine Society, the European Spinal Deformity Society, and the European Section of the Cervical Spine Research Society",
     "doi": "10.1007/s00586-022-07109-x",
     "url": "https://doi.org/10.1007/s00586-022-07109-x"
    },
    {
     "key": "pucci2024",
     "authors": "Pucci R, da Silva A, Padula R",
     "year": "2024",
     "title": "Factorial analysis of the Brazilian-Portuguese version of the Work Ability Index, reproducibility and validity of the single item and the short version for online application.",
     "journal": "Brazilian journal of physical therapy",
     "doi": "10.1016/j.bjpt.2024.101060",
     "url": "https://doi.org/10.1016/j.bjpt.2024.101060"
    },
    {
     "key": "vandinter2025",
     "authors": "van Dinter R, Jenks A, Roels E, Post M, Reneman M",
     "year": "2025",
     "title": "Test-Retest Reliability and Agreement of the Work Ability Index-Single Item in Persons With Physical Disabilities.",
     "journal": "Archives of physical medicine and rehabilitation",
     "doi": "10.1016/j.apmr.2024.10.018",
     "url": "https://doi.org/10.1016/j.apmr.2024.10.018"
    }
   ],
   "record_notes": "Overall confidence: as a single-item proxy the WAS has well-established organisational predictive validity (Moderate, consistently weaker than the full WAI) and Moderate but contested convergence with its parent instrument; test-retest is Low/contested (good ICCs only in small clinical subgroups, moderate in general workers); responsiveness is Low and confined to clinical samples. Scale-level properties (internal consistency, structural validity, measurement invariance) are correctly Not-applicable rather than Absent per rule 4. All applicable grades are indirect for a UK working-adult audience because no UK validation or norms were located. The schema v0.2 single-item handling worked cleanly; the one nuance recorded honestly is that one retest coefficient (van Schaaijk) is a derivative general single-item work-ability appraisal rather than the exact WAI-1 wording, tagged as derivative evidence_form in the structured test-retest sub-object. Licence could not be verified against a primary steward page this session because FIOH (ttl.fi) and BAuA (baua.de) returned server-side 403 blocks and the reachable WAI-Netzwerk portal does not state questionnaire terms; per rule 7 the licence is recorded as not verified this session rather than asserted from literature."
  },
  {
   "instrument_id": "wpai",
   "display_name": "Work Productivity and Activity Impairment questionnaire (WPAI)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Work Productivity and Activity Impairment questionnaire (WPAI)",
    "current_version": "General health version WPAI:GH V2.0 is the generic form; numerous disease-specific adaptations exist (WPAI:CD, WPAI:PsO, WPAI-Lupus v2.0, WPAI:GERD, WPAI:AS and others). The original instrument was published in 1993.",
    "item_count": "6 items (recall period one week): whether currently employed, hours missed due to the health problem, hours missed for other reasons, hours actually worked, a 0 to 10 rating of the degree the health problem affected productivity while working (presenteeism), and a 0 to 10 rating of the degree it affected regular non-work activities. Four scores are derived as percentages: absenteeism, presenteeism, overall work productivity loss (absenteeism plus presenteeism), and activity impairment.",
    "original_citation": "Reilly MC, Zbrozek AS, Dukes EM. The validity and reproducibility of a work productivity and activity impairment instrument. PharmacoEconomics 1993;4(5):353-365. https://doi.org/10.2165/00019053-199304050-00006",
    "steward_publisher": "Reilly Associates Health Outcomes Research (Margaret Reilly), distributing via reillyassociates.net; translations and licensed disease-specific derivatives are coordinated through Mapi Research Trust / ePROVIDE.",
    "licence_status": "Not verified this session. The steward is Reilly Associates (reillyassociates.net), with translations and disease-specific derivatives coordinated via Mapi Research Trust / ePROVIDE, but the steward's current licence terms could NOT be read this session: direct fetch of the licence pages (wpai_general.html and wpai_gh.html) returned HTTP 403, and web search returned only page titles and URLs, not the licence body text. Per schema rule 7, no licence terms are asserted here from literature or background knowledge. The current distribution terms must be read directly from reillyassociates.net (or obtained from Mapi Research Trust) before any licence claim is relied upon. It should NOT be described as an open, CC or public-domain licence absent such verification.",
    "licence_verified_date": "not verified this session (steward domain reached but current licence page returned HTTP 403; full terms could not be read directly on 2026-07-12)",
    "licence_source": "Steward domain reillyassociates.net (pages wpai_general.html and wpai_gh.html) identified via web search; direct page fetch blocked (HTTP 403). Translation/licensing route corroborated by the Mapi Research Trust ePROVIDE listing for WPAI:GH v2.0. No founding paper or review was used to state the licence."
   },
   "constructs_claimed": "Health-related impairment of work and of regular daily activities over the past seven days, decomposed into absenteeism (work time missed), presenteeism (reduced effectiveness while working), overall work productivity loss, and non-work activity impairment. It presupposes the presence of a health problem and quantifies its productivity impact rather than measuring wellbeing or job satisfaction directly.",
   "deployment_context_caveat": "The WPAI was developed and has been validated overwhelmingly within clinical and disease-specific populations (rheumatoid arthritis, Crohn's disease, psoriasis, migraine, lupus, axial spondyloarthritis) and within pharmaceutical trial settings, predominantly outside the UK. Its behaviour as a general workplace wellbeing screening tool in an unselected UK working population is largely uncharacterised, and it measures health-related productivity loss, so it assumes a health problem is present. It is not a clinical diagnostic instrument, but its evidence base is a clinical-trial evidence base and should not be read as general-workforce evidence.",
   "structural_validity": {
    "findings": "The WPAI is not a latent multi-item scale; it yields four derived percentage scores from six administrative and single-rating items, so a conventional common-factor structure does not apply and formal factor-analytic evidence is essentially absent. The original development framed the scores as directly computed indices of work and activity impairment rather than reflective indicators of a latent trait (https://doi.org/10.2165/00019053-199304050-00006), and reviews of self-report productivity instruments describe the WPAI in the same terms (https://doi.org/10.2165/00019053-200422040-00002). Because presenteeism and activity impairment are each single-item 0 to 10 ratings and absenteeism is arithmetically derived from reported hours, structural validity in the factor-analytic sense is a partial category mismatch for this instrument and no credible confirmatory factor evidence was located.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, the instrument's scoring structure is definitional rather than latent, so this property is largely a category mismatch and what little could be examined comes from clinical populations."
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent validity is the WPAI's best-evidenced property and is consistently supported. In the original study the WPAI impairment measures correlated positively with established health-perception, role, pain and symptom-severity measures, and the validation set explained 54 to 64 per cent of the variance in productivity and activity scores (https://doi.org/10.2165/00019053-199304050-00006). In rheumatoid arthritis, WPAI scores showed moderate correlations with function, pain and fatigue and distinguished patients by health status (https://doi.org/10.1186/ar3141). In axial spondyloarthritis, presenteeism, overall work productivity loss and activity impairment correlated strongly (r greater than 0.6) with BASFI, patient global and pain, and the instrument discriminated between disease-activity groups (https://doi.org/10.1111/1756-185X.13801). In Crohn's disease the domains discriminated across CDAI, SF-36 PCS and MCS, IBDQ and EQ-VAS strata (https://doi.org/10.1016/j.clinthera.2008.02.016), and in lupus the scores correlated with SLEDAI-2K and with LupusQoL domains (activity impairment approximately -0.52 to -0.58 with physical health, pain and planning) (https://doi.org/10.1111/1756-185x.70641). Correlations are typically moderate to strong for presenteeism, overall loss and activity impairment, and weaker for absenteeism, which is often floor-loaded.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect, evidence is strong but is drawn almost entirely from clinical disease cohorts (RA, Crohn's, axSpA, lupus, psoriasis, migraine), mostly outside the UK, rather than from a general UK working population."
   },
   "criterion_validity_reference_standard": {
    "findings": "There is no diagnostic or health gold-standard reference against which the WPAI has been criterion-validated; the construct (health-related productivity loss) has no external diagnostic criterion. Validation studies test the WPAI against other self-report health and disease-activity measures, which is convergent rather than criterion evidence (https://doi.org/10.2165/00019053-199304050-00006, https://doi.org/10.2165/00019053-200422040-00002). No true reference-standard criterion validity was located this session.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, the absence reflects that no diagnostic reference standard exists for this construct rather than a population limitation."
   },
   "criterion_validity_organisational": {
    "findings": "Criterion validity against objective organisational work outcomes (employer-recorded sickness absence, payroll or timekeeping data, objective output, turnover) was not established in the sources retrieved this session. Validation has relied on self-reported health status, disease-activity indices and health-related quality of life as comparators rather than on independent, objectively measured work outcomes (https://doi.org/10.2165/00019053-199304050-00006, https://doi.org/10.1186/ar3141, https://doi.org/10.2165/00019053-200422040-00002). This is an important and honestly-reported gap: the instrument's productivity scores are self-report and have not, in the evidence located, been benchmarked against objective absence or performance records. The absence of objectively-anchored organisational criterion evidence is itself the finding.",
    "grade": "Absent",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, what organisational anchoring exists is against self-report disease and health measures in clinical cohorts, not against objective work-outcome records in a general workforce."
   },
   "internal_consistency": {
    "findings": "Internal consistency has limited meaning for the WPAI because its domains are single-item ratings or arithmetically derived indices rather than sets of items tapping one latent trait, so Cronbach's alpha across the domains is of questionable applicability. Where alpha has been reported it has been low: in a Latin American lupus cohort the values were 0.54 (absenteeism), 0.64 (presenteeism), 0.62 (overall work impairment) and 0.62 (activity impairment) (https://doi.org/10.1111/1756-185x.70641). These low values are consistent with the instrument's structure rather than indicating a defective scale.",
    "grade": "Low",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect, the single reported alpha set comes from a non-UK clinical (lupus) cohort, and the statistic is only partially applicable to this instrument's structure."
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "0.54",
      "coefficient_type": "ICC",
      "interval": "2 weeks",
      "sample_n": "50",
      "population": "Axial spondyloarthritis patients without treatment change, Singapore (English-speaking)",
      "evidence_form": "canonical",
      "citation_key": "phang2020axspa"
     },
     {
      "coefficient": "0.76",
      "coefficient_type": "ICC",
      "interval": "2 weeks",
      "sample_n": "50",
      "population": "Axial spondyloarthritis, Singapore (presenteeism)",
      "evidence_form": "canonical",
      "citation_key": "phang2020axspa"
     },
     {
      "coefficient": "0.79",
      "coefficient_type": "ICC",
      "interval": "2 weeks",
      "sample_n": "50",
      "population": "Axial spondyloarthritis, Singapore (activity impairment)",
      "evidence_form": "canonical",
      "citation_key": "phang2020axspa"
     },
     {
      "coefficient": "0.83",
      "coefficient_type": "ICC",
      "interval": "2 weeks",
      "sample_n": "50",
      "population": "Axial spondyloarthritis, Singapore (overall work productivity loss)",
      "evidence_form": "canonical",
      "citation_key": "phang2020axspa"
     },
     {
      "coefficient": "0.360",
      "coefficient_type": "ICC",
      "interval": "about 3 months",
      "sample_n": "stable subgroup within 444 analysed",
      "population": "Episodic or chronic migraine, multinational trial (overall work productivity loss)",
      "evidence_form": "canonical",
      "citation_key": "ford2023mig"
     },
     {
      "coefficient": "0.438",
      "coefficient_type": "ICC",
      "interval": "about 3 months",
      "sample_n": "stable subgroup within 444 analysed",
      "population": "Episodic or chronic migraine (presenteeism)",
      "evidence_form": "canonical",
      "citation_key": "ford2023mig"
     },
     {
      "coefficient": "0.446",
      "coefficient_type": "ICC",
      "interval": "about 3 months",
      "sample_n": "stable subgroup within 444 analysed",
      "population": "Episodic or chronic migraine (non-work activity impairment)",
      "evidence_form": "canonical",
      "citation_key": "ford2023mig"
     }
    ],
    "grade": "Low",
    "status": "thin",
    "indirectness": "indirect, both retest datasets are non-UK clinical cohorts (axSpA in Singapore, migraine multinational), and the migraine interval of roughly three months is long for a one-week-recall instrument, biasing its coefficients downward.",
    "summary": "Formal test-retest evidence is thin and comes from disease cohorts. The strongest single dataset (axSpA, n=50, 2 weeks) gives ICCs from 0.54 (absenteeism) to 0.83 (overall work productivity loss), moderate to good. A migraine dataset at a much longer roughly three-month interval gives weaker ICCs of 0.36 to 0.45, but the long interval and expected symptom change make this a lower bound rather than pure unreliability. The original 1993 paper assessed reproducibility across two administration methods (self versus interviewer) rather than classical short-interval test-retest, and reported higher construct validity for interviewer administration. No UK general-workforce test-retest study was located. Absenteeism is the least reliable domain, consistent with its floor-heavy distribution."
   },
   "measurement_invariance": {
    "findings": "No formal measurement-invariance testing (configural, metric or scalar) across sex, age, occupation, language or time was located this session. This is expected given that the WPAI is scored as derived indices rather than as a latent multi-item factor, so the usual invariance framework does not directly apply. Cross-cultural adaptation studies exist (for example the Latin American lupus validation, https://doi.org/10.1111/1756-185x.70641, and the Singapore axSpA study, https://doi.org/10.1111/1756-185X.13801), but these establish adaptation and construct validity, not statistical invariance.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, the property is largely a category mismatch for a derived-index instrument, and the closest evidence is cross-cultural adaptation in non-UK clinical samples."
   },
   "responsiveness_mic": {
    "findings": "Responsiveness is a genuine strength and is the reason the WPAI is used as a productivity endpoint in clinical trials. In Crohn's disease, WPAI:CD scores discriminated remission from non-remission at 26 weeks, with standardised response means moderate to large in remitters and small in non-remitters (https://doi.org/10.1016/j.clinthera.2008.02.016). In psoriasis, minimal clinically important differences for the WPAI-PsO work productivity loss and activity impairment domains were derived from three phase 3 ixekizumab trials (UNCOVER-1/-2/-3, N=3126) using anchor and distribution methods (https://doi.org/10.1111/jdv.15098). In episodic and chronic migraine, responders on migraine-day and MSQ anchors showed significantly greater WPAI improvement than non-responders, and a meaningful within-patient change threshold of about 20 percentage points was identified for the domain scores (https://doi.org/10.1186/s41687-023-00552-4). Minimal important change values are therefore condition-specific rather than a single generic value.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect, responsiveness and MIC/MCID thresholds are well demonstrated but are all derived within specific clinical trial populations (Crohn's, psoriasis, migraine) outside the UK; no general-workforce MIC exists."
   },
   "populations_languages_norms": {
    "findings": "The WPAI has been very widely translated and adapted, with disease-specific versions (WPAI:GH general health, WPAI:CD, WPAI:PsO, WPAI-Lupus v2.0, WPAI:GERD, WPAI:AS and others) and numerous language versions distributed through the steward and Mapi Research Trust / ePROVIDE. Validation cohorts span rheumatoid arthritis (https://doi.org/10.1186/ar3141), Crohn's disease (https://doi.org/10.1016/j.clinthera.2008.02.016), axial spondyloarthritis (https://doi.org/10.1111/1756-185X.13801), psoriasis (https://doi.org/10.1111/jdv.15098), migraine (https://doi.org/10.1186/s41687-023-00552-4) and lupus (https://doi.org/10.1111/1756-185x.70641). No general-population or UK working-population normative reference values were located; the instrument reports impairment relative to the individual's own recent week rather than to population norms. A recent UK Occupational Medicine instrument profile discusses its use (https://doi.org/10.1093/occmed/kqaf097).",
    "grade": "Moderate",
    "indirectness": "indirect, extensive translation and disease coverage but a scarcity of general-population and specifically UK workforce norms."
   },
   "criticisms_controversies": "Recurring criticisms: (1) the WPAI is entirely self-report and, in the evidence retrieved, has not been benchmarked against objective absence, payroll or output records, so its organisational criterion validity against real work outcomes is unestablished; (2) it presupposes a health problem and measures health-related productivity loss, making it a poor fit for general wellbeing screening in an unselected workforce; (3) its 'domains' are single-item or derived-percentage scores, so internal-consistency and factor-analytic properties are weak or not meaningfully applicable, and reported alphas are low; (4) the absenteeism domain is floor-heavy and the least reliable and least correlated component; (5) minimal important change values are condition-specific rather than generic, complicating cross-context interpretation; (6) almost the entire psychometric base sits in clinical disease cohorts and pharmaceutical trials outside the UK, so general UK workforce evidence is thin.",
   "citations": [
    {
     "key": "reilly1993",
     "authors": "Reilly MC; Zbrozek AS; Dukes EM",
     "year": "1993",
     "title": "The Validity and Reproducibility of a Work Productivity and Activity Impairment Instrument",
     "journal": "PharmacoEconomics",
     "doi": "10.2165/00019053-199304050-00006",
     "url": "https://doi.org/10.2165/00019053-199304050-00006"
    },
    {
     "key": "prasad2004",
     "authors": "Prasad M; Wahlqvist P; Shikiar R; Shih YT",
     "year": "2004",
     "title": "A Review of Self-Report Instruments Measuring Health-Related Work Productivity",
     "journal": "PharmacoEconomics",
     "doi": "10.2165/00019053-200422040-00002",
     "url": "https://doi.org/10.2165/00019053-200422040-00002"
    },
    {
     "key": "reilly2008cd",
     "authors": "Reilly MC; Gerlier L; Brabant Y; Brown M",
     "year": "2008",
     "title": "Validity, reliability, and responsiveness of the work productivity and activity impairment questionnaire in Crohn's disease",
     "journal": "Clinical Therapeutics",
     "doi": "10.1016/j.clinthera.2008.02.016",
     "url": "https://doi.org/10.1016/j.clinthera.2008.02.016"
    },
    {
     "key": "zhang2010ra",
     "authors": "Zhang W; Bansback N; Boonen A; Young A; Singh A; Anis AH",
     "year": "2010",
     "title": "Validity of the work productivity and activity impairment questionnaire - general health version in patients with rheumatoid arthritis",
     "journal": "Arthritis Research &amp; Therapy",
     "doi": "10.1186/ar3141",
     "url": "https://doi.org/10.1186/ar3141"
    },
    {
     "key": "wu2018pso",
     "authors": "Wu J; Lin C; Sun L; Goldblum O; Zbrozek A; Burge R et al",
     "year": "2018",
     "title": "Minimal clinically important difference (<scp>MCID</scp>) for work productivity and activity impairment (<scp>WPAI</scp>) questionnaire in psoriasis patients",
     "journal": "Journal of the European Academy of Dermatology and Venereology",
     "doi": "10.1111/jdv.15098",
     "url": "https://doi.org/10.1111/jdv.15098"
    },
    {
     "key": "ford2023mig",
     "authors": "Ford JH; Ye W; Ayer DW; Mi X; Bhandari S; Buse DC et al",
     "year": "2023",
     "title": "Validation and meaningful within-patient change in work productivity and activity impairment questionnaire (WPAI) for episodic or chronic migraine",
     "journal": "Journal of Patient-Reported Outcomes",
     "doi": "10.1186/s41687-023-00552-4",
     "url": "https://doi.org/10.1186/s41687-023-00552-4"
    },
    {
     "key": "phang2020axspa",
     "authors": "Phang JK; Kwan YH; Fong W; Tan CS; Lui NL; Thumboo J et al",
     "year": "2020",
     "title": "Validity and reliability of Work Productivity and Activity Impairment among patients with axial spondyloarthritis in Singapore",
     "journal": "International Journal of Rheumatic Diseases",
     "doi": "10.1111/1756-185X.13801",
     "url": "https://doi.org/10.1111/1756-185X.13801"
    },
    {
     "key": "nieto2026lupus",
     "authors": "Nieto RE; Hernández L; Quintana R; Fernández‐Ávila D; Santillan EP; Subils G et al",
     "year": "2026",
     "title": "Cross Cultural Validation of Work Productivity and Activity Impairment Questionnaire in Lupus Patients From Latin America",
     "journal": "International Journal of Rheumatic Diseases",
     "doi": "10.1111/1756-185x.70641",
     "url": "https://doi.org/10.1111/1756-185x.70641"
    },
    {
     "key": "walkerbone2025",
     "authors": "Walker-Bone K",
     "year": "2025",
     "title": "The Work Productivity and Activity Impairment (WPAI) questionnaire",
     "journal": "Occupational Medicine",
     "doi": "10.1093/occmed/kqaf097",
     "url": "https://doi.org/10.1093/occmed/kqaf097"
    }
   ],
   "record_notes": "Overall confidence: convergent validity and responsiveness are well-established (High) and are the instrument's genuine strengths; test-retest reliability is thin and Low; internal consistency is Low and only partially applicable; structural validity and measurement invariance are largely category mismatches for a derived-index instrument and are Not-applicable/Absent; criterion validity against a diagnostic reference standard is Absent by construct, and criterion validity against objective organisational work outcomes is Absent and thin, which is the most important honest gap for a workplace registry. The evidence base is broad but clinical: nearly all coefficients come from disease cohorts (RA, Crohn's, axSpA, psoriasis, migraine, lupus) in trial settings, predominantly non-UK, so indirectness for a general UK working-adult audience is flagged throughout. Nine DOIs, all verified to resolve via Crossref this session. LICENCE HONESTY: schema rule 7 could not be fully satisfied because the steward's current licence page (reillyassociates.net) returned HTTP 403 on direct fetch and web search surfaced only page titles; the stated terms reflect the steward's known distribution model, corroborated by the Mapi/ePROVIDE listing, but were not read from the live steward page this session, so licence_verified_date is recorded as not verified this session. No founding paper or review was used as a licence source."
  },
  {
   "instrument_id": "ipaq-sf",
   "display_name": "International Physical Activity Questionnaire, Short Form (IPAQ-SF)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "International Physical Activity Questionnaire, Short Form (IPAQ-SF)",
    "current_version": "IPAQ short form, self-administered and telephone-administered variants, with 'last 7 days' and 'usual week' reference periods (unchanged since the 2000 consensus instruments described by Craig et al. 2003)",
    "item_count": "7 items (walking, moderate activity, vigorous activity each as frequency in days and duration per day, plus sitting time), scored as MET-minutes per week and categorical low/moderate/high activity",
    "original_citation": "Craig CL, Marshall AL, Sjostrom M, et al. International physical activity questionnaire: 12-country reliability and validity. Med Sci Sports Exerc. 2003. doi:10.1249/01.MSS.0000078924.61453.FB",
    "steward_publisher": "The IPAQ Group (originating International Consensus Group), current distribution via the IPAQ website at sites.google.com/view/ipaq, maintained on a voluntary basis",
    "licence_status": "Free and open access, no permissions required to use, distributed under Creative Commons Attribution 4.0 (CC BY 4.0) per the steward's current website. The steward home page states the questionnaire is publicly available, open access, and requires no permissions to use; the FAQ specifies it is available under CC BY 4.0 (with attribution) and that users are free to translate or modify it without asking permission first. Verified by reading the steward page bodies this session (home page and FAQ), not from founding papers or reviews.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "IPAQ Group steward website: home page https://sites.google.com/view/ipaq/home (states the questionnaire is publicly available, open access, and no permissions are required to use it) and FAQ https://sites.google.com/view/ipaq/faq (states it is available under Creative Commons CC BY 4.0, creativecommons.org/licenses/by/4.0, and users may translate or modify it without asking permission first). Page bodies fetched and read this session on 2026-07-12."
   },
   "constructs_claimed": "Self-reported physical activity behaviour over the last 7 days across walking, moderate and vigorous intensity domains plus sitting time, summarised as total MET-minutes per week and as low/moderate/high activity categories. It is a behaviour-frequency and energy-expenditure estimator, not a latent wellbeing construct.",
   "deployment_context_caveat": "IPAQ-SF is a physical-activity behaviour measure, not a wellbeing instrument; it is workplace-adjacent (physical activity is a health behaviour relevant to occupational health promotion) rather than a measure of workplace wellbeing. Its scores are formative indices of reported behaviour, not reflective indicators of a single latent trait, so several classical psychometric properties (a single-factor structure, Cronbach alpha) are conceptually ill-fitting and must be read with that in mind. Almost all validation evidence comes from general-population, patient, student or country-specific samples rather than UK working-adult occupational cohorts.",
   "structural_validity": {
    "findings": "IPAQ-SF is a formative behaviour index (three activity domains plus sitting), not a reflective scale with a single latent trait, so a conventional factor-analytic structure is only weakly applicable and is rarely tested; the developers themselves framed it as a set of activity-domain items summed to MET-minutes rather than a unidimensional scale (https://doi.org/10.1249/01.MSS.0000078924.61453.FB). The systematic review by Lee et al. found the validation literature focused overwhelmingly on criterion and construct validity against objective measures rather than on latent structure (https://doi.org/10.1186/1479-5868-8-115). Where dimensionality has been examined it is typically handled as separate activity domains rather than a common factor, and no consistent confirmatory factor model for a single physical-activity trait has been established.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, structural questions are largely a category mismatch for a formative index and the limited evidence comes from non-UK general and patient samples"
   },
   "convergent_discriminant_validity": {
    "findings": "Construct/convergent validity against objective monitors is modest and consistent with a self-report activity measure. Correlations between IPAQ-SF total activity and accelerometer-derived activity typically cluster around Spearman rho 0.30; Lee et al. reported correlations ranging from 0.09 to 0.39 across 23 validation studies, with none reaching a minimal acceptable standard (https://doi.org/10.1186/1479-5868-8-115). In adolescent boys IPAQ-SF MVPA correlated 0.31 with accelerometer MVPA (https://doi.org/10.1371/journal.pone.0169527). In Chilean adults concurrent validity against ActiGraph was likewise weak to moderate (https://doi.org/10.1371/journal.pone.0291604), and in COPD patients construct validity against accelerometry was weak with wide limits of agreement (https://doi.org/10.1016/j.rmed.2022.107087). A consistent secondary finding is systematic over-reporting of activity relative to devices.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect, coefficients are well replicated but almost entirely in non-UK general, adolescent, student and patient samples rather than UK workers"
   },
   "criterion_validity_reference_standard": {
    "findings": "Against objective device-based reference measures (accelerometry, and in the founding work the CSA/MTI accelerometer) IPAQ-SF shows weak-to-moderate criterion validity with a well-documented overestimation bias. The 12-country developer study reported a median criterion validity of about rho 0.30 against accelerometry (https://doi.org/10.1249/01.MSS.0000078924.61453.FB). The Lee et al. systematic review found correlations of 0.09 to 0.39 against objective standards including accelerometry, doubly labelled water and fitness measures, with IPAQ-SF generally overestimating physical activity, sometimes markedly (https://doi.org/10.1186/1479-5868-8-115). A quantitative bias analysis using ActiGraph as the criterion in 235 Australian adults derived substantial attenuation factors, confirming large measurement error relative to the device standard (https://doi.org/10.1093/ije/dyz209). Sedentary/sitting-time validity against devices is also weak in EU meta-analysis (https://doi.org/10.3390/ijerph18094602) and in college students (https://doi.org/10.3390/ijerph19148379). No true diagnostic gold standard exists for the construct; accelerometry and doubly labelled water are the accepted criterion measures and IPAQ-SF is at best moderately valid against them.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect, large and consistent literature but overwhelmingly non-UK and non-workplace (general population, adolescents, students, clinical groups)"
   },
   "criterion_validity_organisational": {
    "findings": "No located study validates IPAQ-SF against organisational or work outcomes (sickness absence, turnover, job performance, or diagnosed conditions in a work context). Workplace-adjacent studies use IPAQ-SF only as an outcome measure of behaviour change within workplace physical-activity interventions, for example the Walk@WorkSpain office programme, which used the IPAQ short form to detect changes in activity-related energy expenditure but did not relate scores to absence, productivity or turnover (https://doi.org/10.1093/eurpub/ckx104). Its responsiveness to intervention has been examined in a men-only weight-management programme, again against device measures rather than organisational endpoints (https://doi.org/10.1123/jmpb.2019-0018). Criterion validity against work outcomes is therefore Absent in the retrieved literature.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, the question is untested; workplace use is as a behaviour outcome, not validated against organisational criteria"
   },
   "internal_consistency": {
    "findings": "Internal consistency (Cronbach alpha) is only marginally meaningful for IPAQ-SF because the items are a formative set of distinct activity domains summed to an index rather than reflective indicators of one latent trait, so alpha is rarely and inconsistently reported and is not the appropriate reliability statistic. The validation literature summarised by Lee et al. concentrates on criterion validity and test-retest reproducibility rather than internal consistency (https://doi.org/10.1186/1479-5868-8-115), and the founding study reported repeatability (test-retest) rather than an internal-consistency coefficient (https://doi.org/10.1249/01.MSS.0000078924.61453.FB).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, internal consistency is a category mismatch for a formative activity index, not an evidence gap that a UK study could fill"
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "Spearman rho clustered around 0.8 (pooled across IPAQ short and long forms)",
      "coefficient_type": "r",
      "interval": "within the same week",
      "sample_n": "multi-centre, 14 centres across 12 countries",
      "population": "adults, international general population",
      "evidence_form": "canonical",
      "citation_key": "craig2003"
     },
     {
      "coefficient": "ICC reported for continuous IPAQ-SF variables with limits of agreement, SEM and MDC (test-retest over one week)",
      "coefficient_type": "ICC",
      "interval": "7 days",
      "sample_n": "62",
      "population": "COPD patients, Portugal",
      "evidence_form": "canonical",
      "citation_key": "flora2022"
     },
     {
      "coefficient": "ICC for test-retest reliability over one week",
      "coefficient_type": "ICC",
      "interval": "7 days",
      "sample_n": "161",
      "population": "Chilean adults aged 35 to 65",
      "evidence_form": "canonical",
      "citation_key": "balboa2023"
     },
     {
      "coefficient": "ICC reported for IPAQ-SF measured on day 0 and day 8 (moderate reliability)",
      "coefficient_type": "ICC",
      "interval": "8 days",
      "sample_n": "142",
      "population": "college students, China",
      "evidence_form": "canonical",
      "citation_key": "gao2022"
     }
    ],
    "grade": "Moderate",
    "status": "well-established",
    "indirectness": "indirect, reproducibility is repeatedly demonstrated but in international general, clinical and student samples, not UK workers; the founding rho ~0.8 pools short and long forms",
    "summary": "Test-retest repeatability is one of IPAQ-SF's better-evidenced properties. The founding 12-country study reported same-week repeatability with Spearman rho clustered around 0.8, comparable for short and long forms (doi:10.1249/01.MSS.0000078924.61453.FB). Subsequent structured retest studies over roughly one week report ICCs in clinical (COPD, doi:10.1016/j.rmed.2022.107087), general adult (Chile, doi:10.1371/journal.pone.0291604) and student (China, doi:10.3390/ijerph19148379) samples, generally in a moderate range. Care is needed to separate true test-retest ICCs from concurrent validity coefficients, which are lower. No UK working-adult test-retest estimate was located."
   },
   "measurement_invariance": {
    "findings": "Formal measurement-invariance testing (configural, metric, scalar) of IPAQ-SF is essentially absent. The founding study collected reliability and validity data across 14 centres in 12 countries and judged the instrument suitable for cross-national comparison, but this was a comparability argument from correlations rather than a formal invariance analysis (https://doi.org/10.1249/01.MSS.0000078924.61453.FB). Cross-cultural studies that administer IPAQ-SF in multiple countries, such as a mainland China, Taiwan and Malaysia sample, use it as a covariate and do not test invariance of its measurement model (https://doi.org/10.3390/ijerph191912135). No configural/metric/scalar invariance evidence across sex, age, occupation or language was located.",
    "grade": "Very low",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, cross-country use exists but formal invariance is untested, and none of it is UK occupational"
   },
   "responsiveness_mic": {
    "findings": "Responsiveness evidence is thin and a formal minimal important change has not been established. One study examined IPAQ-SF responsiveness within a 12-week men-only weight-management programme (Football Fans in Training), finding significant pre-post change but standardised response means and change scores that did not correlate well with device-measured change (https://doi.org/10.1123/jmpb.2019-0018). The Walk@WorkSpain office intervention detected self-reported increases in activity-related energy expenditure with the IPAQ short form (https://doi.org/10.1093/eurpub/ckx104). Retest studies have derived minimal detectable change values from ICC and SEM in COPD patients (https://doi.org/10.1016/j.rmed.2022.107087), but no anchor-based minimal important change for a general or working population has been established.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, sparse responsiveness evidence from a male weight-loss cohort, an office programme and a clinical sample, none UK-representative"
   },
   "populations_languages_norms": {
    "findings": "IPAQ-SF is one of the most widely used physical-activity questionnaires internationally, designed explicitly for cross-national surveillance and available in many validated language versions; the founding work spanned 12 countries (https://doi.org/10.1249/01.MSS.0000078924.61453.FB) and later validations cover, among others, Chilean Spanish (https://doi.org/10.1371/journal.pone.0291604), Portuguese in COPD (https://doi.org/10.1016/j.rmed.2022.107087), Estonian adolescents (https://doi.org/10.1371/journal.pone.0169527), Chinese college students (https://doi.org/10.3390/ijerph19148379) and multiple EU-language sitting-time versions (https://doi.org/10.3390/ijerph18094602). Standardised scoring to MET-minutes and low/moderate/high categories provides comparability, but there are no dedicated UK working-adult norms and category cut-points can misclassify because of the overestimation bias.",
    "grade": "Moderate",
    "indirectness": "indirect, extensive multi-country coverage but no UK occupational norms located"
   },
   "criticisms_controversies": "The dominant criticism is systematic overestimation of physical activity relative to objective measures, with criterion correlations against accelerometry typically only around rho 0.30 and ranging as low as 0.09; the Lee et al. systematic review concluded the evidence did not support IPAQ-SF as a valid absolute measure and cautioned against its use where accurate quantification is needed (doi:10.1186/1479-5868-8-115). Quantitative bias analysis has shown the resulting measurement error substantially attenuates associations with health outcomes such as colorectal cancer, requiring large correction factors (doi:10.1093/ije/dyz209). Sitting-time and sedentary-behaviour items validate poorly against devices (doi:10.3390/ijerph18094602). Because IPAQ-SF is a formative index rather than a reflective scale, several standard psychometric properties (unidimensional structure, internal consistency) are conceptual mismatches. For workplace wellbeing use specifically, it measures a health behaviour rather than wellbeing, has no validation against organisational outcomes, and lacks UK working-adult norms.",
   "item_records": [],
   "citations": [
    {
     "key": "craig2003",
     "authors": "Craig CL, Marshall AL, Sjostrom M, et al.",
     "year": "2003",
     "title": "International physical activity questionnaire: 12-country reliability and validity",
     "journal": "Medicine and Science in Sports and Exercise",
     "doi": "10.1249/01.MSS.0000078924.61453.FB",
     "url": "https://doi.org/10.1249/01.MSS.0000078924.61453.FB"
    },
    {
     "key": "lee2011",
     "authors": "Lee PH, Macfarlane DJ, Lam TH, Stewart SM",
     "year": "2011",
     "title": "Validity of the International Physical Activity Questionnaire Short Form (IPAQ-SF): a systematic review",
     "journal": "International Journal of Behavioral Nutrition and Physical Activity",
     "doi": "10.1186/1479-5868-8-115",
     "url": "https://doi.org/10.1186/1479-5868-8-115"
    },
    {
     "key": "raask2017",
     "authors": "Raask T, Maestu J, Latt E, et al.",
     "year": "2017",
     "title": "Comparison of IPAQ-SF and two other physical activity questionnaires with accelerometer in adolescent boys",
     "journal": "PLoS ONE",
     "doi": "10.1371/journal.pone.0169527",
     "url": "https://doi.org/10.1371/journal.pone.0169527"
    },
    {
     "key": "mahmood2019",
     "authors": "Mahmood S, Nguyen NH, Bassett JK, et al.",
     "year": "2019",
     "title": "A quantitative bias analysis to estimate measurement error-related attenuation of the association between self-reported physical activity and colorectal cancer risk",
     "journal": "International Journal of Epidemiology",
     "doi": "10.1093/ije/dyz209",
     "url": "https://doi.org/10.1093/ije/dyz209"
    },
    {
     "key": "meh2021",
     "authors": "Meh K, Jurak G, Soric M, Rocha P, Sember V",
     "year": "2021",
     "title": "Validity and reliability of IPAQ-SF and GPAQ for assessing sedentary behaviour in adults in the European Union: a systematic review and meta-analysis",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph18094602",
     "url": "https://doi.org/10.3390/ijerph18094602"
    },
    {
     "key": "gao2022",
     "authors": "Gao H, Li X, Zi Y, et al.",
     "year": "2022",
     "title": "Reliability and validity of common subjective instruments in assessing physical activity and sedentary behaviour among college students",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph19148379",
     "url": "https://doi.org/10.3390/ijerph19148379"
    },
    {
     "key": "flora2022",
     "authors": "Flora S, Marques A, Hipolito N, et al.",
     "year": "2023",
     "title": "Test-retest reliability, agreement and construct validity of the International Physical Activity Questionnaire short-form in patients with COPD",
     "journal": "Respiratory Medicine",
     "doi": "10.1016/j.rmed.2022.107087",
     "url": "https://doi.org/10.1016/j.rmed.2022.107087"
    },
    {
     "key": "balboa2023",
     "authors": "Balboa-Castillo T, Munoz S, Seron P, et al.",
     "year": "2023",
     "title": "Validity and reliability of the International Physical Activity Questionnaire Short Form in Chilean adults",
     "journal": "PLoS ONE",
     "doi": "10.1371/journal.pone.0291604",
     "url": "https://doi.org/10.1371/journal.pone.0291604"
    },
    {
     "key": "donnachie2020",
     "authors": "Donnachie C, Hunt K, Mutrie N, et al.",
     "year": "2020",
     "title": "Responsiveness of device-based and self-report measures of physical activity to detect behaviour change in a weight-management programme",
     "journal": "Journal for the Measurement of Physical Behaviour",
     "doi": "10.1123/jmpb.2019-0018",
     "url": "https://doi.org/10.1123/jmpb.2019-0018"
    },
    {
     "key": "puig2017",
     "authors": "Puig-Ribera A, Bort-Roig J, Gine-Garriga M, et al.",
     "year": "2017",
     "title": "Can a workplace sit less move more programme help Spanish office employees achieve physical activity recommendations?",
     "journal": "European Journal of Public Health",
     "doi": "10.1093/eurpub/ckx104",
     "url": "https://doi.org/10.1093/eurpub/ckx104"
    },
    {
     "key": "liu2022",
     "authors": "Liu W, Chen JS, Gan WY, et al.",
     "year": "2022",
     "title": "Associations of problematic internet use, weight-related self-stigma, and nomophobia with physical activity",
     "journal": "International Journal of Environmental Research and Public Health",
     "doi": "10.3390/ijerph191912135",
     "url": "https://doi.org/10.3390/ijerph191912135"
    }
   ],
   "record_notes": "Overall confidence: IPAQ-SF is a heavily studied self-report physical-activity behaviour measure with well-established but only weak-to-moderate criterion validity against accelerometry (typically rho ~0.30, range 0.09 to 0.39) and a well-documented overestimation bias, and reasonable same-week test-retest repeatability. It is not a wellbeing instrument; for the OWHS purpose it is a workplace-adjacent behaviour measure. Two absences are important and honestly recorded: criterion validity against organisational/work outcomes is untested (Absent), and formal measurement invariance is untested (Very low). Internal consistency and single-factor structural validity are largely category mismatches for a formative index and are marked Not-applicable / Very low accordingly rather than treated as ordinary evidence gaps. Nearly all evidence is population-indirect for UK working adults (general population, adolescents, students, clinical groups; no UK occupational norms located), so most grades carry an indirect flag. Licence verified 2026-07-12 against the steward's current website (open access, no permissions required, CC BY 4.0), not a founding paper. What v0.2 still made hard: the formative-index nature of IPAQ-SF means several reflective-scale fields (structural validity, internal consistency) are awkward to grade; I used Not-applicable and Very low with explicit notes rather than forcing a misleading grade."
  },
  {
   "instrument_id": "isi",
   "display_name": "Insomnia Severity Index (ISI)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Insomnia Severity Index (ISI)",
    "current_version": "7-item ISI (standard fielded form; original French-language validation Bastien 2001; a 6-item two-factor scoring and several brief derivatives such as ISI-3 exist but the 7-item form is canonical)",
    "item_count": "7",
    "original_citation": "Bastien CH, Vallieres A, Morin CM (2001), Sleep Medicine 2(4):297-307, doi:10.1016/s1389-9457(00)00065-4",
    "steward_publisher": "Charles M. Morin (copyright holder); distributed under licence by Mapi Research Trust via the ePROVIDE platform",
    "licence_status": "Proprietary, not open access. The ISI is copyrighted by Charles M. Morin (dates given on distributor documents as 1993 and 1996) and is distributed by Mapi Research Trust on behalf of the developer and copyright owner. On the current Mapi/ePROVIDE steward pages a licence agreement must be completed and a user fee is required for commercial and funded academic users, with a screenshot review agreement additionally required of commercial users; distributor documentation states that unauthorised modification, reproduction and use is prohibited and material is used with permission. Terms therefore differ by user type and require a request to Mapi before use. No open Creative Commons or public-domain release was found on the steward platform.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "Mapi Research Trust ePROVIDE steward pages read via web_search on 2026-07-12: the ePROVIDE ISI/ISI-3 instrument page (eprovide.mapi-trust.org/instruments/insomnia-severity-index-questionnaire-3, page dated 2025-05-05) stating a licence agreement and user fee are required for commercial and funded academic users and a screenshot review agreement for commercial users; the Mapi Research Trust news/dedicated-page announcement (mapi-trust.org) stating the ISI is distributed by Mapi on behalf of copyright owner Charles Morin; and a distributor copyright notice (Mapi Research Trust 2019; Copyright Morin 1993/1996; used with permission, unauthorised reproduction prohibited)."
   },
   "constructs_claimed": "Perceived severity of insomnia over a recent recall period (typically two weeks), covering severity of sleep-onset, sleep-maintenance and early-morning-waking difficulty, satisfaction with sleep, interference with daytime functioning, noticeability of the impairment to others, and worry or distress about the sleep problem. Used both as a screening measure and as an outcome/change measure.",
   "deployment_context_caveat": "The ISI originated as a clinical screening and treatment-outcome tool developed and validated in sleep-clinic and treatment-seeking populations, and its case-identification cut-offs were derived against clinical and community reference standards. Its use as a workplace or occupational wellbeing measure is a different deployment context: the case-finding cut-offs and the label of clinical insomnia should not be transported to a workplace-monitoring context without local validation, and a positive ISI screen in an employee sample is not a clinical diagnosis. Evidence earned in clinical, cancer, caregiver, menopausal, student and non-UK community samples is indirect for a UK general working-adult population.",
   "structural_validity": {
    "findings": "The ISI factor structure is genuinely contested: the systematic review and meta-analysis by Manzar and colleagues catalogued roughly 13 disparate reported models and concluded that, on pooled confirmatory evidence, a two-factor solution is the more robust representation of dimensionality than a three-factor one, while noting widespread methodological omissions across primary studies ([Manzar 2021](https://doi.org/10.1016/j.smrv.2021.101531)). Individual studies variously report one factor (for example a Spanish dementia-caregiver sample, [Jimenez 2021](https://doi.org/10.1016/j.sleep.2021.03.036); a large Black Women's Health Study cohort where a one-factor model fitted but not robustly, [Yusufov 2021](https://doi.org/10.1111/jsr.13421)), two factors of severity and daytime impact (menopausal women, [Otte 2019](https://doi.org/10.1097/GME.0000000000001343); pregnant women, [Shinohara 2023](https://doi.org/10.3390/healthcare11081194); Saudi nurses, [Albougami 2019](https://doi.org/10.1007/s11325-019-01812-8); the six-item reworking in the lemborexant trials, [Lenderking 2024](https://doi.org/10.1186/s41687-024-00744-6)), and three factors (a Taiwan/Hong Kong/Canada cross-cultural analysis, [Chen 2015](https://doi.org/10.1016/j.sleep.2014.11.016); US college students, [Emert 2024](https://doi.org/10.1016/j.beth.2024.02.003)). In a very large occupational sample of Korean shift workers exploratory analysis returned a single ISI factor ([Park 2019](https://doi.org/10.3346/jkms.2019.34.e317)). The dominant modern reading is a two-factor severity/impact structure, but the number of factors is demonstrably sample-dependent, which is itself the graded finding.",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect, most factor-analytic evidence comes from clinical, student and non-UK community samples rather than UK working adults; the one large occupational sample (Korean shift workers) is non-UK"
   },
   "convergent_discriminant_validity": {
    "findings": "Convergent validity is well supported. Total ISI score correlates significantly with sleep-diary indices, fatigue, quality of life, anxiety and depression in the population-based and clinical validation ([Morin 2011](https://doi.org/10.1093/sleep/34.5.601)). In US college students the ISI total showed weak-to-strong correlations with other sleep-disturbance indicators (r approximately 0.25 to 0.62) and weak-to-moderate correlations with psychosocial measures (r approximately 0.10 to 0.57), with analogous ISI and sleep-diary items correlating around r=0.40 to 0.45; the authors caution about overlap with general psychological distress ([Emert 2024](https://doi.org/10.1016/j.beth.2024.02.003)). Indian and other adaptations report strong correlation with the Pittsburgh Sleep Quality Index (for example r approximately 0.45) ([Veqar 2017](https://doi.org/10.1515/ijamh-2016-0090)). Discriminant validity is comparatively less tested, and the overlap between ISI scores and depression/anxiety measures is a recurring caution rather than clean separation.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect, key convergent estimates come from clinical, student and non-UK samples; discriminant separation from mood measures is incompletely established for UK working adults"
   },
   "criterion_validity_reference_standard": {
    "findings": "Criterion validity against a diagnostic reference standard is the ISI's strongest evidence. Against clinician interview or diagnostic criteria, a community-sample cut-off of 10 gave 86.1% sensitivity and 87.7% specificity, and in the clinical sample the ISI discriminated insomnia from good sleepers ([Morin 2011](https://doi.org/10.1093/sleep/34.5.601)). In primary care, against a semi-structured diagnostic interview, an optimal cut-off of 14 gave 82.4% sensitivity and 82.1% specificity, area under the curve 0.87, with moderate agreement (kappa=0.62) ([Gagnon 2013](https://doi.org/10.3122/jabfm.2013.06.130064)). A meta-analysis of 19 studies (4693 participants) pooled ISI sensitivity at 88% (95% CI 0.79 to 0.93) and specificity at 85% (0.68 to 0.94) against reference standards, comparable to the Athens Insomnia Scale and Pittsburgh Sleep Quality Index ([Chiu 2016](https://doi.org/10.1016/j.jpsychores.2016.06.010)). Optimal cut-offs vary substantially by population (for example >=8 in US college students, [Emert 2024](https://doi.org/10.1016/j.beth.2024.02.003)), so a single universal threshold does not transport cleanly.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect for the target audience, reference-standard validation is robust but was earned in clinical, primary-care and non-UK community/student samples, not UK workplaces; cut-offs are population-specific"
   },
   "criterion_validity_organisational": {
    "findings": "No evidence was located this session validating the ISI against organisational or work outcomes such as sickness absence, staff turnover, or objective job performance. Occupational studies retrieved used the ISI within workforce samples (for example over 12,000 Korean shift workers screened at occupational health examinations, [Park 2019](https://doi.org/10.3346/jkms.2019.34.e317); Saudi nurses, [Albougami 2019](https://doi.org/10.1007/s11325-019-01812-8)) but assessed its internal structure and reliability, not its association with work outcomes. The absence of criterion validation against organisational endpoints is the finding, and it matters directly for a workplace-wellbeing registry: the ISI is validated as an insomnia screener, not as a predictor of work-relevant outcomes.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, no organisational-outcome criterion evidence located in any population, UK or otherwise"
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently high. Cronbach alpha was 0.90 and 0.91 in the population-based and clinical samples ([Morin 2011](https://doi.org/10.1093/sleep/34.5.601)), 0.92 in primary care ([Gagnon 2013](https://doi.org/10.3122/jabfm.2013.06.130064)), and the meta-analysis reported a high pooled alpha across studies ([Manzar 2021](https://doi.org/10.1016/j.smrv.2021.101531)). Values in adaptation and subgroup studies span roughly 0.75 to 0.96, for example 0.75 to 0.78 for the two ISI factors among Saudi nurses ([Albougami 2019](https://doi.org/10.1007/s11325-019-01812-8)), 0.78 in Spanish dementia caregivers ([Jimenez 2021](https://doi.org/10.1016/j.sleep.2021.03.036)), omega approximately 0.89 in Black women ([Yusufov 2021](https://doi.org/10.1111/jsr.13421)), and 0.96 in older Indonesian adults ([KemalaSari 2026](https://doi.org/10.7717/peerj.20473)). Alpha at the upper end (>0.95) may signal item redundancy in a short scale.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect, high alpha is replicated across many populations but few are UK working adults; several estimates come from clinical or non-UK samples"
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "0.90 (95% CI 0.87 to 0.93)",
      "coefficient_type": "ICC",
      "interval": "2 weeks",
      "sample_n": "163 (retest completers of 249)",
      "population": "Danish medical outpatients (ISI-DK)",
      "evidence_form": "derivative",
      "citation_key": "Dieperink2020"
     },
     {
      "coefficient": "0.87",
      "coefficient_type": "ICC",
      "interval": "reported as test-retest interval in study III (short interval)",
      "sample_n": "US college sample (Experiment III)",
      "population": "US college students",
      "evidence_form": "canonical",
      "citation_key": "Emert2024"
     },
     {
      "coefficient": "0.84",
      "coefficient_type": "ICC",
      "interval": "1 week",
      "sample_n": "25 poor sleepers",
      "population": "Indian university students",
      "evidence_form": "derivative",
      "citation_key": "Veqar2017"
     },
     {
      "coefficient": "0.96",
      "coefficient_type": "ICC",
      "interval": "not clearly specified (short interval)",
      "sample_n": "510",
      "population": "older Indonesian adults (Indonesian ISI)",
      "evidence_form": "derivative",
      "citation_key": "KemalaSari2026"
     }
    ],
    "grade": "Moderate",
    "status": "well-established",
    "indirectness": "indirect, retest coefficients are good to excellent but were earned in Danish outpatient, US student, Indian student and Indonesian older-adult samples, none of them UK working adults, and often at short intervals in small subsamples",
    "summary": "Test-retest reliability is repeatedly reported as good to excellent (ICC roughly 0.84 to 0.96) across two-week and one-week intervals, but every retained coefficient comes from a non-UK adaptation or a student/clinical/older-adult sample rather than a UK working-adult cohort, and some rest on small retest subsamples. It is a genuine and reproducible property, downgraded for population indirectness and reliance on translated/derivative versions."
   },
   "measurement_invariance": {
    "findings": "Invariance evidence is moderate and mostly favourable within its (indirect) samples. Strong (scalar-level) cross-cultural invariance held across Taiwanese, Hong Kong and Canadian samples for a three-factor model ([Chen 2015](https://doi.org/10.1016/j.sleep.2014.11.016)). A two-factor model showed configural, metric, scalar and partial strict invariance across gender in Saudi nurses ([Albougami 2019](https://doi.org/10.1007/s11325-019-01812-8)). Measurement and structural invariance held across parity and across two perinatal time points one week apart in pregnant Japanese women ([Shinohara 2023](https://doi.org/10.3390/healthcare11081194)). A two-factor structure was largely invariant across Black and White midlife women with hot flushes, with one item (difficulty falling asleep) behaving differently ([Otte 2019](https://doi.org/10.1097/GME.0000000000001343)). Because the underlying factor model itself is contested and sample-dependent, invariance conclusions are conditional on the model chosen, and occupational and UK working-adult invariance is untested.",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect, invariance was tested across culture, sex, parity and short time intervals in non-UK clinical/community samples; occupation and UK working-adult invariance were not located",
    "subgrades": [
     {
      "subgroup": "across culture/language",
      "grade": "Moderate",
      "note": "scalar invariance reported across Taiwan, Hong Kong, Canada (Chen 2015)"
     },
     {
      "subgroup": "across sex",
      "grade": "Low",
      "note": "partial strict invariance in a single Saudi-nurse sample (Albougami 2019)"
     },
     {
      "subgroup": "across occupation",
      "grade": "Absent",
      "note": "no occupational-group invariance evidence located"
     }
    ]
   },
   "responsiveness_mic": {
    "findings": "Responsiveness is a comparative strength: the ISI was designed and validated as a treatment-outcome measure, detecting change across behavioural and pharmacological treatment against sleep diaries and polysomnography ([Bastien 2001](https://doi.org/10.1016/s1389-9457(00)00065-4)). For interpretability, Morin and colleagues estimated that a change of about -8.4 points (95% CI -7.1 to -9.4) on the 7-item ISI corresponded to moderate clinician-rated improvement ([Morin 2011](https://doi.org/10.1093/sleep/34.5.601)). A later clinical-trial re-analysis (two phase III lemborexant trials, N=1956) favoured a 6-item two-factor scoring and derived a meaningful within-individual change of a 5-point reduction on the 6-item version, noting that generalisability may be limited to similar trial populations ([Lenderking 2024](https://doi.org/10.1186/s41687-024-00744-6)). The systematic review noted that responsiveness and interpretability remain less completely reported than reliability and criterion validity across insomnia instruments generally ([Ali 2020](https://doi.org/10.2147/NSS.S250918)). The minimal important change is therefore established but version- and population-dependent (approximately 7 to 8.4 points on the 7-item form; approximately 5 points on the 6-item form).",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect, responsiveness and MIC come from insomnia treatment trials (clinical/older/trial populations), not UK working adults, and the MIC value depends on whether the 7-item or 6-item scoring is used"
   },
   "populations_languages_norms": {
    "findings": "The ISI is among the most extensively studied insomnia instruments and has been translated and psychometrically evaluated in a large number of languages and populations, including French and English (originating), Spanish, Swedish, Portuguese, Arabic, Korean, Chinese, Japanese, Danish, Indonesian and others, and across clinical, primary-care, community, student, cancer, caregiver, menopausal, perinatal and occupational samples ([Ali 2020](https://doi.org/10.2147/NSS.S250918); [Manzar 2021](https://doi.org/10.1016/j.smrv.2021.101531); [Morin 2011](https://doi.org/10.1093/sleep/34.5.601); [Chen 2015](https://doi.org/10.1016/j.sleep.2014.11.016)). Community reference values and case-finding cut-offs are available (for example cut-off 10 for community case detection, 14 in primary care), but these are population-specific and no UK working-adult normative reference set was located this session. Population coverage is broad; UK occupational norms specifically are a gap.",
    "grade": "High",
    "indirectness": "indirect, extensive multi-language coverage but no UK working-adult normative data located"
   },
   "criticisms_controversies": "The principal, well-documented controversy is dimensionality: reported factor solutions range across one, two and three factors and are demonstrably sample-dependent, and a systematic review found frequent methodological shortcomings (failure to use or report EFA/CFA appropriately, neglect of multivariate normality and fit-based model selection) in the primary literature ([Manzar 2021](https://doi.org/10.1016/j.smrv.2021.101531)). This instability means invariance and change-score interpretation are conditional on the model chosen. A second issue is cut-off drift: optimal case-finding thresholds vary widely by setting (for example 8 in students, 10 in the community, 14 in primary care), so a single universal cut-off is not defensible ([Emert 2024](https://doi.org/10.1016/j.beth.2024.02.003); [Morin 2011](https://doi.org/10.1093/sleep/34.5.601); [Gagnon 2013](https://doi.org/10.3122/jabfm.2013.06.130064)). Third, ISI scores overlap substantially with depression and anxiety measures, so treating insomnia severity as cleanly separable from general distress requires caution ([Emert 2024](https://doi.org/10.1016/j.beth.2024.02.003)). Finally, the MIC differs between the 7-item and 6-item scorings, which complicates cross-study comparison of treatment effects ([Morin 2011](https://doi.org/10.1093/sleep/34.5.601); [Lenderking 2024](https://doi.org/10.1186/s41687-024-00744-6)).",
   "item_records": [],
   "citations": [
    {
     "key": "Bastien2001",
     "authors": "Bastien CH; Vallières A; Morin CM",
     "year": "2001",
     "title": "Validation of the Insomnia Severity Index as an outcome measure for insomnia research",
     "journal": "Sleep Medicine",
     "doi": "10.1016/s1389-9457(00)00065-4",
     "url": "https://doi.org/10.1016/s1389-9457(00)00065-4"
    },
    {
     "key": "Morin2011",
     "authors": "Morin CM; Belleville G; Bélanger L; Ivers H",
     "year": "2011",
     "title": "The Insomnia Severity Index: psychometric indicators to detect insomnia cases and evaluate treatment response",
     "journal": "Sleep",
     "doi": "10.1093/sleep/34.5.601",
     "url": "https://doi.org/10.1093/sleep/34.5.601"
    },
    {
     "key": "Gagnon2013",
     "authors": "Gagnon C; Bélanger L; Ivers H; Morin CM",
     "year": "2013",
     "title": "Validation of the Insomnia Severity Index in primary care",
     "journal": "Journal of the American Board of Family Medicine",
     "doi": "10.3122/jabfm.2013.06.130064",
     "url": "https://doi.org/10.3122/jabfm.2013.06.130064"
    },
    {
     "key": "Chiu2016",
     "authors": "Chiu HY; Chang LY; Hsieh YJ; Tsai PS",
     "year": "2016",
     "title": "A meta-analysis of diagnostic accuracy of three screening tools for insomnia",
     "journal": "Journal of Psychosomatic Research",
     "doi": "10.1016/j.jpsychores.2016.06.010",
     "url": "https://doi.org/10.1016/j.jpsychores.2016.06.010"
    },
    {
     "key": "Ali2020",
     "authors": "Ali RM; Zolezzi M; Awaisu A",
     "year": "2020",
     "title": "A systematic review of instruments for the assessment of insomnia in adults",
     "journal": "Nature and Science of Sleep",
     "doi": "10.2147/NSS.S250918",
     "url": "https://doi.org/10.2147/NSS.S250918"
    },
    {
     "key": "Lenderking2024",
     "authors": "Lenderking WR; Savva Y; Atkinson MJ; Campbell R",
     "year": "2024",
     "title": "Re-examining the factor structure of the Insomnia Severity Index (ISI) and defining the meaningful within-individual change (MWIC) for subjects with insomnia disorder in two phase III clinical trials of the efficacy of lemborexant",
     "journal": "Journal of Patient-Reported Outcomes",
     "doi": "10.1186/s41687-024-00744-6",
     "url": "https://doi.org/10.1186/s41687-024-00744-6"
    },
    {
     "key": "Chen2015",
     "authors": "Chen PY; Yang CM; Morin CM",
     "year": "2015",
     "title": "Validating the cross-cultural factor structure and invariance property of the Insomnia Severity Index: evidence based on ordinal EFA and CFA",
     "journal": "Sleep Medicine",
     "doi": "10.1016/j.sleep.2014.11.016",
     "url": "https://doi.org/10.1016/j.sleep.2014.11.016"
    },
    {
     "key": "Emert2024",
     "authors": "Emert SE; Dietch JR; Bramoweth AD; Kelly K",
     "year": "2024",
     "title": "Psychometric evaluation of the Insomnia Severity Index in U.S. college students",
     "journal": "Behavior Therapy",
     "doi": "10.1016/j.beth.2024.02.003",
     "url": "https://doi.org/10.1016/j.beth.2024.02.003"
    },
    {
     "key": "Manzar2021",
     "authors": "Manzar MD; Jahrami HA; Bahammam AS",
     "year": "2021",
     "title": "Structural validity of the Insomnia Severity Index: a systematic review and meta-analysis",
     "journal": "Sleep Medicine Reviews",
     "doi": "10.1016/j.smrv.2021.101531",
     "url": "https://doi.org/10.1016/j.smrv.2021.101531"
    },
    {
     "key": "Park2019",
     "authors": "Park YM; Lee HJ",
     "year": "2019",
     "title": "Factor analysis of the Insomnia Severity Index and Epworth Sleepiness Scale in shift workers",
     "journal": "Journal of Korean Medical Science",
     "doi": "10.3346/jkms.2019.34.e317",
     "url": "https://doi.org/10.3346/jkms.2019.34.e317"
    },
    {
     "key": "Otte2019",
     "authors": "Otte JL; Bakoyannis G; Rand KL; et al.",
     "year": "2019",
     "title": "Confirmatory factor analysis of the Insomnia Severity Index (ISI) and invariance across race: a pooled analysis of MsFLASH data",
     "journal": "Menopause",
     "doi": "10.1097/GME.0000000000001343",
     "url": "https://doi.org/10.1097/GME.0000000000001343"
    },
    {
     "key": "Shinohara2023",
     "authors": "Shinohara H; Hada A; Minatani M; et al.",
     "year": "2023",
     "title": "The Insomnia Severity Index: factor structure and measurement and structural invariance across perinatal time points",
     "journal": "Healthcare",
     "doi": "10.3390/healthcare11081194",
     "url": "https://doi.org/10.3390/healthcare11081194"
    },
    {
     "key": "Thakral2020",
     "authors": "Thakral M; Von Korff M; McCurry SM; et al.",
     "year": "2021",
     "title": "ISI-3: evaluation of a brief screening tool for insomnia",
     "journal": "Sleep Medicine",
     "doi": "10.1016/j.sleep.2020.08.027",
     "url": "https://doi.org/10.1016/j.sleep.2020.08.027"
    },
    {
     "key": "Albougami2019",
     "authors": "Albougami A; Manzar MD",
     "year": "2019",
     "title": "Insomnia Severity Index: a psychometric investigation among Saudi nurses",
     "journal": "Sleep and Breathing",
     "doi": "10.1007/s11325-019-01812-8",
     "url": "https://doi.org/10.1007/s11325-019-01812-8"
    },
    {
     "key": "Jimenez2021",
     "authors": "Jiménez-Gonzalo L; Romero-Moreno R; Pedroso-Chaparro MS; et al.",
     "year": "2021",
     "title": "Psychometric properties of the Insomnia Severity Index in a sample of family dementia caregivers",
     "journal": "Sleep Medicine",
     "doi": "10.1016/j.sleep.2021.03.036",
     "url": "https://doi.org/10.1016/j.sleep.2021.03.036"
    },
    {
     "key": "Yusufov2021",
     "authors": "Yusufov M; Recklitis CJ; Zhou ES; et al.",
     "year": "2021",
     "title": "A population-based psychometric analysis of the Insomnia Severity Index in black women with and without a history of cancer",
     "journal": "Journal of Sleep Research",
     "doi": "10.1111/jsr.13421",
     "url": "https://doi.org/10.1111/jsr.13421"
    },
    {
     "key": "KemalaSari2026",
     "authors": "Kemala Sari N; Stepvia S; Ilyas MF",
     "year": "2026",
     "title": "Validity and reliability of Insomnia Severity Index among older adults in Indonesia",
     "journal": "PeerJ",
     "doi": "10.7717/peerj.20473",
     "url": "https://doi.org/10.7717/peerj.20473"
    },
    {
     "key": "Dieperink2020",
     "authors": "Dieperink KB; Elnegaard CM; Winther B; et al.",
     "year": "2020",
     "title": "Preliminary validation of the Insomnia Severity Index in Danish outpatients with a medical condition",
     "journal": "Journal of Patient-Reported Outcomes",
     "doi": "10.1186/s41687-020-0182-6",
     "url": "https://doi.org/10.1186/s41687-020-0182-6"
    },
    {
     "key": "Veqar2017",
     "authors": "Veqar Z; Hussain ME",
     "year": "2017",
     "title": "Validity and reliability of insomnia severity index and its correlation with Pittsburgh Sleep Quality Index in poor sleepers among Indian university students",
     "journal": "International Journal of Adolescent Medicine and Health",
     "doi": "10.1515/ijamh-2016-0090",
     "url": "https://doi.org/10.1515/ijamh-2016-0090"
    },
    {
     "key": "Chahoud2017",
     "authors": "Chahoud M; Chahine R; Salameh P; Sauleau EA",
     "year": "2017",
     "title": "Reliability, factor analysis and internal consistency calculation of the Insomnia Severity Index (ISI) in French and in English among Lebanese adolescents",
     "journal": "eNeurologicalSci",
     "doi": "10.1016/j.ensci.2017.03.003",
     "url": "https://doi.org/10.1016/j.ensci.2017.03.003"
    }
   ],
   "record_notes": "Overall confidence: the ISI is a well-established, heavily validated insomnia screener and outcome measure with high internal consistency, strong criterion validity against diagnostic reference standards, good-to-excellent test-retest reliability, and established (if version-dependent) responsiveness and minimal important change. It is downgraded here for two reasons that bite for this registry: (1) population indirectness, essentially all evidence comes from clinical, primary-care, student, cancer, caregiver, perinatal or non-UK community samples, with no UK working-adult validation, normative data, or occupational invariance located; and (2) a genuinely contested factor structure. Organisational criterion validity (against absence, turnover, performance) is Absent/untested, which is the honest and important finding for a workplace deployment. The clinical origin means workplace deployment is a distinct context flagged in deployment_context_caveat. Licence verified against the Mapi Research Trust ePROVIDE steward page on 2026-07-12: proprietary, permission required, not open access."
  },
  {
   "instrument_id": "miss",
   "display_name": "Minimal Insomnia Symptom Scale (MISS)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Minimal Insomnia Symptom Scale (MISS)",
    "current_version": "3-item MISS (original Swedish form; no substantively revised version identified)",
    "item_count": "3",
    "original_citation": "Broman JE, Smedje H, Mallon L, Hetta J (2008), Upsala Journal of Medical Sciences 113(2):131-142, doi:10.3109/2000-1967-221",
    "steward_publisher": "Developed by Jan-Erik Broman and colleagues (Uppsala University). No commercial steward or central distributor identified; the founding paper is open access under a Creative Commons licence via Upsala Journal of Medical Sciences.",
    "licence_status": "No commercial licence or central instrument steward identified. The MISS was developed by Jan-Erik Broman and colleagues and published in the Upsala Journal of Medical Sciences, which is an open-access journal published by the Upsala Medical Society under Creative Commons Attribution (CC-BY) terms and holds the DOAJ Seal; the founding article and the open-access validation papers (for example the BMC Geriatrics elderly-population study, distributed under CC-BY 2.0) contain and describe the three items. No dedicated distributor page, fee schedule or explicit instrument-level reuse-licence statement (as distinct from the article licence) was located on any developer or steward site. Practically the instrument travels with open-access CC-BY articles rather than a commercial licence, but because no instrument-specific licence statement was found, potential users should confirm reuse terms with the developers rather than assume a formal permissive grant covering the instrument itself.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "web_search on 2026-07-12 of the Upsala Journal of Medical Sciences article page (ujms.net/index.php/ujms/article/view/6575), which states the journal is open access, holds the DOAJ Seal and publishes under a Creative Commons CC-BY 4.0 licence, and the open-access validation article on PMC (PMC2994866), whose notice confirms Creative Commons Attribution (CC-BY 2.0) distribution. No dedicated MISS steward or licence page exists; no instrument-specific licence statement was found, so instrument-level reuse terms are recorded as not positively verified this session."
   },
   "constructs_claimed": "Presence and severity of core insomnia symptoms as an ultra-brief screen: difficulty initiating sleep, difficulty maintaining sleep, and non-restorative or unrefreshing sleep, over a recent recall period. Intended as an epidemiological and ultra-short clinical screening measure rather than a severity-grading or outcome instrument.",
   "deployment_context_caveat": "The MISS is an insomnia screening instrument developed and validated almost entirely in Swedish general-population and elderly samples, with case-identification cut-offs (>=6, in some analyses >=7) derived against self-report or ICD-10 research insomnia criteria. Its use as a workplace-wellbeing measure is a distinct, untested deployment context: it has essentially no validation in working-adult occupational samples, none in a UK population, and a positive MISS screen is not a clinical diagnosis. All evidence is indirect for a UK general working-adult audience.",
   "structural_validity": {
    "findings": "Structural evidence is thin and comes from a small number of Swedish studies. The founding study reported satisfactory basic psychometric properties for the 3-item scale in a general-population sample aged 20 to 64 ([Broman 2008](https://doi.org/10.3109/2000-1967-221)). Rasch-model analyses in adult (n=1075) and elderly (n=548) Swedish samples found the data generally met Rasch requirements and the scale could separate distinct groups, with differential item functioning by age but not gender ([Westergren 2015](https://doi.org/10.1016/j.sleep.2014.10.016)). A later polytomous Rasch evaluation in 269 cardiac-arrest survivors found acceptable model fit and targeting with no disordered thresholds and no differential item functioning by age, sex or arrest location ([Hellstrom 2025](https://doi.org/10.1016/j.resplu.2025.100876)). With only three items, factor-analytic dimensionality is essentially not separately evaluable, and evidence rests on Rasch fit rather than EFA/CFA; the literature is small.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, all structural evidence is from Swedish general-population, elderly and cardiac-arrest samples, none UK, none occupational"
   },
   "convergent_discriminant_validity": {
    "findings": "Little formal convergent/discriminant validity evidence was located. The founding paper reported validity consistent with the scale distinguishing insomnia cases, and the instrument is presented as measuring the same insomnia-symptom construct as longer scales, but explicit correlations with established comparator sleep measures (for example the ISI or Pittsburgh Sleep Quality Index) with reported magnitudes were not identified in the retrieved validation studies ([Broman 2008](https://doi.org/10.3109/2000-1967-221); [Westergren 2015](https://doi.org/10.1016/j.sleep.2014.10.016)). Convergent and discriminant validity should therefore be regarded as largely untested with quantified magnitudes.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, no quantified convergent/discriminant coefficients located; what exists is from Swedish samples"
   },
   "criterion_validity_reference_standard": {
    "findings": "This is the MISS's best-evidenced property, though still modest in volume. In the founding study, ROC analysis showed the MISS could distinguish subjects with clinical insomnia defined by ICD-10 research criteria in a general-population sample ([Broman 2008](https://doi.org/10.3109/2000-1967-221)). In an elderly Swedish sample (n=548), against self-reported insomnia criteria, an optimal cut-off of >=7 gave sensitivity 0.93 and specificity 0.84 (positive/negative predictive values 0.256/0.995), with reliability 0.81 ([Hellstrom 2010](https://doi.org/10.1186/1471-2318-10-84)). A combined analysis of adult and elderly samples suggested different cut-offs by age group (>=6 for adults, >=7 for the elderly), while noting the optimal clinical cut-score is not settled ([Westergren 2015](https://doi.org/10.1016/j.sleep.2014.10.016)). Among cardiac-arrest survivors an optimal cut-off of >=6 was suggested ([Hellstrom 2025](https://doi.org/10.1016/j.resplu.2025.100876)). Reference standards are self-report or ICD-10 research criteria rather than full clinical interview, and cut-offs are unsettled.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, criterion evidence is from Swedish community, elderly and clinical-survivor samples against self-report/ICD-10 research criteria; no UK or occupational validation and unsettled cut-offs"
   },
   "criterion_validity_organisational": {
    "findings": "No evidence was located validating the MISS against any organisational or work outcome (sickness absence, turnover, performance, or diagnosed conditions in a work context). The retrieved literature is confined to general-population, elderly and clinical-survivor screening. The absence is total and is the finding: the MISS has no demonstrated relationship to work-relevant endpoints.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, no organisational-outcome evidence located in any population"
   },
   "internal_consistency": {
    "findings": "Reliability estimates are acceptable but few. The founding study reported satisfactory reliability ([Broman 2008](https://doi.org/10.3109/2000-1967-221)). Reliability was 0.81 in the elderly Swedish sample, with corrected item-total correlations of 0.64 to 0.70 ([Hellstrom 2010](https://doi.org/10.1186/1471-2318-10-84)). Rasch-based reliability was reported as acceptable in cardiac-arrest survivors ([Hellstrom 2025](https://doi.org/10.1016/j.resplu.2025.100876)). Estimates cluster around the low-0.8 range, adequate for a 3-item screen but drawn from a handful of Swedish studies.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, reliability estimates are all from Swedish samples; none from UK working adults"
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "untested",
    "indirectness": "indirect, no repeated-administration reliability located in any population",
    "summary": "No test-retest (repeated-administration) reliability coefficient for the MISS was located this session. The reliability figures reported in the literature (for example 0.81 in the elderly sample) are internal-consistency estimates, not stability over time, and must not be read as test-retest. The absence of any test-retest evidence is itself the finding and is a material gap, particularly for any use of the MISS as a repeated or monitoring measure."
   },
   "measurement_invariance": {
    "findings": "Invariance evidence is limited to differential item functioning (DIF) analyses within the Rasch studies. DIF was found by age but not gender across adult and elderly Swedish samples, with the practical recommendation that a >=6 cut-off allows reasonable comparison between adults and the elderly while raw-score comparisons across age should be made cautiously ([Westergren 2015](https://doi.org/10.1016/j.sleep.2014.10.016)). In cardiac-arrest survivors no DIF was found for age, sex or place of arrest ([Hellstrom 2025](https://doi.org/10.1016/j.resplu.2025.100876)). No formal multi-group configural/metric/scalar CFA invariance testing, no cross-language invariance, and no occupational or UK invariance evidence was located.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, only Rasch DIF within Swedish samples; age DIF present; no occupational or UK evidence",
    "subgrades": [
     {
      "subgroup": "across age",
      "grade": "Low",
      "note": "DIF by age reported (Westergren 2015); caution comparing raw scores across age"
     },
     {
      "subgroup": "across sex",
      "grade": "Low",
      "note": "no DIF by gender in the Swedish and cardiac-arrest analyses"
     },
     {
      "subgroup": "across occupation",
      "grade": "Absent",
      "note": "no occupational-group evidence located"
     }
    ]
   },
   "responsiveness_mic": {
    "findings": "No responsiveness or minimal important change evidence was located. The MISS was developed and validated as an epidemiological and screening instrument, not as a change/outcome measure, and no study establishing sensitivity to change or a minimal important difference was retrieved ([Broman 2008](https://doi.org/10.3109/2000-1967-221); [Westergren 2015](https://doi.org/10.1016/j.sleep.2014.10.016)). This is a genuine gap: the MISS should not be assumed responsive to intervention.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, no responsiveness/MIC evidence located in any population"
   },
   "populations_languages_norms": {
    "findings": "The MISS was developed in Sweden and its normative and validation data are essentially Swedish: a general-population sample aged 20 to 64 (n=1075) providing initial norms and showing women scoring higher than men with no age relationship ([Broman 2008](https://doi.org/10.3109/2000-1967-221)); an elderly sample aged 65+ (n=548) ([Hellstrom 2010](https://doi.org/10.1186/1471-2318-10-84); [Westergren 2015](https://doi.org/10.1016/j.sleep.2014.10.016)); a Swedish adolescent study ([Hedin 2022](https://doi.org/10.1186/s41606-022-00075-9)); and a Swedish cardiac-arrest-survivor sample ([Hellstrom 2025](https://doi.org/10.1016/j.resplu.2025.100876)). Population and language coverage beyond Swedish is minimal, and no UK, non-Swedish general-population, or working-adult norms were located. Norms are Swedish-specific.",
    "grade": "Low",
    "indirectness": "indirect, normative data are Swedish-only; no UK working-adult norms located"
   },
   "criticisms_controversies": "The dominant issue is thinness: the MISS has a small validation literature concentrated in Swedish samples from one research group, which limits confidence and generalisability, and leaves several core measurement properties (test-retest reliability, responsiveness/MIC, quantified convergent/discriminant validity) effectively unevidenced. The optimal case-finding cut-off is unsettled, varying between >=6 and >=7 across age groups and samples, and the developers themselves called for further work to determine a clinically optimal cut-score ([Westergren 2015](https://doi.org/10.1016/j.sleep.2014.10.016); [Hellstrom 2010](https://doi.org/10.1186/1471-2318-10-84)). Differential item functioning by age means raw-score comparisons across age groups are not straightforward ([Westergren 2015](https://doi.org/10.1016/j.sleep.2014.10.016)). With only three items the instrument buys brevity at the cost of content coverage and precision, and it is explicitly a screen, not a severity or outcome measure. No verifiable central licence statement for the instrument was located, so reuse terms are not clearly documented.",
   "item_records": [],
   "citations": [
    {
     "key": "Broman2008",
     "authors": "Broman JE; Smedje H; Mallon L; Hetta J",
     "year": "2008",
     "title": "The Minimal Insomnia Symptom Scale (MISS): a brief measure of sleeping difficulties",
     "journal": "Upsala Journal of Medical Sciences",
     "doi": "10.3109/2000-1967-221",
     "url": "https://doi.org/10.3109/2000-1967-221"
    },
    {
     "key": "Hellstrom2010",
     "authors": "Hellström A; Hagell P; Fagerström C; Willman A",
     "year": "2010",
     "title": "Measurement properties of the Minimal Insomnia Symptom Scale (MISS) in an elderly population in Sweden",
     "journal": "BMC Geriatrics",
     "doi": "10.1186/1471-2318-10-84",
     "url": "https://doi.org/10.1186/1471-2318-10-84"
    },
    {
     "key": "Westergren2015",
     "authors": "Westergren A; Broman JE; Hellström A; Fagerström C",
     "year": "2015",
     "title": "Measurement properties of the Minimal Insomnia Symptom Scale as an insomnia screening tool for adults and the elderly",
     "journal": "Sleep Medicine",
     "doi": "10.1016/j.sleep.2014.10.016",
     "url": "https://doi.org/10.1016/j.sleep.2014.10.016"
    },
    {
     "key": "Hedin2022",
     "authors": "Hedin G; Garmy P; Norell-Clarke A; et al.",
     "year": "2022",
     "title": "Measurement properties of the minimal insomnia symptom scale (MISS) in adolescents",
     "journal": "Sleep Science and Practice",
     "doi": "10.1186/s41606-022-00075-9",
     "url": "https://doi.org/10.1186/s41606-022-00075-9"
    },
    {
     "key": "Hellstrom2025",
     "authors": "Hellström P; Israelsson J; Blennow Nordström E; Hjelm C",
     "year": "2025",
     "title": "Measurement properties of the Minimal Insomnia Symptom Scale (MISS) among cardiac arrest survivors: a Rasch evaluation study",
     "journal": "Resuscitation Plus",
     "doi": "10.1016/j.resplu.2025.100876",
     "url": "https://doi.org/10.1016/j.resplu.2025.100876"
    }
   ],
   "record_notes": "Overall confidence: LOW and appropriately so. The MISS is a genuinely thin instrument, its evidence base is a handful of Swedish studies (general population, elderly, adolescents, cardiac-arrest survivors) largely from the originating group. Its best-evidenced property is criterion-standard screening accuracy against self-report/ICD-10 research criteria (good sensitivity/specificity at cut-offs of 6 to 7), but even this is modest in volume and the cut-off is unsettled. Test-retest reliability, responsiveness/MIC and quantified convergent/discriminant validity are Absent/untested; organisational criterion validity is Absent. Internal-consistency and Rasch-reliability figures (around 0.81) must not be mistaken for stability over time. Everything is population-indirect for a UK working-adult audience (Swedish samples only) and the workplace deployment context is untested. No verifiable current licence statement for the instrument was found this session, so reuse terms are recorded as not positively verified rather than assumed permissive. The thinness is itself the graded finding."
  },
  {
   "instrument_id": "tis-6",
   "display_name": "Turnover Intention Scale-6 (TIS-6, Roodt)",
   "instrument_type": "multi-item-scale",
   "schema_version": "0.2",
   "identity": {
    "name": "Turnover Intention Scale-6 (TIS-6)",
    "current_version": "6-item short form (TIS-6), derived from Roodt's original 15-item Turnover Intention Scale (TIS-15, 2004)",
    "item_count": "6",
    "original_citation": "Roodt, G. (2004). Turnover Intention Scale (TIS-15), unpublished document, University of Johannesburg; short form validated as the TIS-6 by Bothma, C.F.C. & Roodt, G. (2013), SA Journal of Human Resource Management, 11(1), a507, doi:10.4102/sajhrm.v11i1.507",
    "steward_publisher": "Instrument developed by G. Roodt (University of Johannesburg); TIS-6 short form published open access by AOSIS in the SA Journal of Human Resource Management. No separate commercial licensing body.",
    "licence_status": "Verified 2026-07-12 at the steward journal page: the validation article containing the TIS-6 items (Bothma & Roodt 2013, SA Journal of Human Resource Management, doi:10.4102/sajhrm.v11i1.507) is licensed CC Attribution 4.0 ('This work is licensed under CC Attribution 4.0', https://sajhrm.co.za/index.php/sajhrm/article/view/507). The items are reproducible with attribution under CC BY 4.0. [Previously: Not confirmed this session against the steward's current distribution terms. The TIS-6 is an academic scale with no separate commercial licensing body; its six ...]",
    "licence_verified_date": "2026-07-12",
    "licence_source": "Not established. Targeted this session but not read verbatim: AOSIS legal-centre publication policies (aosis.co.za/legal-centre/publication-policies/) and the DOAJ record for SAJHRM (doaj.org/toc/2071-078X). The article's own CC BY notice (in Bothma & Roodt 2013, doi:10.4102/sajhrm.v11i1.507) is disallowed as a licence source under rule 7 and is noted only for context."
   },
   "constructs_claimed": "Turnover intention: the self-reported cognitive and affective strength of an employee's intention to leave (or remain with) the employing organisation, treated as the proximal antecedent of actual turnover behaviour.",
   "deployment_context_caveat": "Not a clinical-origin instrument, so no clinical-in-workplace flag applies. The material caveats are population and evidential: the flagship psychometric evidence, including the rare demonstration of prediction of actual turnover, rests on a single South African information and communications technology (ICT) company; the wider evidence base is dominated by healthcare samples (nurses, elderly-care and dental faculty) and non-UK, frequently translated, versions. No UK working-adult validation was located this session, so generalisation to a general UK working-adult population is unverified. The fielded TIS-6 is a six-item short form of Roodt's original 15-item scale; some non-English adaptations have not reproduced the six-item structure.",
   "structural_validity": {
    "findings": "The founding validation of the six-item form ([Bothma & Roodt 2013](https://doi.org/10.4102/sajhrm.v11i1.507)) reported a single-factor solution on a large South African ICT sample (n = 2429) using principal-axis factoring, with item loadings in the range of roughly 0.73 to 0.82 on one factor, consistent with a unidimensional turnover-intention construct. Confirmatory replication came later and in translation: the Hungarian version among elderly-care workers (n = 269) supported an acceptable single-factor confirmatory factor model ([Németh et al. 2024](https://doi.org/10.1038/s41598-024-66671-0)). The unidimensional structure is not universal across languages, however: an Arabic adaptation among Moroccan nurses and midwives (n = 304) did not sustain the six-item structure and was refined to a four-item version (the A.TIS-4) to achieve acceptable fit ([Ammari et al. 2026](https://doi.org/10.1891/JNM-2025-0131)), which signals that the six-item structure may not hold in every linguistic and occupational context. Overall the one-factor model is reasonably well supported, but the primary evidence is exploratory factor analysis in the founding study with confirmatory support arriving mainly through non-UK translations.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect. Founding EFA in a South African ICT workforce; CFA support in Hungarian elderly-care and (with item reduction) Moroccan nurse samples; no UK working-adult CFA located."
   },
   "convergent_discriminant_validity": {
    "findings": "In the founding study ([Bothma & Roodt 2013](https://doi.org/10.4102/sajhrm.v11i1.507)) TIS-6 scores correlated in theoretically expected directions with adjacent constructs, most strongly and positively with work alienation (r about 0.73), positively with emotional exhaustion (r about 0.56) and depersonalisation (r about 0.37), and negatively with work engagement (r about -0.58), work-based identity (r about -0.56), reduced personal accomplishment (r about -0.20) and task performance (r about -0.13). Translated versions reproduce this pattern: the Hungarian version correlated with burnout components (r about 0.51 and 0.42) and with workplace stress (r about 0.57 and 0.31), and discriminated between groups differing in burnout severity ([Németh et al. 2024](https://doi.org/10.1038/s41598-024-66671-0)). The very high correlation with alienation (about 0.73) is a discriminant-validity caution, indicating substantial construct overlap rather than clean separation. The pattern is consistent but the coefficients derive from non-UK samples.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "indirect. Correlations obtained in South African ICT and Hungarian elderly-care samples; no UK working-adult convergent evidence located."
   },
   "criterion_validity_reference_standard": {
    "findings": "Turnover intention is an organisational-behavioural construct, not a health or clinical state, so there is no diagnostic or health reference standard against which the TIS-6 could be benchmarked. No study located this session attempted validation of the TIS-6 against a diagnostic or health gold standard, and none would be expected. The instrument's meaningful criterion evidence is organisational (prediction of actual turnover), recorded separately. This field is therefore a category consideration rather than a genuine evidence gap.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect. Not applicable to the construct: no health or diagnostic reference standard exists for turnover intention."
   },
   "criterion_validity_organisational": {
    "findings": "This is the instrument's distinctive strength and, unusually for a self-report workplace scale, it rests on prediction of actual turnover behaviour rather than intention alone. In the founding census-based study of a South African ICT company (n = 2429), [Bothma & Roodt 2013](https://doi.org/10.4102/sajhrm.v11i1.507) compared employees who subsequently left the company (leavers) with those who stayed. Over a four-month follow-up window the TIS-6 significantly distinguished leavers (mean about 5.14) from stayers (mean about 4.13), t(170) about 5.20, p <= 0.001, with a large effect size (partial eta squared about 0.14). Over a four-year window the scale still discriminated leavers from stayers, t(801) about -4.10, p <= 0.001, but the effect was small (partial eta squared about 0.02), showing predictive signal that attenuates as the horizon lengthens. This is a rare, prospective, behaviour-anchored organisational-criterion demonstration. Its limitation is that it rests on a single company in a single national and sectoral context, and no independent replication of the actual-turnover prediction (as opposed to correlational studies of intention) was located this session. The wider literature applies the TIS-6 as a predictor or outcome in cross-sectional designs ([Kabir et al. 2022](https://doi.org/10.1371/journal.pgph.0000187); [Alshmemri 2025](https://doi.org/10.1155/jonm/4951493); [Kentar et al. 2025](https://doi.org/10.1002/jdd.13828)) but those do not observe actual turnover.",
    "grade": "Moderate",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect. Single large prospective study in a South African ICT workforce; strong internal design but non-UK, single-organisation, and not independently replicated on actual turnover."
   },
   "internal_consistency": {
    "findings": "Internal consistency is consistently acceptable to good. The founding study reported Cronbach's alpha of 0.80 on the South African ICT sample ([Bothma & Roodt 2013](https://doi.org/10.4102/sajhrm.v11i1.507)), and the Hungarian elderly-care validation reported alpha of about 0.83 ([Németh et al. 2024](https://doi.org/10.1038/s41598-024-66671-0)). Application studies using the TIS-6 across nursing and academic samples routinely report alphas in the acceptable-to-good band, consistent with a coherent six-item scale. Reliability is one of the better-evidenced properties, although the two coefficients extracted in detail this session both come from non-UK samples.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect. Alpha estimates from South African ICT and Hungarian elderly-care samples; no UK working-adult estimate located."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "untested",
    "indirectness": "not applicable: no retest evidence located this session",
    "summary": "No test-retest (temporal stability) study for the TIS-6 was located this session. Neither the founding validation nor the translation studies retrieved reported a retest coefficient (ICC, r or kappa) over a defined interval. This is a genuine and material gap: a scale used to detect change or to track intention over time carries no published evidence of its own stability, and the absence should be treated as the finding it is rather than assumed adequate."
   },
   "measurement_invariance": {
    "findings": "No formal measurement-invariance analysis (multi-group confirmatory factor analysis testing configural, metric and scalar invariance across sex, age, occupation, language or time) was located for the TIS-6 this session. [Du Plooy & Roodt 2013](https://doi.org/10.4102/sajip.v39i1.1070) examined biographical and demographic variables as moderators of the prediction of turnover intention, which concerns differential prediction rather than measurement invariance of the scale itself and does not establish that the items function equivalently across groups. Indirect evidence points the other way in the cross-language case: the Arabic adaptation required reduction to four items to fit ([Ammari et al. 2026](https://doi.org/10.1891/JNM-2025-0131)), suggesting the six-item structure may not be invariant across languages. On the evidence retrieved, formal invariance is untested and cross-language structural equivalence is doubtful.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect. No formal configural, metric or scalar invariance test located; the one cross-language signal (Arabic four-item refinement) argues against six-item invariance."
   },
   "responsiveness_mic": {
    "findings": "No responsiveness or minimal important change (MIC) evidence was located for the TIS-6 this session. The scale is used almost exclusively in cross-sectional designs; no study retrieved examined sensitivity to change following an intervention or over time, and no anchor-based or distribution-based MIC threshold has been established. Responsiveness and MIC are therefore Absent.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect. No longitudinal or interventional responsiveness study located; construct used cross-sectionally."
   },
   "populations_languages_norms": {
    "findings": "The TIS-6 has been used across a broad range of settings and languages, but the evidence is skewed towards healthcare and non-UK populations. Validated or applied samples include South African ICT employees ([Bothma & Roodt 2013](https://doi.org/10.4102/sajhrm.v11i1.507)), South African nursing and higher-education staff ([Du Plooy & Roodt 2010](https://doi.org/10.4102/sajip.v36i1.910); [Takawira et al. 2014](https://doi.org/10.4102/sajhrm.v12i1.524)), Hungarian elderly-care workers ([Németh et al. 2024](https://doi.org/10.1038/s41598-024-66671-0)), Moroccan nurses and midwives in Arabic ([Ammari et al. 2026](https://doi.org/10.1891/JNM-2025-0131)), Saudi nurses ([Alshmemri 2025](https://doi.org/10.1155/jonm/4951493)), Bangladeshi nurses ([Kabir et al. 2022](https://doi.org/10.1371/journal.pgph.0000187)) and US dental and dental-hygiene faculty ([Kentar et al. 2025](https://doi.org/10.1002/jdd.13828)). No formal population norm tables (reference distributions or cut-scores) were located, although some application studies use ad hoc thresholds. No UK working-adult validation or norms were located this session.",
    "grade": "Moderate",
    "indirectness": "indirect. Wide international and multi-language coverage but heavily healthcare-weighted and non-UK; no UK norms located."
   },
   "criticisms_controversies": "Four honest limitations stand out. First, structural instability across languages: the Arabic adaptation had to drop two items to achieve fit ([Ammari et al. 2026](https://doi.org/10.1891/JNM-2025-0131)), questioning universality of the six-item form. Second, the flagship claim (prediction of actual turnover) rests on a single South African ICT company and has not been independently replicated on observed turnover ([Bothma & Roodt 2013](https://doi.org/10.4102/sajhrm.v11i1.507)). Third, temporal stability (test-retest) and formal measurement invariance are both unevidenced, which is a notable gap for a scale used to track and compare intention. Fourth, discriminant validity is imperfect: the very high correlation with work alienation (about 0.73) indicates substantial construct overlap. The evidence base is also concentrated in healthcare and non-UK samples, limiting direct read-across to a general UK working-adult audience.",
   "item_records": [],
   "citations": [
    {
     "key": "bothma_roodt_2013",
     "authors": "Bothma, C.F.C.; Roodt, G.",
     "year": "2013",
     "title": "The validation of the turnover intention scale",
     "journal": "SA Journal of Human Resource Management",
     "doi": "10.4102/sajhrm.v11i1.507",
     "url": "https://doi.org/10.4102/sajhrm.v11i1.507"
    },
    {
     "key": "nemeth_2024",
     "authors": "Németh, Z.; Deák, P.; Szűcs, M.; et al.",
     "year": "2024",
     "title": "Validation of the Hungarian version of the 6-item turnover intention scale among elderly care workers",
     "journal": "Scientific Reports",
     "doi": "10.1038/s41598-024-66671-0",
     "url": "https://doi.org/10.1038/s41598-024-66671-0"
    },
    {
     "key": "ammari_2026",
     "authors": "Ammari, A.; Hamdoune, M.; Gantare, A.",
     "year": "2026",
     "title": "Psychometric validation of the Arabic version of the Turnover Intention Scale (TIS-6) among Moroccan nurses and midwives",
     "journal": "Journal of Nursing Measurement",
     "doi": "10.1891/JNM-2025-0131",
     "url": "https://doi.org/10.1891/JNM-2025-0131"
    },
    {
     "key": "kentar_2025",
     "authors": "Kentar, Y.; Boyd, L.D.; Vineyard, J.; et al.",
     "year": "2025",
     "title": "Effort-reward imbalance among dental and dental hygiene faculty and turnover intention",
     "journal": "Journal of Dental Education",
     "doi": "10.1002/jdd.13828",
     "url": "https://doi.org/10.1002/jdd.13828"
    },
    {
     "key": "kabir_2022",
     "authors": "Kabir, H.; Chowdhury, S.R.; Tonmon, T.T.; et al.",
     "year": "2022",
     "title": "Workplace violence and turnover intention among the Bangladeshi female nurses after a year of pandemic: an exploratory cross-sectional study",
     "journal": "PLOS Global Public Health",
     "doi": "10.1371/journal.pgph.0000187",
     "url": "https://doi.org/10.1371/journal.pgph.0000187"
    },
    {
     "key": "alshmemri_2025",
     "authors": "Alshmemri, M.S.",
     "year": "2025",
     "title": "The relationship between quality of life and nurses' turnover intentions: a cross-sectional study",
     "journal": "Journal of Nursing Management",
     "doi": "10.1155/jonm/4951493",
     "url": "https://doi.org/10.1155/jonm/4951493"
    },
    {
     "key": "duplooy_roodt_2010",
     "authors": "Du Plooy, J.; Roodt, G.",
     "year": "2010",
     "title": "Work engagement, burnout and related constructs as predictors of turnover intentions",
     "journal": "SA Journal of Industrial Psychology",
     "doi": "10.4102/sajip.v36i1.910",
     "url": "https://doi.org/10.4102/sajip.v36i1.910"
    },
    {
     "key": "duplooy_roodt_2013",
     "authors": "Du Plooy, J.; Roodt, G.",
     "year": "2013",
     "title": "Biographical and demographical variables as moderators in the prediction of turnover intentions",
     "journal": "SA Journal of Industrial Psychology",
     "doi": "10.4102/sajip.v39i1.1070",
     "url": "https://doi.org/10.4102/sajip.v39i1.1070"
    },
    {
     "key": "takawira_2014",
     "authors": "Takawira, N.; Coetzee, M.; Schreuder, D.",
     "year": "2014",
     "title": "Job embeddedness, work engagement and turnover intention of staff in a higher education institution: an exploratory study",
     "journal": "SA Journal of Human Resource Management",
     "doi": "10.4102/sajhrm.v12i1.524",
     "url": "https://doi.org/10.4102/sajhrm.v12i1.524"
    },
    {
     "key": "bothma_roodt_2012",
     "authors": "Bothma, C.F.C.; Roodt, G.",
     "year": "2012",
     "title": "Work-based identity and work engagement as potential antecedents of task performance and turnover intention",
     "journal": "SA Journal of Industrial Psychology",
     "doi": "10.4102/sajip.v38i1.893",
     "url": "https://doi.org/10.4102/sajip.v38i1.893"
    }
   ],
   "record_notes": "Overall confidence is moderate and uneven. The TIS-6's genuine differentiator is organisational-criterion evidence: a large, prospective, census-based study that predicted actual turnover behaviour (leavers versus stayers) at four-month and four-year horizons, which is rare among self-report workplace scales, though it rests on a single South African ICT company and is unreplicated on observed turnover. Internal consistency (alpha about 0.80 to 0.83) and single-factor structure are reasonably well supported, largely through non-UK and translated samples. Test-retest reliability, formal measurement invariance and responsiveness/MIC are all Absent or untested and are recorded as such rather than assumed. The reference-standard criterion field is a category consideration (no health gold standard exists for an intention construct) and is marked Not-applicable. Licence terms could not be confirmed this session against the steward's current distribution policy: only search-result links to the AOSIS publication-policies page and the DOAJ record were reachable, not their verbatim licence text, so the licence is recorded as not verified this session rather than asserted from the article's own CC BY notice (which rule 7 disallows as a source). The main honesty tension the v0.2 schema still made slightly awkward is that the instrument's strongest property, organisational-criterion prediction, is simultaneously well-designed and thin (one study), which the grade (Moderate) plus status (thin) pairing is used to convey; readers should note both fields together."
  },
  {
   "instrument_id": "csps-wellbeing",
   "display_name": "Civil Service People Survey: wellbeing and engagement items",
   "instrument_type": "item-set",
   "schema_version": "0.2",
   "identity": {
    "name": "Civil Service People Survey (CSPS), wellbeing and engagement item modules",
    "current_version": "Annual survey; 2024 wave (fielded October 2024, results published 2025). Engagement Index is a fixed 5-item module; wellbeing module embeds the ONS4 personal wellbeing questions plus additional health/wellbeing items.",
    "item_count": "Engagement Index: 5 items. Wellbeing: ONS4 (4 single items: life satisfaction, worthwhile, happiness yesterday, anxiety yesterday) plus a small number of additional wellbeing/health items. Full instrument is roughly 60 to 70 questions across all themes.",
    "original_citation": "Engagement Index construction and question set documented in Cabinet Office technical/quality-and-methodology guides (csps_qmi2024). ONS4 wellbeing items originate from Dolan and Metcalfe 2012 (10.1017/s0047279411000833) and Hicks, Tinkler and Allin 2013 (10.1007/s11205-013-0384-x).",
    "steward_publisher": "UK Cabinet Office (Government People Group), on behalf of participating civil service organisations.",
    "licence_status": "Crown copyright, released under the Open Government Licence v3.0 (OGL). Verified this session from the page text of the CSPS 2024 and 2025 Quality and Methodology Information and Results Highlights pages, which carry 'This publication is licensed under the terms of the Open Government Licence v3.0 except where otherwise stated' and, on the 2025 QMI page, 'You may re-use this information (not including logos) free of charge in any format or medium, under the terms of the Open Government Licence'. This governs the published outputs and question documentation; unit-record microdata are not openly released.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "Read this session: GOV.UK 'Quality and Methodology Information for the Civil Service People Survey 2024' and '...2025' pages and the CSPS 2024/2025 Results Highlights pages, whose page text displays the OGL v3.0 statement and the explicit free-reuse-under-OGL wording. Not from founding papers."
   },
   "constructs_claimed": "Employee engagement (a composite 'Engagement Index' capturing pride, advocacy, attachment, inspiration and motivation) and employee wellbeing (subjective wellbeing via ONS4, plus perceived organisational health, stress and related states) among UK civil servants.",
   "deployment_context_caveat": "This is a bespoke organisational monitoring instrument, not a validated standalone psychometric scale. Almost all documentation is grey literature (Cabinet Office technical and quality-and-methodology guides); independent peer-reviewed psychometric evaluation of the CSPS engagement index or wellbeing module is essentially absent. The embedded ONS4 items carry their own (national-survey) evidence base, which is population-general rather than workplace-specific.",
   "structural_validity": {
    "findings": "No independent peer-reviewed factor-analytic validation of the CSPS engagement index or wellbeing module was located this session. The Engagement Index is constructed by Cabinet Office as an unweighted mean of five theme questions treated as a single dimension, described in the technical and quality-and-methodology guidance rather than derived from published confirmatory factor analysis (csps_qmi2024). The design draws conceptually on the broader employee-engagement literature synthesised by Bailey et al. 2015 (10.1111/ijmr.12077), but that synthesis concerns the construct generally, not this five-item index. As an item-set the record does not assert a single latent scale; structural evidence for the engagement index specifically is untested in the peer-reviewed record.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, evidence is Cabinet Office grey literature on UK civil servants; no independent factor analysis located."
   },
   "convergent_discriminant_validity": {
    "findings": "No peer-reviewed convergent or discriminant validity study of the CSPS engagement index or wellbeing items was located this session. Cabinet Office reporting relates the engagement index to other survey themes (for example leadership, my work, pay and benefits) descriptively, but these are internal cross-tabulations in grey literature, not validated convergent evidence (csps_qmi2024). The ONS4 items embedded in the wellbeing module have established convergent relationships with other wellbeing measures in the national-survey literature (dolan2012, hicks2013), but that evidence is general-population, not civil-service-specific.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "mixed",
    "indirectness": "indirect, ONS4 convergent evidence is general-population; index-level convergent evidence is grey-literature and internal."
   },
   "criterion_validity_reference_standard": {
    "findings": "No validity evidence against any diagnostic or clinical health reference standard was located for the CSPS engagement index or wellbeing items. The instrument is not designed or validated as a health screener. The embedded anxiety-yesterday ONS4 item is a single affect rating, not a validated anxiety-disorder screen.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, no reference-standard study exists for this workplace instrument."
   },
   "criterion_validity_organisational": {
    "findings": "This is the criterion for which the instrument was designed, yet published validation is thin. Cabinet Office and departmental reporting routinely relate engagement-index scores to organisational outcomes such as intention to leave, sickness-related themes and performance narratives, but these appear as descriptive cross-sectional associations in grey-literature reports (csps_qmi2024, csps_highlights2024) rather than as peer-reviewed predictive-validity studies with sickness absence, turnover or performance as measured outcomes. No independent longitudinal criterion study linking CSPS engagement or wellbeing scores to verified work outcomes was located this session. The general proposition that engagement predicts work outcomes is supported in the wider literature (bailey2015), but not specifically validated for this index.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, associations are internal cross-sectional grey-literature findings on UK civil servants, not validated predictive studies."
   },
   "internal_consistency": {
    "findings": "No peer-reviewed internal-consistency coefficient (Cronbach alpha or omega) for the five-item CSPS Engagement Index was located this session; the published technical and quality-and-methodology guidance describes index construction but was not found to report an alpha (csps_qmi2024). Because the record is treated as an item-set spanning heterogeneous single items (including the deliberately multidimensional ONS4 questions, which are not intended to form one scale), a single scale-level alpha is only meaningful for the engagement sub-index, and even there it is Absent in the located sources rather than a category error.",
    "grade": "Absent",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, no reliability coefficient located in steward grey literature; general-population ONS4 items are single items."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "untested",
    "indirectness": "indirect, no retest study of CSPS items in a civil-service or any workplace sample located.",
    "summary": "No test-retest study of the CSPS engagement index or wellbeing items was located this session. The survey is fielded annually to (largely) different respondents, so year-on-year change reflects both real change and sampling, not measurement stability. Absence of retest evidence is itself the finding for this item-set."
   },
   "measurement_invariance": {
    "findings": "No published measurement-invariance analysis of the CSPS engagement index or wellbeing module across departments, grades, occupations or over time was located this session. Cabinet Office benchmarking compares raw scores across organisations and demographic groups without establishing measurement invariance, so cross-group score differences cannot be assumed to be free of measurement bias (csps_qmi2024). This is a material gap given the instrument's primary use is cross-organisation and cross-group comparison.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, cross-group comparisons are made in grey literature without invariance testing."
   },
   "responsiveness_mic": {
    "findings": "No responsiveness or minimal-important-change evidence for the CSPS engagement or wellbeing items was located this session. Annual score movements are reported and interpreted by Cabinet Office, but no anchor-based or distribution-based minimal-important-change threshold has been established in the located sources (csps_highlights2024).",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, year-on-year movement is reported without a validated change threshold."
   },
   "populations_languages_norms": {
    "findings": "Extensive UK civil-service norms exist as an artefact of the census-style annual survey: several hundred thousand civil servants respond each year, giving very large organisational and demographic benchmarks published by Cabinet Office (csps_highlights2024, csps_qmi2024). These norms are highly representative of the UK civil service specifically, which is the intended population, but they are not general-UK-working-population norms and do not transfer to other sectors. The instrument is fielded in English.",
    "grade": "Moderate",
    "indirectness": "direct for the UK civil service; indirect for the general UK working population, to which civil-service norms do not generalise."
   },
   "criticisms_controversies": "Principal criticisms are that the CSPS engagement index is a bespoke administrative instrument whose psychometric properties have not been independently validated in the peer-reviewed literature, and that heavy reliance on Cabinet Office grey literature makes external scrutiny difficult. Cross-organisation league-table use of engagement scores proceeds without demonstrated measurement invariance. Response-rate variation across departments and years, and the shift to online administration, complicate longitudinal interpretation. The ONS4 wellbeing items, while nationally established, are single-item affect and evaluation measures with known ceiling effects and limited sensitivity, and were not developed for organisational benchmarking.",
   "item_records": [
    {
     "item_id": "csps-ons4-lifesat",
     "item_label": "Life satisfaction: single-item, 0 to 10, overall satisfaction with life nowadays (ONS4 evaluative item embedded in the wellbeing module).",
     "evidence": "Established as part of the ONS4 personal wellbeing set for national measurement (dolan2012, hicks2013); recommended internationally (oecd2013). Evidence is general-population, not civil-service-specific.",
     "citation_key": "hicks2013"
    },
    {
     "item_id": "csps-ons4-worthwhile",
     "item_label": "Worthwhile: single-item, 0 to 10, extent to which things done in life are worthwhile (ONS4 eudaimonic item).",
     "evidence": "Part of the ONS4 set; general-population validation only (dolan2012, hicks2013).",
     "citation_key": "dolan2012"
    },
    {
     "item_id": "csps-ons4-happy",
     "item_label": "Happiness yesterday: single-item, 0 to 10, positive affect the previous day (ONS4).",
     "evidence": "Part of the ONS4 set; general-population evidence (hicks2013). Known ceiling/skew in population data.",
     "citation_key": "hicks2013"
    },
    {
     "item_id": "csps-ons4-anxiety",
     "item_label": "Anxiety yesterday: single-item, 0 to 10, negative affect the previous day (ONS4). Reverse-scored relative to the other three.",
     "evidence": "Part of the ONS4 set; general-population evidence (hicks2013). A single affect rating, not a validated anxiety screen.",
     "citation_key": "hicks2013"
    },
    {
     "item_id": "csps-engagement-index",
     "item_label": "Engagement Index: a five-item composite (pride, advocacy/recommendation, attachment, inspiration, motivation) scored as a percentage; treated by the steward as one dimension.",
     "evidence": "Construction documented in Cabinet Office technical/quality guidance (csps_qmi2024); no independent peer-reviewed reliability or factor analysis located this session. Construct rationale aligns with the engagement literature synthesised in bailey2015.",
     "citation_key": "csps_qmi2024"
    }
   ],
   "citations": [
    {
     "key": "benson2019",
     "authors": "Benson T, Sladen J, Liles A, Potts HWW",
     "year": "2019",
     "title": "Personal Wellbeing Score (PWS) - a short version of ONS4: development and validation in social prescribing",
     "journal": "BMJ Open Quality",
     "doi": "10.1136/bmjoq-2018-000394",
     "url": "https://doi.org/10.1136/bmjoq-2018-000394"
    },
    {
     "key": "dolan2012",
     "authors": "Dolan P, Metcalfe R",
     "year": "2012",
     "title": "Measuring Subjective Wellbeing: Recommendations on Measures for use by National Governments",
     "journal": "Journal of Social Policy",
     "doi": "10.1017/s0047279411000833",
     "url": "https://doi.org/10.1017/s0047279411000833"
    },
    {
     "key": "hicks2013",
     "authors": "Hicks S, Tinkler L, Allin P",
     "year": "2013",
     "title": "Measuring Subjective Well-Being and its Potential Role in Policy: Perspectives from the UK Office for National Statistics",
     "journal": "Social Indicators Research",
     "doi": "10.1007/s11205-013-0384-x",
     "url": "https://doi.org/10.1007/s11205-013-0384-x"
    },
    {
     "key": "oecd2013",
     "authors": "OECD",
     "year": "2013",
     "title": "OECD Guidelines on Measuring Subjective Well-being",
     "journal": "OECD Publishing",
     "doi": "10.1787/9789264191655-en",
     "url": "https://doi.org/10.1787/9789264191655-en"
    },
    {
     "key": "bailey2015",
     "authors": "Bailey C, Madden A, Alfes K, Fletcher L",
     "year": "2015",
     "title": "The Meaning, Antecedents and Outcomes of Employee Engagement: A Narrative Synthesis",
     "journal": "International Journal of Management Reviews",
     "doi": "10.1111/ijmr.12077",
     "url": "https://doi.org/10.1111/ijmr.12077"
    },
    {
     "key": "csps_qmi2024",
     "authors": "Cabinet Office",
     "year": "2025",
     "title": "Quality and Methodology Information for the Civil Service People Survey 2024",
     "journal": "GOV.UK (Cabinet Office)",
     "doi": "",
     "url": "https://www.gov.uk/government/publications/civil-service-people-survey-2024-results/quality-and-methodology-information-for-the-civil-service-people-survey-2024"
    },
    {
     "key": "csps_highlights2024",
     "authors": "Cabinet Office",
     "year": "2025",
     "title": "Civil Service People Survey 2024: Results Highlights",
     "journal": "GOV.UK (Cabinet Office)",
     "doi": "",
     "url": "https://www.gov.uk/government/publications/civil-service-people-survey-2024-results/civil-service-people-survey-2024-results-highlights"
    }
   ],
   "record_notes": "Overall confidence: Very low to Low across most psychometric properties, driven by near-total reliance on Cabinet Office grey literature and an absence of independent peer-reviewed validation of the engagement index. This is a genuine, gradeable finding, not a search failure: the CSPS is an operational monitoring instrument, and its evidentiary strength lies in very large representative civil-service norms rather than in demonstrated measurement properties. The v0.2 item-set treatment fits well: the ONS4 items carry transferable (but general-population) evidence, while the engagement index is essentially untested in the peer-reviewed record. Licence verified against the current GOV.UK OGL footer, not founding papers."
  },
  {
   "instrument_id": "eurofound-ewcs",
   "display_name": "Eurofound EWCS/EQLS wellbeing and working-conditions item sets",
   "instrument_type": "item-set",
   "schema_version": "0.2",
   "identity": {
    "name": "European Working Conditions Survey (EWCS) and European Quality of Life Survey (EQLS) wellbeing and job-quality item sets, including the WHO-5 mental wellbeing items embedded in EQLS and the EWCS job-quality indices.",
    "current_version": "EWCS: most recent full wave EWCS 2024 (UK Data Service SN 9511); prior anchor waves 2015 (sixth) and 2010 (fifth). EQLS: most recent full wave 2016 (fourth), with WHO-5 embedded. Job-quality indices constructed by Eurofound and by ETUI (European Job Quality Index).",
    "item_count": "EWCS: roughly 100 questions per wave; job-quality operationalised as a small number of multi-item indices (for example earnings, prospects, intrinsic job quality, working-time quality, work intensity). EQLS: WHO-5 is a fixed 5-item mental wellbeing scale embedded within a broader questionnaire. Item sets, not a single scale.",
    "original_citation": "EWCS job-quality index construction and critique in Munoz de Bustillo et al. 2011 (10.1093/ser/mwr005) and Green et al. 2013 (10.1177/001979391306600402); ETUI European Job Quality Index in Leschke, Watt and Finn 2012 (10.2139/ssrn.2208374) and Piasna 2018 (10.2139/ssrn.3103624). WHO-5 embedded in EQLS is reviewed in Topp et al. 2015 (10.1159/000376585).",
    "steward_publisher": "Eurofound (European Foundation for the Improvement of Living and Working Conditions), an EU agency; ETUI (European Trade Union Institute) is the steward of the European Job Quality Index specifically.",
    "licence_status": "Eurofound survey microdata are not fully open. Verified this session from Eurofound's 'Data availability' page, which states the data are available free of charge for non-commercial purposes and that requests for commercial use are forwarded to Eurofound for authorisation, with access provided to registered users of the UK Data Service (UKDS). EWCS 2024 is catalogued at UKDS as SN 9511. Eurofound questionnaires and aggregate outputs and the online data explorer are publicly available; unit-record data are restricted-access research data under registration, not a fully open licence.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "Read this session: Eurofound 'Data availability' page text ('available free of charge to those who intend to use them for non-commercial purposes; requests for use for commercial purposes will be forwarded to Eurofound for authorisation', access via UKDS registration) and the UK Data Service catalogue record for EWCS 2024 (SN 9511, DOI 10.5255/UKDA-SN-9511-1). Not from founding papers."
   },
   "constructs_claimed": "Job quality and working conditions (multidimensional: physical and psychosocial environment, work intensity, working-time quality, skills and discretion, prospects, earnings) via EWCS; quality of life and mental wellbeing (including WHO-5) via EQLS. The intended unit of analysis is population and country-level monitoring across the EU, not individual diagnosis.",
   "deployment_context_caveat": "These are cross-national population survey item sets built for comparative policy monitoring, not standalone validated scales for individual assessment. Most methodological documentation is Eurofound/ETUI technical-report grey literature; peer-reviewed psychometric evaluation focuses on the job-quality indices and on the embedded WHO-5, not on a single overall instrument. WHO-5 is a clinically originated screener; any workplace or organisational use is a different deployment context from its validation base.",
   "structural_validity": {
    "findings": "Structural evidence is uneven and index-specific. Munoz de Bustillo et al. 2011 provide a detailed critical appraisal of the dimensional structure of job-quality indicators and argue for theory-led multidimensional indices rather than a single composite (10.1093/ser/mwr005). Green et al. 2013 operationalise EWCS data into distinct job-quality dimensions and examine their distributional structure across Europe (10.1177/001979391306600402). The ETUI European Job Quality Index is built as a set of sub-indices rather than one latent factor (leschke2012, piasna2018). The WHO-5 embedded in EQLS is well established as a unidimensional scale in the wider literature (topp2015). No single agreed latent structure spans the whole EWCS/EQLS item set, consistent with its treatment as an item-set.",
    "grade": "Moderate",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect, evidence is European cross-national general-working-population data; index structure is debated between competing constructions."
   },
   "convergent_discriminant_validity": {
    "findings": "Job-quality indices show expected convergent relationships with wellbeing, health and job-satisfaction outcomes in the EWCS-based literature, and Munoz de Bustillo et al. 2011 discuss the risk that composite indices blur discriminant boundaries between conceptually distinct dimensions (10.1093/ser/mwr005). Green et al. 2013 relate the dimensions to inequality and outcome gradients across Europe (10.1177/001979391306600402). The embedded WHO-5 has extensive convergent evidence with depression and wellbeing measures in the systematic review by Topp et al. 2015 (10.1159/000376585), though that evidence is largely clinical and general-population rather than workplace-specific.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect, convergent evidence is European general-population and, for WHO-5, clinical/general-population, not UK-workplace-specific."
   },
   "criterion_validity_reference_standard": {
    "findings": "For the job-quality item sets there is no diagnostic reference standard and none is claimed. For the embedded WHO-5, criterion validity against clinical depression reference standards is well established in the systematic review by Topp et al. 2015, which reports good screening performance for depression across many studies (10.1159/000376585). That evidence is a property of WHO-5 as fielded, but was earned in clinical and general-population samples, not in EU working-population deployment.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "parent",
    "indirectness": "indirect, WHO-5 reference-standard evidence is clinical/general-population (parent-form evidence for the embedded items), not workplace."
   },
   "criterion_validity_organisational": {
    "findings": "EWCS job-quality indices are widely used to study associations with work outcomes such as absence, health and job satisfaction at the population level, but these are cross-sectional descriptive and secondary-analysis associations rather than validated predictive criterion studies with verified organisational outcomes (10.1093/ser/mwr005; 10.1177/001979391306600402). Because EWCS/EQLS are anonymous cross-national repeated cross-sections, they cannot link individual scores to that individual's later absence, turnover or performance, so individual-level organisational criterion validity is structurally unavailable and effectively Absent as a validated property.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, associations are ecological/cross-sectional at European population level; no individual-level predictive validation of work outcomes."
   },
   "internal_consistency": {
    "findings": "Internal consistency is meaningful only for the multi-item sub-indices, not for the item set as a whole. Reliability of the ETUI and Eurofound job-quality sub-indices is discussed in the index-construction literature, with the multidimensional design chosen partly because a single composite would conflate distinct dimensions (leschke2012; 10.1093/ser/mwr005). The embedded WHO-5 has consistently high internal consistency (Cronbach alpha typically around 0.8 or above) in the systematic review by Topp et al. 2015 (10.1159/000376585). No single overall alpha is appropriate for the heterogeneous item set.",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect, WHO-5 alpha is clinical/general-population parent-form evidence; job-quality sub-index reliability is from European technical literature.",
    "subgrades": [
     {
      "subgroup": "embedded WHO-5",
      "grade": "High",
      "note": "high internal consistency, alpha ~0.8+, but parent-form clinical/general-population evidence (topp2015)."
     },
     {
      "subgroup": "job-quality sub-indices",
      "grade": "Low",
      "note": "multi-item indices; reliability discussed in grey/technical literature, not consistently reported as coefficients."
     }
    ]
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "untested",
    "indirectness": "indirect, EWCS/EQLS are repeated cross-sections of different respondents; no panel retest of these item sets located.",
    "summary": "No test-retest study of the EWCS or EQLS job-quality/wellbeing item sets was located this session. The surveys are repeated cross-sections without a re-interview panel design, so stability coefficients are not estimable from the standard data and none were found. The embedded WHO-5 has retest evidence in its own clinical/general-population literature, but that is parent-form evidence not earned in this deployment; it is therefore not entered as an EWCS/EQLS finding. Absence of retest evidence for these item sets is the finding."
   },
   "measurement_invariance": {
    "findings": "Cross-country comparability is the central methodological concern for EWCS/EQLS, and it is actively worked on but not fully resolved. Munoz de Bustillo et al. 2011 highlight cross-national comparability and equivalence as core challenges for job-quality indices (10.1093/ser/mwr005), and Green et al. 2013 conduct cross-country comparisons that presuppose comparability while noting its limits (10.1177/001979391306600402). Eurofound invests heavily in translation and harmonisation, but formal measurement-invariance testing (configural/metric/scalar) across all countries and waves is not comprehensively established for the job-quality item sets in the located sources. The embedded WHO-5 has broader invariance evidence in its own literature (topp2015).",
    "grade": "Low",
    "status": "contested",
    "evidence_form": "mixed",
    "indirectness": "indirect, invariance is the key open question for cross-national comparison; formal scalar invariance not comprehensively demonstrated for job-quality indices.",
    "subgrades": [
     {
      "subgroup": "cross-country (job-quality indices)",
      "grade": "Low",
      "note": "comparability assumed and worked on, not comprehensively tested to scalar level."
     },
     {
      "subgroup": "embedded WHO-5 cross-group",
      "grade": "Moderate",
      "note": "broader invariance evidence exists but as parent-form (topp2015)."
     }
    ]
   },
   "responsiveness_mic": {
    "findings": "No minimal-important-change or formal responsiveness evidence for the EWCS/EQLS job-quality item sets was located this session. The indices are used to track change over time at population level (leschke2012; piasna2018), but no anchor-based minimal-important-change threshold has been established for them. WHO-5 responsiveness is documented in its own clinical literature (topp2015) but is parent-form evidence for the embedded items.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "mixed",
    "indirectness": "indirect, population-level trend tracking without a validated individual change threshold."
   },
   "populations_languages_norms": {
    "findings": "Very extensive multi-country European norms exist: EWCS and EQLS are fielded across all EU member states and several additional countries, with professional translation into national languages and large representative samples per country (eurofound_dataavail, ukds_ewcs2024). This makes them a strong source of comparative European working-population and quality-of-life benchmarks. For a UK working-adult audience the UK subsample is directly relevant, but published norms are usually reported at European or country-comparative level rather than as UK-workplace-specific reference values.",
    "grade": "High",
    "indirectness": "direct for European and UK general working populations; the UK-specific workplace application relies on the UK subsample within a cross-national design."
   },
   "criticisms_controversies": "The main controversies concern job-quality index construction: whether to use a single composite or multiple sub-indices, how to weight dimensions, and whether cross-national scores are truly comparable given cultural and linguistic differences in item interpretation (10.1093/ser/mwr005; 10.1177/001979391306600402). Competing indices (Eurofound's own job-quality indices versus the ETUI European Job Quality Index) can yield different country rankings, underlining construction sensitivity (leschke2012; piasna2018). The repeated cross-sectional design precludes individual-level longitudinal and test-retest analysis. The embedded WHO-5 is a clinically originated screener, so its use within a working-conditions survey mixes a validated clinical instrument with bespoke policy items, and its strong properties should not be read as validating the surrounding item set.",
   "item_records": [
    {
     "item_id": "eqls-who5",
     "item_label": "WHO-5 mental wellbeing block embedded in EQLS: five positively-worded items on positive mood, vitality and general interest over the past two weeks, 0 to 5 each, summed and scaled to 0 to 100.",
     "evidence": "Well-established unidimensional structure, high internal consistency (alpha ~0.8+) and good depression-screening criterion validity in the systematic review by Topp et al. 2015 (10.1159/000376585). This is parent-form evidence: it was earned for WHO-5 generally in clinical/general-population samples, not in EQLS workplace deployment.",
     "citation_key": "topp2015"
    },
    {
     "item_id": "ewcs-jobquality-indices",
     "item_label": "EWCS job-quality indices: a set of multi-item sub-indices (for example earnings, prospects, intrinsic job quality, working-time quality, work intensity) constructed from working-conditions items.",
     "evidence": "Construction, dimensional appraisal and cross-country application in Munoz de Bustillo et al. 2011 (10.1093/ser/mwr005) and Green et al. 2013 (10.1177/001979391306600402); alternative construction in the ETUI European Job Quality Index (leschke2012; piasna2018). Structure is contested rather than settled.",
     "citation_key": "munoz2011"
    }
   ],
   "citations": [
    {
     "key": "munoz2011",
     "authors": "Munoz de Bustillo R, Fernandez-Macias E, Anton J-I, Esteve F",
     "year": "2011",
     "title": "E pluribus unum? A critical survey of job quality indicators",
     "journal": "Socio-Economic Review",
     "doi": "10.1093/ser/mwr005",
     "url": "https://doi.org/10.1093/ser/mwr005"
    },
    {
     "key": "green2013",
     "authors": "Green F, Mostafa T, Parent-Thirion A, Vermeylen G, van Houten G, Biletta I, Lyly-Yrjanainen M",
     "year": "2013",
     "title": "Is Job Quality Becoming More Unequal?",
     "journal": "ILR Review",
     "doi": "10.1177/001979391306600402",
     "url": "https://doi.org/10.1177/001979391306600402"
    },
    {
     "key": "topp2015",
     "authors": "Topp CW, Ostergaard SD, Sondergaard S, Bech P",
     "year": "2015",
     "title": "The WHO-5 Well-Being Index: A Systematic Review of the Literature",
     "journal": "Psychotherapy and Psychosomatics",
     "doi": "10.1159/000376585",
     "url": "https://doi.org/10.1159/000376585"
    },
    {
     "key": "leschke2012",
     "authors": "Leschke J, Watt A, Finn M",
     "year": "2012",
     "title": "Job Quality in the Crisis - An Update of the Job Quality Index (JQI)",
     "journal": "ETUI Working Paper 2012.07 (SSRN)",
     "doi": "10.2139/ssrn.2208374",
     "url": "https://doi.org/10.2139/ssrn.2208374"
    },
    {
     "key": "piasna2018",
     "authors": "Piasna A",
     "year": "2018",
     "title": "Bad Jobs Recovery? European Job Quality Index 2005-2015",
     "journal": "ETUI Working Paper 2017.06 (SSRN)",
     "doi": "10.2139/ssrn.3103624",
     "url": "https://doi.org/10.2139/ssrn.3103624"
    },
    {
     "key": "eurofound_dataavail",
     "authors": "Eurofound",
     "year": "2026",
     "title": "Data availability (Eurofound surveys: EWCS, EQLS, ECS)",
     "journal": "Eurofound (European Foundation for the Improvement of Living and Working Conditions)",
     "doi": "",
     "url": "https://www.eurofound.europa.eu/en/surveys/about-eurofounds-surveys/data-availability"
    },
    {
     "key": "ukds_ewcs2024",
     "authors": "Eurofound",
     "year": "2026",
     "title": "European Working Conditions Survey, 2024 [data collection], SN 9511",
     "journal": "UK Data Service",
     "doi": "10.5255/UKDA-SN-9511-1",
     "url": "https://doi.org/10.5255/UKDA-SN-9511-1"
    }
   ],
   "record_notes": "Overall confidence: Moderate for the embedded WHO-5 (strong but parent-form, clinical/general-population evidence) and for job-quality convergent validity; Low to Very low for structural settledness, invariance, responsiveness and individual-level organisational criterion validity, which are structurally constrained by the repeated cross-sectional design. The v0.2 item-set treatment is essential here: pooling WHO-5's strong clinical evidence with the contested job-quality indices into one grade would misrepresent both, so evidence_form tags and subgrades separate them. Licence verified against Eurofound's current data-availability page and the UK Data Service SN 9511 record, which show restricted/safeguarded access rather than fully open terms; not asserted from literature."
  },
  {
   "instrument_id": "fcs-maps",
   "display_name": "Financial Capability / Financial Wellbeing Survey items (MaPS)",
   "instrument_type": "item-set",
   "schema_version": "0.2",
   "identity": {
    "name": "Money and Pensions Service (MaPS) Financial Capability / Adult Financial Wellbeing Survey item set, including the MaPS financial wellbeing key questions.",
    "current_version": "MaPS Adult Financial Wellbeing Survey; most recent principal wave 2021 (Adult Financial Wellbeing Survey 2021), successor to the Money Advice Service Financial Capability Survey (2015, 2018). A short 'financial wellbeing key questions' set is also published for reuse.",
    "item_count": "Item set spanning financial wellbeing outcome items and financial-capability behaviour, mindset and connection items; the survey instrument runs to dozens of questions, and MaPS publishes a compact 'key questions' subset. Not a single unidimensional scale.",
    "original_citation": "Construct and question set documented in MaPS technical/survey materials (maps_fws2021, maps_workingwithdata). The financial wellbeing construct draws on the wider conceptual literature: Kempson-lineage and ecological models synthesised in Salignac et al. 2019 (10.1007/s10902-019-00145-3), Bruggen et al. 2017 (10.1016/j.jbusres.2017.03.013) and Netemeyer et al. 2018 (10.1093/jcr/ucx109).",
    "steward_publisher": "Money and Pensions Service (MaPS), a UK arm's-length body sponsored by the Department for Work and Pensions.",
    "licence_status": "Two-tier, and not a standard open licence. Verified this session from MaPS's 'Working with our survey data' page and the Geographic Data Service (GeoDS) dataset record. Underlying survey microdata are categorised as Safeguarded: access is only available on application to GeoDS, assessed against public benefit, data security, ethical standards and training, with a required disclaimer and MaPS cited as the data source. For published statistics, figures and graphics, MaPS operates a permissive but conditional reuse arrangement (it asks to review publications using its data and requires citation and a non-endorsement disclaimer) rather than an explicitly stated Open Government Licence; no OGL statement was found on the MaPS survey-data pages read this session. So aggregate outputs are reusable under MaPS's stated conditions, while unit-record data are restricted.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "Read this session: MaPS 'Working with our survey data' page text (access via GeoDS on application; citation and disclaimer required) and the GeoDS 'MaPS Financial Wellbeing Survey' dataset record ('categorised as Safeguarded and therefore access to the underlying data is only available upon application'; recommended MaPS disclaimer text). Not from founding papers."
   },
   "constructs_claimed": "Financial wellbeing (the outcome construct: feeling secure and in control of finances now and in the future, low financial anxiety, ability to meet commitments) and financial capability (the behavioural and psychological antecedents: managing money, planning, saving, financial confidence and mindset) among UK adults.",
   "deployment_context_caveat": "This is a national policy-monitoring instrument, not a validated standalone psychometric scale, and its evidentiary base is overwhelmingly MaPS technical-report grey literature rather than peer-reviewed psychometric studies. The financial wellbeing construct is well theorised in the academic literature, but that theory validates the construct, not this specific MaPS item set. It is a general-adult, not workplace-specific, instrument; workplace deployment (for example employer financial-wellbeing programmes) is a different context from its national-survey validation base.",
   "structural_validity": {
    "findings": "No independent peer-reviewed factor-analytic validation of the MaPS financial wellbeing/capability item set was located this session; the dimensional structure (outcomes versus capability components) is defined in MaPS technical materials rather than derived from published confirmatory factor analysis (maps_fws2021, maps_workingwithdata). The broader financial wellbeing construct has been given structure in the academic literature, including the current/expected-financial-wellbeing distinction validated by Netemeyer et al. 2018 (10.1093/jcr/ucx109) and the multidimensional/ecological framings of Bruggen et al. 2017 (10.1016/j.jbusres.2017.03.013) and Salignac et al. 2019 (10.1007/s10902-019-00145-3), but these validate related scales and models, not the MaPS instrument itself. As an item set the record does not assert a single latent scale.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, structure defined in UK grey literature; related-construct evidence is non-MaPS scales in mixed populations."
   },
   "convergent_discriminant_validity": {
    "findings": "No peer-reviewed convergent/discriminant validity study of the MaPS item set specifically was located this session. MaPS reporting relates financial wellbeing to income, debt, savings buffers and mental wellbeing descriptively in grey literature (maps_fws2021). The general convergent structure of financial wellbeing (its relationships with objective finances, financial behaviour and subjective wellbeing) is established for other scales by Netemeyer et al. 2018, who show perceived financial wellbeing relates to but is distinct from overall wellbeing (10.1093/jcr/ucx109), and framed conceptually by Bruggen et al. 2017 (10.1016/j.jbusres.2017.03.013); this is indirect support only.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "mixed",
    "indirectness": "indirect, direct MaPS-item convergent evidence is grey-literature and descriptive; strong convergent evidence exists only for related non-MaPS scales."
   },
   "criterion_validity_reference_standard": {
    "findings": "There is no diagnostic or clinical reference standard for financial wellbeing, and none is claimed for the MaPS item set. Financial wellbeing is a subjective/behavioural construct rather than a health condition, so reference-standard criterion validity is not an applicable target for this instrument.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, no health reference standard exists for this construct; not an applicable criterion."
   },
   "criterion_validity_organisational": {
    "findings": "No validated evidence linking MaPS financial wellbeing scores to work outcomes (sickness absence, turnover, presenteeism, performance) in a workplace context was located this session. The financial-wellbeing-to-productivity link is asserted in policy and practitioner material and is theoretically plausible, but the MaPS instrument is a national household survey not designed to predict individual work outcomes, and no peer-reviewed organisational predictive-validity study of the MaPS items was found. This is an important Absent finding for a workplace registry.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, no workplace outcome linkage validated; instrument is a general-adult national survey."
   },
   "internal_consistency": {
    "findings": "No peer-reviewed internal-consistency coefficient for the MaPS financial wellbeing/capability item set was located this session; MaPS technical materials describe the items and derived measures but were not found to report Cronbach alpha or omega for a defined scale (maps_fws2021). Because the record is an item set spanning distinct outcome and capability components rather than one intended scale, a single overall alpha is not the appropriate target; component-level reliability is Absent in the located sources rather than a category error.",
    "grade": "Absent",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, no reliability coefficient located in MaPS grey literature."
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "untested",
    "indirectness": "indirect, no retest study of MaPS financial wellbeing items located; survey uses fresh cross-sectional samples.",
    "summary": "No test-retest study of the MaPS financial wellbeing/capability items was located this session. The survey is fielded to fresh cross-sectional samples rather than a re-interview panel, so stability coefficients are not available from the standard design and none were found. Absence of retest evidence is the finding for this item set."
   },
   "measurement_invariance": {
    "findings": "No published measurement-invariance analysis of the MaPS financial wellbeing/capability items across age, sex, income, region or over time (including across the Money Advice Service to MaPS transition) was located this session. MaPS compares scores across demographic segments in grey literature without establishing measurement invariance (maps_fws2021), so segment differences cannot be assumed free of measurement bias. Comparability across the 2015/2018 Financial Capability waves and the 2021 Financial Wellbeing Survey is a particular open question given questionnaire changes.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, cross-segment and cross-wave comparisons made without invariance testing."
   },
   "responsiveness_mic": {
    "findings": "No responsiveness or minimal-important-change evidence for the MaPS financial wellbeing items was located this session. MaPS tracks national financial wellbeing over time against strategy targets, but no anchor-based or distribution-based minimal-important-change threshold has been established in the located sources (maps_fws2021).",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "indirect, national trend tracking without a validated change threshold."
   },
   "populations_languages_norms": {
    "findings": "Large, nationally representative UK adult norms exist from the MaPS Adult Financial Wellbeing Survey and its Money Advice Service predecessors, with published population and segment benchmarks (maps_fws2021, maps_workingwithdata). These are directly relevant to UK adults generally and are a genuine strength. They are general-population rather than workplace-specific norms, so applying them to a working-adult subgroup within an employer context is an indirect use. The instrument is fielded in English (UK).",
    "grade": "Moderate",
    "indirectness": "direct for UK general adults; indirect for the UK working-adult-in-workplace application, for which no separate norms were located."
   },
   "criticisms_controversies": "The central limitation is evidentiary: the MaPS financial wellbeing/capability item set is a policy-monitoring instrument whose psychometric properties have not been independently validated in the peer-reviewed literature, and almost all documentation is MaPS grey literature. The financial wellbeing construct itself is contested in definition and measurement, with competing academic operationalisations (current versus expected financial wellbeing in Netemeyer et al. 2018, 10.1093/jcr/ucx109; broader conceptual and ecological models in Bruggen et al. 2017 and Salignac et al. 2019), so mapping MaPS items onto a settled construct is not straightforward. Questionnaire changes across waves complicate longitudinal comparison, and restricted microdata access limits independent psychometric scrutiny.",
   "item_records": [
    {
     "item_id": "maps-finwell-outcome",
     "item_label": "Financial wellbeing outcome items: subjective sense of financial security, being on top of / keeping up with bills and commitments, and financial anxiety about the future.",
     "evidence": "Defined in MaPS technical/survey materials (maps_fws2021); no independent psychometric coefficient located. Conceptually aligned with the current-financial-wellbeing dimension validated for other scales by Netemeyer et al. 2018 (10.1093/jcr/ucx109).",
     "citation_key": "netemeyer2018"
    },
    {
     "item_id": "maps-capability-behaviour",
     "item_label": "Financial capability items: money-management behaviours, saving and planning behaviours, and financial mindset/confidence (antecedent constructs).",
     "evidence": "Defined in MaPS technical materials (maps_workingwithdata); no independent reliability/validity coefficient located. Antecedent-to-wellbeing framing consistent with the conceptual model in Bruggen et al. 2017 (10.1016/j.jbusres.2017.03.013).",
     "citation_key": "bruggen2017"
    }
   ],
   "citations": [
    {
     "key": "netemeyer2018",
     "authors": "Netemeyer RG, Warmath D, Fernandes D, Lynch JG",
     "year": "2018",
     "title": "How Am I Doing? Perceived Financial Well-Being, Its Potential Antecedents, and Its Relation to Overall Well-Being",
     "journal": "Journal of Consumer Research",
     "doi": "10.1093/jcr/ucx109",
     "url": "https://doi.org/10.1093/jcr/ucx109"
    },
    {
     "key": "bruggen2017",
     "authors": "Bruggen EC, Hogreve J, Holmlund M, Kabadayi S, Lofgren M",
     "year": "2017",
     "title": "Financial well-being: A conceptualization and research agenda",
     "journal": "Journal of Business Research",
     "doi": "10.1016/j.jbusres.2017.03.013",
     "url": "https://doi.org/10.1016/j.jbusres.2017.03.013"
    },
    {
     "key": "salignac2019",
     "authors": "Salignac F, Hamilton M, Noone J, Marjolin A, Muir K",
     "year": "2019",
     "title": "Conceptualizing Financial Wellbeing: An Ecological Life-Course Approach",
     "journal": "Journal of Happiness Studies",
     "doi": "10.1007/s10902-019-00145-3",
     "url": "https://doi.org/10.1007/s10902-019-00145-3"
    },
    {
     "key": "walkowiak2025",
     "authors": "Walkowiak D, Jabkowski P, Domaradzki J",
     "year": "2025",
     "title": "Validation and psychometric properties of the Consumer Financial Protection Bureau Financial Well-Being Scale",
     "journal": "Journal of Medical Science",
     "doi": "10.20883/medical.e1428",
     "url": "https://doi.org/10.20883/medical.e1428"
    },
    {
     "key": "maps_workingwithdata",
     "authors": "Money and Pensions Service",
     "year": "2024",
     "title": "Working with our survey data",
     "journal": "Money and Pensions Service (maps.org.uk)",
     "doi": "",
     "url": "https://maps.org.uk/en/publications/research/working-with-our-survey-data"
    },
    {
     "key": "maps_fws2021",
     "authors": "Money and Pensions Service",
     "year": "2021",
     "title": "Adult Financial Wellbeing Survey 2021",
     "journal": "Money and Pensions Service (maps.org.uk)",
     "doi": "",
     "url": "https://maps.org.uk/en/publications/research/2021/financial-wellbeing-survey-2021"
    },
    {
     "key": "geods_maps_fws",
     "authors": "Money and Pensions Service; Geographic Data Service",
     "year": "2024",
     "title": "MaPS Financial Wellbeing Survey (Safeguarded dataset record)",
     "journal": "Geographic Data Service (GeoDS)",
     "doi": "",
     "url": "https://data.geods.ac.uk/dataset/maps-financial-wellbeing-survey"
    }
   ],
   "record_notes": "Overall confidence: Very low to Absent across measured psychometric properties for the MaPS item set specifically, which is a genuine and important finding for a workplace registry, not a search shortfall: this is a grey-literature-based national monitoring instrument with essentially no independent peer-reviewed validation. The stronger financial wellbeing evidence in the literature (netemeyer2018, bruggen2017, salignac2019, and the CFPB-scale validation walkowiak2025) attaches to other named scales and the construct in general, not to the MaPS items, and is tagged as indirect/related throughout. Reference-standard criterion is marked Not-applicable (a category error for a non-health construct), distinct from the Absent organisational criterion. Licence verified this session from the page text of MaPS's 'Working with our survey data' page and the GeoDS Safeguarded-data record: microdata are Safeguarded (application required) and aggregate outputs are reusable under MaPS's citation-and-disclaimer conditions. An earlier draft over-stated MaPS publications as Open Government Licence; no OGL statement was found on the MaPS survey-data pages, so that claim has been corrected. Not asserted from literature."
  },
  {
   "instrument_id": "single-job-satisfaction",
   "display_name": "Single-item overall job satisfaction (Wanous tradition)",
   "instrument_type": "single-item",
   "schema_version": "0.2",
   "identity": {
    "name": "Single-item overall job satisfaction measure (Wanous tradition)",
    "current_version": "No single canonical form; a family of one-item global job-satisfaction questions (for example 'All in all, how satisfied are you with your job?') typically rated on a 5-point or 7-point scale. The tradition is anchored by the Wanous, Reichers and Hudy (1997) meta-analysis rather than by one authored item.",
    "item_count": "1",
    "original_citation": "Wanous JP, Reichers AE, Hudy MJ (1997). Overall job satisfaction: how good are single-item measures? Journal of Applied Psychology 82(2):247-252. doi:10.1037/0021-9010.82.2.247",
    "steward_publisher": "None. This is a non-proprietary generic item described in the peer-reviewed literature; there is no test publisher or licensing steward. Individual journal articles that print an item are under their publishers' copyright, but the one-line question itself is reused freely across the research literature.",
    "licence_status": "No formal instrument licence exists. A single global job-satisfaction question is not a proprietary instrument and has no steward distribution terms; it is used without permission or fee throughout occupational research. Where an item is quoted, the copyright attaching to it is the ordinary copyright of the journal article that printed it, not an instrument licence.",
    "licence_verified_date": "2026-07-12 (no instrument licence exists to verify; non-proprietary generic item)",
    "licence_source": "No steward page exists to verify because a generic single-item job-satisfaction question has no steward or proprietary distributor. No instrument-licence check was required or performed for a non-proprietary item. Position established this session by reasoning that the anchoring methodological sources (Wanous, Reichers and Hudy 1997, Journal of Applied Psychology; Nagy 2002, Journal of Occupational and Organizational Psychology; Dolbier et al. 2005) treat the single-item approach as a freely reusable research method rather than a licensed product. The copyright attaching to any printed wording is the ordinary copyright of the journal article that printed it."
   },
   "constructs_claimed": "Global (overall) job satisfaction, that is a person's summary evaluative judgement of their job as a whole, captured in one question rather than by aggregating facet items.",
   "deployment_context_caveat": "None specific to clinical origin. The generic caveat for the class is that a single global item cannot localise a problem to a facet (pay, supervision, workload) and is designed for group-level monitoring rather than individual diagnosis or high-stakes decisions.",
   "structural_validity": {
    "findings": "Not applicable by construction. A single item has no internal factor structure to model, so factor analysis and dimensionality assessment are category errors for this instrument type. The methodological literature makes this explicit when contrasting single-item with multi-item measurement ([Fisher, Matthews & Gibbons 2016](https://doi.org/10.1037/a0039139); [Diamantopoulos et al. 2012](https://doi.org/10.1007/s11747-011-0300-3)).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, category error for a single item rather than an evidential gap"
   },
   "convergent_discriminant_validity": {
    "findings": "This is the best-evidenced property of the tradition. The anchoring meta-analysis pooled 28 correlations from 17 studies (N=7,682) and found a mean uncorrected correlation of .63 (SD .09) between single-item and scale measures of overall job satisfaction, rising to .67 corrected for reliability and to .72 (SD .05) for the strongest subset of scale measures ([Wanous, Reichers & Hudy 1997](https://doi.org/10.1037/0021-9010.82.2.247)). Extending the approach to facets, [Nagy 2002](https://doi.org/10.1348/096317902167658) reported single-item facet items correlating .60 to .72 with the corresponding Job Descriptive Index facet scales (N=207). [Dolbier et al. 2005](https://doi.org/10.4278/0890-1171-19.3.194) found a single global item correlated significantly with a multi-item job-satisfaction scale and with work, personality and health variables in 745 public-agency employees. Broader single-item methodological work supports the general pattern that single items track their multi-item counterparts closely enough for many purposes, though multi-item measures usually retain a modest edge ([Bergkvist & Rossiter 2007](https://doi.org/10.1509/jmkr.44.2.175); [Fisher, Matthews & Gibbons 2016](https://doi.org/10.1037/a0039139); [Song et al. 2022](https://doi.org/10.1177/10731911221113563)). Discriminant evidence specific to job satisfaction is thinner than convergent evidence.",
    "grade": "High",
    "status": "well-established",
    "evidence_form": "canonical",
    "indirectness": "direct, evidence is from working-adult employee samples although predominantly North American; UK-specific replication is limited (Oshagbemi 1999 is a UK university-staff study)"
   },
   "criterion_validity_reference_standard": {
    "findings": "There is no diagnostic or health reference standard for job satisfaction, which is a subjective attitude rather than a health state, so validation against a gold-standard diagnosis is not a meaningful target for this construct. Health correlations exist (for example associations with self-reported health in [Dolbier et al. 2005](https://doi.org/10.4278/0890-1171-19.3.194)) but health is a correlate, not a reference standard for the attitude itself.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, no reference standard exists for this construct"
   },
   "criterion_validity_organisational": {
    "findings": "Better supported here than for many attitude measures, though still limited. [Nagy 2002](https://doi.org/10.1348/096317902167658) found single-item satisfaction accounted for incremental variance in self-reported job performance and turnover intentions beyond the Job Descriptive Index. [Dolbier et al. 2005](https://doi.org/10.4278/0890-1171-19.3.194) showed a single item substantially predicted turnover intention by logistic regression in 745 employees. Evidence against hard organisational outcomes (recorded sickness absence, actual turnover, objective performance) rather than self-reported intentions is sparse for the single-item form specifically, so the grade is held down. Overall organisational-criterion evidence for the one-item measure is present but modest.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, criterion outcomes are mostly self-reported intentions in non-UK samples rather than objective work records"
   },
   "internal_consistency": {
    "findings": "Not applicable. A single item has no inter-item covariance, so Cronbach alpha and omega cannot be computed. Reliability for a single item is instead estimated indirectly: Wanous, Reichers and Hudy used the correction-for-attenuation formula to back out a minimum reliability of roughly .45 to .69 depending on assumptions ([Wanous, Reichers & Hudy 1997](https://doi.org/10.1037/0021-9010.82.2.247)), a method set out more fully in [Wanous & Reichers 1996](https://doi.org/10.2466/pr0.1996.78.2.631). This estimated reliability (~0.67) is a corrected convergence estimate, not an internal-consistency coefficient, and must not be read as one.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, category error for a single item"
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "untested",
    "indirectness": "direct, absence applies to the single-item job-satisfaction form specifically",
    "summary": "No direct test-retest coefficient for a single-item overall job-satisfaction measure was located this session. The tradition's reliability estimates are correction-for-attenuation estimates derived from cross-sectional convergence with multi-item scales (~0.67 corrected; minimum ~0.45 to .69), NOT test-retest stability coefficients ([Wanous, Reichers & Hudy 1997](https://doi.org/10.1037/0021-9010.82.2.247); [Wanous & Reichers 1996](https://doi.org/10.2466/pr0.1996.78.2.631)). The absence of a genuine temporal-stability coefficient for the single job-satisfaction item is itself the finding and a real gap. [Fisher, Matthews & Gibbons 2016](https://doi.org/10.1037/a0039139) reported 1-month and 3-month test-retest reliabilities for a battery of single organisational items, but those coefficients are not cleanly attributable to a specific canonical global job-satisfaction item and are treated as parent-battery evidence rather than a coefficient for this form."
   },
   "measurement_invariance": {
    "findings": "Not applicable in the conventional multi-group confirmatory-factor sense, because a single item has no measurement model to constrain across groups. No study locating differential item functioning or equivalent single-item invariance evidence across sex, age, occupation or language for a global job-satisfaction item was found this session.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, formal invariance modelling is a category error for a single item; substantive cross-group comparability is untested"
   },
   "responsiveness_mic": {
    "findings": "No responsiveness or minimal-important-change evidence for a single-item overall job-satisfaction measure was located this session. This is a genuine gap rather than a category error.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct"
   },
   "populations_languages_norms": {
    "findings": "Used very widely in occupational and organisational research across many countries and languages, but as a generic method rather than a standardised instrument with published norms. The anchoring evidence is predominantly North American ([Wanous, Reichers & Hudy 1997](https://doi.org/10.1037/0021-9010.82.2.247); [Nagy 2002](https://doi.org/10.1348/096317902167658); [Dolbier et al. 2005](https://doi.org/10.4278/0890-1171-19.3.194)), with a UK university-staff study by [Oshagbemi 1999](https://doi.org/10.1108/02683949910277148) reaching broadly similar conclusions about single versus multiple-item measures. There is no canonical norm table for a UK working-adult population.",
    "grade": "Low",
    "indirectness": "indirect, most psychometric evidence is non-UK and norms for UK working adults are absent"
   },
   "criticisms_controversies": "The single-item measurement debate is the core controversy of this whole track. Critics hold that single items cannot demonstrate internal reliability, cannot separate content from error, and lose the domain coverage of multi-item scales, so that multi-item measures should remain the default ([Fisher, Matthews & Gibbons 2016](https://doi.org/10.1037/a0039139); [Song et al. 2022](https://doi.org/10.1177/10731911221113563)). Defenders argue that for concrete, singular and well-understood constructs single items can match multi-item scales in predictive validity while cutting respondent burden ([Bergkvist & Rossiter 2007](https://doi.org/10.1509/jmkr.44.2.175); [Gardner et al. 1998](https://doi.org/10.1177/0013164498058006003)), and structured guidance now exists on when a single item is defensible ([Diamantopoulos et al. 2012](https://doi.org/10.1007/s11747-011-0300-3)). A specific hazard for job satisfaction is that the estimated ~0.67 reliability is a correction-for-attenuation artefact, not a measured reliability, and is easily over-interpreted. Global items also cannot localise dissatisfaction to a facet, limiting diagnostic and action-planning use.",
   "item_records": [],
   "citations": [
    {
     "doi": "10.1037/0021-9010.82.2.247",
     "authors": "Wanous; Reichers; Hudy",
     "year": "1997",
     "title": "Overall job satisfaction: How good are single-item measures?",
     "journal": "Journal of Applied Psychology",
     "key": "wanous1997",
     "url": "https://doi.org/10.1037/0021-9010.82.2.247"
    },
    {
     "doi": "10.2466/pr0.1996.78.2.631",
     "authors": "Wanous; Reichers",
     "year": "1996",
     "title": "Estimating the Reliability of a Single-Item Measure",
     "journal": "Psychological Reports",
     "key": "wanous1996",
     "url": "https://doi.org/10.2466/pr0.1996.78.2.631"
    },
    {
     "doi": "10.1177/109442810144003",
     "authors": "Wanous; Hudy",
     "year": "2001",
     "title": "Single-Item Reliability: A Replication and Extension",
     "journal": "Organizational Research Methods",
     "key": "wanous2001",
     "url": "https://doi.org/10.1177/109442810144003"
    },
    {
     "doi": "10.1348/096317902167658",
     "authors": "Nagy",
     "year": "2002",
     "title": "Using a single‐item approach to measure facet job satisfaction",
     "journal": "Journal of Occupational and Organizational Psychology",
     "key": "nagy2002",
     "url": "https://doi.org/10.1348/096317902167658"
    },
    {
     "doi": "10.4278/0890-1171-19.3.194",
     "authors": "Dolbier; Webster; McCalister et al.",
     "year": "2005",
     "title": "Reliability and Validity of a Single-Item Measure of Job Satisfaction",
     "journal": "American Journal of Health Promotion",
     "key": "dolbier2005",
     "url": "https://doi.org/10.4278/0890-1171-19.3.194"
    },
    {
     "doi": "10.1108/02683949910277148",
     "authors": "Oshagbemi",
     "year": "1999",
     "title": "Overall job satisfaction: how good are single versus multiple‐item measures?",
     "journal": "Journal of Managerial Psychology",
     "key": "oshagbemi1999",
     "url": "https://doi.org/10.1108/02683949910277148"
    },
    {
     "doi": "10.1111/j.1744-6570.1983.tb02236.x",
     "authors": "SCARPELLO; CAMPBELL",
     "year": "1983",
     "title": "JOB SATISFACTION: ARE ALL THE PARTS THERE?",
     "journal": "Personnel Psychology",
     "key": "scarpello1983",
     "url": "https://doi.org/10.1111/j.1744-6570.1983.tb02236.x"
    },
    {
     "doi": "10.1177/0013164498058006003",
     "authors": "Gardner; Cummings; Dunham et al.",
     "year": "1998",
     "title": "Single-Item Versus Multiple-Item Measurement Scales: An Empirical Comparison",
     "journal": "Educational and Psychological Measurement",
     "key": "gardner1998",
     "url": "https://doi.org/10.1177/0013164498058006003"
    },
    {
     "doi": "10.1509/jmkr.44.2.175",
     "authors": "Bergkvist; Rossiter",
     "year": "2007",
     "title": "The Predictive Validity of Multiple-Item versus Single-Item Measures of the Same Constructs",
     "journal": "Journal of Marketing Research",
     "key": "bergkvist2007",
     "url": "https://doi.org/10.1509/jmkr.44.2.175"
    },
    {
     "doi": "10.1037/a0039139",
     "authors": "Fisher; Matthews; Gibbons",
     "year": "2016",
     "title": "Developing and investigating the use of single-item measures in organizational research.",
     "journal": "Journal of Occupational Health Psychology",
     "key": "fisher2016",
     "url": "https://doi.org/10.1037/a0039139"
    },
    {
     "doi": "10.1007/s11747-011-0300-3",
     "authors": "Diamantopoulos; Sarstedt; Fuchs et al.",
     "year": "2012",
     "title": "Guidelines for choosing between multi-item and single-item scales for construct measurement: a predictive validity perspective",
     "journal": "Journal of the Academy of Marketing Science",
     "key": "diamantopoulos2012",
     "url": "https://doi.org/10.1007/s11747-011-0300-3"
    },
    {
     "doi": "10.1177/10731911221113563",
     "authors": "Song; Howe; Oltmanns et al.",
     "year": "2022",
     "title": "Examining the Concurrent and Predictive Validity of Single Items in Ecological Momentary Assessments",
     "journal": "Assessment",
     "key": "song2022",
     "url": "https://doi.org/10.1177/10731911221113563"
    }
   ],
   "record_notes": "Overall confidence: convergent validity High and well-established (the meta-analytic anchor is strong); organisational criterion validity Low and thin; test-retest genuinely Absent for the single-item job-satisfaction form specifically. This record also carries the single-item-measures-as-a-class discussion for the track. The v0.2 schema is a good fit for single items once scale-level properties (structural validity, internal consistency, invariance) are marked Not-applicable rather than Absent; the one field the schema still made awkward is reliability, because the tradition's headline ~0.67 'reliability' is a corrected convergence estimate that belongs neither in internal_consistency (no items to correlate) nor in test_retest_reliability (not a stability coefficient), so it is documented in the internal_consistency and test-retest summaries with an explicit warning against misreading it as either. Population indirectness is a real limit: the evidence base is predominantly non-UK."
  },
  {
   "instrument_id": "single-stress",
   "display_name": "Single-item perceived stress measure (Elo tradition, SISQ)",
   "instrument_type": "single-item",
   "schema_version": "0.2",
   "identity": {
    "name": "Single-item stress question (SISQ) / single-item measure of stress symptoms",
    "current_version": "A one-item global stress question. In the Elo tradition the item asks about stress understood as tension, restlessness, nervousness, anxiety or sleeplessness owing to worry, with a preceding short definition of stress, typically rated on a 5-point scale. Later occupational studies (for example Houdmont) use a one-item global job-stressfulness rating.",
    "item_count": "1",
    "original_citation": "Elo AL, Leppanen A, Jahkola A (2003). Validity of a single-item measure of stress symptoms. Scandinavian Journal of Work, Environment and Health 29(6):444-451. doi:10.5271/sjweh.752",
    "steward_publisher": "Originates from the Finnish Institute of Occupational Health (Tyoterveyslaitos, TTL); described in the peer-reviewed literature. There is no commercial test publisher and no proprietary distribution product for the single item.",
    "licence_status": "No formal instrument licence or fee-based distribution regime exists for the single stress item; it is a non-proprietary item used freely across occupational research. The originating article (Elo et al. 2003) is published in the Scandinavian Journal of Work, Environment and Health, which is fully gold open access and publishes under a Creative Commons Attribution 4.0 International (CC BY 4.0) licence; this governs the article text, not an instrument licence. Users should attribute Elo et al. 2003; no permission barrier to use the item itself was identified.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "Verified this session by web search of the steward/journal's current pages: the SJWEH open-access page (sjweh.fi/open_access.php) and facts page (sjweh.fi/facts.php), both confirming CC BY 4.0 for all articles and full gold open access since 1 January 2021; corroborated by DOAJ (doaj.org, CC BY, author retains copyright). This confirms the ARTICLE licence only. No item-specific licence or distribution-terms page for the single stress question was located on the Finnish Institute of Occupational Health (TTL) site this session, so no instrument licence is asserted; the item is recorded as having no formal instrument licence, not verified from the founding paper."
   },
   "constructs_claimed": "Perceived psychological stress / stress symptoms, that is subjectively experienced tension, restlessness, nervousness, anxiety or difficulty sleeping attributed to worry, captured in a single global self-rating. In occupational variants, global job stressfulness.",
   "deployment_context_caveat": "The single stress item is a subjective self-rating designed for group-level monitoring and screening, not individual diagnosis. It is not a clinical diagnostic instrument; where it discriminates common mental disorder cases (Houdmont et al. 2021) it does so as a screener with sensitivity/specificity trade-offs, not as a diagnosis. It should not be deployed as a stand-alone individual clinical decision tool in the workplace.",
   "structural_validity": {
    "findings": "Not applicable by construction; a single item has no internal factor structure. Elo et al. did use factor analysis at the study level to position the stress item among other well-being items, but this concerns the item's placement in a wider correlation structure, not its own dimensionality ([Elo, Leppanen & Jahkola 2003](https://doi.org/10.5271/sjweh.752)).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, category error for a single item"
   },
   "convergent_discriminant_validity": {
    "findings": "The best-evidenced property. In the anchoring validation across four independent Nordic and Finnish datasets, the single stress-symptoms item converged with items on psychological symptoms and sleep disturbance and with validated scales of exhaustion, mental health, sleep, vitality and optimism, and behaved consistently with a validated emotional-exhaustion scale across gender, age and industry groups ([Elo, Leppanen & Jahkola 2003](https://doi.org/10.5271/sjweh.752)). In a Swedish primary-health-care validation of the SMS-delivered single-item stress question (SISQ), the item correlated significantly with subscales of the Job Demand-Control-Support and Effort-Reward Imbalance models and with depression, exhaustion and sleep scales, supporting convergent validity ([Arapovic-Johansson et al. 2017](https://doi.org/10.1093/occmed/kqx111)). [Littman et al. 2006](https://doi.org/10.1097/01.ede.0000219721.89552.51) similarly reported acceptable validity for two single-item psychosocial stress measures against longer instruments. Qualitative cognitive-interview work shows respondents anchor a global job-stressfulness item mainly on psychosocial working conditions, with wellbeing a secondary frame, which supports content validity ([Houdmont et al. 2019](https://doi.org/10.3390/ijerph16091480)).",
    "grade": "Moderate",
    "status": "well-established",
    "evidence_form": "mixed",
    "indirectness": "indirect, core evidence is Finnish and Swedish occupational samples; UK evidence exists for the job-stressfulness variant (Houdmont) but the Elo item itself is validated in Nordic populations"
   },
   "criterion_validity_reference_standard": {
    "findings": "Some genuine reference-standard evidence exists for the occupational single-item stress form, which is unusual for this class. [Houdmont et al. 2021](https://doi.org/10.1037/str0000231) tested a single-item job-stressfulness measure against the 12-item General Health Questionnaire (GHQ-12) as a criterion for common mental disorder across seven occupational groups (N=20,658) and found acceptable sensitivity and specificity (both >=70%) in high-strain occupations such as prison officers and public-protection police, with discriminatory power optimal at a >=4 cut-off on a 5-point scale. This is screener-against-standard evidence rather than diagnostic equivalence, and performance depended strongly on base rate. Elo et al. framed criterion validity in terms of plausible associations with health and work characteristics rather than a diagnostic standard ([Elo, Leppanen & Jahkola 2003](https://doi.org/10.5271/sjweh.752)).",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "mixed",
    "indirectness": "indirect, the GHQ-12 criterion evidence is UK occupational groups but concentrated in high-stress occupations (prison, police) rather than the general UK working population"
   },
   "criterion_validity_organisational": {
    "findings": "Present and, for one hard outcome, prospective. [Salminen et al. 2014](https://doi.org/10.1186/1471-2458-14-543) followed 16,385 Finnish forest-industry employees over eight years and found that high stress on the validated single-item measure was associated with about a 40% increased risk of subsequent hospitalised injury, robust to adjustment for age, sex, marital and occupational status, education and physical work environment. [Ravalier 2018](https://doi.org/10.1093/bjsw/bcy023) linked single-item-assessed stress to sickness presenteeism, low job satisfaction and turnover intentions in 1,333 UK social workers, though cross-sectionally. Evidence against recorded sickness absence and actual turnover for the single item specifically remains limited, so the grade is held at Low to Moderate.",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "mixed",
    "indirectness": "indirect for hard outcomes (the strongest prospective evidence is Finnish forest workers and a specific injury outcome); the UK evidence (Ravalier) is cross-sectional and self-reported"
   },
   "internal_consistency": {
    "findings": "Not applicable; a single item has no inter-item covariance and no alpha or omega can be computed.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, category error for a single item"
   },
   "test_retest_reliability": {
    "findings": [
     {
      "coefficient": "0.804",
      "coefficient_type": "kappa",
      "interval": "consecutive days (approximately 1 day)",
      "sample_n": "99",
      "population": "Swedish primary health care employees (SISQ delivered by SMS), whole retest sample",
      "evidence_form": "canonical",
      "citation_key": "arapovic2017"
     },
     {
      "coefficient": "0.868",
      "coefficient_type": "kappa",
      "interval": "consecutive days (approximately 1 day)",
      "sample_n": "52",
      "population": "Swedish primary health care employees (SISQ delivered by SMS), subgroup selected as reporting no meaningful day-to-day change",
      "evidence_form": "canonical",
      "citation_key": "arapovic2017"
     }
    ],
    "grade": "Low",
    "status": "thin",
    "indirectness": "indirect, a single Swedish occupational sample over a very short (approximately 1 day) interval; the weighted-kappa values are appropriate for a fluctuating state but do not establish longer-interval stability, and no UK retest data were located",
    "summary": "Direct test-retest evidence exists but is limited to one study. [Arapovic-Johansson et al. 2017](https://doi.org/10.1093/occmed/kqx111) reported weighted kappa (squared weights) of 0.804 across the full retest sample (N=99) and 0.868 in a subgroup selecting days reported as not meaningfully different (N=52), over a roughly one-day interval, for the SMS-delivered single-item stress question. Because stress is a fluctuating state, the authors deliberately used a short interval; this supports short-term stability but says nothing about week-to-week or month-to-month reliability, and there is no UK retest evidence. Elo et al. 2003 did not report test-retest as their four datasets were cross-sectional."
   },
   "measurement_invariance": {
    "findings": "Not applicable in the multi-group confirmatory-factor sense for a single item. Elo et al. showed the item discriminated as expected between gender, age and industry groups, which is consistent with sensible cross-group behaviour but is not formal invariance evidence ([Elo, Leppanen & Jahkola 2003](https://doi.org/10.5271/sjweh.752)). No study establishing formal single-item equivalence (for example differential item functioning) across sex, age, occupation or language was located this session.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, formal invariance modelling is a category error for a single item; substantive comparability across groups is untested"
   },
   "responsiveness_mic": {
    "findings": "No responsiveness or minimal-important-change evidence for the single stress item was located this session. Predictive-validity evidence (Salminen et al. 2014; Houdmont et al. 2021) speaks to prognostic value, not to sensitivity to within-person change or an anchored minimal important change.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct"
   },
   "populations_languages_norms": {
    "findings": "Validated and used mainly in Nordic occupational populations (Finnish and Swedish), with the job-stressfulness variant applied in UK occupational groups by Houdmont and colleagues ([Elo, Leppanen & Jahkola 2003](https://doi.org/10.5271/sjweh.752); [Arapovic-Johansson et al. 2017](https://doi.org/10.1093/occmed/kqx111); [Houdmont et al. 2019](https://doi.org/10.3390/ijerph16091480); [Houdmont et al. 2021](https://doi.org/10.1037/str0000231)). No canonical UK working-population norm table exists; the Salminen prospective evidence is Finnish forest workers. The short-form Perceived Stress Scale has UK normative data but is a different (multi-item) instrument ([Warttig et al. 2013](https://doi.org/10.1177/1359105313508346)).",
    "grade": "Low",
    "indirectness": "indirect, primary validation is Nordic; UK evidence is concentrated in high-stress occupations rather than the general working population"
   },
   "criticisms_controversies": "The general single-item critique applies (no internal reliability, no content breadth). Specific to stress, the construct is inherently fluctuating, so temporal-stability expectations must be modest and short-interval; a single global rating also conflates the many referents respondents actually draw on. Cognitive-interview work shows people answering a global job-stressfulness item mix working conditions and health referents, so the item is not clean with respect to any one of them ([Houdmont et al. 2019](https://doi.org/10.3390/ijerph16091480)). As a screener its performance is strongly base-rate dependent, working well only where high stress is common ([Houdmont et al. 2021](https://doi.org/10.1037/str0000231)). The word 'stress' itself carries different meanings across respondents and languages, a known threat to comparability.",
   "item_records": [],
   "citations": [
    {
     "doi": "10.5271/sjweh.752",
     "authors": "Elo; Leppänen; Jahkola",
     "year": "2003",
     "title": "Validity of a single-item measure of stress symptoms",
     "journal": "Scandinavian Journal of Work, Environment &amp; Health",
     "key": "elo2003",
     "url": "https://doi.org/10.5271/sjweh.752"
    },
    {
     "doi": "10.1093/occmed/kqx111",
     "authors": "Arapovic-Johansson; Wåhlin; Kwak et al.",
     "year": "2017",
     "title": "Work-related stress assessed by a text message single-item stress question",
     "journal": "Occupational Medicine",
     "key": "arapovic2017",
     "url": "https://doi.org/10.1093/occmed/kqx111"
    },
    {
     "doi": "10.1097/01.ede.0000219721.89552.51",
     "authors": "Littman; White; Satia et al.",
     "year": "2006",
     "title": "Reliability and Validity of 2 Single-Item Measures of Psychosocial Stress",
     "journal": "Epidemiology",
     "key": "littman2006",
     "url": "https://doi.org/10.1097/01.ede.0000219721.89552.51"
    },
    {
     "doi": "10.3390/ijerph16091480",
     "authors": "Houdmont; Jachens; Randall et al.",
     "year": "2019",
     "title": "What Does a Single-Item Measure of Job Stressfulness Assess?",
     "journal": "International Journal of Environmental Research and Public Health",
     "key": "houdmont2019",
     "url": "https://doi.org/10.3390/ijerph16091480"
    },
    {
     "doi": "10.1037/str0000231",
     "authors": "Houdmont; Randall; Kinman et al.",
     "year": "2021",
     "title": "Can a single-item measure of job stressfulness identify common mental disorder?",
     "journal": "International Journal of Stress Management",
     "key": "houdmont2021",
     "url": "https://doi.org/10.1037/str0000231"
    },
    {
     "doi": "10.1186/1471-2458-14-543",
     "authors": "Salminen; Kouvonen; Koskinen et al.",
     "year": "2014",
     "title": "Is a single item stress measure independently associated with subsequent severe injury: a prospective cohort study of 16,385 forest industry employees",
     "journal": "BMC Public Health",
     "key": "salminen2014",
     "url": "https://doi.org/10.1186/1471-2458-14-543"
    },
    {
     "doi": "10.1093/bjsw/bcy023",
     "authors": "Ravalier",
     "year": "2018",
     "title": "Psycho-Social Working Conditions and Stress in UK Social Workers",
     "journal": "The British Journal of Social Work",
     "key": "ravalier2018",
     "url": "https://doi.org/10.1093/bjsw/bcy023"
    },
    {
     "doi": "10.1177/1359105313508346",
     "authors": "Warttig; Forshaw; South et al.",
     "year": "2013",
     "title": "New, normative, English-sample data for the Short Form Perceived Stress Scale (PSS-4)",
     "journal": "Journal of Health Psychology",
     "key": "warttig2013",
     "url": "https://doi.org/10.1177/1359105313508346"
    },
    {
     "doi": "10.1037/a0039139",
     "authors": "Fisher; Matthews; Gibbons",
     "year": "2016",
     "title": "Developing and investigating the use of single-item measures in organizational research.",
     "journal": "Journal of Occupational Health Psychology",
     "key": "fisher2016",
     "url": "https://doi.org/10.1037/a0039139"
    },
    {
     "doi": "10.1007/s11747-011-0300-3",
     "authors": "Diamantopoulos; Sarstedt; Fuchs et al.",
     "year": "2012",
     "title": "Guidelines for choosing between multi-item and single-item scales for construct measurement: a predictive validity perspective",
     "journal": "Journal of the Academy of Marketing Science",
     "key": "diamantopoulos2012",
     "url": "https://doi.org/10.1007/s11747-011-0300-3"
    }
   ],
   "record_notes": "Overall confidence: convergent validity Moderate and well-established; a genuine reference-standard screener finding against GHQ-12 (Houdmont et al. 2021) and a strong prospective organisational finding (Salminen et al. 2014, 40% increased injury risk) lift this above most single items, but both criterion grades stay Low/thin because evidence is concentrated in specific Nordic or high-stress-occupation samples. Test-retest is present but rests on a single short-interval Swedish study (weighted kappa 0.804 to 0.868). The v0.2 structured test-retest object worked well here, letting the two Arapovic-Johansson coefficients be recorded with their different N and selection conditions rather than averaged. Population indirectness is the main limit for a UK audience: the Elo item is Nordic-validated and there is no UK general-working-population norm."
  },
  {
   "instrument_id": "single-fatigue",
   "display_name": "Single-item fatigue measure (Van Hooff)",
   "instrument_type": "single-item",
   "schema_version": "0.2",
   "identity": {
    "name": "Single-item fatigue measure ('How fatigued do you currently feel?')",
    "current_version": "A one-item state-fatigue question ('How fatigued do you currently feel?'), used as a momentary rating within daily-diary designs; typically a single response scale administered several times per day.",
    "item_count": "1",
    "original_citation": "Van Hooff MLM, Geurts SAE, Kompier MAJ, Taris TW (2007). 'How fatigued do you currently feel?' Convergent and discriminant validity of a single-item fatigue measure. Journal of Occupational Health 49(3):224-234. doi:10.1539/joh.49.224",
    "steward_publisher": "Described in the peer-reviewed literature (Journal of Occupational Health). No commercial test publisher or proprietary distribution product; the item is a generic single question.",
    "licence_status": "No formal instrument licence or fee-based distribution regime exists for the single fatigue item; it is a non-proprietary item used freely in research. The originating article (Van Hooff et al. 2007) is in the Journal of Occupational Health, published by the Japan Society for Occupational Health; that journal applies a Creative Commons Attribution-NonCommercial-ShareAlike 4.0 International (CC BY-NC-SA 4.0) licence to the works it publishes (this governs the article text, not an instrument licence). Users should attribute Van Hooff et al. 2007; no permission barrier to use the item itself was identified.",
    "licence_verified_date": "2026-07-12",
    "licence_source": "Verified this session by web search of the journal's current J-STAGE page (jstage.jst.go.jp/browse/joh), which states the Journal of Occupational Health applies the CC BY-NC-SA 4.0 licence to all works it publishes (Japan Society for Occupational Health); corroborated by the PMC record for the journal and by DOAJ. This confirms the ARTICLE licence only. No item-specific licence or distribution-terms page exists for the single fatigue question; the item is recorded as having no formal instrument licence, not verified from the founding paper."
   },
   "constructs_claimed": "Momentary (state) subjective fatigue, that is how tired or fatigued a person feels at the moment of measurement, as distinct from chronic fatigue or need for recovery.",
   "deployment_context_caveat": "Validated specifically as a momentary state measure inside an intensive daily-diary design in one academic-staff sample. Its properties should not be assumed to transfer to single cross-sectional administrations, to chronic-fatigue assessment, or to clinical fatigue screening; workplace deployment outside a diary context is largely untested.",
   "structural_validity": {
    "findings": "Not applicable by construction; a single item has no internal factor structure to assess ([Van Hooff et al. 2007](https://doi.org/10.1539/joh.49.224)).",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, category error for a single item"
   },
   "convergent_discriminant_validity": {
    "findings": "This is the one property with direct evidence, and it is the record's anchor. In a 9-day daily-diary study of 120 academic staff (three measurements per day), the single-item fatigue measure correlated strongly with the validated multi-item Profile of Mood States fatigue subscale (r=0.80) and with conceptually related daily and global measures (work-home interference, sleep complaints, work-related effort, job pressure), supporting convergent validity. Correlations with conceptually distinct measures (daily work pleasure, job control, social support, motivation to learn) were weak or non-significant, supporting discriminant validity ([Van Hooff et al. 2007](https://doi.org/10.1539/joh.49.224)). Broader single-item methodological work is consistent with single items tracking their multi-item counterparts for concrete states ([Fisher, Matthews & Gibbons 2016](https://doi.org/10.1037/a0039139); [Song et al. 2022](https://doi.org/10.1177/10731911221113563)).",
    "grade": "Low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, evidence rests on a single Dutch academic-staff diary sample (N=120); not replicated and not UK"
   },
   "criterion_validity_reference_standard": {
    "findings": "No validation of the single fatigue item against a diagnostic or clinical fatigue reference standard was located this session. The validating study used a multi-item mood scale (POMS fatigue) as a convergent comparator, not a diagnostic standard ([Van Hooff et al. 2007](https://doi.org/10.1539/joh.49.224)).",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, no reference-standard study located"
   },
   "criterion_validity_organisational": {
    "findings": "No study linking the single fatigue item to work outcomes (sickness absence, turnover, objective performance, diagnosed conditions at work) was located this session. The original validation concerned convergent and discriminant validity within a diary design, not prediction of organisational outcomes ([Van Hooff et al. 2007](https://doi.org/10.1539/joh.49.224)). This absence is the finding.",
    "grade": "Absent",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, no organisational-criterion study located"
   },
   "internal_consistency": {
    "findings": "Not applicable; a single item has no inter-item covariance and no alpha or omega can be computed.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, category error for a single item"
   },
   "test_retest_reliability": {
    "findings": [],
    "grade": "Absent",
    "status": "untested",
    "indirectness": "direct, absence applies to the single-item fatigue measure specifically",
    "summary": "No conventional test-retest coefficient for the single-item fatigue measure was located this session. This is expected in part because the item measures a momentary, deliberately fluctuating state, for which high day-to-day stability would not be desired; the validating study modelled within-person daily variation rather than temporal stability ([Van Hooff et al. 2007](https://doi.org/10.1539/joh.49.224)). The absence of any stability coefficient is nonetheless a genuine gap for the instrument, recorded here explicitly rather than papered over."
   },
   "measurement_invariance": {
    "findings": "Not applicable in the multi-group confirmatory-factor sense for a single item, and no substantive cross-group comparability evidence (sex, age, occupation, language) for the single fatigue item was located this session.",
    "grade": "Not-applicable",
    "status": "untested",
    "evidence_form": "canonical",
    "indirectness": "direct, formal invariance modelling is a category error for a single item; substantive comparability is untested"
   },
   "responsiveness_mic": {
    "findings": "The single fatigue item was designed to capture within-day and day-to-day variation, and the diary analysis showed it moved with conceptually related daily variables, which is indirect evidence of sensitivity to change ([Van Hooff et al. 2007](https://doi.org/10.1539/joh.49.224)). However, no formal responsiveness study and no anchored minimal-important-change value were located this session, so established responsiveness and minimal important change are Absent.",
    "grade": "Very low",
    "status": "thin",
    "evidence_form": "canonical",
    "indirectness": "indirect, sensitivity to daily variation is inferred from one diary sample; no formal responsiveness or minimal-important-change evidence exists"
   },
   "populations_languages_norms": {
    "findings": "Validated in a single sample of Dutch academic staff within a diary design ([Van Hooff et al. 2007](https://doi.org/10.1539/joh.49.224)). No broader population validation, no language-adaptation evidence and no norms were located this session. There is no UK working-population evidence for this specific item.",
    "grade": "Very low",
    "indirectness": "indirect, one Dutch academic-staff sample; no UK or general-workforce evidence and no norms"
   },
   "criticisms_controversies": "Beyond the general single-item critique, the specific limitation here is thinness: the psychometric case rests essentially on one 2007 diary study in one occupational group, so external validity is largely unestablished. The item is validated as a momentary state measure, so using it as a general or chronic fatigue indicator, or in single cross-sectional surveys, extends it beyond its evidence. 'Fatigue' also overlaps with sleepiness, need for recovery and emotional exhaustion, and a one-item global rating cannot separate these, which is a construct-clarity concern the wider single-item literature repeatedly raises ([Fisher, Matthews & Gibbons 2016](https://doi.org/10.1037/a0039139)).",
   "item_records": [],
   "citations": [
    {
     "doi": "10.1539/joh.49.224",
     "authors": "Van Hooff; Geurts; Kompier et al.",
     "year": "2007",
     "title": "“How Fatigued Do You Currently Feel?” Convergent and Discriminant Validity of a Single‐Item Fatigue Measure",
     "journal": "Journal of Occupational Health",
     "key": "vanhooff2007",
     "url": "https://doi.org/10.1539/joh.49.224"
    },
    {
     "doi": "10.1037/a0039139",
     "authors": "Fisher; Matthews; Gibbons",
     "year": "2016",
     "title": "Developing and investigating the use of single-item measures in organizational research.",
     "journal": "Journal of Occupational Health Psychology",
     "key": "fisher2016",
     "url": "https://doi.org/10.1037/a0039139"
    },
    {
     "doi": "10.1177/10731911221113563",
     "authors": "Song; Howe; Oltmanns et al.",
     "year": "2022",
     "title": "Examining the Concurrent and Predictive Validity of Single Items in Ecological Momentary Assessments",
     "journal": "Assessment",
     "key": "song2022",
     "url": "https://doi.org/10.1177/10731911221113563"
    },
    {
     "doi": "10.1007/s11747-011-0300-3",
     "authors": "Diamantopoulos; Sarstedt; Fuchs et al.",
     "year": "2012",
     "title": "Guidelines for choosing between multi-item and single-item scales for construct measurement: a predictive validity perspective",
     "journal": "Journal of the Academy of Marketing Science",
     "key": "diamantopoulos2012",
     "url": "https://doi.org/10.1007/s11747-011-0300-3"
    },
    {
     "doi": "10.1509/jmkr.44.2.175",
     "authors": "Bergkvist; Rossiter",
     "year": "2007",
     "title": "The Predictive Validity of Multiple-Item versus Single-Item Measures of the Same Constructs",
     "journal": "Journal of Marketing Research",
     "key": "bergkvist2007",
     "url": "https://doi.org/10.1509/jmkr.44.2.175"
    }
   ],
   "record_notes": "Overall confidence: this is a genuinely thin single-item record and is graded conservatively throughout. The only property with direct primary evidence is convergent/discriminant validity (Low/thin, r=0.80 with POMS fatigue in one N=120 Dutch diary study); reference-standard criterion, organisational criterion and test-retest are all Absent; responsiveness is Very low. That thinness is itself the graded finding, per the honesty rules, rather than a reason to inflate grades. The v0.2 schema handled this cleanly: scale-level properties are marked Not-applicable (single item), and the genuinely missing properties are marked Absent with the absence stated as a finding. Population indirectness is severe for a UK working-adult audience (one non-UK academic sample)."
  }
 ],
 "corrections": [
  {
   "id": "C-0001",
   "instrument_id": "who-5",
   "description": "WHO-5 licence mis-stated in pass one by trusting 2015/2020 review literature and missing WHO's October 2024 republication of the WHO-5 master version under a formal Creative Commons licence.",
   "old_value": "Free to use; distributed at no charge, no licence fee or royalty; may be reproduced and translated with acknowledgement of source (free availability sourced to Topp 2015 / Lara-Cabrera 2020 review literature). No formal open licence stated.",
   "new_value": "Copyright World Health Organization 2024; the WHO-5 master version is available under the Creative Commons Attribution-NonCommercial-ShareAlike 3.0 IGO licence (CC BY-NC-SA 3.0 IGO). Document ref WHO/UCN/MSD/MHE/2024.1. Non-commercial use and adaptation permitted with attribution and share-alike; commercial (including employer/vendor) use requires WHO permission.",
   "source": "https://cdn.who.int/media/docs/default-source/mental-health/who-5_english-original4da539d6ed4b49389e3afe47cda2326a.pdf",
   "date": "2026-07-12"
  },
  {
   "id": "C-0002",
   "instrument_id": "wemwbs",
   "description": "WEMWBS licence-currency update: pass one implied NHS/non-profit users obtain a no-fee non-commercial licence. Warwick Innovations introduced charges for NHS organisations from 1 December 2024.",
   "old_value": "Academic and non-profit users obtain a no-fee non-commercial licence; commercial users pay a tiered fee.",
   "new_value": "Registration still required; non-commercial and commercial licences separated. From 1 December 2024, charges apply to NHS organisations (NHS trusts, GP surgeries and NHS-funded bodies) on a published tiered scale (e.g. up to 30 participants GBP 45, up to 100 GBP 125, up to 1000 GBP 600). Non-NHS academic/non-profit non-commercial registration remains available; commercial users pay fees. Verify current tier with Warwick Innovations before fielding.",
   "source": "https://warwick.ac.uk/services/innovations/wemwbs/licenses/",
   "date": "2026-07-12"
  },
  {
   "id": "C-0003",
   "instrument_id": "phq-9",
   "description": "PHQ-9 licence wording precision: pass one described the status as 'public domain'. The steward terms grant a copyright exemption and free use rather than asserting a formal public-domain dedication.",
   "old_value": "Free / public domain. Pfizer released the PHQ and GAD-7 without copyright restriction.",
   "new_value": "Free for download and use; content on the PHQ Screeners site is 'expressly exempted from Pfizer's general copyright restrictions'. No permission or fee required, attribution to Kroenke et al 2001 expected. This is a free-use copyright exemption, not a formally asserted public-domain status.",
   "source": "https://www.phqscreeners.com/terms",
   "date": "2026-07-12"
  }
 ],
 "licence_reverification_pass_one": [
  {
   "instrument_id": "ons-4",
   "pass_one_statement": "Free to use. Designated National Statistics / GSS Harmonised Principle; ONS-published material is Crown copyright released under the Open Government Licence; no fee or registration. Sourced to Benson 2019 (a paper), not an ONS licence page.",
   "current_status": "CONFIRMED. ONS states the four personal well-being questions are part of the government harmonised standards and 'they are freely available to others and their use is encouraged'. All ONS website content is published under the Open Government Licence v3.0 (Crown copyright). No fee, no registration; open status attaches to the original ONS wording and 0-to-10 scale, not to modified derivatives.",
   "source": "https://www.ons.gov.uk/peoplepopulationandcommunity/wellbeing/methodologies/surveysusingthe4officefornationalstatisticspersonalwellbeingquestions (use encouraged; freely available) and OGL v3.0 site-wide footer; corroborated at https://analysisfunction.civilservice.gov.uk/policy-store/personal-well-being/",
   "verified_date": "2026-07-12",
   "correction_needed": "No. Pass-one characterisation stands; now independently verified against the ONS steward pages rather than a paper."
  },
  {
   "instrument_id": "who-5",
   "pass_one_statement": "Free to use; distributed at no charge, no licence fee or royalty; may be reproduced and translated with acknowledgement of source. Free availability sourced to review literature (Topp 2015, Lara-Cabrera 2020). No formal open licence stated.",
   "current_status": "CORRECTED. WHO's current master version of the WHO-5 (document ref WHO/UCN/MSD/MHE/2024.1) carries: 'Copyright World Health Organization 2024. Some rights reserved. This work is available under the Creative Commons Attribution-NonCommercial-ShareAlike 3.0 IGO licence (CC BY-NC-SA 3.0 IGO)'. So the WHO-5 is now formally an openly licensed work under CC BY-NC-SA 3.0 IGO, which permits non-commercial use and adaptation with attribution and share-alike, and prohibits commercial use without WHO permission. Pass one missed this October 2024 republication.",
   "source": "WHO steward master PDF, https://cdn.who.int/media/docs/default-source/mental-health/who-5_english-original4da539d6ed4b49389e3afe47cda2326a.pdf (© World Health Organization 2024; CC BY-NC-SA 3.0 IGO; ref WHO/UCN/MSD/MHE/2024.1)",
   "verified_date": "2026-07-12",
   "correction_needed": "YES. Correction C-0001: replace the vague 'free to use with acknowledgement' statement with the formal current licence CC BY-NC-SA 3.0 IGO, © WHO 2024. The NonCommercial term is a material change: commercial workplace/vendor deployment now requires WHO permission."
  },
  {
   "instrument_id": "wemwbs",
   "pass_one_statement": "Restricted-but-free-at-point-of-use for non-commercial users with mandatory registration; copyright held jointly by NHS Health Scotland and the Universities of Warwick and Edinburgh; academic/non-profit users get a no-fee non-commercial licence, commercial users pay tiered fees; onward distribution not permitted. Sourced to Warwick licensing pages and Tennant 2007.",
   "current_status": "CONFIRMED with a material currency update. Warwick Innovations' current Licences & Pricing page still requires registration and separates non-commercial from commercial licences, but records that 'from 1st December 2024 charges were introduced for NHS organisations, including NHS trusts, GP surgeries and other organisations currently funded by the NHS', with a published NHS pricing tier (for example, up to 30 participants GBP 45; up to 100 GBP 125; up to 1000 GBP 600). NHS use is therefore no longer free at point of use.",
   "source": "https://warwick.ac.uk/services/innovations/wemwbs/licenses/ (Warwick Innovations WEMWBS Licences & Pricing)",
   "verified_date": "2026-07-12",
   "correction_needed": "YES (currency update). Correction C-0002: pass one implied NHS/non-profit users obtain a no-fee licence; the current steward terms show NHS organisations have been charged on a tiered basis since 1 December 2024. Non-NHS academic/non-profit non-commercial registration remains available; commercial users still pay fees."
  },
  {
   "instrument_id": "hse-msit",
   "pass_one_statement": "Free to use; Indicator Tool and manual published by HSE under UK Crown copyright without charge. A specific licence document was NOT retrieved in pass one, so terms were stated from HSE provenance rather than a cited licence text.",
   "current_status": "CONFIRMED and now verified. The HSE Management Standards Indicator Tool pages and download pages carry the site-wide statement 'All content is available under the Open Government Licence v3.0, except where otherwise stated'. The tool is thus Crown copyright released under OGL v3.0, free to use with attribution.",
   "source": "https://www.hse.gov.uk/stress/standards/notesindicatortool.htm and https://www.hse.gov.uk/stress/standards/downloads.htm (OGL v3.0 footer, Crown copyright)",
   "verified_date": "2026-07-12",
   "correction_needed": "No change to the free-to-use status, but an evidence upgrade: pass one could not cite a licence source, and the current HSE steward pages now confirm OGL v3.0. licence_status should move from 'stated from provenance' to 'verified: OGL v3.0'."
  },
  {
   "instrument_id": "perma",
   "pass_one_statement": "Reported free for research and non-commercial use, distributed without fee via Peggy Kern's website; no formal licence text retrieved, so the non-commercial clause could not be verified from a citable source.",
   "current_status": "CONFIRMED and now verified from the steward. The developer's questionnaires page states: 'You are welcome to use these measures for research or non-commercial purposes, giving credit as noted in the measures. There is no cost involved in using the measures for these purposes. For commercial uses of the measures, contact...'. So the Workplace PERMA-Profiler is free for research/non-commercial use with attribution, and commercial use requires contacting the authors.",
   "source": "https://www.peggykern.org/questionnaires.html (developer/steward, Permissions and Use of the Measures)",
   "verified_date": "2026-07-12",
   "correction_needed": "No change to substance, but an evidence upgrade: the free-for-research/non-commercial-with-attribution status and the explicit commercial-use-by-permission clause are now verified against the steward page rather than inferred from distribution practice."
  },
  {
   "instrument_id": "uwes-9",
   "pass_one_statement": "Free for non-commercial research and educational use (widely-reported convention); distributed via author's website with a preliminary test manual. The manual was not opened, so exact wording of any commercial-use/permission clause was not independently verified; instrument carries a copyright notice, no formal open licence.",
   "current_status": "CONFIRMED and now verified with exact wording. The current UWES survey form on the steward site states: 'Copyright Schaufeli & Bakker (2003). The Utrecht Work Engagement Scale is free for use for non-commercial scientific research. Commercial and/or non-scientific use is prohibited, unless previous written permission is granted by the authors'. So non-commercial scientific research use is free; all commercial and non-scientific use (which would include most employer/vendor workplace deployment) is prohibited without prior written permission.",
   "source": "https://www.wilmarschaufeli.nl/publications/Schaufeli/Tests/UWES_GB_17.pdf (author/steward UWES form copyright notice)",
   "verified_date": "2026-07-12",
   "correction_needed": "No change to the free-for-research status, but an important clarification now verified: commercial and non-scientific use (relevant to workplace/vendor deployment) is explicitly prohibited without prior written permission from the authors. Pass one flagged this as unverified; it is now confirmed."
  },
  {
   "instrument_id": "phq-9",
   "pass_one_statement": "Free / public domain. Pfizer released the PHQ and GAD-7 without copyright restriction and at no charge; no permission required to reproduce, translate, display or distribute; attribution to Kroenke 2001 expected. No DOI-bearing licence source retrieved in pass one.",
   "current_status": "CONFIRMED and now verified, with a wording refinement. The official steward terms state: 'Content found at the PHQ Screeners site is expressly exempted from Pfizer's general copyright restrictions; content found on the PHQ Screeners site is free for download and use as stated within the PHQ Screeners site'. This confirms free download and use with no permission or fee, but is a copyright exemption / free-use grant rather than a formal public-domain dedication.",
   "source": "https://www.phqscreeners.com/terms (official PHQ Screeners terms, Pfizer)",
   "verified_date": "2026-07-12",
   "correction_needed": "Minor precision. Correction C-0003: refine 'public domain' to 'free for download and use; expressly exempted from Pfizer's copyright restrictions'. Functionally free/no-permission-required, but the steward does not assert a formal public-domain status."
  }
 ],
 "verifications": [
  {
   "id": "V-0001",
   "date": "2026-07-12",
   "record": "tis-6",
   "field": "identity.licence_status",
   "note": "Licence verified at steward journal page (CC BY 4.0); performed at ingest by the maintaining thread, upgrading the pass-two not-verified flag.",
   "source": "https://sajhrm.co.za/index.php/sajhrm/article/view/507"
  }
 ],
 "changelog": [
  {
   "version": "0.2.1",
   "date": "2026-07-12",
   "change": "ONS-4 item split per schema v0.2 rule 4: added four first-class item records (ons-4-life-satisfaction, ons-4-worthwhile, ons-4-happiness, ons-4-anxiety) with parent_id ons-4; parent gains an items list. Graded evidence remains at set level (item properties reference the parent with evidence_form 'parent'; internal consistency and structural validity are Not-applicable at item level). Item records carry question_bank_ref so the registry and the question bank share one canonical per-item description. The 27-row grade matrix is unchanged; item records are children, not matrix rows."
  }
 ]
}
