From 3e599211fd29695a0e2dcba103a2cab76e0051df Mon Sep 17 00:00:00 2001 From: bakodramane Date: Wed, 3 Jun 2026 22:07:01 +0200 Subject: [PATCH] =?UTF-8?q?feat:=20Phase=202=20=E2=80=94=20recommender=20e?= =?UTF-8?q?ngine=20with=2015+=20test=20scenarios?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pure function recommend(DataAvailability) → Recommendation[] that maps user data flags to a ranked list of SAE methods with rationale and caveats. Implements §3 priority table from docs/SAE-CATALOGUE.md: - Filters methods by required inputs and target type - Scores methods by scenario (poverty, binary/proportion, count, continuous) - Handles outlier, spatial, out-of-sample, and weights sub-scenarios - Always appends direct estimator last as a benchmark - Sets stataV14Warning for methods with stataMinVersion > 14 - Builds whyApplicable strings and inferred caveats per scenario 380 tests pass (56 new recommender tests + 324 existing). Co-Authored-By: Claude Sonnet 4.6 --- src/engine/recommender.test.ts | 309 ++++++++++++++++++++++++ src/engine/recommender.ts | 418 +++++++++++++++++++++++++++++++++ 2 files changed, 727 insertions(+) create mode 100644 src/engine/recommender.test.ts create mode 100644 src/engine/recommender.ts diff --git a/src/engine/recommender.test.ts b/src/engine/recommender.test.ts new file mode 100644 index 0000000..53819a4 --- /dev/null +++ b/src/engine/recommender.test.ts @@ -0,0 +1,309 @@ +import { describe, it, expect } from 'vitest' +import { recommend } from './recommender.js' +import type { DataAvailability } from '../types/index.js' + +// Baseline availability object — override individual flags per test. +const base: DataAvailability = { + hasMicrodata: false, + hasAreaAggregates: false, + hasWeights: false, + hasCensusAuxiliaries: false, + hasContiguityMatrix: false, + hasCoordinates: false, + hasOutOfSampleAreas: false, + targetType: 'continuous', + likelyOutliers: false, + likelySpatialCorrelation: false, +} + +function ids(recs: ReturnType): string[] { + return recs.map(r => r.entry.id) +} + +function rankOf(recs: ReturnType, id: string): number { + const rec = recs.find(r => r.entry.id === id) + if (!rec) throw new Error(`Method '${id}' not found in recommendations`) + return rec.rank +} + +// ── Scenario 1: Area aggregates only, continuous target → FH ranks first ─────── +describe('Scenario 1 — area aggregates only, continuous target', () => { + const avail: DataAvailability = { + ...base, + hasAreaAggregates: true, + hasCensusAuxiliaries: true, + targetType: 'continuous', + } + const recs = recommend(avail) + + it('returns results', () => expect(recs.length).toBeGreaterThan(0)) + it('FH-EBLUP ranks first', () => expect(recs[0].entry.id).toBe('fh-eblup')) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) + it('every recommendation has non-empty whyApplicable', () => + recs.forEach(r => expect(r.whyApplicable.length).toBeGreaterThan(0))) +}) + +// ── Scenario 2: Microdata + area auxiliaries, continuous → BHF first ─────────── +describe('Scenario 2 — microdata + area auxiliaries, continuous target', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasCensusAuxiliaries: true, + targetType: 'continuous', + } + const recs = recommend(avail) + + it('BHF-EBLUP ranks first', () => expect(recs[0].entry.id).toBe('bhf-eblup')) + it('BHF-EBLUP appears', () => expect(ids(recs)).toContain('bhf-eblup')) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) + it('every recommendation has non-empty whyApplicable', () => + recs.forEach(r => expect(r.whyApplicable.length).toBeGreaterThan(0))) +}) + +// ── Scenario 3: Poverty target + microdata + census → EBP first, ELL present ── +describe('Scenario 3 — poverty target, microdata + census', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasWeights: true, + hasCensusAuxiliaries: true, + targetType: 'poverty', + } + const recs = recommend(avail) + + it('EBP/CensusEB ranks first', () => expect(recs[0].entry.id).toBe('ebp-censuseb')) + it('ELL is present', () => expect(ids(recs)).toContain('ell')) + it('EBP ranks above ELL', () => expect(rankOf(recs, 'ebp-censuseb')).toBeLessThan(rankOf(recs, 'ell'))) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) +}) + +// ── Scenario 4: Binary target + microdata → GLMM-binary above FH ─────────────── +describe('Scenario 4 — binary target with microdata', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasAreaAggregates: true, + hasCensusAuxiliaries: true, + targetType: 'binary', + } + const recs = recommend(avail) + + it('GLMM-binary ranks first', () => expect(recs[0].entry.id).toBe('glmm-binary')) + it('FH-EBLUP is present', () => expect(ids(recs)).toContain('fh-eblup')) + it('GLMM-binary ranks above FH-EBLUP', () => + expect(rankOf(recs, 'glmm-binary')).toBeLessThan(rankOf(recs, 'fh-eblup'))) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) +}) + +// ── Scenario 5: Count target → GLMM-count appears; two-part flagged ──────────── +describe('Scenario 5 — count target', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasCensusAuxiliaries: true, + targetType: 'count', + } + const recs = recommend(avail) + + it('GLMM-count appears', () => expect(ids(recs)).toContain('glmm-count')) + it('two-part / ZI model appears', () => expect(ids(recs)).toContain('two-part-zinfl')) + it('GLMM-count ranks above two-part', () => + expect(rankOf(recs, 'glmm-count')).toBeLessThan(rankOf(recs, 'two-part-zinfl'))) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) +}) + +// ── Scenario 6: Outlier flag, continuous → M-quantile and REBLUP above EBLUP ── +describe('Scenario 6 — outlier flag with continuous target', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasCensusAuxiliaries: true, + targetType: 'continuous', + likelyOutliers: true, + } + const recs = recommend(avail) + + it('M-quantile appears', () => expect(ids(recs)).toContain('m-quantile')) + it('REBLUP appears', () => expect(ids(recs)).toContain('reblup')) + it('BHF-EBLUP appears', () => expect(ids(recs)).toContain('bhf-eblup')) + it('M-quantile ranks above BHF-EBLUP', () => + expect(rankOf(recs, 'm-quantile')).toBeLessThan(rankOf(recs, 'bhf-eblup'))) + it('REBLUP ranks above BHF-EBLUP', () => + expect(rankOf(recs, 'reblup')).toBeLessThan(rankOf(recs, 'bhf-eblup'))) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) + it('BHF-EBLUP has an outlier caveat', () => { + const bhf = recs.find(r => r.entry.id === 'bhf-eblup')! + expect(bhf.caveats.some(c => c.toLowerCase().includes('outlier'))).toBe(true) + }) +}) + +// ── Scenario 7: Spatial flag → Spatial-FH present and ranked above FH ────────── +describe('Scenario 7 — spatial flag, area aggregates available', () => { + const avail: DataAvailability = { + ...base, + hasAreaAggregates: true, + hasCensusAuxiliaries: true, + hasContiguityMatrix: true, + targetType: 'continuous', + likelySpatialCorrelation: true, + } + const recs = recommend(avail) + + it('Spatial-FH is present', () => expect(ids(recs)).toContain('spatial-fh')) + it('FH-EBLUP is present', () => expect(ids(recs)).toContain('fh-eblup')) + it('Spatial-FH ranks above FH-EBLUP', () => + expect(rankOf(recs, 'spatial-fh')).toBeLessThan(rankOf(recs, 'fh-eblup'))) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) +}) + +// ── Scenario 8: No microdata, no area aggregates → only direct returned ───────── +describe('Scenario 8 — no data available', () => { + const avail: DataAvailability = { ...base } + const recs = recommend(avail) + + it('returns exactly one recommendation', () => expect(recs).toHaveLength(1)) + it('the only result is direct', () => expect(recs[0].entry.id).toBe('direct')) + it('direct has a degraded caveat', () => + expect(recs[0].caveats.some(c => c.toLowerCase().includes('unavailable'))).toBe(true)) +}) + +// ── Scenario 9: Out-of-sample areas → EBP preferred; OOS caveats present ─────── +describe('Scenario 9 — out-of-sample areas flag', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasCensusAuxiliaries: true, + hasOutOfSampleAreas: true, + targetType: 'continuous', + } + const recs = recommend(avail) + + it('EBP/CensusEB ranks first', () => expect(recs[0].entry.id).toBe('ebp-censuseb')) + it('every result has an OOS caveat', () => + recs.forEach(r => + expect(r.caveats.some(c => c.toLowerCase().includes('out-of-sample'))).toBe(true) + )) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) +}) + +// ── Scenario 10: Weights present → GREG appears ───────────────────────────────── +describe('Scenario 10 — survey weights available', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasWeights: true, + hasCensusAuxiliaries: true, + targetType: 'continuous', + } + const recs = recommend(avail) + + it('GREG appears in results', () => expect(ids(recs)).toContain('greg')) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) +}) + +// ── Scenario 11: Method needing Stata v17 carries stataV14Warning ────────────── +describe('Scenario 11 — Stata v14 warning for high-version methods', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasWeights: true, + hasCensusAuxiliaries: true, + targetType: 'poverty', + } + const recs = recommend(avail) + + it('EBP/CensusEB (stataMinVersion 17) carries stataV14Warning', () => { + const ebp = recs.find(r => r.entry.id === 'ebp-censuseb')! + expect(ebp.stataV14Warning).toBe(true) + }) + it('ELL (stataMinVersion 17) carries stataV14Warning', () => { + const ell = recs.find(r => r.entry.id === 'ell')! + expect(ell.stataV14Warning).toBe(true) + }) + it('methods with stataMinVersion 14 do not carry stataV14Warning', () => { + recs + .filter(r => r.entry.stataMinVersion <= 14) + .forEach(r => expect(r.stataV14Warning).toBe(false)) + }) +}) + +// ── Scenario 12: All results have non-empty whyApplicable ────────────────────── +describe('Scenario 12 — whyApplicable always non-empty', () => { + const scenarios: DataAvailability[] = [ + { ...base, hasAreaAggregates: true, hasCensusAuxiliaries: true }, + { ...base, hasMicrodata: true, hasCensusAuxiliaries: true }, + { ...base, hasMicrodata: true, hasWeights: true, hasCensusAuxiliaries: true, targetType: 'poverty' }, + { ...base, hasMicrodata: true, hasCensusAuxiliaries: true, targetType: 'binary' }, + { ...base, hasMicrodata: true, hasCensusAuxiliaries: true, targetType: 'count' }, + { ...base, hasMicrodata: true, hasCensusAuxiliaries: true, likelyOutliers: true }, + ] + + scenarios.forEach((avail, i) => { + it(`Scenario ${i + 1}: every recommendation has a non-empty whyApplicable`, () => { + const recs = recommend(avail) + recs.forEach(r => + expect(r.whyApplicable.trim().length, `${r.entry.id} has empty whyApplicable`).toBeGreaterThan(0) + ) + }) + }) +}) + +// ── Scenario 13: Spatial flag with microdata → spatial method ranks above non-spatial +describe('Scenario 13 — spatial flag with microdata and coordinates', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasAreaAggregates: true, + hasCensusAuxiliaries: true, + hasContiguityMatrix: true, + hasCoordinates: true, + targetType: 'continuous', + likelySpatialCorrelation: true, + } + const recs = recommend(avail) + + it('a spatial method appears', () => + expect(recs.some(r => r.entry.spatial)).toBe(true)) + it('spatial method ranks above BHF-EBLUP (non-spatial)', () => { + const bestSpatial = recs.find(r => r.entry.spatial)! + expect(bestSpatial.rank).toBeLessThan(rankOf(recs, 'bhf-eblup')) + }) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) +}) + +// ── Scenario 14: Outliers + spatial + coordinates → MQGWR ranks first ────────── +describe('Scenario 14 — outliers and spatial correlation with coordinates', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasCensusAuxiliaries: true, + hasCoordinates: true, + targetType: 'continuous', + likelyOutliers: true, + likelySpatialCorrelation: true, + } + const recs = recommend(avail) + + it('MQGWR ranks first', () => expect(recs[0].entry.id).toBe('mqgwr')) + it('M-quantile is present', () => expect(ids(recs)).toContain('m-quantile')) + it('REBLUP is present', () => expect(ids(recs)).toContain('reblup')) + it('direct estimator ranks last', () => expect(recs[recs.length - 1].entry.id).toBe('direct')) +}) + +// ── Scenario 15: recommend() is pure — same inputs always produce same output ── +describe('Scenario 15 — purity and determinism', () => { + const avail: DataAvailability = { + ...base, + hasMicrodata: true, + hasCensusAuxiliaries: true, + targetType: 'continuous', + likelyOutliers: true, + } + + it('two calls with identical inputs return identical results', () => { + const a = recommend(avail) + const b = recommend(avail) + expect(ids(a)).toEqual(ids(b)) + expect(a.map(r => r.rank)).toEqual(b.map(r => r.rank)) + }) +}) diff --git a/src/engine/recommender.ts b/src/engine/recommender.ts new file mode 100644 index 0000000..05b67b6 --- /dev/null +++ b/src/engine/recommender.ts @@ -0,0 +1,418 @@ +import type { CatalogueEntry, DataAvailability } from '../types/index.js' +import { catalogue } from '../catalogue/index.js' + +export interface Recommendation { + entry: CatalogueEntry + rank: number + whyApplicable: string + caveats: string[] + stataV14Warning: boolean +} + +const DIRECT_ID = 'direct' + +// ── Eligibility ──────────────────────────────────────────────────────────────── + +function isEligible(entry: CatalogueEntry, availability: DataAvailability): boolean { + // Direct estimator is always included as a benchmark, even when inputs are missing. + if (entry.id === DIRECT_ID) return true + + const req = entry.requiredInputs + + // Target type must be supported (skip check when target is unknown). + if ( + availability.targetType !== 'unknown' && + !entry.targetTypes.includes(availability.targetType) + ) { + return false + } + + if (req.microdata && !availability.hasMicrodata) return false + if (req.areaAggregates && !availability.hasAreaAggregates) return false + if (req.weights && !availability.hasWeights) return false + if (req.contiguityMatrix && !availability.hasContiguityMatrix) return false + if (req.coordinates && !availability.hasCoordinates) return false + if (req.censusAuxiliaries !== 'none' && !availability.hasCensusAuxiliaries) return false + + return true +} + +// ── Priority scoring ─────────────────────────────────────────────────────────── +// Lower score = higher rank. Direct is always 9999 (last). + +function computeScore(entry: CatalogueEntry, availability: DataAvailability): number { + const { id } = entry + const { + targetType, + hasMicrodata, + hasAreaAggregates, + hasWeights, + hasCensusAuxiliaries, + hasContiguityMatrix, + hasCoordinates, + hasOutOfSampleAreas, + likelyOutliers, + likelySpatialCorrelation, + } = availability + + if (id === DIRECT_ID) return 9999 + + // ── Poverty target ────────────────────────────────────────────────────────── + if (targetType === 'poverty') { + if (id === 'ebp-censuseb') return 10 + if (id === 'ell') return 20 + return 500 + } + + // ── Binary / Proportion target ────────────────────────────────────────────── + if (targetType === 'binary' || targetType === 'proportion') { + if (hasMicrodata) { + if (id === 'glmm-binary') return 10 + if (id === 'hb-unit') return 20 + if (id === 'fh-eblup') return 30 + if (id === 'hb-fh') return 40 + } else { + if (id === 'fh-eblup') return 10 + if (id === 'hb-fh') return 20 + } + return 500 + } + + // ── Count target ──────────────────────────────────────────────────────────── + if (targetType === 'count') { + if (id === 'glmm-count') return 10 + if (id === 'two-part-zinfl') return 20 + if (id === 'hb-unit') return 30 + return 500 + } + + // ── Continuous (or unknown) target ────────────────────────────────────────── + + // Area aggregates only — no microdata + if (!hasMicrodata) { + const spatialActive = likelySpatialCorrelation && hasContiguityMatrix + if (spatialActive) { + if (id === 'spatial-fh') return 10 + if (id === 'fh-eblup') return 20 + if (id === 'robust-fh') return likelyOutliers ? 22 : 35 + if (id === 'hb-fh') return 25 + return 500 + } + if (likelyOutliers) { + if (id === 'fh-eblup') return 10 + if (id === 'robust-fh') return 12 + if (id === 'hb-fh') return 20 + if (id === 'spatial-fh') return 30 + return 500 + } + // Standard area-level + if (id === 'fh-eblup') return 10 + if (id === 'hb-fh') return 20 + if (id === 'spatial-fh') return 30 + if (id === 'robust-fh') return 35 + return 500 + } + + // Microdata available + + // Outliers + Spatial + if (likelyOutliers && likelySpatialCorrelation) { + if (id === 'mqgwr' && hasCoordinates) return 10 + if (id === 'm-quantile') return 20 + if (id === 'reblup') return 30 + if (id === 'hb-unit') return 50 + if (id === 'bhf-eblup') return 200 + return 500 + } + + // Outliers only + if (likelyOutliers) { + if (id === 'm-quantile') return 10 + if (id === 'reblup') return 20 + if (id === 'mqgwr' && hasCoordinates) return 30 + if (id === 'hb-unit') return 50 + if (id === 'bhf-eblup') return 200 + return 500 + } + + // Spatial only + if (likelySpatialCorrelation) { + if (id === 'spatial-fh' && hasAreaAggregates) return 10 + if (id === 'mqgwr' && hasCoordinates) return 15 + if (id === 'bhf-eblup') return 20 + if (id === 'fh-eblup' && hasAreaAggregates) return 25 + if (id === 'm-quantile') return 30 + if (id === 'hb-unit') return 35 + return 500 + } + + // Out-of-sample areas (§3: EBP first, FH second) + if (hasOutOfSampleAreas) { + if (id === 'ebp-censuseb' && hasCensusAuxiliaries) return 10 + if (id === 'fh-eblup' && hasAreaAggregates) return 15 + if (id === 'bhf-eblup') return 20 + if (id === 'm-quantile') return 25 + if (id === 'reblup') return 30 + if (id === 'hb-unit') return 35 + if (id === 'greg' && hasWeights) return 40 + return 500 + } + + // Standard microdata + continuous (§3: BHF first, EBP second) + if (id === 'bhf-eblup') return 10 + if (id === 'ebp-censuseb') return hasCensusAuxiliaries ? 15 : 500 + if (id === 'greg' && hasWeights) return 20 + if (id === 'm-quantile') return 25 + if (id === 'reblup') return 30 + if (id === 'hb-unit') return 35 + if (id === 'greg') return 40 + return 500 +} + +// ── Human-readable explanation ───────────────────────────────────────────────── + +function buildWhyApplicable(entry: CatalogueEntry, availability: DataAvailability): string { + const { id } = entry + const { + targetType, + hasMicrodata, + hasAreaAggregates, + hasCensusAuxiliaries, + likelyOutliers, + likelySpatialCorrelation, + hasOutOfSampleAreas, + hasContiguityMatrix, + hasCoordinates, + hasWeights, + } = availability + + if (id === DIRECT_ID) { + return 'Use as a benchmark to validate model-based estimates against the raw survey data.' + } + + // Poverty-specific explanations + if (targetType === 'poverty') { + if (id === 'ebp-censuseb') { + return 'Best-practice method for poverty mapping: simulates welfare distributions onto census microdata to estimate non-linear poverty and inequality indicators (headcount, gap, Gini).' + } + if (id === 'ell') { + return 'Alternative poverty-mapping method that decomposes residuals into location and household components; suitable when EBP normality assumptions are harder to satisfy.' + } + } + + // Binary / proportion + if (targetType === 'binary' || targetType === 'proportion') { + if (id === 'glmm-binary') { + return 'Correctly models binary or proportion outcomes using a logistic random-intercept model on your survey microdata; respects the [0,1] range of proportions.' + } + if (id === 'hb-unit') { + return 'Bayesian unit-level model providing exact posterior distributions for binary/proportion targets, especially valuable when area sample sizes are very small.' + } + if (id === 'fh-eblup') { + return hasMicrodata + ? 'Area-level Fay–Herriot model using your pre-computed direct estimates; works alongside your microdata for a design-consistent comparison.' + : 'Standard area-level method combining direct estimates with auxiliary regression; no microdata required and compatible with Stata 14+.' + } + if (id === 'hb-fh') { + return 'Bayesian Fay–Herriot model providing exact small-sample inference for binary/proportion targets; no microdata required.' + } + } + + // Count + if (targetType === 'count') { + if (id === 'glmm-count') { + return 'Correctly models count outcomes using a Poisson mixed model; captures between-area heterogeneity beyond the Poisson mean.' + } + if (id === 'two-part-zinfl') { + return 'Two-part model for count data with excess zeros: a separate model for whether an outcome occurs and for its magnitude when non-zero.' + } + if (id === 'hb-unit') { + return 'Bayesian unit-level Poisson model providing exact posterior distributions for count outcomes; suited to sparse data with very small area samples.' + } + } + + // Area aggregates only (no microdata) + if (!hasMicrodata && hasAreaAggregates) { + const spatialActive = likelySpatialCorrelation && hasContiguityMatrix + if (id === 'spatial-fh') { + return spatialActive + ? 'Extends Fay–Herriot with a spatial autoregressive (SAR) model for area random effects; ranked first because spatial correlation is indicated and a contiguity matrix is available.' + : 'Spatial Fay–Herriot model; eligible because a contiguity matrix is available — consider promoting to rank 1 if spatial correlation is confirmed.' + } + if (id === 'fh-eblup') { + return spatialActive + ? 'Standard Fay–Herriot model; spatial correlation is indicated, so Spatial-FH is preferred above, but FH remains a useful non-spatial baseline.' + : 'Standard area-level SAE method combining direct estimates with auxiliary regression; the canonical starting point when only area-level data are available.' + } + if (id === 'robust-fh') { + return likelyOutliers + ? 'Robustified Fay–Herriot that downweights outlying areas; ranked highly because you indicated likely outliers in the direct estimates.' + : 'Robustified Fay–Herriot; ranked below standard FH here, but useful if diagnostic plots reveal influential area observations.' + } + if (id === 'hb-fh') { + return 'Bayesian Fay–Herriot model providing exact small-sample inference; no microdata required — particularly suited when areas have very small sample sizes (ni < 5).' + } + } + + // Microdata available + if (hasMicrodata) { + if (likelyOutliers && likelySpatialCorrelation) { + if (id === 'mqgwr' && hasCoordinates) { + return 'Geographically weighted M-quantile regression handles both outliers and spatial non-stationarity simultaneously; ranked first because both flags are set and coordinates are available.' + } + if (id === 'm-quantile') { + return 'Robust M-quantile estimator that handles outliers without strong distributional assumptions; consider MQGWR if coordinates are available for full spatial adjustment.' + } + if (id === 'reblup') { + return 'Robust EBLUP using Huber influence functions to downweight outlying units; mixed-model framework alternative to M-quantile.' + } + } + + if (likelyOutliers) { + if (id === 'm-quantile') { + return 'Robust, distribution-free estimator using M-quantile regression; ranked first because you indicated likely outliers in the survey data.' + } + if (id === 'reblup') { + return 'Robust EBLUP using Huber influence functions; ranked highly because outliers are indicated — downweights influential units within a mixed-model framework.' + } + if (id === 'mqgwr' && hasCoordinates) { + return 'M-quantile GWR adds spatial non-stationarity to robust estimation; available because coordinates are present, though spatial correlation was not flagged.' + } + if (id === 'bhf-eblup') { + return 'Standard BHF EBLUP included for reference; caution — outliers are indicated, and M-quantile or REBLUP are more appropriate alternatives.' + } + } + + if (likelySpatialCorrelation) { + if (id === 'spatial-fh' && hasAreaAggregates) { + return 'Spatial Fay–Herriot borrows strength from neighbouring areas; ranked first for the spatial scenario because area-level data and a contiguity matrix are both available.' + } + if (id === 'mqgwr' && hasCoordinates) { + return 'M-quantile GWR handles spatial non-stationarity using a distance-weighted local regression; ranked because spatial correlation is indicated and coordinates are available.' + } + if (id === 'bhf-eblup') { + return 'Standard unit-level EBLUP; spatial correlation is indicated — consider Spatial-FH or MQGWR if the spatial structure is important.' + } + if (id === 'fh-eblup' && hasAreaAggregates) { + return 'Standard Fay–Herriot included as a non-spatial baseline; Spatial-FH is preferred when spatial correlation is indicated.' + } + if (id === 'm-quantile') { + return 'M-quantile estimator; spatial correlation is indicated, but spatial extensions (MQGWR) are preferred if coordinates are available.' + } + if (id === 'hb-unit') { + return 'Bayesian unit-level model; spatial correlation is indicated — consider extending to a spatial HB model if available in your software.' + } + } + + if (hasOutOfSampleAreas) { + if (id === 'ebp-censuseb') { + return 'EBP/CensusEB ranked first because you have out-of-sample areas and census microdata; out-of-sample areas are predicted via a synthetic predictor using the fitted model.' + } + if (id === 'fh-eblup' && hasAreaAggregates) { + return 'Fay–Herriot provides synthetic predictions for out-of-sample areas using fixed-effect predictions; a good area-level fallback when microdata coverage is incomplete.' + } + if (id === 'bhf-eblup') { + return 'BHF EBLUP handles out-of-sample areas via a synthetic predictor (fixed effects only); population means of auxiliaries are required for all target areas.' + } + } + + // Standard microdata + continuous + if (id === 'bhf-eblup') { + return 'Standard unit-level EBLUP combining individual survey records with census population means; the canonical first choice for continuous microdata with area auxiliaries.' + } + if (id === 'ebp-censuseb') { + return hasCensusAuxiliaries + ? 'Empirical Best Predictor using census microdata; can estimate non-linear indicators and handles out-of-sample areas, making it a strong alternative to BHF.' + : 'EBP/CensusEB requires unit-level census microdata, which is not indicated as available.' + } + if (id === 'greg' && hasWeights) { + return 'GREG estimator uses your survey weights for design-consistent estimation, calibrated against known area-level population totals; included because weights are available.' + } + if (id === 'greg') { + return 'GREG estimator provides design-consistent estimates calibrated against population totals; requires survey weights.' + } + if (id === 'm-quantile') { + return 'Robust M-quantile estimator; a useful alternative or robustness check when distributional assumptions of the standard EBLUP may be questionable.' + } + if (id === 'reblup') { + return 'Robust EBLUP using Huber influence functions; preferred over M-quantile when a mixed-model framework is desired and robustness to unit-level outliers is needed.' + } + if (id === 'hb-unit') { + return 'Bayesian unit-level model providing exact posterior inference; particularly suited to very small area sample sizes or when probability statements about estimates are needed.' + } + } + + return entry.whyChooseThis +} + +// ── Caveats ──────────────────────────────────────────────────────────────────── + +function buildCaveats(entry: CatalogueEntry, availability: DataAvailability): string[] { + const caveats: string[] = [...(entry.caveats ?? [])] + const { likelyOutliers, likelySpatialCorrelation, hasOutOfSampleAreas } = availability + + // Outlier warning for non-robust methods + if (likelyOutliers && !entry.robust && entry.id !== DIRECT_ID) { + caveats.push( + 'You indicated likely outliers — M-quantile or REBLUP are more robust alternatives.' + ) + } + + // Spatial warning for non-spatial methods + if (likelySpatialCorrelation && !entry.spatial && entry.id !== DIRECT_ID) { + caveats.push( + 'You indicated likely spatial correlation — Spatial Fay–Herriot or MQGWR model this explicitly.' + ) + } + + // Out-of-sample caveat + if (hasOutOfSampleAreas) { + caveats.push( + 'Out-of-sample areas: estimates rely on a synthetic predictor (fixed effects only); model assumptions are especially critical for these areas.' + ) + } + + // Unit-level census requirement + if (entry.requiredInputs.censusAuxiliaries === 'unit') { + caveats.push( + 'Requires unit-level (individual record) census microdata — area-level population means alone are not sufficient.' + ) + } + + // Stata version caveat + if (entry.stataMinVersion > 14) { + caveats.push( + `Stata implementation requires v${entry.stataMinVersion}+. For Stata 14, use the R script or the fallback command shown in the .do file.` + ) + } + + // Direct estimator degraded warning + if (entry.id === DIRECT_ID && (!availability.hasMicrodata || !availability.hasWeights)) { + caveats.push( + 'Survey microdata and/or sampling weights are unavailable; direct estimation cannot be computed in the standard sense.' + ) + } + + return caveats +} + +// ── Public API ───────────────────────────────────────────────────────────────── + +export function recommend(availability: DataAvailability): Recommendation[] { + const eligible = catalogue.filter(entry => isEligible(entry, availability)) + + const scored = eligible.map(entry => ({ + entry, + score: computeScore(entry, availability), + })) + + scored.sort((a, b) => a.score - b.score) + + return scored.map((item, index) => ({ + entry: item.entry, + rank: index + 1, + whyApplicable: buildWhyApplicable(item.entry, availability), + caveats: buildCaveats(item.entry, availability), + stataV14Warning: item.entry.stataMinVersion > 14, + })) +}