Skip to content

API reference

Functions

Compute the stratification index proposed in Zhou (2012).

Call with arrays — strat(outcome, strata, weights=w) — or with a DataFrame / mapping of columns first: strat(df, outcome="income", strata="big_class", weights="weight", group="education"). In data mode group_name defaults to the group column name.

Parameters:

Name Type Description Default
outcome

Numeric array of outcomes (or a column name in data mode).

None
strata

Array of the same length indicating strata membership. pandas Categorical keeps its category order.

None
weights

Optional numeric array of sampling weights.

None
ordered bool

If True, strata are taken as pre-ordered ascendingly (by their level order); otherwise they are ordered by average percentile rank.

False
group

Optional grouping factor. If supplied (with more than one level), the result includes a between-/within-group decomposition of the overall stratification.

None
group_name str | None

Label used for the group in printed output (R derives it from the expression passed as group; Python uses the column name in data mode, else "group", unless given explicitly).

None
se_method str

"approx" (default) — the Goodman & Kruskal (1963) approximation, as in the R package; "bootstrap" — standard deviation of the index over n_boot resamples of the complete cases (percentile ranks and stratum order are recomputed in every replicate).

'approx'
n_boot int

Number of bootstrap replicates (se_method="bootstrap" only).

200
random_state

Seed or numpy.random.Generator for the bootstrap.

None

Returns:

Type Description
StratResult

Overall index with approximate standard error (Goodman & Kruskal 1963), per-stratum information, and the group decomposition when a group is supplied.

References

Zhou, Xiang. 2012. "A Nonparametric Index of Stratification." Sociological Methodology, 42(1): 365-389.

Source code in src/stratindex/core.py
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
def strat(
    *args,
    outcome=None,
    strata=None,
    weights=None,
    ordered: bool = False,
    group=None,
    group_name: str | None = None,
    se_method: str = "approx",
    n_boot: int = 200,
    random_state=None,
) -> StratResult:
    """Compute the stratification index proposed in Zhou (2012).

    Call with arrays — ``strat(outcome, strata, weights=w)`` — or with a
    DataFrame / mapping of columns first:
    ``strat(df, outcome="income", strata="big_class", weights="weight",
    group="education")``. In data mode ``group_name`` defaults to the group
    column name.

    Parameters
    ----------
    outcome:
        Numeric array of outcomes (or a column name in data mode).
    strata:
        Array of the same length indicating strata membership. pandas
        Categorical keeps its category order.
    weights:
        Optional numeric array of sampling weights.
    ordered:
        If True, strata are taken as pre-ordered ascendingly (by their level
        order); otherwise they are ordered by average percentile rank.
    group:
        Optional grouping factor. If supplied (with more than one level), the
        result includes a between-/within-group decomposition of the overall
        stratification.
    group_name:
        Label used for the group in printed output (R derives it from the
        expression passed as ``group``; Python uses the column name in data
        mode, else "group", unless given explicitly).
    se_method:
        ``"approx"`` (default) — the Goodman & Kruskal (1963) approximation,
        as in the R package; ``"bootstrap"`` — standard deviation of the
        index over ``n_boot`` resamples of the complete cases (percentile
        ranks and stratum order are recomputed in every replicate).
    n_boot:
        Number of bootstrap replicates (``se_method="bootstrap"`` only).
    random_state:
        Seed or ``numpy.random.Generator`` for the bootstrap.

    Returns
    -------
    StratResult
        Overall index with approximate standard error (Goodman & Kruskal
        1963), per-stratum information, and the group decomposition when a
        group is supplied.

    References
    ----------
    Zhou, Xiang. 2012. "A Nonparametric Index of Stratification."
    Sociological Methodology, 42(1): 365-389.
    """
    outcome, strata, weights, group, group_name = _resolve_inputs(
        "strat", args, outcome, strata, weights, group, group_name
    )
    if group_name is None:
        group_name = "group"
    if not isinstance(ordered, bool):
        raise ValueError("ordered has to be a valid logical scalar")
    if se_method not in ("approx", "bootstrap"):
        raise ValueError('se_method has to be "approx" or "bootstrap"')
    if se_method == "bootstrap" and (not isinstance(n_boot, int) or n_boot < 2):
        raise ValueError("n_boot has to be an integer >= 2")

    cd = clean(outcome, strata, weights=weights, group=group)
    strata_info = _summarize(cd)
    y, r, w, order = _sorted_kernel_input(
        cd.prank, cd.strata_codes, cd.weights, ordered, strata_info["s_prank"]
    )

    decomposition = None
    within_group = None
    with np.errstate(invalid="ignore", divide="ignore"):
        if cd.group_codes is None or len(cd.group_levels) == 1:
            deno, nume = pair_sums(y, r, w)
            index = np.divide(nume, deno)
        else:
            c = cd.group_codes[order]
            sums = pair_sums_by(y, r, w, c, len(cd.group_levels))
            deno = sums["deno_within"] + sums["deno_between"]
            index = np.divide(sums["nume_within"] + sums["nume_between"], deno)
            decomposition = {
                "within": {
                    "weight": np.divide(sums["deno_within"], deno),
                    "strat": np.divide(sums["nume_within"], sums["deno_within"]),
                },
                "between": {
                    "weight": np.divide(sums["deno_between"], deno),
                    "strat": np.divide(sums["nume_between"], sums["deno_between"]),
                },
            }
            within_group = {
                group_name: cd.group_levels,
                "weight": sums["deno_by"] / sums["deno_within"],
                "strat": sums["nume_by"] / sums["deno_by"],
            }

    if se_method == "bootstrap":
        std_error = _bootstrap_se(cd, ordered, n_boot, random_state)
    else:
        # approximate standard error (Goodman & Kruskal 1963)
        with np.errstate(invalid="ignore", divide="ignore"):
            arg = deno / (1.0 - index**2) / cd.n
            std_error = 1.0 / np.sqrt(arg)

    return StratResult(
        strat=float(index),
        std_error=float(std_error),
        strata_info=strata_info,
        decomposition=decomposition,
        within_group=within_group,
        group_name=group_name,
    )

Rank strata by the average percentile rank of their members.

Call with arrays — srank(outcome, strata, weights=w) — or with a DataFrame / mapping of columns first: srank(df, outcome="income", strata="big_class", weights="weight").

Parameters:

Name Type Description Default
outcome

Numeric array of outcomes (or a column name in data mode).

None
strata

Array of the same length indicating strata membership. pandas Categorical keeps its category order.

None
weights

Optional numeric array of sampling weights.

None
group

Optional grouping factor, carried through to raw.

None

Returns:

Type Description
SrankResult

raw (complete cases with percentile ranks) and summary (per-stratum share and average percentile rank).

Source code in src/stratindex/core.py
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
def srank(*args, outcome=None, strata=None, weights=None, group=None) -> SrankResult:
    """Rank strata by the average percentile rank of their members.

    Call with arrays — ``srank(outcome, strata, weights=w)`` — or with a
    DataFrame / mapping of columns first:
    ``srank(df, outcome="income", strata="big_class", weights="weight")``.

    Parameters
    ----------
    outcome:
        Numeric array of outcomes (or a column name in data mode).
    strata:
        Array of the same length indicating strata membership. pandas
        Categorical keeps its category order.
    weights:
        Optional numeric array of sampling weights.
    group:
        Optional grouping factor, carried through to ``raw``.

    Returns
    -------
    SrankResult
        ``raw`` (complete cases with percentile ranks) and ``summary``
        (per-stratum share and average percentile rank).
    """
    outcome, strata, weights, group, _ = _resolve_inputs(
        "srank", args, outcome, strata, weights, group, None
    )
    cd = clean(outcome, strata, weights=weights, group=group)
    return SrankResult(raw=_raw(cd), summary=_summarize(cd))

Load the cpsmarch2015 dataset.

Parameters:

Name Type Description Default
as_pandas bool

If True, return a pandas DataFrame (requires pandas); otherwise a dict of NumPy arrays keyed by column name.

False

Returns:

Type Description
dict[str, ndarray] | DataFrame

Columns: income (float, personal market income in US dollars), big_class (str), micro_class (int), education (str), weight (float, CPS sampling weight).

Source code in src/stratindex/datasets.py
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
def load_cpsmarch2015(as_pandas: bool = False):
    """Load the cpsmarch2015 dataset.

    Parameters
    ----------
    as_pandas:
        If True, return a pandas DataFrame (requires pandas); otherwise a
        dict of NumPy arrays keyed by column name.

    Returns
    -------
    dict[str, numpy.ndarray] | pandas.DataFrame
        Columns: ``income`` (float, personal market income in US dollars),
        ``big_class`` (str), ``micro_class`` (int), ``education`` (str),
        ``weight`` (float, CPS sampling weight).
    """
    resource = files("stratindex.data").joinpath("cpsmarch2015.csv.gz")
    with resource.open("rb") as raw, gzip.open(raw, "rt", newline="") as fh:
        reader = csv.reader(fh)
        header = tuple(next(reader))
        assert header == _COLUMNS
        rows = list(reader)

    columns = list(zip(*rows, strict=True))
    data = {
        "income": np.array(columns[0], dtype=float),
        "big_class": np.array(columns[1]),
        "micro_class": np.array(columns[2], dtype=int),
        "education": np.array(columns[3]),
        "weight": np.array(columns[4], dtype=float),
    }
    if as_pandas:
        import pandas as pd

        return pd.DataFrame(data)
    return data

Result types

The stratification index and its approximate standard error.

Attributes:

Name Type Description
strat float

The overall stratification index.

std_error float

Approximate standard error (Goodman & Kruskal 1963).

strata_info dict[str, ndarray]

Per-stratum table — dict with strata, share and s_prank.

decomposition dict[str, dict[str, float]] | None

None unless a group was supplied; otherwise a dict with rows within / between, each a dict with weight and strat.

within_group dict[str, ndarray] | None

None unless a group was supplied; otherwise a dict with the group levels (keyed by group_name) and per-group weight / strat arrays.

Source code in src/stratindex/results.py
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
@dataclass
class StratResult:
    """The stratification index and its approximate standard error.

    Attributes
    ----------
    strat:
        The overall stratification index.
    std_error:
        Approximate standard error (Goodman & Kruskal 1963).
    strata_info:
        Per-stratum table — dict with ``strata``, ``share`` and ``s_prank``.
    decomposition:
        ``None`` unless a group was supplied; otherwise a dict with rows
        ``within`` / ``between``, each a dict with ``weight`` and ``strat``.
    within_group:
        ``None`` unless a group was supplied; otherwise a dict with the group
        levels (keyed by ``group_name``) and per-group ``weight`` / ``strat``
        arrays.
    """

    strat: float
    std_error: float
    strata_info: dict[str, np.ndarray]
    decomposition: dict[str, dict[str, float]] | None = None
    within_group: dict[str, np.ndarray] | None = None
    group_name: str = "group"

    @property
    def overall(self) -> dict[str, float]:
        return {"strat": self.strat, "std_error": self.std_error}

    def to_pandas(self):
        """Return ``strata_info`` (and ``within_group`` if any) as DataFrames."""
        import pandas as pd

        strata_info = pd.DataFrame(self.strata_info)
        if self.within_group is None:
            return strata_info
        return strata_info, pd.DataFrame(self.within_group)

    def _overall_rows(self) -> dict[str, list]:
        return {name: [value] for name, value in self.overall.items()}

    def format(self, digits: int = 3) -> str:
        lines = [
            "overall stratification:",
            "",
            _table(self._overall_rows(), digits),
        ]
        if self.decomposition is not None:
            lines += [
                "",
                f"decomposition by {self.group_name}:",
                "",
                _table(self._decomposition_rows(), digits),
            ]
        return "\n".join(lines)

    def _decomposition_rows(self) -> dict[str, list]:
        return {
            "": [f"within {self.group_name}", f"between {self.group_name}"],
            "weight": [
                self.decomposition["within"]["weight"],
                self.decomposition["between"]["weight"],
            ],
            "strat": [
                self.decomposition["within"]["strat"],
                self.decomposition["between"]["strat"],
            ],
        }

    def __str__(self) -> str:
        return self.format()

    def _repr_html_(self) -> str:
        digits = 3
        parts = [_html_table(self._overall_rows(), digits, caption="overall stratification")]
        if self.decomposition is not None:
            parts.append(
                _html_table(
                    self._decomposition_rows(),
                    digits,
                    caption=f"decomposition by {self.group_name}",
                )
            )
        parts.append(_html_table(self.strata_info, digits, caption="strata"))
        return "<div>" + "".join(parts) + "</div>"

to_pandas()

Return strata_info (and within_group if any) as DataFrames.

Source code in src/stratindex/results.py
124
125
126
127
128
129
130
131
def to_pandas(self):
    """Return ``strata_info`` (and ``within_group`` if any) as DataFrames."""
    import pandas as pd

    strata_info = pd.DataFrame(self.strata_info)
    if self.within_group is None:
        return strata_info
    return strata_info, pd.DataFrame(self.within_group)

Stratum-specific information: population share and average percentile rank.

Attributes:

Name Type Description
raw dict[str, ndarray]

Complete cases of all inputs — dict with prank, strata, weights (and group when supplied), each an ndarray of length n.

summary dict[str, ndarray]

Per-stratum table — dict with strata (level labels), share and s_prank arrays.

Source code in src/stratindex/results.py
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
@dataclass
class SrankResult:
    """Stratum-specific information: population share and average percentile rank.

    Attributes
    ----------
    raw:
        Complete cases of all inputs — dict with ``prank``, ``strata``,
        ``weights`` (and ``group`` when supplied), each an ndarray of length n.
    summary:
        Per-stratum table — dict with ``strata`` (level labels), ``share``
        and ``s_prank`` arrays.
    """

    raw: dict[str, np.ndarray] = field(repr=False)
    summary: dict[str, np.ndarray]

    def to_pandas(self):
        """Return ``(raw, summary)`` as pandas DataFrames (requires pandas)."""
        import pandas as pd

        return pd.DataFrame(self.raw), pd.DataFrame(self.summary)

    def format(self, digits: int = 3) -> str:
        return _table(self.summary, digits)

    def __str__(self) -> str:
        return self.format()

    def _repr_html_(self) -> str:
        return _html_table(self.summary, digits=3)

to_pandas()

Return (raw, summary) as pandas DataFrames (requires pandas).

Source code in src/stratindex/results.py
76
77
78
79
80
def to_pandas(self):
    """Return ``(raw, summary)`` as pandas DataFrames (requires pandas)."""
    import pandas as pd

    return pd.DataFrame(self.raw), pd.DataFrame(self.summary)