Skip to content

API Reference

Top-level package for EduBehaviors-Kit.

Assertion = Literal['sentence_addresses_the_whole_class', 'sentence_answers_a_question', 'sentence_calls_on_student_by_name', 'sentence_checks_for_understanding_or_agreement', 'sentence_evaluates_a_student_response', 'sentence_expresses_certainty_or_emphasis', 'sentence_expresses_confusion_or_requests_help', 'sentence_expresses_emotion_or_humor', 'sentence_expresses_personal_stance_or_thinking_aloud', 'sentence_grants_or_requests_permission', 'sentence_has_a_directive_or_instruction', 'sentence_has_a_question', 'sentence_has_a_rhetorical_question', 'sentence_has_acknowledgment', 'sentence_has_agreement_or_affirmation', 'sentence_has_answer_to_a_math_problem', 'sentence_has_apology', 'sentence_has_comparison_terms', 'sentence_has_counting_sequence', 'sentence_has_disagreement_or_challenge', 'sentence_has_explanation_or_reasoning', 'sentence_has_fraction_terms', 'sentence_has_informal_language', 'sentence_has_math_terms', 'sentence_has_measurement_terms', 'sentence_has_negation_or_denial', 'sentence_has_number', 'sentence_has_politeness_marker', 'sentence_has_praise_or_encouragement', 'sentence_has_time_reference', 'sentence_includes_student_name', 'sentence_invites_participation', 'sentence_is_a_declarative_statement_or_description', 'sentence_is_a_short_utterance', 'sentence_is_incomplete_or_trails_off', 'sentence_manages_classroom_behavior_or_attention', 'sentence_narrates_ongoing_action', 'sentence_poses_a_hypothetical_or_scenario', 'sentence_quotes_or_reads_text_aloud', 'sentence_references_classroom_materials_or_visuals', 'sentence_references_prior_learning_or_lesson', 'sentence_references_student_behavior_or_work', 'sentence_references_task_procedure_or_logistics', 'sentence_repeats_or_revoices_prior_speech', 'sentence_seeks_or_gives_clarification', 'sentence_shows_realization_or_insight', 'sentence_shows_uncertainty', 'sentence_summarizes_or_reviews', 'sentence_uses_collaborative_or_inclusive_language'] module-attribute

Type for assertions with a published model.

DEFAULT_WORDS = ('you', 'to', 'the', 'and', 'so', 'okay', 'i', 'what', 'is', 'a', 'that', 'do', 'of', 'this', 'one', 'going', 'we', 'it', 'your', 'are', 'right', 'have', 'in', 'two', 'be', 'how', 'can', 'three', 'all', 'on', 'number', 'go', 'for', 'then', 'were', 'if', "you're", 'up', 'not', 'think', 'would', 'about', "i'm", 'know', 'just', 'here', 'with', 'its', "it's", 'did', 'need', 'thats', 'want', 'at', 'when', 'now', 'there', 'get', 'four', 'like', 'me', 'because', 'or', 'our', 'lets', 'see', 'but', 'times', 'by', 'five', 'my', 'guys', 'answer', 'make', 'many', 'good', 'write', 'out', 'use', 'was', 'numbers', 'these', 'those', 'them', 'they', 'down', 'got', 'will', 'say', 'why', 'does', 'work', 'add', 'more', 'some', 'no', 'tell', 'math', 'give', 'he') module-attribute

Default word list for annotation with WordAnnotator

EXISTING_ASSERTIONS = get_args(Assertion) module-attribute

The assertions with a published model.

AssertionAnnotator

Scores text against one or more assertions using their SetFit classifiers.

Attributes:

Name Type Description
assertions list[Assertion]

The assertions this annotator scores, in output column order.

models dict[Assertion, SetFitModel]

The loaded models, keyed by assertion. Empty when lazy is set.

device device

The device inference runs on.

lazy bool

Whether models are loaded per call rather than held for the annotator's lifetime.

Source code in src/edubehaviors/annotation.py
class AssertionAnnotator:
    """Scores text against one or more assertions using their SetFit classifiers.

    Attributes:
        assertions: The assertions this annotator scores, in output column order.
        models: The loaded models, keyed by assertion. Empty when `lazy` is set.
        device: The device inference runs on.
        lazy: Whether models are loaded per call rather than held for the annotator's lifetime.
    """

    assertions: list[Assertion]
    models: dict[Assertion, SetFitModel]
    device: torch.device
    lazy: bool

    def _validate_assertions(self) -> None:
        """Check that `self.assertions` is non-empty, duplicate-free, and published.

        Raises:
            ValueError: If no assertions were passed, an assertion appears twice, or an
                assertion has no published model.
        """
        if not self.assertions:
            raise ValueError("No assertions passed")
        valid_assertions: set[str] = set()
        invalid_assertions: set[str] = set()
        for assertion in self.assertions:
            if assertion in valid_assertions or assertion in invalid_assertions:
                raise ValueError(f"Duplicate assertion '{assertion}' passed in list: {self.assertions}")
            if assertion not in EXISTING_ASSERTIONS:
                invalid_assertions.add(assertion)
            else:
                valid_assertions.add(assertion)
        if invalid_assertions:
            raise ValueError(f"Invalid assertions passed: {list(invalid_assertions)}")

    def _load_model(self, assertion: Assertion) -> SetFitModel:
        """Load one assertion's model onto the CPU.
        `_predict_single` moves it to `self.device` while scoring.

        Args:
            assertion: The assertion whose model to load.

        Returns:
            The model, on the CPU.
        """
        return SetFitModel.from_pretrained(f"StanfordSCALE/assertion_{assertion}", device="cpu")

    def _load_models(self) -> None:
        """Load every model in `self.assertions` that is not already loaded."""
        for assertion in tqdm(self.assertions):
            if assertion in self.models:
                continue
            self.models[assertion] = self._load_model(assertion)

    def __init__(
        self,
        assertions: Iterable[Assertion] | None = None,
        *,
        device: str | torch.device | None = None,
        lazy: bool = True,
    ) -> None:
        """Validate the requested assertions and, unless `lazy` is set, load their models.

        Args:
            assertions: The assertions to score. Defaults to `EXISTING_ASSERTIONS`, every
                published assertion, each of which downloads its own model when it is scored.
            device: The device to run inference on. Defaults to the best available
                accelerator.
            lazy: Whether to load each model only for as long as it is being used, instead of
                holding every model for the annotator's lifetime.

        Raises:
            TypeError: If `assertions` is a single assertion rather than a list of assertions.
            ValueError: If `assertions` is empty, contains duplicates, or names an assertion
                with no published model.
        """
        if isinstance(assertions, str):
            hint = "pass None for all existing assertions" if assertions == ALL else f"pass ['{assertions}'] instead"
            raise TypeError(f"assertions must be a list of assertions or None, not a single assertion; {hint}")
        self.assertions = list(EXISTING_ASSERTIONS if assertions is None else assertions)
        self._validate_assertions()
        self.device = torch.device(device) if device is not None else torch.device(get_device_name())
        self.lazy = lazy
        self.models = {}
        if not self.lazy:
            self._load_models()

    def _predict_single(
        self, inputs: list[str], assertion: Assertion, *, batch_size: int = 32, show_progress_bar: bool | None = None
    ) -> pd.Series:
        """Score inputs against a single assertion.

        Models are stored on CPU and moved to to `self.device` during use.
        If lazy, models are deleted from memory after runtime.

        Args:
            inputs: The texts to score.
            assertion: The assertion to score against.
            batch_size: The batch size to encode with.
            show_progress_bar: Whether to show a progress bar while encoding.

        Returns:
            The positive-class probability per input, named after the assertion.
        """
        model = self.models[assertion] if assertion in self.models else self._load_model(assertion)
        model.to(self.device)
        try:
            output = model.predict_proba(inputs, batch_size=batch_size, show_progress_bar=show_progress_bar)
        finally:
            if not self.lazy:
                self.models[assertion] = model.to("cpu")
        output = output[:, 1].tolist()
        del model
        if self.lazy:
            # a SetFitModel references itself through its model card, so dropping the last
            # reference does not free it; collect now so the caching allocator can reuse its
            # blocks for the next model
            gc.collect()
        return pd.Series(output, name=assertion)

    def predict_proba(
        self, inputs: list[str] | pd.Series, *, batch_size: int = 32, show_progress_bar: bool | None = None
    ) -> pd.DataFrame:
        """Score inputs against every assertion.

        Args:
            inputs: The texts to score.
            batch_size: The batch size to encode with.
            show_progress_bar: Whether to show a progress bar while encoding.

        Returns:
            The positive-class probability per input, with one column per assertion.

        Raises:
            TypeError: If `inputs` holds something other than strings.
            ValueError: If `inputs` is empty or holds missing values.
        """
        prepared = _prepare_inputs(inputs)
        texts = prepared.to_list()
        preds = {
            a: self._predict_single(texts, assertion=a, batch_size=batch_size, show_progress_bar=show_progress_bar)
            for a in self.assertions
        }
        return pd.DataFrame(preds).add_prefix("assertion__").set_axis(prepared.index)

    def predict(
        self,
        inputs: list[str] | pd.Series,
        *,
        batch_size: int = 32,
        show_progress_bar: bool | None = None,
        threshold: float = 0.5,
    ) -> pd.DataFrame:
        """Predict whether each input fires each assertion.

        Args:
            inputs: The texts to score.
            batch_size: The batch size to encode with.
            show_progress_bar: Whether to show a progress bar while encoding.
            threshold: The probability at or above which an assertion fires.

        Returns:
            Whether each input fires each assertion, with one column per assertion, sharing the
            index of `inputs`.

        Raises:
            TypeError: If `inputs` holds something other than strings.
            ValueError: If `inputs` is empty or holds missing values.
        """
        return self.predict_proba(inputs, batch_size=batch_size, show_progress_bar=show_progress_bar) >= threshold

    def annotate(
        self,
        inputs: list[str] | pd.Series,
        *,
        batch_size: int = 32,
        show_progress_bar: bool | None = None,
        threshold: float = 0.5,
    ) -> pd.DataFrame:
        """Annotate inputs for every assertion, the way `WordAnnotator.annotate` does.

        Args:
            inputs: The texts to score.
            batch_size: The batch size to encode with.
            show_progress_bar: Whether to show a progress bar while encoding.
            threshold: The probability at or above which an assertion fires.

        Returns:
            Whether each input fires each assertion, with one column per assertion, sharing the
            index of `inputs`.

        Raises:
            TypeError: If `inputs` holds something other than strings.
            ValueError: If `inputs` is empty or holds missing values.
        """
        return self.predict(inputs, batch_size=batch_size, show_progress_bar=show_progress_bar, threshold=threshold)

__init__(assertions=None, *, device=None, lazy=True)

Validate the requested assertions and, unless lazy is set, load their models.

Parameters:

Name Type Description Default
assertions Iterable[Assertion] | None

The assertions to score. Defaults to EXISTING_ASSERTIONS, every published assertion, each of which downloads its own model when it is scored.

None
device str | device | None

The device to run inference on. Defaults to the best available accelerator.

None
lazy bool

Whether to load each model only for as long as it is being used, instead of holding every model for the annotator's lifetime.

True

Raises:

Type Description
TypeError

If assertions is a single assertion rather than a list of assertions.

ValueError

If assertions is empty, contains duplicates, or names an assertion with no published model.

Source code in src/edubehaviors/annotation.py
def __init__(
    self,
    assertions: Iterable[Assertion] | None = None,
    *,
    device: str | torch.device | None = None,
    lazy: bool = True,
) -> None:
    """Validate the requested assertions and, unless `lazy` is set, load their models.

    Args:
        assertions: The assertions to score. Defaults to `EXISTING_ASSERTIONS`, every
            published assertion, each of which downloads its own model when it is scored.
        device: The device to run inference on. Defaults to the best available
            accelerator.
        lazy: Whether to load each model only for as long as it is being used, instead of
            holding every model for the annotator's lifetime.

    Raises:
        TypeError: If `assertions` is a single assertion rather than a list of assertions.
        ValueError: If `assertions` is empty, contains duplicates, or names an assertion
            with no published model.
    """
    if isinstance(assertions, str):
        hint = "pass None for all existing assertions" if assertions == ALL else f"pass ['{assertions}'] instead"
        raise TypeError(f"assertions must be a list of assertions or None, not a single assertion; {hint}")
    self.assertions = list(EXISTING_ASSERTIONS if assertions is None else assertions)
    self._validate_assertions()
    self.device = torch.device(device) if device is not None else torch.device(get_device_name())
    self.lazy = lazy
    self.models = {}
    if not self.lazy:
        self._load_models()

annotate(inputs, *, batch_size=32, show_progress_bar=None, threshold=0.5)

Annotate inputs for every assertion, the way WordAnnotator.annotate does.

Parameters:

Name Type Description Default
inputs list[str] | Series

The texts to score.

required
batch_size int

The batch size to encode with.

32
show_progress_bar bool | None

Whether to show a progress bar while encoding.

None
threshold float

The probability at or above which an assertion fires.

0.5

Returns:

Type Description
DataFrame

Whether each input fires each assertion, with one column per assertion, sharing the

DataFrame

index of inputs.

Raises:

Type Description
TypeError

If inputs holds something other than strings.

ValueError

If inputs is empty or holds missing values.

Source code in src/edubehaviors/annotation.py
def annotate(
    self,
    inputs: list[str] | pd.Series,
    *,
    batch_size: int = 32,
    show_progress_bar: bool | None = None,
    threshold: float = 0.5,
) -> pd.DataFrame:
    """Annotate inputs for every assertion, the way `WordAnnotator.annotate` does.

    Args:
        inputs: The texts to score.
        batch_size: The batch size to encode with.
        show_progress_bar: Whether to show a progress bar while encoding.
        threshold: The probability at or above which an assertion fires.

    Returns:
        Whether each input fires each assertion, with one column per assertion, sharing the
        index of `inputs`.

    Raises:
        TypeError: If `inputs` holds something other than strings.
        ValueError: If `inputs` is empty or holds missing values.
    """
    return self.predict(inputs, batch_size=batch_size, show_progress_bar=show_progress_bar, threshold=threshold)

predict(inputs, *, batch_size=32, show_progress_bar=None, threshold=0.5)

Predict whether each input fires each assertion.

Parameters:

Name Type Description Default
inputs list[str] | Series

The texts to score.

required
batch_size int

The batch size to encode with.

32
show_progress_bar bool | None

Whether to show a progress bar while encoding.

None
threshold float

The probability at or above which an assertion fires.

0.5

Returns:

Type Description
DataFrame

Whether each input fires each assertion, with one column per assertion, sharing the

DataFrame

index of inputs.

Raises:

Type Description
TypeError

If inputs holds something other than strings.

ValueError

If inputs is empty or holds missing values.

Source code in src/edubehaviors/annotation.py
def predict(
    self,
    inputs: list[str] | pd.Series,
    *,
    batch_size: int = 32,
    show_progress_bar: bool | None = None,
    threshold: float = 0.5,
) -> pd.DataFrame:
    """Predict whether each input fires each assertion.

    Args:
        inputs: The texts to score.
        batch_size: The batch size to encode with.
        show_progress_bar: Whether to show a progress bar while encoding.
        threshold: The probability at or above which an assertion fires.

    Returns:
        Whether each input fires each assertion, with one column per assertion, sharing the
        index of `inputs`.

    Raises:
        TypeError: If `inputs` holds something other than strings.
        ValueError: If `inputs` is empty or holds missing values.
    """
    return self.predict_proba(inputs, batch_size=batch_size, show_progress_bar=show_progress_bar) >= threshold

predict_proba(inputs, *, batch_size=32, show_progress_bar=None)

Score inputs against every assertion.

Parameters:

Name Type Description Default
inputs list[str] | Series

The texts to score.

required
batch_size int

The batch size to encode with.

32
show_progress_bar bool | None

Whether to show a progress bar while encoding.

None

Returns:

Type Description
DataFrame

The positive-class probability per input, with one column per assertion.

Raises:

Type Description
TypeError

If inputs holds something other than strings.

ValueError

If inputs is empty or holds missing values.

Source code in src/edubehaviors/annotation.py
def predict_proba(
    self, inputs: list[str] | pd.Series, *, batch_size: int = 32, show_progress_bar: bool | None = None
) -> pd.DataFrame:
    """Score inputs against every assertion.

    Args:
        inputs: The texts to score.
        batch_size: The batch size to encode with.
        show_progress_bar: Whether to show a progress bar while encoding.

    Returns:
        The positive-class probability per input, with one column per assertion.

    Raises:
        TypeError: If `inputs` holds something other than strings.
        ValueError: If `inputs` is empty or holds missing values.
    """
    prepared = _prepare_inputs(inputs)
    texts = prepared.to_list()
    preds = {
        a: self._predict_single(texts, assertion=a, batch_size=batch_size, show_progress_bar=show_progress_bar)
        for a in self.assertions
    }
    return pd.DataFrame(preds).add_prefix("assertion__").set_axis(prepared.index)

ClassificationPipeline

Takes in a labeled DataFrame, annotates it, splits into train/test, trains a standard classifier, and scores the test set.

pipeline = ClassificationPipeline(data, words="all", label_column="label_press_for_reasoning")
print(pipeline.report())

Word features are counted with WordAnnotator and assertion features are scored with AssertionAnnotator. To use all default words or existing annotators, pass "all". Otherwise, pass a list of words or assertions, or None to skip an annotator. At least one of words or assertions must be passed.

Labels are used exactly as they appear in label_column, so a boolean column gives a binary problem and a column of names gives a multiclass one, with no relabelling behind your back.

Attributes:

Name Type Description
word_annotator WordAnnotator | None

The word annotator, or None when word features are off.

assertion_annotator AssertionAnnotator | None

The assertion annotator, or None when assertion features are off.

text_column str

The column holding the text that was annotated.

label_column str

The column holding the labels being predicted.

group_column str | None

The column that was kept out of both splits at once, or None.

word_method WordMethod

Whether word features count occurrences or flag containment.

assertion_method AssertionMethod

Whether assertion features are probabilities or booleans.

test_size float

The fraction of the data held out for testing.

stratify bool

Whether the split preserved the label distribution.

random_state int | RandomState | None

The seed the split and the classifier used.

batch_size int

The batch size assertion scoring encoded with.

show_progress_bar bool | None

Whether assertion scoring showed a progress bar.

data DataFrame

The frame the pipeline was built from.

features DataFrame

The feature matrix, sharing the index of data.

outcome Series

The labels, as they appeared in data.

split Series

Whether each row is in the train or test split.

split_strategy SplitStrategy

How the split was produced.

groups Series | None

The group of each row, or None when group_column is not set.

classifier LogisticRegressionCV

The classifier, fitted on the train split.

metrics dict[str, float]

How the classifier scored on the test split.

Source code in src/edubehaviors/pipeline.py
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
class ClassificationPipeline:
    """Takes in a labeled DataFrame, annotates it, splits into train/test, trains a
    standard classifier, and scores the test set.

        pipeline = ClassificationPipeline(data, words="all", label_column="label_press_for_reasoning")
        print(pipeline.report())

    Word features are counted with `WordAnnotator` and assertion features are scored with
    `AssertionAnnotator`. To use all default words or existing annotators, pass `"all"`.
    Otherwise, pass a list of words or assertions, or `None` to skip an annotator. At
    least one of `words` or `assertions` must be passed.

    Labels are used exactly as they appear in `label_column`, so a boolean column gives a binary
    problem and a column of names gives a multiclass one, with no relabelling behind your back.

    Attributes:
        word_annotator: The word annotator, or None when word features are off.
        assertion_annotator: The assertion annotator, or None when assertion features are off.
        text_column: The column holding the text that was annotated.
        label_column: The column holding the labels being predicted.
        group_column: The column that was kept out of both splits at once, or None.
        word_method: Whether word features count occurrences or flag containment.
        assertion_method: Whether assertion features are probabilities or booleans.
        test_size: The fraction of the data held out for testing.
        stratify: Whether the split preserved the label distribution.
        random_state: The seed the split and the classifier used.
        batch_size: The batch size assertion scoring encoded with.
        show_progress_bar: Whether assertion scoring showed a progress bar.
        data: The frame the pipeline was built from.
        features: The feature matrix, sharing the index of `data`.
        outcome: The labels, as they appeared in `data`.
        split: Whether each row is in the train or test split.
        split_strategy: How the split was produced.
        groups: The group of each row, or None when `group_column` is not set.
        classifier: The classifier, fitted on the train split.
        metrics: How the classifier scored on the test split.
    """

    word_annotator: WordAnnotator | None
    assertion_annotator: AssertionAnnotator | None
    text_column: str
    label_column: str
    group_column: str | None
    word_method: WordMethod
    assertion_method: AssertionMethod
    test_size: float
    stratify: bool
    random_state: int | RandomState | None
    batch_size: int
    show_progress_bar: bool | None

    data: pd.DataFrame
    features: pd.DataFrame
    outcome: pd.Series
    split: pd.Series
    split_strategy: SplitStrategy
    groups: pd.Series | None
    classifier: LogisticRegressionCV
    metrics: dict[str, float]
    _split_predictions: dict[Split, pd.Series]

    def __init__(
        self,
        data: pd.DataFrame,
        words: Iterable[str] | Literal["all"] | None = None,
        assertions: Iterable[Assertion] | Literal["all"] | None = None,
        *,
        text_column: str = "sentence",
        label_column: str = "label",
        group_column: str | None = None,
        word_method: WordMethod = "count",
        case: bool = False,
        assertion_method: AssertionMethod = "proba",
        lazy: bool = True,
        test_size: float = 0.2,
        stratify: bool = True,
        random_state: int | RandomState | None = None,
        batch_size: int = 32,
        show_progress_bar: bool | None = None,
        features: pd.DataFrame | None = None,
    ) -> None:
        """Annotate `data`, split it, fit the classifier, and score the held-out split.

        Args:
            data: The frame holding the text column, the label column, and the group column
                when one is given.
            words: The words to count, `"all"` for `DEFAULT_WORDS`, or None to skip word
                features. An empty iterable also skips them.
            assertions: The assertions to score, `"all"` for `EXISTING_ASSERTIONS`, or None to
                skip assertion features. An empty iterable also skips them. Every assertion
                scored downloads its own model, so `"all"` downloads one per published
                assertion.
            text_column: The column holding the text to annotate.
            label_column: The column holding the labels to predict. Its values are used as they
                are, so binarize beforehand if that is what you want.
            group_column: The column whose values must not straddle the split, for example a
                transcript id, so that rows from one transcript stay on one side.
            word_method: Whether word features count occurrences or flag containment.
            case: Whether word matching is case-sensitive.
            assertion_method: Whether assertion features are positive-class probabilities
                (`"proba"`) or booleans thresholded at 0.5 (`"binary"`).
            lazy: Whether assertion models are loaded per call rather than held for the
                pipeline's lifetime. Pass False when calling `predict` repeatedly afterwards.
            test_size: The fraction of the data to hold out. With `group_column` set this is
                approximate, since whole groups move together: the closest of
                `round(1 / test_size)` grouped folds is used, and uneven groups make some folds
                much larger than others.
            stratify: Whether the split preserves the label distribution. With `group_column`
                set this is best-effort, since whole groups move together.
            random_state: The seed the split and the classifier use.
            batch_size: The batch size assertion scoring encodes with.
            show_progress_bar: Whether assertion scoring shows a progress bar.
            features: A feature matrix from another pipeline's `features` or `annotate(data)`,
                to skip re-annotating when only the split or the seed changed.

        Raises:
            KeyError: If a configured column is missing from `data`.
            TypeError: If `words` or `assertions` is a single value, or the texts are not
                strings.
            ValueError: If neither words nor assertions were requested, `assertion_method` is
                not supported, `test_size` is not a fraction, `data` has a duplicated index or
                fewer than two distinct labels, `features` does not share the index of `data`,
                or the split leaves a single label in the train split.

        Warns:
            UserWarning: If the test split holds a single label, which makes its metrics
                degenerate.
        """
        requested_words: list[str] | None = None
        if isinstance(words, str):
            if words != ALL:
                raise TypeError(f"words must be a list, '{ALL}', or None, not a single value; pass ['{words}'] instead")
            requested_words = list(DEFAULT_WORDS)
        elif words is not None:
            requested_words = list(words) or None
        requested_assertions: list[Assertion] | None = None
        if isinstance(assertions, str):
            if assertions != ALL:
                raise TypeError(
                    f"assertions must be a list, '{ALL}', or None, not a single value; pass ['{assertions}'] instead"
                )
            requested_assertions = list(EXISTING_ASSERTIONS)
        elif assertions is not None:
            requested_assertions = list(assertions) or None
        if requested_words is None and requested_assertions is None:
            raise ValueError(f"No features requested; pass words, assertions, or both (either can be '{ALL}')")
        if assertion_method not in EXISTING_ASSERTION_METHODS:
            raise ValueError(f"assertion_method must be one of {EXISTING_ASSERTION_METHODS}, got '{assertion_method}'")
        if not 0 < test_size < 1:
            raise ValueError(f"test_size must be strictly between 0 and 1, got {test_size}")
        if data.index.has_duplicates:
            duplicates = data.index[data.index.duplicated()].unique().tolist()[:5]
            raise ValueError(f"data has a duplicated index, which misaligns features; duplicates include {duplicates}")
        self.word_annotator = None if requested_words is None else WordAnnotator(requested_words, case=case)
        self.assertion_annotator = (
            None if requested_assertions is None else AssertionAnnotator(requested_assertions, lazy=lazy)
        )
        self.text_column = text_column
        self.label_column = label_column
        self.group_column = group_column
        self.word_method = word_method
        self.assertion_method = assertion_method
        self.test_size = test_size
        self.stratify = stratify
        self.random_state = random_state
        self.batch_size = batch_size
        self.show_progress_bar = show_progress_bar

        self.data = data
        self.outcome = data[label_column]
        if self.outcome.isna().sum() > 0:
            raise ValueError(
                f"label column '{label_column}' contains missing values;"
                " drop or fill them before instantiating pipeline"
            )
        if self.outcome.nunique() < 2:
            raise ValueError(
                f"label column '{label_column}' has fewer than 2 distinct labels; at least 2 are needed to fit"
            )
        self.groups = None if group_column is None else data[group_column]
        if features is None:
            features = self.annotate(data)
        elif not features.index.equals(data.index):
            raise ValueError("features must share the index of data; pass a frame from this pipeline's annotate()")
        self.features = features
        self.split, self.split_strategy = self._make_split()
        is_train = self.split == "train"
        if self.outcome[is_train].nunique() < 2:
            raise ValueError("train split holds a single label; adjust test_size, the seed, or the grouping")
        if self.outcome[~is_train].nunique() < 2:
            warn("test split holds a single label, so its metrics are degenerate", UserWarning, stacklevel=2)
        self.classifier = standard_classifier(random_state=random_state).fit(
            features.loc[is_train], self.outcome[is_train]
        )
        self._split_predictions = {}
        self.metrics = self.evaluate()

    def annotate(self, data: pd.DataFrame | pd.Series | list[str]) -> pd.DataFrame:
        """Build the feature matrix for `data`.

        Args:
            data: A frame holding the text column, or the texts themselves.

        Returns:
            The word features followed by the assertion features, sharing the index of `data`.

        Raises:
            KeyError: If `data` is a frame without the text column.
            TypeError: If the texts are not strings.
            ValueError: If `data` is empty, or any text is missing.
        """
        text = data[self.text_column] if isinstance(data, pd.DataFrame) else data
        frames = []
        if self.word_annotator is not None:
            frames.append(self.word_annotator.annotate(text, method=self.word_method))
        if self.assertion_annotator is not None:
            score = (
                self.assertion_annotator.predict_proba
                if self.assertion_method == "proba"
                else self.assertion_annotator.predict
            )
            frames.append(score(text, batch_size=self.batch_size, show_progress_bar=self.show_progress_bar))
        return pd.concat(frames, axis=1)

    def _make_split(self) -> tuple[pd.Series, SplitStrategy]:
        """Assign every row to the train or test split.

        Returns:
            The split of each row, and how it was produced.

        Raises:
            ValueError: If there are too few groups for `test_size`.
        """
        index = self.outcome.index
        if self.groups is None:
            train_positions, test_positions = train_test_split(
                np.arange(len(index)),
                test_size=self.test_size,
                shuffle=True,
                random_state=self.random_state,
                stratify=self.outcome if self.stratify else None,
            )
            strategy: SplitStrategy = "stratified" if self.stratify else "random"
        else:
            rows = np.zeros((len(index), 1))
            if self.stratify:
                # fold sizes vary as much as group sizes do
                n_splits = max(2, round(1 / self.test_size))
                n_groups = int(self.groups.nunique())
                if n_groups < n_splits:
                    raise ValueError(
                        f"test_size={self.test_size} needs at least {n_splits} groups in "
                        f"'{self.group_column}', got {n_groups}; raise test_size or pass stratify=False"
                    )
                splitter = StratifiedGroupKFold(n_splits=n_splits, shuffle=True, random_state=self.random_state)
                folds = list(splitter.split(rows, self.outcome, groups=self.groups))
                train_positions, test_positions = min(
                    folds, key=lambda fold: abs(len(fold[1]) / len(index) - self.test_size)
                )
                strategy = "stratified_grouped"
            else:
                splitter = GroupShuffleSplit(n_splits=1, test_size=self.test_size, random_state=self.random_state)
                train_positions, test_positions = next(splitter.split(rows, self.outcome, groups=self.groups))
                strategy = "grouped"
        split = pd.Series("train", index=index, name=SPLIT_COLUMN, dtype="object")
        split.iloc[test_positions] = "test"
        return split, strategy

    def _predict_split(self, split: Split) -> pd.Series:
        """Predict the labels of one split with the fitted classifier, once.

        Args:
            split: The split to predict.

        Returns:
            The predicted label of each row in that split.

        Raises:
            ValueError: If `split` is not a supported split.
        """
        if split not in EXISTING_SPLITS:
            raise ValueError(f"split must be one of {EXISTING_SPLITS}, got '{split}'")
        if split not in self._split_predictions:
            rows = self.features.loc[self.split == split]
            self._split_predictions[split] = pd.Series(
                self.classifier.predict(rows), index=rows.index, name="prediction"
            )
        return self._split_predictions[split]

    def evaluate(self, *, split: Split = "test") -> dict[str, float]:
        """Get macro-averaged metrics for specified split using fitted classifier.

        Args:
            split: The split to score.

        Returns:
            Accuracy, macro precision, macro recall and macro F1, with the size of each split.

        Raises:
            ValueError: If `split` is not a supported split.
        """
        predicted = self._predict_split(split)
        true = self.outcome[self.split == split]
        precision, recall, f1, _ = precision_recall_fscore_support(true, predicted, average="macro", zero_division=0)
        return {
            "accuracy": float(accuracy_score(true, predicted)),
            "precision_macro": float(precision),
            "recall_macro": float(recall),
            "f1_macro": float(f1),
            "n_train": int((self.split == "train").sum()),
            "n_test": int((self.split == "test").sum()),
        }

    def report(self, *, split: Split = "test", digits: int = 2) -> str:
        """Return classification report for specified split.

        Args:
            split: The split to report on.
            digits: The number of decimal places to report.

        Returns:
            SKlearn classification report

        Raises:
            ValueError: If `split` is not a supported split.
        """
        predicted = self._predict_split(split)
        true = self.outcome[self.split == split]
        return classification_report(true, predicted, digits=digits, zero_division=0)

    def predict(self, data: pd.DataFrame | pd.Series | list[str]) -> pd.Series:
        """Predict labels for text the pipeline has not seen.

        Args:
            data: A frame holding the text column, or the texts themselves.

        Returns:
            The predicted label per row, sharing the index of `data`.
        """
        features = self.annotate(data)[self.features.columns]
        return pd.Series(self.classifier.predict(features), index=features.index, name="prediction")

    def predict_proba(self, data: pd.DataFrame | pd.Series | list[str]) -> pd.DataFrame:
        """Predict label probabilities for text the pipeline has not seen.

        Args:
            data: A frame holding the text column, or the texts themselves.

        Returns:
            One column per label, sharing the index of `data`.
        """
        features = self.annotate(data)[self.features.columns]
        return pd.DataFrame(
            self.classifier.predict_proba(features), index=features.index, columns=self.classifier.classes_
        )

    @property
    def predictions(self) -> pd.Series:
        """The predicted label of each held-out row, which `metrics` is computed from."""
        return self._predict_split("test")

    @property
    def predicted(self) -> pd.DataFrame:
        """The held-out rows with their true and predicted labels."""
        is_test = self.split == "test"
        predictions = self.predictions
        y_true = self.outcome[is_test]
        return pd.DataFrame(
            {
                self.text_column: self.data.loc[is_test, self.text_column],
                self.label_column: y_true,
                "prediction": predictions,
                "correct": y_true == predictions,
            }
        )

    @property
    def annotated(self) -> pd.DataFrame:
        """The data with its features and split assignment, for inspection or export.

        Note:
            A `split` column already in the data is replaced by the pipeline's own.
        """
        data = self.data.drop(columns=[SPLIT_COLUMN], errors="ignore")
        return pd.concat([data, self.features, self.split], axis=1)

    @property
    def coefficients(self) -> pd.Series | pd.DataFrame:
        """The fitted coefficients, indexed by feature name. Binary problems give a
        Series; multiclass problems give one column per label.
        """
        coefficients = self.classifier.coef_
        if coefficients.shape[0] == 1:
            return pd.Series(coefficients[0], index=self.features.columns, name="coefficient")
        return pd.DataFrame(coefficients.T, index=self.features.columns, columns=self.classifier.classes_)

annotated property

The data with its features and split assignment, for inspection or export.

Note

A split column already in the data is replaced by the pipeline's own.

coefficients property

The fitted coefficients, indexed by feature name. Binary problems give a Series; multiclass problems give one column per label.

predicted property

The held-out rows with their true and predicted labels.

predictions property

The predicted label of each held-out row, which metrics is computed from.

__init__(data, words=None, assertions=None, *, text_column='sentence', label_column='label', group_column=None, word_method='count', case=False, assertion_method='proba', lazy=True, test_size=0.2, stratify=True, random_state=None, batch_size=32, show_progress_bar=None, features=None)

Annotate data, split it, fit the classifier, and score the held-out split.

Parameters:

Name Type Description Default
data DataFrame

The frame holding the text column, the label column, and the group column when one is given.

required
words Iterable[str] | Literal['all'] | None

The words to count, "all" for DEFAULT_WORDS, or None to skip word features. An empty iterable also skips them.

None
assertions Iterable[Assertion] | Literal['all'] | None

The assertions to score, "all" for EXISTING_ASSERTIONS, or None to skip assertion features. An empty iterable also skips them. Every assertion scored downloads its own model, so "all" downloads one per published assertion.

None
text_column str

The column holding the text to annotate.

'sentence'
label_column str

The column holding the labels to predict. Its values are used as they are, so binarize beforehand if that is what you want.

'label'
group_column str | None

The column whose values must not straddle the split, for example a transcript id, so that rows from one transcript stay on one side.

None
word_method WordMethod

Whether word features count occurrences or flag containment.

'count'
case bool

Whether word matching is case-sensitive.

False
assertion_method AssertionMethod

Whether assertion features are positive-class probabilities ("proba") or booleans thresholded at 0.5 ("binary").

'proba'
lazy bool

Whether assertion models are loaded per call rather than held for the pipeline's lifetime. Pass False when calling predict repeatedly afterwards.

True
test_size float

The fraction of the data to hold out. With group_column set this is approximate, since whole groups move together: the closest of round(1 / test_size) grouped folds is used, and uneven groups make some folds much larger than others.

0.2
stratify bool

Whether the split preserves the label distribution. With group_column set this is best-effort, since whole groups move together.

True
random_state int | RandomState | None

The seed the split and the classifier use.

None
batch_size int

The batch size assertion scoring encodes with.

32
show_progress_bar bool | None

Whether assertion scoring shows a progress bar.

None
features DataFrame | None

A feature matrix from another pipeline's features or annotate(data), to skip re-annotating when only the split or the seed changed.

None

Raises:

Type Description
KeyError

If a configured column is missing from data.

TypeError

If words or assertions is a single value, or the texts are not strings.

ValueError

If neither words nor assertions were requested, assertion_method is not supported, test_size is not a fraction, data has a duplicated index or fewer than two distinct labels, features does not share the index of data, or the split leaves a single label in the train split.

Warns:

Type Description
UserWarning

If the test split holds a single label, which makes its metrics degenerate.

Source code in src/edubehaviors/pipeline.py
def __init__(
    self,
    data: pd.DataFrame,
    words: Iterable[str] | Literal["all"] | None = None,
    assertions: Iterable[Assertion] | Literal["all"] | None = None,
    *,
    text_column: str = "sentence",
    label_column: str = "label",
    group_column: str | None = None,
    word_method: WordMethod = "count",
    case: bool = False,
    assertion_method: AssertionMethod = "proba",
    lazy: bool = True,
    test_size: float = 0.2,
    stratify: bool = True,
    random_state: int | RandomState | None = None,
    batch_size: int = 32,
    show_progress_bar: bool | None = None,
    features: pd.DataFrame | None = None,
) -> None:
    """Annotate `data`, split it, fit the classifier, and score the held-out split.

    Args:
        data: The frame holding the text column, the label column, and the group column
            when one is given.
        words: The words to count, `"all"` for `DEFAULT_WORDS`, or None to skip word
            features. An empty iterable also skips them.
        assertions: The assertions to score, `"all"` for `EXISTING_ASSERTIONS`, or None to
            skip assertion features. An empty iterable also skips them. Every assertion
            scored downloads its own model, so `"all"` downloads one per published
            assertion.
        text_column: The column holding the text to annotate.
        label_column: The column holding the labels to predict. Its values are used as they
            are, so binarize beforehand if that is what you want.
        group_column: The column whose values must not straddle the split, for example a
            transcript id, so that rows from one transcript stay on one side.
        word_method: Whether word features count occurrences or flag containment.
        case: Whether word matching is case-sensitive.
        assertion_method: Whether assertion features are positive-class probabilities
            (`"proba"`) or booleans thresholded at 0.5 (`"binary"`).
        lazy: Whether assertion models are loaded per call rather than held for the
            pipeline's lifetime. Pass False when calling `predict` repeatedly afterwards.
        test_size: The fraction of the data to hold out. With `group_column` set this is
            approximate, since whole groups move together: the closest of
            `round(1 / test_size)` grouped folds is used, and uneven groups make some folds
            much larger than others.
        stratify: Whether the split preserves the label distribution. With `group_column`
            set this is best-effort, since whole groups move together.
        random_state: The seed the split and the classifier use.
        batch_size: The batch size assertion scoring encodes with.
        show_progress_bar: Whether assertion scoring shows a progress bar.
        features: A feature matrix from another pipeline's `features` or `annotate(data)`,
            to skip re-annotating when only the split or the seed changed.

    Raises:
        KeyError: If a configured column is missing from `data`.
        TypeError: If `words` or `assertions` is a single value, or the texts are not
            strings.
        ValueError: If neither words nor assertions were requested, `assertion_method` is
            not supported, `test_size` is not a fraction, `data` has a duplicated index or
            fewer than two distinct labels, `features` does not share the index of `data`,
            or the split leaves a single label in the train split.

    Warns:
        UserWarning: If the test split holds a single label, which makes its metrics
            degenerate.
    """
    requested_words: list[str] | None = None
    if isinstance(words, str):
        if words != ALL:
            raise TypeError(f"words must be a list, '{ALL}', or None, not a single value; pass ['{words}'] instead")
        requested_words = list(DEFAULT_WORDS)
    elif words is not None:
        requested_words = list(words) or None
    requested_assertions: list[Assertion] | None = None
    if isinstance(assertions, str):
        if assertions != ALL:
            raise TypeError(
                f"assertions must be a list, '{ALL}', or None, not a single value; pass ['{assertions}'] instead"
            )
        requested_assertions = list(EXISTING_ASSERTIONS)
    elif assertions is not None:
        requested_assertions = list(assertions) or None
    if requested_words is None and requested_assertions is None:
        raise ValueError(f"No features requested; pass words, assertions, or both (either can be '{ALL}')")
    if assertion_method not in EXISTING_ASSERTION_METHODS:
        raise ValueError(f"assertion_method must be one of {EXISTING_ASSERTION_METHODS}, got '{assertion_method}'")
    if not 0 < test_size < 1:
        raise ValueError(f"test_size must be strictly between 0 and 1, got {test_size}")
    if data.index.has_duplicates:
        duplicates = data.index[data.index.duplicated()].unique().tolist()[:5]
        raise ValueError(f"data has a duplicated index, which misaligns features; duplicates include {duplicates}")
    self.word_annotator = None if requested_words is None else WordAnnotator(requested_words, case=case)
    self.assertion_annotator = (
        None if requested_assertions is None else AssertionAnnotator(requested_assertions, lazy=lazy)
    )
    self.text_column = text_column
    self.label_column = label_column
    self.group_column = group_column
    self.word_method = word_method
    self.assertion_method = assertion_method
    self.test_size = test_size
    self.stratify = stratify
    self.random_state = random_state
    self.batch_size = batch_size
    self.show_progress_bar = show_progress_bar

    self.data = data
    self.outcome = data[label_column]
    if self.outcome.isna().sum() > 0:
        raise ValueError(
            f"label column '{label_column}' contains missing values;"
            " drop or fill them before instantiating pipeline"
        )
    if self.outcome.nunique() < 2:
        raise ValueError(
            f"label column '{label_column}' has fewer than 2 distinct labels; at least 2 are needed to fit"
        )
    self.groups = None if group_column is None else data[group_column]
    if features is None:
        features = self.annotate(data)
    elif not features.index.equals(data.index):
        raise ValueError("features must share the index of data; pass a frame from this pipeline's annotate()")
    self.features = features
    self.split, self.split_strategy = self._make_split()
    is_train = self.split == "train"
    if self.outcome[is_train].nunique() < 2:
        raise ValueError("train split holds a single label; adjust test_size, the seed, or the grouping")
    if self.outcome[~is_train].nunique() < 2:
        warn("test split holds a single label, so its metrics are degenerate", UserWarning, stacklevel=2)
    self.classifier = standard_classifier(random_state=random_state).fit(
        features.loc[is_train], self.outcome[is_train]
    )
    self._split_predictions = {}
    self.metrics = self.evaluate()

annotate(data)

Build the feature matrix for data.

Parameters:

Name Type Description Default
data DataFrame | Series | list[str]

A frame holding the text column, or the texts themselves.

required

Returns:

Type Description
DataFrame

The word features followed by the assertion features, sharing the index of data.

Raises:

Type Description
KeyError

If data is a frame without the text column.

TypeError

If the texts are not strings.

ValueError

If data is empty, or any text is missing.

Source code in src/edubehaviors/pipeline.py
def annotate(self, data: pd.DataFrame | pd.Series | list[str]) -> pd.DataFrame:
    """Build the feature matrix for `data`.

    Args:
        data: A frame holding the text column, or the texts themselves.

    Returns:
        The word features followed by the assertion features, sharing the index of `data`.

    Raises:
        KeyError: If `data` is a frame without the text column.
        TypeError: If the texts are not strings.
        ValueError: If `data` is empty, or any text is missing.
    """
    text = data[self.text_column] if isinstance(data, pd.DataFrame) else data
    frames = []
    if self.word_annotator is not None:
        frames.append(self.word_annotator.annotate(text, method=self.word_method))
    if self.assertion_annotator is not None:
        score = (
            self.assertion_annotator.predict_proba
            if self.assertion_method == "proba"
            else self.assertion_annotator.predict
        )
        frames.append(score(text, batch_size=self.batch_size, show_progress_bar=self.show_progress_bar))
    return pd.concat(frames, axis=1)

evaluate(*, split='test')

Get macro-averaged metrics for specified split using fitted classifier.

Parameters:

Name Type Description Default
split Split

The split to score.

'test'

Returns:

Type Description
dict[str, float]

Accuracy, macro precision, macro recall and macro F1, with the size of each split.

Raises:

Type Description
ValueError

If split is not a supported split.

Source code in src/edubehaviors/pipeline.py
def evaluate(self, *, split: Split = "test") -> dict[str, float]:
    """Get macro-averaged metrics for specified split using fitted classifier.

    Args:
        split: The split to score.

    Returns:
        Accuracy, macro precision, macro recall and macro F1, with the size of each split.

    Raises:
        ValueError: If `split` is not a supported split.
    """
    predicted = self._predict_split(split)
    true = self.outcome[self.split == split]
    precision, recall, f1, _ = precision_recall_fscore_support(true, predicted, average="macro", zero_division=0)
    return {
        "accuracy": float(accuracy_score(true, predicted)),
        "precision_macro": float(precision),
        "recall_macro": float(recall),
        "f1_macro": float(f1),
        "n_train": int((self.split == "train").sum()),
        "n_test": int((self.split == "test").sum()),
    }

predict(data)

Predict labels for text the pipeline has not seen.

Parameters:

Name Type Description Default
data DataFrame | Series | list[str]

A frame holding the text column, or the texts themselves.

required

Returns:

Type Description
Series

The predicted label per row, sharing the index of data.

Source code in src/edubehaviors/pipeline.py
def predict(self, data: pd.DataFrame | pd.Series | list[str]) -> pd.Series:
    """Predict labels for text the pipeline has not seen.

    Args:
        data: A frame holding the text column, or the texts themselves.

    Returns:
        The predicted label per row, sharing the index of `data`.
    """
    features = self.annotate(data)[self.features.columns]
    return pd.Series(self.classifier.predict(features), index=features.index, name="prediction")

predict_proba(data)

Predict label probabilities for text the pipeline has not seen.

Parameters:

Name Type Description Default
data DataFrame | Series | list[str]

A frame holding the text column, or the texts themselves.

required

Returns:

Type Description
DataFrame

One column per label, sharing the index of data.

Source code in src/edubehaviors/pipeline.py
def predict_proba(self, data: pd.DataFrame | pd.Series | list[str]) -> pd.DataFrame:
    """Predict label probabilities for text the pipeline has not seen.

    Args:
        data: A frame holding the text column, or the texts themselves.

    Returns:
        One column per label, sharing the index of `data`.
    """
    features = self.annotate(data)[self.features.columns]
    return pd.DataFrame(
        self.classifier.predict_proba(features), index=features.index, columns=self.classifier.classes_
    )

report(*, split='test', digits=2)

Return classification report for specified split.

Parameters:

Name Type Description Default
split Split

The split to report on.

'test'
digits int

The number of decimal places to report.

2

Returns:

Type Description
str

SKlearn classification report

Raises:

Type Description
ValueError

If split is not a supported split.

Source code in src/edubehaviors/pipeline.py
def report(self, *, split: Split = "test", digits: int = 2) -> str:
    """Return classification report for specified split.

    Args:
        split: The split to report on.
        digits: The number of decimal places to report.

    Returns:
        SKlearn classification report

    Raises:
        ValueError: If `split` is not a supported split.
    """
    predicted = self._predict_split(split)
    true = self.outcome[self.split == split]
    return classification_report(true, predicted, digits=digits, zero_division=0)

WordAnnotator

Counts occurrences of a list of words in text.

Attributes:

Name Type Description
words list[str]

The words this annotator counts, in output column order.

case bool

Whether matching is case-sensitive.

Source code in src/edubehaviors/annotation.py
class WordAnnotator:
    """Counts occurrences of a list of words in text.

    Attributes:
        words: The words this annotator counts, in output column order.
        case: Whether matching is case-sensitive.
    """

    words: list[str]
    case: bool

    @staticmethod
    def _validate_words(words: list[str], *, case: bool) -> None:
        """Check that `words` is a non-empty list of distinct, non-empty strings.

        Args:
            words: The words to check.
            case: Whether matching is case-sensitive. When it is not, words differing only by
                case are duplicates, since they would produce identical columns.

        Raises:
            ValueError: If `words` is empty, contains an empty string, or contains duplicates.
        """
        if not words:
            raise ValueError("No words passed")
        if "" in words:
            raise ValueError("Empty string '' passed in word list")
        keys = words if case else [word.casefold() for word in words]
        duplicates = sorted({key for key, count in Counter(keys).items() if count > 1})
        if duplicates:
            raise ValueError(f"Duplicate words passed in list: {duplicates}")

    def __init__(self, words: Iterable[str] | None = None, *, case: bool = False) -> None:
        """Validate the words to count.

        Args:
            words: The words to count. Defaults to `DEFAULT_WORDS`.
            case: Whether matching is case-sensitive. Defaults to False, so the lowercase
                `DEFAULT_WORDS` match text of any case.

        Raises:
            TypeError: If `words` is a single word rather than a list of words.
            ValueError: If `words` is empty, contains an empty string, or contains duplicates.
        """
        if isinstance(words, str):
            hint = "pass None for default word list" if words == ALL else f"pass ['{words}'] instead"
            raise TypeError(f"words must be a list of strings or None, not a single word; {hint}")
        words = list(DEFAULT_WORDS if words is None else words)
        self._validate_words(words, case=case)
        self.words = words
        self.case = case

    @staticmethod
    def _annotate_word(inputs: pd.Series, word: str, *, method: WordMethod, case: bool) -> pd.Series:
        """Annotate inputs for a single word.

        Args:
            inputs: The texts to annotate.
            word: The word to match.
            method: Whether to count occurrences or flag containment.
            case: Whether matching is case-sensitive.

        Returns:
            Count, named after the word.

        Note:
            A word starting or ending with a non-word character (for example "'em") never
            matches, because the word boundary on that side cannot be satisfied.
        """
        pattern = rf"\b{re.escape(word)}\b"
        flags = 0 if case else re.IGNORECASE
        if method == "count":
            return inputs.str.count(pattern, flags=flags).rename(f"word_count__{word}")
        return inputs.str.contains(pattern, flags=flags).rename(f"word_contains__{word}")

    def annotate(
        self, inputs: list[str] | pd.Series, *, method: WordMethod = "count", allow_na: bool = False
    ) -> pd.DataFrame:
        """Annotate inputs for every word in `self.words`.

        Args:
            inputs: The texts to annotate.
            method: Whether to count occurrences or flag containment.
            allow_na: Whether missing values are allowed. When they are, a missing input yields
                `pd.NA` rather than zero occurrences.

        Returns:
            One column per word, in `self.words` order, sharing the index of `inputs`.

        Raises:
            TypeError: If `inputs` holds something other than strings.
            ValueError: If `method` is not a supported method, or if `inputs` holds missing
                values and `allow_na` is not set.
        """
        if method not in EXISTING_WORD_METHODS:
            raise ValueError(f"method must be one of {EXISTING_WORD_METHODS}, got '{method}'")
        prepared = _prepare_inputs(inputs, allow_na=allow_na)
        annotated = [self._annotate_word(prepared, word=w, method=method, case=self.case) for w in self.words]
        return pd.concat(annotated, axis=1)

__init__(words=None, *, case=False)

Validate the words to count.

Parameters:

Name Type Description Default
words Iterable[str] | None

The words to count. Defaults to DEFAULT_WORDS.

None
case bool

Whether matching is case-sensitive. Defaults to False, so the lowercase DEFAULT_WORDS match text of any case.

False

Raises:

Type Description
TypeError

If words is a single word rather than a list of words.

ValueError

If words is empty, contains an empty string, or contains duplicates.

Source code in src/edubehaviors/annotation.py
def __init__(self, words: Iterable[str] | None = None, *, case: bool = False) -> None:
    """Validate the words to count.

    Args:
        words: The words to count. Defaults to `DEFAULT_WORDS`.
        case: Whether matching is case-sensitive. Defaults to False, so the lowercase
            `DEFAULT_WORDS` match text of any case.

    Raises:
        TypeError: If `words` is a single word rather than a list of words.
        ValueError: If `words` is empty, contains an empty string, or contains duplicates.
    """
    if isinstance(words, str):
        hint = "pass None for default word list" if words == ALL else f"pass ['{words}'] instead"
        raise TypeError(f"words must be a list of strings or None, not a single word; {hint}")
    words = list(DEFAULT_WORDS if words is None else words)
    self._validate_words(words, case=case)
    self.words = words
    self.case = case

annotate(inputs, *, method='count', allow_na=False)

Annotate inputs for every word in self.words.

Parameters:

Name Type Description Default
inputs list[str] | Series

The texts to annotate.

required
method WordMethod

Whether to count occurrences or flag containment.

'count'
allow_na bool

Whether missing values are allowed. When they are, a missing input yields pd.NA rather than zero occurrences.

False

Returns:

Type Description
DataFrame

One column per word, in self.words order, sharing the index of inputs.

Raises:

Type Description
TypeError

If inputs holds something other than strings.

ValueError

If method is not a supported method, or if inputs holds missing values and allow_na is not set.

Source code in src/edubehaviors/annotation.py
def annotate(
    self, inputs: list[str] | pd.Series, *, method: WordMethod = "count", allow_na: bool = False
) -> pd.DataFrame:
    """Annotate inputs for every word in `self.words`.

    Args:
        inputs: The texts to annotate.
        method: Whether to count occurrences or flag containment.
        allow_na: Whether missing values are allowed. When they are, a missing input yields
            `pd.NA` rather than zero occurrences.

    Returns:
        One column per word, in `self.words` order, sharing the index of `inputs`.

    Raises:
        TypeError: If `inputs` holds something other than strings.
        ValueError: If `method` is not a supported method, or if `inputs` holds missing
            values and `allow_na` is not set.
    """
    if method not in EXISTING_WORD_METHODS:
        raise ValueError(f"method must be one of {EXISTING_WORD_METHODS}, got '{method}'")
    prepared = _prepare_inputs(inputs, allow_na=allow_na)
    annotated = [self._annotate_word(prepared, word=w, method=method, case=self.case) for w in self.words]
    return pd.concat(annotated, axis=1)

standard_classifier(*, Cs=DEFAULT_C_GRID, scoring='f1_macro', class_weight='balanced', cv=5, **kwargs)

Factory method for customized sklearn.linear_model.LogisticRegressionCV.

Parameters:

Name Type Description Default
Cs int | Sequence[float]

The inverse regularization strengths to choose between, or how many to space logarithmically between 1e-4 and 1e4.

DEFAULT_C_GRID
scoring str | Callable

The metric the penalty is chosen against, as a scorer name or a callable.

'f1_macro'
class_weight Mapping | Literal['balanced'] | None

How much each class counts toward the loss.

'balanced'
cv int

The number of StratifiedKFold folds used.

5
**kwargs Any

Keyword arguments to pass to LogisticRegressionCV.

{}

Returns:

Type Description
LogisticRegressionCV

An unfitted LogisticRegressionCV.

Source code in src/edubehaviors/models.py
def standard_classifier(
    *,
    Cs: int | Sequence[float] = DEFAULT_C_GRID,
    scoring: str | Callable = "f1_macro",
    class_weight: Mapping | Literal["balanced"] | None = "balanced",
    cv: int = 5,
    **kwargs: Any,
) -> LogisticRegressionCV:
    """Factory method for customized sklearn.linear_model.LogisticRegressionCV.

    Args:
        Cs: The inverse regularization strengths to choose between, or how many to space
            logarithmically between 1e-4 and 1e4.
        scoring: The metric the penalty is chosen against, as a scorer name or a callable.
        class_weight: How much each class counts toward the loss.
        cv: The number of StratifiedKFold folds used.
        **kwargs: Keyword arguments to pass to LogisticRegressionCV.

    Returns:
        An unfitted `LogisticRegressionCV`.
    """
    return LogisticRegressionCV(Cs=Cs, scoring=scoring, class_weight=class_weight, cv=cv, **kwargs)