Source code for geoestimate.types

"""Public result types for :mod:`geoestimate`."""

from dataclasses import dataclass
from typing import Literal

import pandas as pd

InferenceMethod = Literal["design", "iid", "cluster", "bootstrap"]
ResolvedInferenceMethod = Literal["iid", "cluster", "bootstrap"]


[docs] @dataclass(frozen=True, slots=True) class Diagnostics: """Analytic diagnostics for one estimate. Attributes: iid_standard_error: Standard error that treats rows as independent. cluster_standard_error: Cluster-sandwich standard error, when declared. design_effect: Cluster variance divided by iid variance, when defined. effective_sample_size: Observation count divided by the design effect. cluster_sizes: Number of observations in each declared cluster. effective_cluster_count: Kish effective count of declared clusters. """ iid_standard_error: float cluster_standard_error: float | None design_effect: float | None effective_sample_size: float cluster_sizes: tuple[int, ...] | None effective_cluster_count: float | None
[docs] @dataclass(frozen=True, slots=True) class Estimate: """A scalar estimate with uncertainty and design diagnostics.""" estimand: str variables: tuple[str, ...] estimate: float standard_error: float confidence_interval: tuple[float, float] confidence_level: float inference_method: ResolvedInferenceMethod n_observations: int n_clusters: int | None population_size: int | None diagnostics: Diagnostics
[docs] def summary(self) -> str: """Return a compact, human-readable summary.""" level = f"{100 * self.confidence_level:g}%" variable_text = ", ".join(self.variables) lines = [ "=" * 60, "geoestimate: Estimate", "=" * 60, f"Estimand: {self.estimand}", f"Variables: {variable_text}", f"Observations: {self.n_observations}", ] if self.n_clusters is not None: lines.append(f"Clusters: {self.n_clusters}") if self.population_size is not None: lines.append(f"Population size: {self.population_size}") lines.extend( [ f"Estimate: {self.estimate:.6g}", f"Standard error: {self.standard_error:.6g}", ( f"{level} confidence interval: " f"[{self.confidence_interval[0]:.6g}, " f"{self.confidence_interval[1]:.6g}]" ), f"Inference: {self.inference_method}", "=" * 60, ] ) return "\n".join(lines)
[docs] def to_frame(self) -> pd.DataFrame: """Return the scalar result as a one-row tidy table.""" is_univariate = self.estimand in {"mean", "total"} return pd.DataFrame( [ { "estimand": self.estimand, "variable": self.variables[0] if is_univariate else None, "numerator": None if is_univariate else self.variables[0], "denominator": None if is_univariate else self.variables[1], "estimate": self.estimate, "standard_error": self.standard_error, "conf_low": self.confidence_interval[0], "conf_high": self.confidence_interval[1], "confidence_level": self.confidence_level, "inference_method": self.inference_method, "n_observations": self.n_observations, "n_clusters": self.n_clusters, "population_size": self.population_size, } ] )
def __repr__(self) -> str: """Return a concise representation.""" return ( f"Estimate(estimand={self.estimand!r}, estimate={self.estimate:.6g}, " f"standard_error={self.standard_error:.6g}, " f"inference_method={self.inference_method!r})" )