skimpy

These details have not been verified by PyPI

Project links

Development Status
- 3 - Alpha
License
- OSI Approved :: MIT License
Programming Language

Project description

Skimpy

A light weight tool for creating summary statistics from dataframes.

png

skimpy is a light weight tool that provides summary statistics about variables in data frames within the console or your interactive Python window. Think of it as a super-charged version of df.describe().

You can find the documentation here.

Quickstart

skim a dataframe and produce summary statistics within the console using:

from skimpy import skim

skim(df)

where df is a dataframe.

If you need to a dataset to try skimpy out on, you can use the built-in test dataframe:

#| output: asis
from skimpy import skim, generate_test_data

df = generate_test_data()
skim(df)

╭───────────────────────────────────── skimpy summary ──────────────────────────────────────╮
│          Data Summary                Data Types               Categories                  │
│ ┏━━━━━━━━━━━━━━━━━━━┳━━━━━━━━┓ ┏━━━━━━━━━━━━━┳━━━━━━━┓ ┏━━━━━━━━━━━━━━━━━━━━━━━┓          │
│ ┃ dataframe         ┃ Values ┃ ┃ Column Type ┃ Count ┃ ┃ Categorical Variables ┃          │
│ ┡━━━━━━━━━━━━━━━━━━━╇━━━━━━━━┩ ┡━━━━━━━━━━━━━╇━━━━━━━┩ ┡━━━━━━━━━━━━━━━━━━━━━━━┩          │
│ │ Number of rows    │ 1000   │ │ float64     │ 3     │ │ class                 │          │
│ │ Number of columns │ 10     │ │ category    │ 2     │ │ location              │          │
│ └───────────────────┴────────┘ │ datetime64  │ 2     │ └───────────────────────┘          │
│                                │ int64       │ 1     │                                    │
│                                │ bool        │ 1     │                                    │
│                                │ string      │ 1     │                                    │
│                                └─────────────┴───────┘                                    │
│                                          number                                           │
│ ┏━━━━━━━━━━┳━━━━━━━━━┳━━━━━━━━━┳━━━━━━━┳━━━━━━┳━━━━━━━━━┳━━━━━━━┳━━━━━━┳━━━━━━┳━━━━━━━━┓  │
│ ┃ column_n ┃         ┃ complet ┃       ┃      ┃         ┃       ┃      ┃      ┃        ┃  │
│ ┃ ame      ┃ missing ┃ e %     ┃ mean  ┃ sd   ┃ p0      ┃ p25   ┃ p75  ┃ p100 ┃ hist   ┃  │
│ ┡━━━━━━━━━━╇━━━━━━━━━╇━━━━━━━━━╇━━━━━━━╇━━━━━━╇━━━━━━━━━╇━━━━━━━╇━━━━━━╇━━━━━━╇━━━━━━━━┩  │
│ │ length   │       0 │       1 │   0.5 │ 0.36 │ 1.6e-06 │  0.13 │ 0.86 │    1 │ █▃▃▃▄█ │  │
│ │ width    │       0 │       1 │     2 │  1.9 │  0.0021 │   0.6 │    3 │   14 │  █▃▁   │  │
│ │ depth    │       0 │       1 │    10 │  3.2 │       2 │     8 │   12 │   20 │ ▁▄█▆▃▁ │  │
│ │ rnd      │     120 │    0.88 │ -0.02 │    1 │    -2.8 │ -0.74 │ 0.66 │  3.7 │ ▁▄█▅▁  │  │
│ └──────────┴─────────┴─────────┴───────┴──────┴─────────┴───────┴──────┴──────┴────────┘  │
│                                         category                                          │
│ ┏━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━┳━━━━━━━━━━━━┓  │
│ ┃ column_name         ┃ missing       ┃ complete %         ┃ ordered      ┃ unique     ┃  │
│ ┡━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━╇━━━━━━━━━━━━┩  │
│ │ class               │             0 │                  1 │ False        │          2 │  │
│ │ location            │             1 │                  1 │ False        │          5 │  │
│ └─────────────────────┴───────────────┴────────────────────┴──────────────┴────────────┘  │
│                                         datetime                                          │
│ ┏━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━┳━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━┳━━━━━━━━━━━┓  │
│ ┃ column_name     ┃ missing   ┃ complete %   ┃ first        ┃ last         ┃ frequency ┃  │
│ ┡━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━╇━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━╇━━━━━━━━━━━┩  │
│ │ date            │         0 │            1 │  2018-01-31  │  2101-04-30  │ M         │  │
│ │ date_no_freq    │         3 │            1 │  1992-01-05  │  2023-03-04  │ None      │  │
│ └─────────────────┴───────────┴──────────────┴──────────────┴──────────────┴───────────┘  │
│                                          string                                           │
│ ┏━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━┓  │
│ ┃ column_name      ┃ missing    ┃ complete %     ┃ words per row      ┃ total words    ┃  │
│ ┡━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━┩  │
│ │ text             │          6 │           0.99 │                5.8 │           5800 │  │
│ └──────────────────┴────────────┴────────────────┴────────────────────┴────────────────┘  │
│                                           bool                                            │
│ ┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━┓  │
│ ┃ column_name                 ┃ true        ┃ true rate              ┃ hist            ┃  │
│ ┡━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━┩  │
│ │ booly_col                   │         520 │                   0.52 │     █    █      │  │
│ └─────────────────────────────┴─────────────┴────────────────────────┴─────────────────┘  │
╰─────────────────────────────────────────── End ───────────────────────────────────────────╯

It is recommended that you set your datatypes before using skimpy (for example converting any text columns to pandas string datatype), as this will produce richer statistical summaries. However, the skim function will try and guess what the datatypes of your columns are.

skimpy also comes with a clean_columns function as a convenience. This slugifies column names. For example,

import pandas as pd
from rich import print
from skimpy import clean_columns

columns = [
    "bs lncs;n edbn ",
    "Nín hǎo. Wǒ shì zhōng guó rén",
    "___This is a test___",
    "ÜBER Über German Umlaut",
]
messy_df = pd.DataFrame(columns=columns, index=[0], data=[range(len(columns))])
print("Column names:")
print(list(messy_df.columns))

Column names:

[
    'bs lncs;n edbn ',
    'Nín hǎo. Wǒ shì zhōng guó rén',
    '___This is a test___',
    'ÜBER Über German Umlaut'
]

Now let's clean these—by default what we get back is in snake case:

clean_df = clean_columns(messy_df)
print(list(clean_df.columns))

4 column names have been cleaned

[
    'bs_lncs_n_edbn',
    'nin_hao_wo_shi_zhong_guo_ren',
    'this_is_a_test',
    'uber_uber_german_umlaut'
]

Other naming conventions are available, for example camel case:

clean_df = clean_columns(messy_df, case="camel")
print(list(clean_df.columns))

4 column names have been cleaned

['bsLncsNEdbn', 'ninHaoWoShiZhongGuoRen', 'thisIsATest', 'uberUberGermanUmlaut']

Requirements

You can find a full list of requirements in the pyproject.toml file. The main requirements are:

python >=3.7.1,<4.0.0

click 7.1.2

rich >=10.9,<12.0

pandas ^1.3.2

Pygments ^2.10.0

typeguard ^2.12.1

jupyter ^1.0.0

ipykernel ^6.7.0

You can try this package out right now in your browser using this Google Colab notebook (requires a Google account). Note that the Google Colab notebook uses the latest package released on PyPI (rather than the development release).

Installation

You can install the latest release of skimpy via pip from PyPI:

$ pip install skimpy

To install the development version from git, use:

$ pip install git+https://github.com/aeturrell/skimpy.git

For development, see the Contributor Guide.

Usage

This package is mostly designed to be used within an interactive console session or Jupyter notebook

from skimpy import skim

skim(df)

However, you can also use it on the command line:

$ skimpy file.csv

Features

Support for boolean, numeric, datetime, string, and category datatypes
Command line interface in addition to interactive console functionality
Light weight, with results printed to terminal using the rich package.
Support for different colours for different types of output
Rounds numerical output to 2 significant figures

skim accepts keyword arguments that change the colour of the top level column headers. For example, to change the colour to magenta, it's

skim(df, header_style="italic magenta")

╭───────────────────────────────────── skimpy summary ──────────────────────────────────────╮
│          Data Summary                Data Types               Categories                  │
│ ┏━━━━━━━━━━━━━━━━━━━┳━━━━━━━━┓ ┏━━━━━━━━━━━━━┳━━━━━━━┓ ┏━━━━━━━━━━━━━━━━━━━━━━━┓          │
│ ┃ dataframe         ┃ Values ┃ ┃ Column Type ┃ Count ┃ ┃ Categorical Variables ┃          │
│ ┡━━━━━━━━━━━━━━━━━━━╇━━━━━━━━┩ ┡━━━━━━━━━━━━━╇━━━━━━━┩ ┡━━━━━━━━━━━━━━━━━━━━━━━┩          │
│ │ Number of rows    │ 1000   │ │ float64     │ 3     │ │ class                 │          │
│ │ Number of columns │ 10     │ │ category    │ 2     │ │ location              │          │
│ └───────────────────┴────────┘ │ datetime64  │ 2     │ └───────────────────────┘          │
│                                │ int64       │ 1     │                                    │
│                                │ bool        │ 1     │                                    │
│                                │ string      │ 1     │                                    │
│                                └─────────────┴───────┘                                    │
│                                          number                                           │
│ ┏━━━━━━━━━━┳━━━━━━━━━┳━━━━━━━━━┳━━━━━━━┳━━━━━━┳━━━━━━━━━┳━━━━━━━┳━━━━━━┳━━━━━━┳━━━━━━━━┓  │
│ ┃ column_n ┃         ┃ complet ┃       ┃      ┃         ┃       ┃      ┃      ┃        ┃  │
│ ┃ ame      ┃ missing ┃ e %     ┃ mean  ┃ sd   ┃ p0      ┃ p25   ┃ p75  ┃ p100 ┃ hist   ┃  │
│ ┡━━━━━━━━━━╇━━━━━━━━━╇━━━━━━━━━╇━━━━━━━╇━━━━━━╇━━━━━━━━━╇━━━━━━━╇━━━━━━╇━━━━━━╇━━━━━━━━┩  │
│ │ length   │       0 │       1 │   0.5 │ 0.36 │ 1.6e-06 │  0.13 │ 0.86 │    1 │ █▃▃▃▄█ │  │
│ │ width    │       0 │       1 │     2 │  1.9 │  0.0021 │   0.6 │    3 │   14 │  █▃▁   │  │
│ │ depth    │       0 │       1 │    10 │  3.2 │       2 │     8 │   12 │   20 │ ▁▄█▆▃▁ │  │
│ │ rnd      │     120 │    0.88 │ -0.02 │    1 │    -2.8 │ -0.74 │ 0.66 │  3.7 │ ▁▄█▅▁  │  │
│ └──────────┴─────────┴─────────┴───────┴──────┴─────────┴───────┴──────┴──────┴────────┘  │
│                                         category                                          │
│ ┏━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━┳━━━━━━━━━━━━┓  │
│ ┃ column_name         ┃ missing       ┃ complete %         ┃ ordered      ┃ unique     ┃  │
│ ┡━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━╇━━━━━━━━━━━━┩  │
│ │ class               │             0 │                  1 │ False        │          2 │  │
│ │ location            │             1 │                  1 │ False        │          5 │  │
│ └─────────────────────┴───────────────┴────────────────────┴──────────────┴────────────┘  │
│                                         datetime                                          │
│ ┏━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━┳━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━┳━━━━━━━━━━━┓  │
│ ┃ column_name     ┃ missing   ┃ complete %   ┃ first        ┃ last         ┃ frequency ┃  │
│ ┡━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━╇━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━╇━━━━━━━━━━━┩  │
│ │ date            │         0 │            1 │  2018-01-31  │  2101-04-30  │ M         │  │
│ │ date_no_freq    │         3 │            1 │  1992-01-05  │  2023-03-04  │ None      │  │
│ └─────────────────┴───────────┴──────────────┴──────────────┴──────────────┴───────────┘  │
│                                          string                                           │
│ ┏━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━┓  │
│ ┃ column_name      ┃ missing    ┃ complete %     ┃ words per row      ┃ total words    ┃  │
│ ┡━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━┩  │
│ │ text             │          6 │           0.99 │                5.8 │           5800 │  │
│ └──────────────────┴────────────┴────────────────┴────────────────────┴────────────────┘  │
│                                           bool                                            │
│ ┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━┓  │
│ ┃ column_name                 ┃ true        ┃ true rate              ┃ hist            ┃  │
│ ┡━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━┩  │
│ │ booly_col                   │         520 │                   0.52 │     █    █      │  │
│ └─────────────────────────────┴─────────────┴────────────────────────┴─────────────────┘  │
╰─────────────────────────────────────────── End ───────────────────────────────────────────╯

Contributing

Contributions are very welcome. To learn more, see the Contributor Guide.

Note that you will need Quarto and Make installed to build the docs. You can preview the docs using poetry run quarto preview --execute. You can build them with make.

License

Distributed under the terms of the MIT license, skimpy is free and open source software. You can find the license here

Issues

If you encounter any problems, please file an issue along with a detailed description.

Credits

This project was generated from @cjolowicz's Hypermodern Python Cookiecutter template.

skimpy was inspired by the R package skimr and by exploratory Python packages including pandas_profiling and dataprep, from which the clean_columns function comes.

The package is built with poetry, while the documentation is built with Quarto. Tests are run with nox.

Project details

These details have not been verified by PyPI

Project links

Development Status
- 3 - Alpha
License
- OSI Approved :: MIT License
Programming Language

Release history Release notifications | RSS feed

0.0.20

Jan 3, 2026

0.0.19

Nov 4, 2025

0.0.18

Jan 6, 2025

0.0.17

Jan 3, 2025

0.0.16

Dec 18, 2024

0.0.15

May 15, 2024

0.0.14

Jan 23, 2024

0.0.13

Jan 23, 2024

0.0.12

Jan 17, 2024

0.0.11

Sep 11, 2023

0.0.10

Aug 25, 2023

0.0.9

Jul 16, 2023

0.0.8

Jan 15, 2023

0.0.7

Oct 23, 2022

This version

0.0.6

Jun 22, 2022

0.0.5

Dec 3, 2021

0.0.4

Oct 6, 2021

0.0.3

Sep 9, 2021

0.0.2

Sep 4, 2021

Download files

Download the file for your platform. If you're not sure which to choose, learn more about installing packages.

Source Distribution

skimpy-0.0.6.tar.gz (22.5 kB view details)

Uploaded Jun 22, 2022 Source

Built Distribution

If you're not sure about the file name format, learn more about wheel file names.

The dropdown lists show the available interpreters, ABIs, and platforms. Enable javascript to be able to filter the list of wheel files.

skimpy-0.0.6-py3-none-any.whl (15.0 kB view details)

Uploaded Jun 22, 2022 Python 3

File details

Details for the file skimpy-0.0.6.tar.gz.

File metadata

Download URL: skimpy-0.0.6.tar.gz
Upload date: Jun 22, 2022
Size: 22.5 kB
Tags: Source
Uploaded using Trusted Publishing? No
Uploaded via: twine/4.0.1 CPython/3.9.13

File hashes

Hashes for skimpy-0.0.6.tar.gz
Algorithm	Hash digest
SHA256	`35f07e617083c1a2755f707fbb0bb5e619494c1184debd010dea90a7eb41e1f9`
MD5	`15bab2a7728e00db01d3bebad136767d`
BLAKE2b-256	`7baa6ad748e836c05237525dec738972a309d29909bc0a1284718620ac25c992`

See more details on using hashes here.

File details

Details for the file skimpy-0.0.6-py3-none-any.whl.

File metadata

Download URL: skimpy-0.0.6-py3-none-any.whl
Upload date: Jun 22, 2022
Size: 15.0 kB
Tags: Python 3
Uploaded using Trusted Publishing? No
Uploaded via: twine/4.0.1 CPython/3.9.13

File hashes

Hashes for skimpy-0.0.6-py3-none-any.whl
Algorithm	Hash digest
SHA256	`9431e76ae4dffaec936d0f55db09f90c713850f9aebfb55b26036d823344656c`
MD5	`b701b1c961156b3650474ce8069e9172`
BLAKE2b-256	`6ba7412a814fb43f488d84044128a3e8e05f712da6892f9dfa1b7b4971c1e5af`

See more details on using hashes here.

skimpy 0.0.6

Navigation

Verified details

Maintainers

Unverified details

Project links

Meta

Classifiers

Project description

Skimpy

Quickstart

Requirements

Installation

Usage

Features

Contributing

License

Issues

Credits

Project details

Verified details

Maintainers

Unverified details

Project links

Meta

Classifiers

Release history Release notifications | RSS feed

Download files

Source Distribution

Built Distribution

File details

File metadata

File hashes

File details

File metadata

File hashes