contentintech
Beginner

Python for Data Cheatsheet

Quick-reference Python syntax for data work: types, comprehensions, file/CSV/JSON I/O, and virtual-env commands at a glance.

pythondata-typesfile-iogenerators
NotesCheatsheet

Data Types & Literals

Built-ins

n = 42                    # int
x = 3.14                  # float
s = "text"                # str
flag = True               # bool
items = [1, 2, 3]         # list  (mutable)
point = (1, 2)            # tuple (immutable)
row = {"k": "v"}          # dict
tags = {"a", "b"}         # set   (unique)
nothing = None            # NoneType

Conversions

int("42"), float("3.14"), str(42)
list("abc")               # ['a','b','c']
"".join(["a", "b"])       # 'ab'
"a,b,c".split(",")        # ['a','b','c']
dict(zip(keys, values))   # pair -> dict

Strings & Formatting

name, pct = "Ada", 0.9142
f"{name}: {pct:.1%}"      # 'Ada: 91.4%'
f"{42:>6}"                # right-align width 6
f"{3.14159:.2f}"          # '3.14'
f"{1234567:,}"            # '1,234,567'
s.strip(), s.lower(), s.upper()
s.replace("a", "b")
s.startswith("pre"), s.endswith(".csv")
"x" in s                  # membership

Comprehensions

[n*n for n in nums]                 # map
[n for n in nums if n > 0]          # filter
[a if a > 0 else 0 for a in nums]   # ternary
{k: v for k, v in pairs}            # dict comp
{w[0] for w in words}               # set comp
(n*n for n in nums)                 # generator (lazy)

Collections

List ops

lst.append(x); lst.extend([a, b])
lst.insert(0, x); lst.pop(); lst.pop(0)
lst.sort(); lst.sort(key=len, reverse=True)
sorted(lst, key=lambda r: r["age"])
lst[1:4], lst[::-1], lst[::2]        # slicing
len(lst), sum(lst), min(lst), max(lst)

Dict ops

d.get("k", default)
d.keys(); d.values(); d.items()
d.setdefault("k", [])
d.update(other)
{**a, **b}                          # merge
from collections import Counter, defaultdict
Counter(words)                      # frequency map
defaultdict(list)                   # auto-init values

Functions

def f(a, b=1, *args, **kwargs) -> int:
    return a + b

square = lambda x: x * x            # anonymous
list(map(square, nums))
list(filter(lambda n: n > 0, nums))
from functools import reduce
reduce(lambda a, b: a + b, nums)
enumerate(items)                    # (i, val) pairs
zip(a, b)                           # parallel iterate

File I/O

from pathlib import Path
p = Path("data") / "file.txt"
p.write_text(text, encoding="utf-8")
text = p.read_text(encoding="utf-8")

with p.open(encoding="utf-8") as f:
    for line in f:
        print(line.rstrip())

with open("out.txt", "w") as f:
    f.write("line\n")

p.exists(); p.glob("*.csv"); p.stem; p.suffix

CSV & JSON

csv module

import csv
with open("f.csv", newline="") as f:
    for row in csv.DictReader(f):
        print(row["col"])

with open("f.csv", "w", newline="") as f:
    w = csv.DictWriter(f, fieldnames=cols)
    w.writeheader(); w.writerows(rows)

json module

import json
obj  = json.loads(text)             # str -> obj
text = json.dumps(obj, indent=2)    # obj -> str
obj  = json.load(file)              # file -> obj
json.dump(obj, file)                # obj -> file

Control Flow

for i, v in enumerate(items):
    if v: continue
    if done: break
else:                               # runs if no break
    ...

try:
    risky()
except (ValueError, KeyError) as e:
    print(e)
finally:
    cleanup()

match status:                       # 3.10+ pattern match
    case 200: ok()
    case _:   fallback()

Data Type Quick Pick

Need Use
Ordered, editablelist
Fixed recordtuple
Lookup by keydict
Uniqueness / membershipset
Numeric arraysnp.array
Tabular datapd.DataFrame

Virtual Envs & pip

python -m venv .venv
source .venv/bin/activate      # mac/linux
.venv\Scripts\activate         # windows
pip install numpy pandas
pip freeze > requirements.txt
pip install -r requirements.txt
deactivate

# uv (fast, modern)
uv venv && uv pip install pandas

Section navigation