0% found this document useful (0 votes)
3 views10 pages

Pandas Series and DataFrame Basics

The document is a Jupyter notebook that introduces pandas Series and DataFrame data structures. It demonstrates how to create Series and DataFrames from various data, index them, query them, and handle missing data. Examples include creating Series from lists, dictionaries, and NumPy arrays; indexing and selecting values from Series and DataFrames; loading data from CSV files into DataFrames; and using DataFrame methods like dropna, fillna, and reset_index.

Uploaded by

Walid Maheri
Copyright
© All Rights Reserved
We take content rights seriously. If you suspect this is your content, claim it here.
Available Formats
Download as PDF, TXT or read online on Scribd
0% found this document useful (0 votes)
3 views10 pages

Pandas Series and DataFrame Basics

The document is a Jupyter notebook that introduces pandas Series and DataFrame data structures. It demonstrates how to create Series and DataFrames from various data, index them, query them, and handle missing data. Examples include creating Series from lists, dictionaries, and NumPy arrays; indexing and selecting values from Series and DataFrames; loading data from CSV files into DataFrames; and using DataFrame methods like dropna, fillna, and reset_index.

Uploaded by

Walid Maheri
Copyright
© All Rights Reserved
We take content rights seriously. If you suspect this is your content, claim it here.
Available Formats
Download as PDF, TXT or read online on Scribd

03/03/2019 Week 2

You are currently looking at version 1.0 of this notebook. To download notebooks and datafiles, as well as get
help on Jupyter notebooks in the Coursera platform, visit the Jupyter Notebook FAQ
([Link] course resource.

The Series Data Structure


In [ ]:

import pandas as pd
[Link]?

In [ ]:

animals = ['Tiger', 'Bear', 'Moose']


[Link](animals)

In [ ]:

numbers = [1, 2, 3]
[Link](numbers)

In [ ]:

animals = ['Tiger', 'Bear', None]


[Link](animals)

In [ ]:

numbers = [1, 2, None]


[Link](numbers)

In [ ]:

import numpy as np
[Link] == None

In [ ]:

[Link] == [Link]

In [ ]:

[Link]([Link])

[Link] 1/10
03/03/2019 Week 2

In [2]:

import pandas as pd
sports = {'Archery': 'Bhutan',
'Golf': 'Scotland',
'Sumo': 'Japan',
'Taekwondo': 'South Korea'}
s = [Link](sports)
s

Out[2]:

Archery Bhutan
Golf Scotland
Sumo Japan
Taekwondo South Korea
dtype: object

In [5]:

[Link]['Golf']

Out[5]:
'Scotland'

In [ ]:

s = [Link](['Tiger', 'Bear', 'Moose'], index=['India', 'America', 'Canada'])


s

In [ ]:

sports = {'Archery': 'Bhutan',


'Golf': 'Scotland',
'Sumo': 'Japan',
'Taekwondo': 'South Korea'}
s = [Link](sports, index=['Golf', 'Sumo', 'Hockey'])
s

Querying a Series
In [ ]:

sports = {'Archery': 'Bhutan',


'Golf': 'Scotland',
'Sumo': 'Japan',
'Taekwondo': 'South Korea'}
s = [Link](sports)
s

In [ ]:

[Link][3]

In [ ]:

[Link]['Golf']

[Link] 2/10
03/03/2019 Week 2

In [ ]:

s[3]

In [ ]:

s['Golf']

In [ ]:

sports = {99: 'Bhutan',


100: 'Scotland',
101: 'Japan',
102: 'South Korea'}
s = [Link](sports)

In [ ]:

s[0] #This won't call [Link][0] as one might expect, it generates an error instead

In [ ]:

s = [Link]([100.00, 120.00, 101.00, 3.00])


s

In [ ]:

total = 0
for item in s:
total+=item
print(total)

In [ ]:

import numpy as np

total = [Link](s)
print(total)

In [ ]:

#this creates a big series of random numbers


s = [Link]([Link](0,1000,10000))
[Link]()

In [ ]:

len(s)

In [ ]:

%%timeit -n 100
summary = 0
for item in s:
summary+=item

[Link] 3/10
03/03/2019 Week 2

In [ ]:

%%timeit -n 100
summary = [Link](s)

In [ ]:

s+=2 #adds two to each item in s using broadcasting


[Link]()

In [ ]:

for label, value in [Link]():


s.set_value(label, value+2)
[Link]()

In [ ]:

%%timeit -n 10
s = [Link]([Link](0,1000,10000))
for label, value in [Link]():
[Link][label]= value+2

In [ ]:

%%timeit -n 10
s = [Link]([Link](0,1000,10000))
s+=2

In [ ]:

s = [Link]([1, 2, 3])
[Link]['Animal'] = 'Bears'
s

In [ ]:

original_sports = [Link]({'Archery': 'Bhutan',


'Golf': 'Scotland',
'Sumo': 'Japan',
'Taekwondo': 'South Korea'})
cricket_loving_countries = [Link](['Australia',
'Barbados',
'Pakistan',
'England'],
index=['Cricket',
'Cricket',
'Cricket',
'Cricket'])
all_countries = original_sports.append(cricket_loving_countries)

In [ ]:

original_sports

[Link] 4/10
03/03/2019 Week 2

In [ ]:

cricket_loving_countries

In [ ]:

all_countries

In [ ]:

all_countries.loc['Cricket']

The DataFrame Data Structure


In [ ]:

import pandas as pd
purchase_1 = [Link]({'Name': 'Chris',
'Item Purchased': 'Dog Food',
'Cost': 22.50})
purchase_2 = [Link]({'Name': 'Kevyn',
'Item Purchased': 'Kitty Litter',
'Cost': 2.50})
purchase_3 = [Link]({'Name': 'Vinod',
'Item Purchased': 'Bird Seed',
'Cost': 5.00})
df = [Link]([purchase_1, purchase_2, purchase_3], index=['Store 1', 'Store 1', 'Store
[Link]()

In [ ]:

[Link]['Store 2']

In [ ]:

type([Link]['Store 2'])

In [ ]:

[Link]['Store 1']

In [ ]:

[Link]['Store 1', 'Cost']

In [ ]:

df.T

In [ ]:

[Link]['Cost']

In [ ]:

df['Cost']

[Link] 5/10
03/03/2019 Week 2

In [ ]:

[Link]['Store 1']['Cost']

In [ ]:

[Link][:,['Name', 'Cost']]

In [ ]:

[Link]('Store 1')

In [ ]:

df

In [ ]:

copy_df = [Link]()
copy_df = copy_df.drop('Store 1')
copy_df

In [ ]:

copy_df.drop?

In [ ]:

del copy_df['Name']
copy_df

In [ ]:

df['Location'] = None
df

Dataframe Indexing and Loading


In [ ]:

costs = df['Cost']
costs

In [ ]:

costs+=2
costs

In [ ]:

df

In [ ]:

!cat [Link]

[Link] 6/10
03/03/2019 Week 2

In [ ]:

df = pd.read_csv('[Link]')
[Link]()

In [ ]:

df = pd.read_csv('[Link]', index_col = 0, skiprows=1)


[Link]()

In [ ]:

[Link]

In [ ]:

for col in [Link]:


if col[:2]=='01':
[Link](columns={col:'Gold' + col[4:]}, inplace=True)
if col[:2]=='02':
[Link](columns={col:'Silver' + col[4:]}, inplace=True)
if col[:2]=='03':
[Link](columns={col:'Bronze' + col[4:]}, inplace=True)
if col[:1]=='№':
[Link](columns={col:'#' + col[1:]}, inplace=True)

[Link]()

Querying a DataFrame
In [ ]:

df['Gold'] > 0

In [ ]:

only_gold = [Link](df['Gold'] > 0)


only_gold.head()

In [ ]:

only_gold['Gold'].count()

In [ ]:

df['Gold'].count()

In [ ]:

only_gold = only_gold.dropna()
only_gold.head()

In [ ]:

only_gold = df[df['Gold'] > 0]


only_gold.head()

[Link] 7/10
03/03/2019 Week 2

In [ ]:

len(df[(df['Gold'] > 0) | (df['Gold.1'] > 0)])

In [ ]:

df[(df['Gold.1'] > 0) & (df['Gold'] == 0)]

Indexing Dataframes
In [ ]:

[Link]()

In [ ]:

df['country'] = [Link]
df = df.set_index('Gold')
[Link]()

In [ ]:

df = df.reset_index()
[Link]()

In [ ]:

df = pd.read_csv('[Link]')
[Link]()

In [ ]:

df['SUMLEV'].unique()

In [ ]:

df=df[df['SUMLEV'] == 50]
[Link]()

[Link] 8/10
03/03/2019 Week 2

In [ ]:

columns_to_keep = ['STNAME',
'CTYNAME',
'BIRTHS2010',
'BIRTHS2011',
'BIRTHS2012',
'BIRTHS2013',
'BIRTHS2014',
'BIRTHS2015',
'POPESTIMATE2010',
'POPESTIMATE2011',
'POPESTIMATE2012',
'POPESTIMATE2013',
'POPESTIMATE2014',
'POPESTIMATE2015']
df = df[columns_to_keep]
[Link]()

In [ ]:

df = df.set_index(['STNAME', 'CTYNAME'])
[Link]()

In [ ]:

[Link]['Michigan', 'Washtenaw County']

In [ ]:

[Link][ [('Michigan', 'Washtenaw County'),


('Michigan', 'Wayne County')] ]

Missing values
In [ ]:

df = pd.read_csv('[Link]')
df

In [ ]:

[Link]?

In [ ]:

df = df.set_index('time')
df = df.sort_index()
df

In [ ]:

df = df.reset_index()
df = df.set_index(['time', 'user'])
df

[Link] 9/10
03/03/2019 Week 2

In [ ]:

df = [Link](method='ffill')
[Link]()

[Link] 10/10

You might also like