import warnings
warnings.filterwarnings('ignore')
import missingno as msno
import pandas as pd
from pandas import DataFrame
import matplotlib.pyplot as plt 
import seaborn as sns
import numpy as np

import pandas as pd
from pandas import DataFrame, Series
import matplotlib.pyplot as plt 
path = '../data/tianchi/'
# .read_csv返回DataFrame形式的数据
train = pd.read_csv(path+'train.csv')
test = pd.read_csv(path+'testA.csv')
# 1首先要看下数据集的分布情况
train.head().append(train.tail())
idheartbeat_signalslabel
000.9912297987616655,0.9435330436439665,0.764677...0.0
110.9714822034884503,0.9289687459588268,0.572932...0.0
221.0,0.9591487564065292,0.7013782792997189,0.23...2.0
330.9757952826275774,0.9340884687738161,0.659636...0.0
440.0,0.055816398940721094,0.26129357194994196,0...2.0
99995999951.0,0.677705342021188,0.22239242747868546,0.25...0.0
99996999960.9268571578157265,0.9063471198026871,0.636993...2.0
99997999970.9258351628306013,0.5873839035878395,0.633226...3.0
99998999981.0,0.9947621698382489,0.8297017704865509,0.45...2.0
99999999990.9259994004527861,0.916476635326053,0.4042900...0.0
test.head().append(test.tail())
idheartbeat_signals
01000000.9915713654170097,1.0,0.6318163407681274,0.13...
11000010.6075533139615096,0.5417083883163654,0.340694...
21000020.9752726292239277,0.6710965234906665,0.686758...
31000030.9956348033996116,0.9170249621481004,0.521096...
41000041.0,0.8879490481178918,0.745564725322326,0.531...
199951199951.0,0.8330283177934747,0.6340472606311671,0.63...
199961199961.0,0.8259705825857048,0.4521053488322387,0.08...
199971199970.951744840752379,0.9162611283848351,0.6675251...
199981199980.9276692903808186,0.6771898159607004,0.242906...
199991199990.6653212231837624,0.527064114047737,0.5166625...
# 2,利用describel()和info()总揽数据分布和数据类型
# 个数count、平均值mean、方差std、最小值min、中位数25% 50% 75% 、以及最大值
train.describe()
idlabel
count100000.000000100000.000000
mean49999.5000000.856960
std28867.6577971.217084
min0.0000000.000000
25%24999.7500000.000000
50%49999.5000000.000000
75%74999.2500002.000000
max99999.0000003.000000
test.describe()
id
count20000.000000
mean109999.500000
std5773.647028
min100000.000000
25%104999.750000
50%109999.500000
75%114999.250000
max119999.000000
train.info()
<class 'pandas.core.frame.DataFrame'>
RangeIndex: 100000 entries, 0 to 99999
Data columns (total 3 columns):
 #   Column             Non-Null Count   Dtype  
---  ------             --------------   -----  
 0   id                 100000 non-null  int64  
 1   heartbeat_signals  100000 non-null  object 
 2   label              100000 non-null  float64
dtypes: float64(1), int64(1), object(1)
memory usage: 2.3+ MB
test.info()
<class 'pandas.core.frame.DataFrame'>
RangeIndex: 20000 entries, 0 to 19999
Data columns (total 2 columns):
 #   Column             Non-Null Count  Dtype 
---  ------             --------------  ----- 
 0   id                 20000 non-null  int64 
 1   heartbeat_signals  20000 non-null  object
dtypes: int64(1), object(1)
memory usage: 312.6+ KB
# 3,缺失值判断
train.isnull().sum()
id                   0
heartbeat_signals    0
label                0
dtype: int64
test.isnull().sum()
id                   0
heartbeat_signals    0
dtype: int64
# 4,查看分布情况
## 1) 总体分布概况(无界约翰逊分布等)
import scipy.stats as st
y = train['label']
plt.figure(1); plt.title('Default')
sns.distplot(y, rug=True, bins=20)
plt.figure(2); plt.title('Normal')
sns.distplot(y, kde=False, fit=st.norm)
plt.figure(3); plt.title('Log Normal')
sns.distplot(y, kde=False, fit=st.lognorm)
<matplotlib.axes._subplots.AxesSubplot at 0x199f1cf9948>

png
png
png

# 2)查看skewness and kurtosis
sns.distplot(train['label']);
print("Skewness: %f" % train['label'].skew())
print("Kurtosis: %f" % train['label'].kurt())
Skewness: 0.871005
Kurtosis: -1.009573

png

train.skew(), train.kurt()
(id       0.000000
 label    0.871005
 dtype: float64,
 id      -1.200000
 label   -1.009573
 dtype: float64)
sns.distplot(train.kurt(),color='orange',axlabel ='Kurtness')
<matplotlib.axes._subplots.AxesSubplot at 0x199f1f40688>

png

## 3) 查看预测值的具体频数
plt.hist(train['label'], orientation = 'vertical',histtype = 'bar', color ='red')
plt.show()

png

Logo

DAMO开发者矩阵,由阿里巴巴达摩院和中国互联网协会联合发起,致力于探讨最前沿的技术趋势与应用成果,搭建高质量的交流与分享平台,推动技术创新与产业应用链接,围绕“人工智能与新型计算”构建开放共享的开发者生态。

更多推荐