Two datasets...
Post-workday evaluations of work motivation related stuff. Large spread across how many time points people collected; few decent-ish ones.
Traditional questionnaires from the same participants.
data=="data/moti_feasibility_james_daily.csv"¶import pandas as pd
import numpy as np
import seaborn as sns
import matplotlib.pyplot as plt
from matplotlib.colors import LinearSegmentedColormap
import session_info
from jmspack.NLTSA import (flatten,
ts_levels,
fluctuation_intensity,
distribution_uniformity,
complexity_resonance,
complexity_resonance_diagram,
cumulative_complexity_peaks,
cumulative_complexity_peaks_plot)
from sklearn.preprocessing import MinMaxScaler
# To use this experimental feature, we need to explicitly ask for it:
from sklearn.experimental import enable_iterative_imputer # noqa
from sklearn.impute import IterativeImputer
from sklearn.tree import DecisionTreeRegressor
from sklearn.ensemble import ExtraTreesRegressor
from sklearn.neighbors import KNeighborsRegressor
from sklearn.pipeline import make_pipeline
from pyunicorn.timeseries import RecurrencePlot
pyunicorn: Package netCDF4 could not be loaded. Some functionality in class Data might not be available! pyunicorn: Package netCDF4 could not be loaded. Some functionality in class NetCDFDictionary might not be available!
Show the session information of the packages used in this analysis
session_info.show(write_req_file=False,
req_file_name="work_motivation_EDA_requirements.txt",)
----- jmspack 0.0.3 matplotlib 3.3.4 numpy 1.19.2 pandas 1.2.3 pyunicorn NA seaborn 0.11.1 session_info 1.0.0 sklearn 0.24.1 -----
PIL 8.1.2 appnope 0.1.2 backcall 0.2.0 cffi 1.14.5 colorama 0.4.4 cycler 0.10.0 cython_runtime NA dateutil 2.8.1 decorator 4.4.2 igraph 0.9.1 ipykernel 5.3.4 ipython_genutils 0.2.0 ipywidgets 7.6.3 jedi 0.17.2 joblib 0.17.0 kiwisolver 1.3.1 mpl_toolkits NA parso 0.7.0 pexpect 4.8.0 pickleshare 0.7.5 pkg_resources NA prompt_toolkit 3.0.8 ptyprocess 0.7.0 pyexpat NA pygments 2.8.1 pyparsing 2.4.7 pytz 2021.1 scipy 1.5.3 six 1.15.0 statsmodels 0.12.2 storemagic NA texttable 1.6.3 tornado 6.1 traitlets 5.0.5 wcwidth 0.2.5 zmq 20.0.0
----- IPython 7.21.0 jupyter_client 6.1.7 jupyter_core 4.7.1 jupyterlab 2.2.6 notebook 6.2.0 ----- Python 3.9.2 (default, Mar 3 2021, 11:58:52) [Clang 10.0.0 ] macOS-10.16-x86_64-i386-64bit ----- Session information updated at 2021-06-15 11:14
_df = pd.read_csv("data/moti_feasibility_james_daily.csv")
_df = _df.assign(date=pd.to_datetime(_df["Date (submitted)"].str.split("T", expand=True)[0], format="%Y-%m-%dT%H:%M:%S"))
def convert_numeric_where_possible(x):
try:
return x.astype(np.number)
except:
return x
df = (_df
.loc[:, ["User", "Field", "Value", "date"]]
.pivot(index=["User", "date"], columns="Field")
.droplevel(level=0, axis=1)
.apply(convert_numeric_where_possible)
)
df.head(2)
| Field | DailySituation | absorption | amotivation | autonomy | competence | dedication | emotional_drain | enjoyment | external pressure | howami_participation_ended | howami_participation_started | importance | interest | internal pressure | productivity_work | relatedness | satisfaction_work | strategies | time_pressure | vigor | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| User | date | ||||||||||||||||||||
| Moti105 | 2020-03-03 | NaN | NaN | NaN | NaN | NaN | NaN | NaN | NaN | NaN | NaN | 2020-03-03 | NaN | NaN | NaN | NaN | NaN | NaN | NaN | NaN | NaN |
| 2020-03-09 | NaN | 68.0 | 66.0 | 33.0 | 62.0 | 69.0 | NaN | 31.0 | 52.0 | NaN | NaN | 67.0 | 74.0 | 60.0 | 72.0 | 65.0 | 42.0 | emphasize_autonomy;reflect_desire;identifyas_r... | NaN | 64.0 |
df.info()
<class 'pandas.core.frame.DataFrame'>
MultiIndex: 1674 entries, ('Moti105', Timestamp('2020-03-03 00:00:00')) to ('Moti218', Timestamp('2020-05-15 00:00:00'))
Data columns (total 20 columns):
# Column Non-Null Count Dtype
--- ------ -------------- -----
0 DailySituation 1173 non-null object
1 absorption 1487 non-null float64
2 amotivation 1487 non-null float64
3 autonomy 1487 non-null float64
4 competence 1487 non-null float64
5 dedication 1487 non-null float64
6 emotional_drain 1126 non-null float64
7 enjoyment 1487 non-null float64
8 external pressure 1487 non-null float64
9 howami_participation_ended 11 non-null object
10 howami_participation_started 101 non-null object
11 importance 1487 non-null float64
12 interest 1487 non-null float64
13 internal pressure 1487 non-null float64
14 productivity_work 1487 non-null float64
15 relatedness 1487 non-null float64
16 satisfaction_work 1487 non-null float64
17 strategies 1301 non-null object
18 time_pressure 1126 non-null float64
19 vigor 1487 non-null float64
dtypes: float64(16), object(4)
memory usage: 274.8+ KB
df.columns
Index(['DailySituation', 'absorption', 'amotivation', 'autonomy', 'competence',
'dedication', 'emotional_drain', 'enjoyment', 'external pressure',
'howami_participation_ended', 'howami_participation_started',
'importance', 'interest', 'internal pressure', 'productivity_work',
'relatedness', 'satisfaction_work', 'strategies', 'time_pressure',
'vigor'],
dtype='object', name='Field')
date_range = pd.date_range(df.reset_index().date.min(),
df.reset_index().date.max())
users_range = np.repeat(df.reset_index().User.unique(), repeats=len(date_range))
dates_repeat_range = np.tile(date_range, reps=df.reset_index().User.nunique())
users_dates_df = (pd.DataFrame({"User": users_range,
"date": dates_repeat_range})
.assign(day = lambda x: x["date"].dt.strftime("%A"))
)
df = (users_dates_df
.loc[users_dates_df["day"].isin(['Monday', 'Tuesday', 'Wednesday', 'Thursday', 'Friday']), :]
.merge(df, right_index=True, left_on=["User", "date"], how="left")
.set_index(["User", "date"])
)
_ = df.reset_index().to_csv("data/moti_feasibility_james_daily_pivot.csv")
The aim of this is to assess whether there is enough data to do some of the more complext NLTSA methods
row_amount_df = (df
.reset_index()
.groupby("User")
.count()
.loc[:, ["absorption"]]
.rename(columns={"absorption": "row_amount"})
.sort_values(by="row_amount")
.reset_index())
row_amount_df.tail(1)
| User | row_amount | |
|---|---|---|
| 101 | Moti151 | 51 |
_ = plt.figure(figsize=(20, 4))
_ = sns.barplot(data=row_amount_df, x="User", y="row_amount")
_ = plt.xticks(rotation=90)
_ = plt.axhline(30, c="red", ls="--", label="30 day cutoff")
_ = plt.title("Row amounts per user")
_ = plt.legend()
The aim of this is to assess the amount of missingness time wise (i.e. the user could have a lot of data, spread over a long period with large gaps in the middle).
row_amount_df.tail(15)
| User | row_amount | |
|---|---|---|
| 87 | Moti114 | 30 |
| 88 | Moti147 | 33 |
| 89 | Moti149 | 33 |
| 90 | Moti164 | 36 |
| 91 | Moti106 | 38 |
| 92 | Moti121 | 38 |
| 93 | Moti137 | 39 |
| 94 | Moti138 | 39 |
| 95 | Moti150 | 41 |
| 96 | Moti157 | 43 |
| 97 | Moti148 | 44 |
| 98 | Moti143 | 44 |
| 99 | Moti140 | 47 |
| 100 | Moti156 | 49 |
| 101 | Moti151 | 51 |
top_users = row_amount_df.tail(15).User.tolist()
top_users
['Moti114', 'Moti147', 'Moti149', 'Moti164', 'Moti106', 'Moti121', 'Moti137', 'Moti138', 'Moti150', 'Moti157', 'Moti148', 'Moti143', 'Moti140', 'Moti156', 'Moti151']
scale_data = True
for user in top_users:
plot_df = df.select_dtypes(np.number).loc[(user, ), :]
# plot_df.index.min(), plot_df.index.max()
# new_index = pd.date_range(start=plot_df.index.min(), end=plot_df.index.max())
# plot_df = plot_df.reindex(new_index)
if scale_data:
plot_df = pd.DataFrame(MinMaxScaler().fit_transform(plot_df), index=plot_df.index, columns=plot_df.columns)
plot_df.index = plot_df.index.astype(str)
_ = plt.figure(figsize=(20, 5))
_ = sns.lineplot(data = plot_df.reset_index().melt(id_vars="date"), x="date", y="value", hue="variable")
_ = sns.scatterplot(data = plot_df.reset_index().melt(id_vars="date"), x="date", y="value", hue="variable", legend=False)
_ = plt.xticks(rotation=90)
_ = plt.title(f"Lineplot of raw values, user == {user}")
_ = plt.figure(figsize=(20, 5))
_ = sns.heatmap(plot_df.T)
_ = plt.title(f"Heatmap of raw values, user == {user}")
<ipython-input-18-a743883b9c93>:13: RuntimeWarning: More than 20 figures have been opened. Figures created through the pyplot interface (`matplotlib.pyplot.figure`) are retained until explicitly closed and may consume too much memory. (To control this warning, see the rcParam `figure.max_open_warning`). _ = plt.figure(figsize=(20, 5))
_ = plt.figure(figsize=(20, 5))
_ = sns.heatmap(plot_df.isna().astype(int).T)
_ = plt.title(f"Heatmap of missing values, user == {user}")
user = top_users[1]
user
'Moti147'
plot_df = df.select_dtypes(np.number).loc[(user, ), :]
# plot_df.index.min(), plot_df.index.max()
new_index = pd.date_range(start=plot_df.index.min(), end=plot_df.index.max())
plot_df = plot_df.reindex(new_index)
# plot_df.index = plot_df.index.astype(str)
feature_selection = plot_df.columns.tolist()
current_feature = "absorption"
# temp = plot_df.isna().astype(int)
temp = plot_df
x = temp[current_feature]
_ = plt.figure(figsize=(20, 5))
_ = plt.plot(x, label=current_feature)
_ = plt.scatter(x.index, x.values, c="green")
_ = plt.xticks(rotation=90)
_ = plt.legend()
impute_estimator = ExtraTreesRegressor(n_estimators=2, random_state=0)
estimator = make_pipeline(
IterativeImputer(random_state=0, estimator=impute_estimator)
)
ts_imp_df = pd.DataFrame(estimator.fit_transform(temp.loc[:, feature_selection].values),
index = temp.index,
columns = feature_selection)
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
ts_imp_df = temp.loc[:, feature_selection].apply(lambda x: x.interpolate(method='polynomial', order=5))
_ = plt.figure(figsize=(20, 5))
_ = plt.plot(ts_imp_df[current_feature], label=current_feature)
_ = plt.scatter(ts_imp_df[current_feature].index, ts_imp_df[current_feature].values, c="red")
_ = plt.scatter(x.index, x.values, c="green")
_ = plt.xticks(rotation=90)
_ = plt.legend()
cmap_name = "neurocast_colours"
colors = [
"#FFFFFF",
"#2768c1"
]
n_bin = 10
cm = LinearSegmentedColormap.from_list(cmap_name, colors, N=n_bin)
x = ts_imp_df[current_feature]
example_rm = RecurrencePlot(time_series=x.values, metric="manhattan", dim=3, tau=2, recurrence_rate=0.05).recurrence_matrix()
# rotate the matrix 90 degrees so it starts at day 0 in the bottom left corner
new_matrix = [[example_rm[j][i] for j in range(len(example_rm))] for i in range(len(example_rm[0])-1,-1,-1)]
fig, axes = plt.subplots(2, 1, figsize=(7, 9), gridspec_kw={'height_ratios': [1, 3]})
sns.despine(left=False)
axe = sns.lineplot(x=np.arange(0, ts_imp_df.loc[:, current_feature].shape[0]),
y=ts_imp_df.loc[:, current_feature],
color = "#2768c1",
ax=axes[0])
# axe.set_ylabel(None)
# axe.set_ylabel(r"$\tilde{u}\left(HT_{n}\right)$")
# axe.set_ylabel(r'$\tilde{u}\left(HT_{n}\right) \, \, \left[\mathrm{ms} \right]$')
axe.set_title(f"Recurrence Plot User == {user}")
yticks = list(np.linspace(0, len(x)-3, 20, dtype=np.int))
_xticklabels = list(np.arange(0, len(x), 1))
xticklabels = [_xticklabels[idx] for idx in yticks]
_yticklabels = list(np.arange(len(x)-3, -1, -1))
yticklabels = [_yticklabels[idx] for idx in yticks]
ax = sns.heatmap(new_matrix, cmap=cm, ax=axes[1], cbar=False, xticklabels = xticklabels, yticklabels = yticklabels
)
_ = ax.set_xticks(yticks)
_ = ax.set_yticks(yticks)
Calculating recurrence plot at fixed recurrence rate... Calculating the manhattan distance matrix...
np.array(top_users)
array(['Moti114', 'Moti147', 'Moti149', 'Moti164', 'Moti106', 'Moti121',
'Moti137', 'Moti138', 'Moti150', 'Moti157', 'Moti148', 'Moti143',
'Moti140', 'Moti156', 'Moti151'], dtype='<U7')
np.array(feature_selection)
array(['absorption', 'amotivation', 'autonomy', 'competence',
'dedication', 'emotional_drain', 'enjoyment', 'external pressure',
'importance', 'interest', 'internal pressure', 'productivity_work',
'relatedness', 'satisfaction_work', 'time_pressure', 'vigor'],
dtype='<U17')
and calculate the fluctuation intensity, distribution uniformity, complexity resonance and cumulative complexity peaks data frames from the scaled imputed data
imp_feature = "absorption"
window_size = 5
item_level_significance = 0.001
time_level_significance = 0.001
complexity_resonance_dfs_dict = dict()
cumulative_complexity_peaks_dfs_dict = dict()
significant_peaks_dfs_dict = dict()
for user in top_users:
print(user)
impute_estimator = ExtraTreesRegressor(n_estimators=10, random_state=0)
estimator = make_pipeline(
IterativeImputer(random_state=0, estimator=impute_estimator)
)
tmp = df.select_dtypes(np.number).loc[(user, ), :]
ts_df = tmp.loc[:, feature_selection]
ts_imp_df = pd.DataFrame(estimator.fit_transform(ts_df.loc[:, feature_selection].values),
index = ts_df.index,
columns = feature_selection)
# ts_imp_df = ts_df.loc[:, feature_selection].apply(lambda x: x.interpolate(method='linear', order=5))
ts_scal_df = pd.DataFrame(MinMaxScaler().fit_transform(ts_imp_df.loc[:, feature_selection].values),
index = ts_imp_df.index,
columns = feature_selection)
_ = plt.figure(figsize=(20, 5))
_ = plt.plot(ts_imp_df[imp_feature], label=imp_feature)
_ = plt.scatter(ts_imp_df[imp_feature].index, ts_imp_df[imp_feature].values, c="red")
_ = plt.scatter(ts_df[imp_feature].index, ts_df[imp_feature].values, c="green")
_ = plt.xticks(rotation=90)
_ = plt.legend()
fluctuation_intensity_df = fluctuation_intensity(df=ts_scal_df,
win=window_size,
xmin=0,
xmax=1,
col_first=1,
col_last=ts_scal_df.shape[1])
distribution_uniformity_df = distribution_uniformity(df=ts_scal_df,
win=window_size,
xmin=0,
xmax=1,
col_first=1,
col_last=ts_scal_df.shape[1])
complexity_resonance_df = complexity_resonance(fluctuation_intensity_df, distribution_uniformity_df)
cumulative_complexity_peaks_df, significant_peaks_df = cumulative_complexity_peaks(df=complexity_resonance_df,
significant_level_item = item_level_significance,
significant_level_time = time_level_significance,)
complexity_resonance_dfs_dict[user] = complexity_resonance_df
cumulative_complexity_peaks_dfs_dict[user] = cumulative_complexity_peaks_df
significant_peaks_dfs_dict[user] = significant_peaks_df
Moti114
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti147
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti149
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti164
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti106
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti121
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti137
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti138
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti150
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti157
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti148
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti143
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti140
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti156
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
Moti151
/opt/miniconda3/envs/general/lib/python3.9/site-packages/sklearn/impute/_iterative.py:685: ConvergenceWarning: [IterativeImputer] Early stopping criterion not reached.
warnings.warn("[IterativeImputer] Early stopping criterion not"
for user in top_users:
_ = complexity_resonance_diagram(df=complexity_resonance_dfs_dict[user],
cmap_n = 12,
plot_title=f'Complexity Resonance Diagram, user == {user}',
labels_n=10,
figsize=(20, 7),)
_ = cumulative_complexity_peaks_plot(cumulative_complexity_peaks_df=cumulative_complexity_peaks_dfs_dict[user],
significant_peaks_df=significant_peaks_dfs_dict[user],
plot_title = f'Cumulative Complexity Peaks Plot, user == {user}',
figsize = (20, 5),
height_ratios = [1, 3],
labels_n = 10)
/opt/miniconda3/envs/general/lib/python3.9/site-packages/jmspack/NLTSA.py:457: RuntimeWarning: More than 20 figures have been opened. Figures created through the pyplot interface (`matplotlib.pyplot.figure`) are retained until explicitly closed and may consume too much memory. (To control this warning, see the rcParam `figure.max_open_warning`). fig, ax = plt.subplots(figsize=figsize)
if len(top_users) > 6:
fig, axs = plt.subplots(figsize=(15,40), nrows=len(top_users), ncols=1)
fig.subplots_adjust(hspace = 0.35, wspace=0.15)
_ = plt.suptitle("The sum of complexity resonance matched\nwith significant cumulative complexity peaks per user", y=0.895)
else:
fig, axs = plt.subplots(figsize=(15,18), nrows=len(top_users), ncols=1)
fig.subplots_adjust(hspace = 0.35, wspace=0.15)
_ = plt.suptitle("The sum of complexity resonance matched\nwith significant cumulative complexity peaks per user", y=0.915)
for i in range(0, len(top_users)):
user = top_users[i]
# print(user)
# _ = plt.figure(figsize=(10, 4))
_ = axs[i].plot(complexity_resonance_dfs_dict[user].sum(axis=1),
label="Complexity Resonance Sum")
_ = axs[i].scatter(x = complexity_resonance_dfs_dict[user].index,
y = complexity_resonance_dfs_dict[user].sum(axis=1),
s=20,
c = "grey",
)
# _ = plt.title(f"Measure Change (Scaled), user == {user}")
_ = axs[i].set_ylabel(f"Measure Change (Scaled)\nuser == {user}")
for sig_ccp in significant_peaks_dfs_dict[user][significant_peaks_dfs_dict[user]["Significant CCPs"] > 0].index.tolist():
if sig_ccp == significant_peaks_dfs_dict[user][significant_peaks_dfs_dict[user]["Significant CCPs"] > 0].index.tolist()[0]:
_ = axs[i].axvline(sig_ccp, c="orange", ls = "--", label="Significant CCPs")
else:
_ = axs[i].axvline(sig_ccp, c="orange", ls = "--")
_ = axs[i].legend()
# _ = plt.savefig(f"images/sum_of_complexity_per_user.png", dpi=400, format="png", bbox_inches='tight')