{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"# Inline graphing\n",
"%matplotlib inline"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"# Import necessary packages\n",
"import os\n",
"import pandas as pd\n",
"import numpy as np\n",
"import seaborn as sns\n",
"import matplotlib.pyplot as plt\n",
"from matplotlib_venn import venn3, venn2\n",
"from scipy import stats\n",
"\n",
"pd.set_option('mode.chained_assignment', None)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [],
"source": [
"# Set working directory to dNef 1 folder\n",
"os.chdir('Path/To/Data')"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [],
"source": [
"# Reading in full data for first dNef experiment\n",
"ua1 = pd.read_csv('UnA/20190612_UnA_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"ua1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"ua1.columns = ['Gene_ua1', 'ProteinInfo_ua1', 'Compartment_ua1', 'Probability_ua1']\n",
"\n",
"ub1 = pd.read_csv('UnB/20190612_UnB_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"ub1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"ub1.columns = ['Gene_ub1', 'ProteinInfo_ub1', 'Compartment_ub1', 'Probability_ub1']\n",
"\n",
"uc1 = pd.read_csv('UnC/20190612_UnC_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"uc1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"uc1.columns = ['Gene_uc1', 'ProteinInfo_uc1', 'Compartment_uc1', 'Probability_uc1']\n",
"\n",
"ia1 = pd.read_csv('IndA/20190612_IndA_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"ia1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"ia1.columns = ['Gene_ia1', 'ProteinInfo_ia1', 'Compartment_ia1', 'Probability_ia1']\n",
"\n",
"ib1 = pd.read_csv('IndB/20190612_IndB_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"ib1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"ib1.columns = ['Gene_ib1', 'ProteinInfo_ib1', 'Compartment_ib1', 'Probability_ib1']\n",
"\n",
"ic1 = pd.read_csv('IndC/20190612_IndC_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"ic1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"ic1.columns = ['Gene_ic1', 'ProteinInfo_ic1', 'Compartment_ic1', 'Probability_ic1']"
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [],
"source": [
"# Set working directory to dNef 2 folder\n",
"os.chdir('Path/To/Data')"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"# Reading in full data for second dNef experiment\n",
"ua2 = pd.read_csv('UnA/20191122_UnA_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"ua2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"ua2.columns = ['Gene_ua2', 'ProteinInfo_ua2', 'Compartment_ua2', 'Probability_ua2']\n",
"\n",
"ub2 = pd.read_csv('UnB/20191122_UnB_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"ub2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"ub2.columns = ['Gene_ub2', 'ProteinInfo_ub2', 'Compartment_ub2', 'Probability_ub2']\n",
"\n",
"uc2 = pd.read_csv('UnC/20191122_UnC_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"uc2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"uc2.columns = ['Gene_uc2', 'ProteinInfo_uc2', 'Compartment_uc2', 'Probability_uc2']\n",
"\n",
"ia2 = pd.read_csv('IndA/20191122_IndA_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"ia2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"ia2.columns = ['Gene_ia2', 'ProteinInfo_ia2', 'Compartment_ia2', 'Probability_ia2']\n",
"\n",
"ib2 = pd.read_csv('IndB/20191122_IndB_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"ib2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"ib2.columns = ['Gene_ib2', 'ProteinInfo_ib2', 'Compartment_ib2', 'Probability_ib2']\n",
"\n",
"ic2 = pd.read_csv('IndC/20191122_IndC_SVCidentified_postiter_threshold.csv', index_col=0)\n",
"ic2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
"ic2.columns = ['Gene_ic2', 'ProteinInfo_ic2', 'Compartment_ic2', 'Probability_ic2']"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"# Set working directory to folder comparing biological replicates\n",
"os.chdir('C:/Users/ooma1/OneDrive - UC San Diego/Guatelli Lab/Cell Fractionation/20191122 Jurkat TREHIV dNef fract MS/ComparingBiolRepl/')"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"# Defining commonly detected protein dataframes for uninduced\n",
"unlist = [ua1, ub1, uc1, ua2, ub2, uc2]\n",
"\n",
"un = pd.concat(unlist, axis=1, join='outer', sort='True')\n",
"\n",
"for item in un.index:\n",
" if un.at[item, 'ProteinInfo_ua1'] is np.nan:\n",
" if un.at[item, 'ProteinInfo_ub1'] is not np.nan:\n",
" un.at[item, 'ProteinInfo_ua1'] = un.at[item, 'ProteinInfo_ub1']\n",
" elif un.at[item, 'ProteinInfo_uc1'] is not np.nan:\n",
" un.at[item, 'ProteinInfo_ua1'] = un.at[item, 'ProteinInfo_uc1']\n",
" elif un.at[item, 'ProteinInfo_ua2'] is not np.nan:\n",
" un.at[item, 'ProteinInfo_ua1'] = un.at[item, 'ProteinInfo_ua2']\n",
" elif un.at[item, 'ProteinInfo_ub2'] is not np.nan:\n",
" un.at[item, 'ProteinInfo_ua1'] = un.at[item, 'ProteinInfo_ub2']\n",
" elif un.at[item, 'ProteinInfo_uc2'] is not np.nan:\n",
" un.at[item, 'ProteinInfo_ua1'] = un.at[item, 'ProteinInfo_uc2']\n",
" \n",
"for item in un.index:\n",
" if un.at[item, 'Gene_ua1'] is np.nan:\n",
" if un.at[item, 'Gene_ub1'] is not np.nan:\n",
" un.at[item, 'Gene_ua1'] = un.at[item, 'Gene_ub1']\n",
" elif un.at[item, 'Gene_uc1'] is not np.nan:\n",
" un.at[item, 'Gene_ua1'] = un.at[item, 'Gene_uc1']\n",
" elif un.at[item, 'Gene_ua2'] is not np.nan:\n",
" un.at[item, 'Gene_ua1'] = un.at[item, 'Gene_ua2']\n",
" elif un.at[item, 'Gene_ub2'] is not np.nan:\n",
" un.at[item, 'Gene_ua1'] = un.at[item, 'Gene_ub2']\n",
" elif un.at[item, 'Gene_uc2'] is not np.nan:\n",
" un.at[item, 'Gene_ua1'] = un.at[item, 'Gene_uc2']\n",
"\n",
"un.drop(columns=['ProteinInfo_ub1', 'ProteinInfo_uc1', 'ProteinInfo_ua2', 'ProteinInfo_ub2', 'ProteinInfo_uc2', \n",
" 'Gene_ub1', 'Gene_uc1', 'Gene_ua2', 'Gene_ub2', 'Gene_uc2'], inplace=True)\n",
"un.rename({'Gene_ua1':'Gene', 'ProteinInfo_ua1':'ProteinInfo'}, axis='columns', inplace=True)\n",
"un.loc[:, 'Compartment_ua1':'Probability_uc2'] = un.loc[:, 'Compartment_ua1':'Probability_uc2'].fillna('ND')\n",
"ndset = ['ND']\n",
"un['NDcount'] = (un.isin(ndset).sum(1))/2"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [],
"source": [
"# Defining commonly detected protein dataframes for induced\n",
"indlist = [ia1, ib1, ic1, ia2, ib2, ic2]\n",
"\n",
"ind = pd.concat(indlist, axis=1, join='outer', sort='True')\n",
"\n",
"for item in ind.index:\n",
" if ind.at[item, 'ProteinInfo_ia1'] is np.nan:\n",
" if ind.at[item, 'ProteinInfo_ib1'] is not np.nan:\n",
" ind.at[item, 'ProteinInfo_ia1'] = ind.at[item, 'ProteinInfo_ib1']\n",
" elif ind.at[item, 'ProteinInfo_ic1'] is not np.nan:\n",
" ind.at[item, 'ProteinInfo_ia1'] = ind.at[item, 'ProteinInfo_ic1']\n",
" elif ind.at[item, 'ProteinInfo_ia2'] is not np.nan:\n",
" ind.at[item, 'ProteinInfo_ia1'] = ind.at[item, 'ProteinInfo_ia2']\n",
" elif ind.at[item, 'ProteinInfo_ib2'] is not np.nan:\n",
" ind.at[item, 'ProteinInfo_ia1'] = ind.at[item, 'ProteinInfo_ib2']\n",
" elif ind.at[item, 'ProteinInfo_ic2'] is not np.nan:\n",
" ind.at[item, 'ProteinInfo_ia1'] = ind.at[item, 'ProteinInfo_ic2']\n",
" \n",
"for item in ind.index:\n",
" if ind.at[item, 'Gene_ia1'] is np.nan:\n",
" if ind.at[item, 'Gene_ib1'] is not np.nan:\n",
" ind.at[item, 'Gene_ia1'] = ind.at[item, 'Gene_ib1']\n",
" elif ind.at[item, 'Gene_ic1'] is not np.nan:\n",
" ind.at[item, 'Gene_ia1'] = ind.at[item, 'Gene_ic1']\n",
" elif ind.at[item, 'Gene_ia2'] is not np.nan:\n",
" ind.at[item, 'Gene_ia1'] = ind.at[item, 'Gene_ia2']\n",
" elif ind.at[item, 'Gene_ib2'] is not np.nan:\n",
" ind.at[item, 'Gene_ia1'] = ind.at[item, 'Gene_ib2']\n",
" elif ind.at[item, 'Gene_ic2'] is not np.nan:\n",
" ind.at[item, 'Gene_ia1'] = ind.at[item, 'Gene_ic2']\n",
"\n",
"ind.drop(columns=['ProteinInfo_ib1', 'ProteinInfo_ic1', 'ProteinInfo_ia2', 'ProteinInfo_ib2', 'ProteinInfo_ic2', \n",
" 'Gene_ib1', 'Gene_ic1', 'Gene_ia2', 'Gene_ib2', 'Gene_ic2'], inplace=True)\n",
"ind.rename({'Gene_ia1':'Gene', 'ProteinInfo_ia1':'ProteinInfo'}, axis='columns', inplace=True)\n",
"ind.loc[:, 'Compartment_ia1':'Probability_ic2'] = ind.loc[:, 'Compartment_ia1':'Probability_ic2'].fillna('ND')\n",
"ndset = ['ND']\n",
"ind['NDcount'] = (ind.isin(ndset).sum(1))/2"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [],
"source": [
"# Creating dataframes with only those proteins classified in all 6 replicates for a given condition (excludes proteins that were markers for\n",
"# both dNef biological replicates)\n",
"un_common = un[un['NDcount'] == 0.0]\n",
"un_common = un_common[((un_common['Probability_ua1'] > 0) & (un_common['Probability_ua2'] > 0))]\n",
"\n",
"ind_common = ind[ind['NDcount'] == 0.0]\n",
"ind_common = ind_common[((ind_common['Probability_ia1'] > 0) & (ind_common['Probability_ia2'] > 0))]"
]
},
{
"cell_type": "code",
"execution_count": 11,
"metadata": {},
"outputs": [],
"source": [
"# Generating dataframes for finding row modes in compartment columns\n",
"un_temp = un_common.drop(columns=['Probability_ua1', 'Probability_ub1', 'Probability_uc1', \n",
" 'Probability_ua2', 'Probability_ub2', 'Probability_uc2'])\n",
"un_temp = un_temp.loc[:, 'Compartment_ua1':'Compartment_uc2']\n",
"\n",
"ind_temp = ind_common.drop(columns=['Probability_ia1', 'Probability_ib1', 'Probability_ic1', \n",
" 'Probability_ia2', 'Probability_ib2', 'Probability_ic2'])\n",
"ind_temp = ind_temp.loc[:, 'Compartment_ia1':'Compartment_ic2']\n",
"\n",
"# Taking row mode and adding this data back onto the dataframes with proteins classified in all 6 replicates for a given condition\n",
"a = un_temp.mode(axis=1).iloc[:, 0:3]\n",
"a.columns = ['Mode1', 'Mode2', 'Mode3']\n",
"un_common = pd.concat([un_common, a], axis=1, join='inner')\n",
"\n",
"b = ind_temp.mode(axis=1).iloc[:, 0:3]\n",
"b.columns = ['Mode1', 'Mode2', 'Mode3']\n",
"ind_common = pd.concat([ind_common, b], axis=1, join='inner')"
]
},
{
"cell_type": "code",
"execution_count": 12,
"metadata": {},
"outputs": [],
"source": [
"# Counting instances of row mode for each protein\n",
"x, y = stats.mode(un_temp, axis=1)\n",
"un_common['ModeCount'] = y\n",
"\n",
"x, y = stats.mode(ind_temp, axis=1)\n",
"ind_common['ModeCount'] = y"
]
},
{
"cell_type": "code",
"execution_count": 13,
"metadata": {},
"outputs": [],
"source": [
"# Generating dataframes for each condition with at least 4, 5, or 6 replicates classified the same\n",
"un6 = un_common[un_common.ModeCount >= 6]\n",
"un6.drop(columns=['NDcount', 'Mode2', 'Mode3'], inplace=True)\n",
"un6.rename({'Mode1':'Mode_un'}, inplace=True, axis='columns')\n",
"\n",
"un5 = un_common[un_common.ModeCount >= 5]\n",
"un5.drop(columns=['NDcount', 'Mode2', 'Mode3'], inplace=True)\n",
"un5.rename({'Mode1':'Mode_un'}, inplace=True, axis='columns')\n",
"\n",
"un4 = un_common[un_common.ModeCount >= 4]\n",
"un4.drop(columns=['NDcount', 'Mode2', 'Mode3'], inplace=True)\n",
"un4.rename({'Mode1':'Mode_un'}, inplace=True, axis='columns')\n",
"\n",
"ind6 = ind_common[ind_common.ModeCount >= 6]\n",
"ind6.drop(columns=['NDcount', 'Mode2', 'Mode3', 'Gene', 'ProteinInfo'], inplace=True)\n",
"ind6.rename({'Mode1':'Mode_ind'}, inplace=True, axis='columns')\n",
"\n",
"ind5 = ind_common[ind_common.ModeCount >= 5]\n",
"ind5.drop(columns=['NDcount', 'Mode2', 'Mode3', 'Gene', 'ProteinInfo'], inplace=True)\n",
"ind5.rename({'Mode1':'Mode_ind'}, inplace=True, axis='columns')\n",
"\n",
"ind4 = ind_common[ind_common.ModeCount >= 4]\n",
"ind4.drop(columns=['NDcount', 'Mode2', 'Mode3', 'Gene', 'ProteinInfo'], inplace=True)\n",
"ind4.rename({'Mode1':'Mode_ind'}, inplace=True, axis='columns')"
]
},
{
"cell_type": "code",
"execution_count": 14,
"metadata": {},
"outputs": [],
"source": [
"# Concatenating uninduced and induced for each cutoff\n",
"common6 = pd.concat([un6, ind6], join='inner', axis=1, sort=True)\n",
"common5 = pd.concat([un5, ind5], join='inner', axis=1, sort=True)\n",
"common4 = pd.concat([un4, ind4], join='inner', axis=1, sort=True)"
]
},
{
"cell_type": "code",
"execution_count": 15,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Gene</th>\n",
" <th>ProteinInfo</th>\n",
" <th>Compartment_ua1</th>\n",
" <th>Probability_ua1</th>\n",
" <th>Compartment_ub1</th>\n",
" <th>Probability_ub1</th>\n",
" <th>Compartment_uc1</th>\n",
" <th>Probability_uc1</th>\n",
" <th>Compartment_ua2</th>\n",
" <th>Probability_ua2</th>\n",
" <th>...</th>\n",
" <th>Compartment_ic1</th>\n",
" <th>Probability_ic1</th>\n",
" <th>Compartment_ia2</th>\n",
" <th>Probability_ia2</th>\n",
" <th>Compartment_ib2</th>\n",
" <th>Probability_ib2</th>\n",
" <th>Compartment_ic2</th>\n",
" <th>Probability_ic2</th>\n",
" <th>Mode_ind</th>\n",
" <th>ModeCount</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" </tbody>\n",
"</table>\n",
"<p>0 rows × 30 columns</p>\n",
"</div>"
],
"text/plain": [
"Empty DataFrame\n",
"Columns: [Gene, ProteinInfo, Compartment_ua1, Probability_ua1, Compartment_ub1, Probability_ub1, Compartment_uc1, Probability_uc1, Compartment_ua2, Probability_ua2, Compartment_ub2, Probability_ub2, Compartment_uc2, Probability_uc2, Mode_un, ModeCount, Compartment_ia1, Probability_ia1, Compartment_ib1, Probability_ib1, Compartment_ic1, Probability_ic1, Compartment_ia2, Probability_ia2, Compartment_ib2, Probability_ib2, Compartment_ic2, Probability_ic2, Mode_ind, ModeCount]\n",
"Index: []\n",
"\n",
"[0 rows x 30 columns]"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Gene</th>\n",
" <th>ProteinInfo</th>\n",
" <th>Compartment_ua1</th>\n",
" <th>Probability_ua1</th>\n",
" <th>Compartment_ub1</th>\n",
" <th>Probability_ub1</th>\n",
" <th>Compartment_uc1</th>\n",
" <th>Probability_uc1</th>\n",
" <th>Compartment_ua2</th>\n",
" <th>Probability_ua2</th>\n",
" <th>...</th>\n",
" <th>Compartment_ic1</th>\n",
" <th>Probability_ic1</th>\n",
" <th>Compartment_ia2</th>\n",
" <th>Probability_ia2</th>\n",
" <th>Compartment_ib2</th>\n",
" <th>Probability_ib2</th>\n",
" <th>Compartment_ic2</th>\n",
" <th>Probability_ic2</th>\n",
" <th>Mode_ind</th>\n",
" <th>ModeCount</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <td>P27448</td>\n",
" <td>MARK3</td>\n",
" <td>MAP/microtubule affinity-regulating kinase 3 (...</td>\n",
" <td>Unclassified</td>\n",
" <td>0.548</td>\n",
" <td>Endosome</td>\n",
" <td>0.894</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>...</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>1</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.808</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.846</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.99</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>5</td>\n",
" </tr>\n",
" <tr>\n",
" <td>Q12959</td>\n",
" <td>DLG1</td>\n",
" <td>Disks large homolog 1 (Synapse-associated prot...</td>\n",
" <td>Peroxisome</td>\n",
" <td>0.838</td>\n",
" <td>Peroxisome</td>\n",
" <td>0.822</td>\n",
" <td>Unclassified</td>\n",
" <td>0.468</td>\n",
" <td>Peroxisome</td>\n",
" <td>0.832</td>\n",
" <td>...</td>\n",
" <td>Plasma membrane</td>\n",
" <td>0.734</td>\n",
" <td>Plasma membrane</td>\n",
" <td>0.902</td>\n",
" <td>Plasma membrane</td>\n",
" <td>0.948</td>\n",
" <td>Plasma membrane</td>\n",
" <td>0.804</td>\n",
" <td>Plasma membrane</td>\n",
" <td>5</td>\n",
" </tr>\n",
" <tr>\n",
" <td>Q9P2E9</td>\n",
" <td>RRBP1</td>\n",
" <td>Ribosome-binding protein 1 (180 kDa ribosome r...</td>\n",
" <td>Unclassified</td>\n",
" <td>0.446</td>\n",
" <td>ER</td>\n",
" <td>0.942</td>\n",
" <td>Unclassified</td>\n",
" <td>0.332</td>\n",
" <td>Unclassified</td>\n",
" <td>0.668</td>\n",
" <td>...</td>\n",
" <td>ER</td>\n",
" <td>0.926</td>\n",
" <td>ER</td>\n",
" <td>0.992</td>\n",
" <td>ER</td>\n",
" <td>0.958</td>\n",
" <td>ER</td>\n",
" <td>0.994</td>\n",
" <td>ER</td>\n",
" <td>5</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"<p>3 rows × 30 columns</p>\n",
"</div>"
],
"text/plain": [
" Gene ProteinInfo \\\n",
"P27448 MARK3 MAP/microtubule affinity-regulating kinase 3 (... \n",
"Q12959 DLG1 Disks large homolog 1 (Synapse-associated prot... \n",
"Q9P2E9 RRBP1 Ribosome-binding protein 1 (180 kDa ribosome r... \n",
"\n",
" Compartment_ua1 Probability_ua1 Compartment_ub1 Probability_ub1 \\\n",
"P27448 Unclassified 0.548 Endosome 0.894 \n",
"Q12959 Peroxisome 0.838 Peroxisome 0.822 \n",
"Q9P2E9 Unclassified 0.446 ER 0.942 \n",
"\n",
" Compartment_uc1 Probability_uc1 Compartment_ua2 Probability_ua2 ... \\\n",
"P27448 Endosome 1 Endosome 1 ... \n",
"Q12959 Unclassified 0.468 Peroxisome 0.832 ... \n",
"Q9P2E9 Unclassified 0.332 Unclassified 0.668 ... \n",
"\n",
" Compartment_ic1 Probability_ic1 Compartment_ia2 \\\n",
"P27448 Large Protein Complex 1 Large Protein Complex \n",
"Q12959 Plasma membrane 0.734 Plasma membrane \n",
"Q9P2E9 ER 0.926 ER \n",
"\n",
" Probability_ia2 Compartment_ib2 Probability_ib2 \\\n",
"P27448 0.808 Large Protein Complex 0.846 \n",
"Q12959 0.902 Plasma membrane 0.948 \n",
"Q9P2E9 0.992 ER 0.958 \n",
"\n",
" Compartment_ic2 Probability_ic2 Mode_ind ModeCount \n",
"P27448 Large Protein Complex 0.99 Large Protein Complex 5 \n",
"Q12959 Plasma membrane 0.804 Plasma membrane 5 \n",
"Q9P2E9 ER 0.994 ER 5 \n",
"\n",
"[3 rows x 30 columns]"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Gene</th>\n",
" <th>ProteinInfo</th>\n",
" <th>Compartment_ua1</th>\n",
" <th>Probability_ua1</th>\n",
" <th>Compartment_ub1</th>\n",
" <th>Probability_ub1</th>\n",
" <th>Compartment_uc1</th>\n",
" <th>Probability_uc1</th>\n",
" <th>Compartment_ua2</th>\n",
" <th>Probability_ua2</th>\n",
" <th>...</th>\n",
" <th>Compartment_ic1</th>\n",
" <th>Probability_ic1</th>\n",
" <th>Compartment_ia2</th>\n",
" <th>Probability_ia2</th>\n",
" <th>Compartment_ib2</th>\n",
" <th>Probability_ib2</th>\n",
" <th>Compartment_ic2</th>\n",
" <th>Probability_ic2</th>\n",
" <th>Mode_ind</th>\n",
" <th>ModeCount</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <td>B0I1T2</td>\n",
" <td>MYO1G</td>\n",
" <td>Unconventional myosin-Ig [Cleaved into: Minor ...</td>\n",
" <td>Peroxisome</td>\n",
" <td>0.772</td>\n",
" <td>Unclassified</td>\n",
" <td>0.51</td>\n",
" <td>Peroxisome</td>\n",
" <td>0.722</td>\n",
" <td>Peroxisome</td>\n",
" <td>0.818</td>\n",
" <td>...</td>\n",
" <td>Unclassified</td>\n",
" <td>0.584</td>\n",
" <td>Unclassified</td>\n",
" <td>0.536</td>\n",
" <td>Unclassified</td>\n",
" <td>0.386</td>\n",
" <td>Plasma membrane</td>\n",
" <td>0.956</td>\n",
" <td>Unclassified</td>\n",
" <td>5</td>\n",
" </tr>\n",
" <tr>\n",
" <td>HIVNL43_GAG</td>\n",
" <td>gag</td>\n",
" <td>Gag</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>1</td>\n",
" <td>Mitochondrion</td>\n",
" <td>1</td>\n",
" <td>Unclassified</td>\n",
" <td>0.7</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>1</td>\n",
" <td>...</td>\n",
" <td>Unclassified</td>\n",
" <td>0.476</td>\n",
" <td>Golgi</td>\n",
" <td>0.78</td>\n",
" <td>Unclassified</td>\n",
" <td>0.508</td>\n",
" <td>Unclassified</td>\n",
" <td>0.576</td>\n",
" <td>Unclassified</td>\n",
" <td>4</td>\n",
" </tr>\n",
" <tr>\n",
" <td>O00560</td>\n",
" <td>SDCB1</td>\n",
" <td>Syntenin-1 (Melanoma differentiation-associate...</td>\n",
" <td>ER</td>\n",
" <td>1</td>\n",
" <td>Unclassified</td>\n",
" <td>0.498</td>\n",
" <td>Unclassified</td>\n",
" <td>0.5</td>\n",
" <td>Unclassified</td>\n",
" <td>0.494</td>\n",
" <td>...</td>\n",
" <td>Unclassified</td>\n",
" <td>0.626</td>\n",
" <td>Endosome</td>\n",
" <td>0.974</td>\n",
" <td>Unclassified</td>\n",
" <td>0.574</td>\n",
" <td>Endosome</td>\n",
" <td>0.814</td>\n",
" <td>Endosome</td>\n",
" <td>4</td>\n",
" </tr>\n",
" <tr>\n",
" <td>O00584</td>\n",
" <td>RNT2</td>\n",
" <td>Ribonuclease T2 (EC 3.1.27.-) (Ribonuclease 6)</td>\n",
" <td>Lysosome</td>\n",
" <td>0.806</td>\n",
" <td>Lysosome</td>\n",
" <td>1</td>\n",
" <td>Lysosome</td>\n",
" <td>0.88</td>\n",
" <td>Lysosome</td>\n",
" <td>1</td>\n",
" <td>...</td>\n",
" <td>Unclassified</td>\n",
" <td>0.57</td>\n",
" <td>Plasma membrane</td>\n",
" <td>0.842</td>\n",
" <td>Unclassified</td>\n",
" <td>0.6</td>\n",
" <td>Lysosome</td>\n",
" <td>0.946</td>\n",
" <td>Unclassified</td>\n",
" <td>4</td>\n",
" </tr>\n",
" <tr>\n",
" <td>O14972</td>\n",
" <td>DSCR3</td>\n",
" <td>Vacuolar protein sorting-associated protein 26...</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>Unclassified</td>\n",
" <td>0.514</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>...</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.854</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.838</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.994</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>4</td>\n",
" </tr>\n",
" <tr>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" <td>...</td>\n",
" </tr>\n",
" <tr>\n",
" <td>Q9UKK6</td>\n",
" <td>NXT1</td>\n",
" <td>NTF2-related export protein 1 (Protein p15)</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.796</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>...</td>\n",
" <td>Unclassified</td>\n",
" <td>0.676</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.916</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.982</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.998</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>4</td>\n",
" </tr>\n",
" <tr>\n",
" <td>Q9ULH0</td>\n",
" <td>KDIS</td>\n",
" <td>Kinase D-interacting substrate of 220 kDa (Ank...</td>\n",
" <td>Golgi</td>\n",
" <td>0.798</td>\n",
" <td>Unclassified</td>\n",
" <td>0.432</td>\n",
" <td>Unclassified</td>\n",
" <td>0.644</td>\n",
" <td>Golgi</td>\n",
" <td>0.714</td>\n",
" <td>...</td>\n",
" <td>Plasma membrane</td>\n",
" <td>0.812</td>\n",
" <td>Plasma membrane</td>\n",
" <td>0.872</td>\n",
" <td>Unclassified</td>\n",
" <td>0.632</td>\n",
" <td>Plasma membrane</td>\n",
" <td>1</td>\n",
" <td>Plasma membrane</td>\n",
" <td>4</td>\n",
" </tr>\n",
" <tr>\n",
" <td>Q9ULV4</td>\n",
" <td>COR1C</td>\n",
" <td>Coronin-1C (Coronin-3) (hCRNN4)</td>\n",
" <td>Golgi</td>\n",
" <td>0.778</td>\n",
" <td>Unclassified</td>\n",
" <td>0.67</td>\n",
" <td>Golgi</td>\n",
" <td>0.996</td>\n",
" <td>Golgi</td>\n",
" <td>0.85</td>\n",
" <td>...</td>\n",
" <td>Unclassified</td>\n",
" <td>0.508</td>\n",
" <td>Endosome</td>\n",
" <td>0.818</td>\n",
" <td>Unclassified</td>\n",
" <td>0.572</td>\n",
" <td>Endosome</td>\n",
" <td>0.95</td>\n",
" <td>Endosome</td>\n",
" <td>4</td>\n",
" </tr>\n",
" <tr>\n",
" <td>Q9Y376</td>\n",
" <td>CAB39</td>\n",
" <td>Calcium-binding protein 39 (MO25alpha) (Protei...</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>Endosome</td>\n",
" <td>0.72</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.854</td>\n",
" <td>...</td>\n",
" <td>Unclassified</td>\n",
" <td>0.556</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>Unclassified</td>\n",
" <td>0.654</td>\n",
" <td>Unclassified</td>\n",
" <td>0.542</td>\n",
" <td>Unclassified</td>\n",
" <td>4</td>\n",
" </tr>\n",
" <tr>\n",
" <td>Q9Y6W5</td>\n",
" <td>WASF2</td>\n",
" <td>Wiskott-Aldrich syndrome protein family member...</td>\n",
" <td>Endosome</td>\n",
" <td>0.992</td>\n",
" <td>Endosome</td>\n",
" <td>0.998</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.908</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>1</td>\n",
" <td>...</td>\n",
" <td>Endosome</td>\n",
" <td>0.834</td>\n",
" <td>Endosome</td>\n",
" <td>1</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.642</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>0.686</td>\n",
" <td>Large Protein Complex</td>\n",
" <td>4</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"<p>67 rows × 30 columns</p>\n",
"</div>"
],
"text/plain": [
" Gene ProteinInfo \\\n",
"B0I1T2 MYO1G Unconventional myosin-Ig [Cleaved into: Minor ... \n",
"HIVNL43_GAG gag Gag \n",
"O00560 SDCB1 Syntenin-1 (Melanoma differentiation-associate... \n",
"O00584 RNT2 Ribonuclease T2 (EC 3.1.27.-) (Ribonuclease 6) \n",
"O14972 DSCR3 Vacuolar protein sorting-associated protein 26... \n",
"... ... ... \n",
"Q9UKK6 NXT1 NTF2-related export protein 1 (Protein p15) \n",
"Q9ULH0 KDIS Kinase D-interacting substrate of 220 kDa (Ank... \n",
"Q9ULV4 COR1C Coronin-1C (Coronin-3) (hCRNN4) \n",
"Q9Y376 CAB39 Calcium-binding protein 39 (MO25alpha) (Protei... \n",
"Q9Y6W5 WASF2 Wiskott-Aldrich syndrome protein family member... \n",
"\n",
" Compartment_ua1 Probability_ua1 Compartment_ub1 \\\n",
"B0I1T2 Peroxisome 0.772 Unclassified \n",
"HIVNL43_GAG Large Protein Complex 1 Mitochondrion \n",
"O00560 ER 1 Unclassified \n",
"O00584 Lysosome 0.806 Lysosome \n",
"O14972 Endosome 1 Endosome \n",
"... ... ... ... \n",
"Q9UKK6 Endosome 1 Endosome \n",
"Q9ULH0 Golgi 0.798 Unclassified \n",
"Q9ULV4 Golgi 0.778 Unclassified \n",
"Q9Y376 Endosome 1 Endosome \n",
"Q9Y6W5 Endosome 0.992 Endosome \n",
"\n",
" Probability_ub1 Compartment_uc1 Probability_uc1 \\\n",
"B0I1T2 0.51 Peroxisome 0.722 \n",
"HIVNL43_GAG 1 Unclassified 0.7 \n",
"O00560 0.498 Unclassified 0.5 \n",
"O00584 1 Lysosome 0.88 \n",
"O14972 1 Unclassified 0.514 \n",
"... ... ... ... \n",
"Q9UKK6 1 Large Protein Complex 0.796 \n",
"Q9ULH0 0.432 Unclassified 0.644 \n",
"Q9ULV4 0.67 Golgi 0.996 \n",
"Q9Y376 1 Endosome 0.72 \n",
"Q9Y6W5 0.998 Large Protein Complex 0.908 \n",
"\n",
" Compartment_ua2 Probability_ua2 ... \\\n",
"B0I1T2 Peroxisome 0.818 ... \n",
"HIVNL43_GAG Large Protein Complex 1 ... \n",
"O00560 Unclassified 0.494 ... \n",
"O00584 Lysosome 1 ... \n",
"O14972 Endosome 1 ... \n",
"... ... ... ... \n",
"Q9UKK6 Endosome 1 ... \n",
"Q9ULH0 Golgi 0.714 ... \n",
"Q9ULV4 Golgi 0.85 ... \n",
"Q9Y376 Large Protein Complex 0.854 ... \n",
"Q9Y6W5 Large Protein Complex 1 ... \n",
"\n",
" Compartment_ic1 Probability_ic1 Compartment_ia2 \\\n",
"B0I1T2 Unclassified 0.584 Unclassified \n",
"HIVNL43_GAG Unclassified 0.476 Golgi \n",
"O00560 Unclassified 0.626 Endosome \n",
"O00584 Unclassified 0.57 Plasma membrane \n",
"O14972 Large Protein Complex 0.854 Endosome \n",
"... ... ... ... \n",
"Q9UKK6 Unclassified 0.676 Large Protein Complex \n",
"Q9ULH0 Plasma membrane 0.812 Plasma membrane \n",
"Q9ULV4 Unclassified 0.508 Endosome \n",
"Q9Y376 Unclassified 0.556 Endosome \n",
"Q9Y6W5 Endosome 0.834 Endosome \n",
"\n",
" Probability_ia2 Compartment_ib2 Probability_ib2 \\\n",
"B0I1T2 0.536 Unclassified 0.386 \n",
"HIVNL43_GAG 0.78 Unclassified 0.508 \n",
"O00560 0.974 Unclassified 0.574 \n",
"O00584 0.842 Unclassified 0.6 \n",
"O14972 1 Large Protein Complex 0.838 \n",
"... ... ... ... \n",
"Q9UKK6 0.916 Large Protein Complex 0.982 \n",
"Q9ULH0 0.872 Unclassified 0.632 \n",
"Q9ULV4 0.818 Unclassified 0.572 \n",
"Q9Y376 1 Unclassified 0.654 \n",
"Q9Y6W5 1 Large Protein Complex 0.642 \n",
"\n",
" Compartment_ic2 Probability_ic2 Mode_ind \\\n",
"B0I1T2 Plasma membrane 0.956 Unclassified \n",
"HIVNL43_GAG Unclassified 0.576 Unclassified \n",
"O00560 Endosome 0.814 Endosome \n",
"O00584 Lysosome 0.946 Unclassified \n",
"O14972 Large Protein Complex 0.994 Large Protein Complex \n",
"... ... ... ... \n",
"Q9UKK6 Large Protein Complex 0.998 Large Protein Complex \n",
"Q9ULH0 Plasma membrane 1 Plasma membrane \n",
"Q9ULV4 Endosome 0.95 Endosome \n",
"Q9Y376 Unclassified 0.542 Unclassified \n",
"Q9Y6W5 Large Protein Complex 0.686 Large Protein Complex \n",
"\n",
" ModeCount \n",
"B0I1T2 5 \n",
"HIVNL43_GAG 4 \n",
"O00560 4 \n",
"O00584 4 \n",
"O14972 4 \n",
"... ... \n",
"Q9UKK6 4 \n",
"Q9ULH0 4 \n",
"Q9ULV4 4 \n",
"Q9Y376 4 \n",
"Q9Y6W5 4 \n",
"\n",
"[67 rows x 30 columns]"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"# Subsetting out proteins that have moved\n",
"common6_moved = common6[common6['Mode_un'] != common6['Mode_ind']]\n",
"display(common6_moved)\n",
"common5_moved = common5[common5['Mode_un'] != common5['Mode_ind']]\n",
"display(common5_moved)\n",
"common4_moved = common4[common4['Mode_un'] != common4['Mode_ind']]\n",
"display(common4_moved)"
]
},
{
"cell_type": "code",
"execution_count": 16,
"metadata": {},
"outputs": [],
"source": [
"# Saving common5_moved and common4_moved\n",
"common5_moved.to_csv('20191123_RelocalizedProteins_5of6repl.csv', sep=',')\n",
"common5_ID = common5_moved.loc[:, 'Gene':'ProteinInfo']\n",
"common5_ID.reset_index(inplace=True)\n",
"common5_ID.drop(['Gene', 'ProteinInfo'], axis=1, inplace=True)\n",
"common5_ID.to_csv('20191123_RelocalizedProteins_5of6repl_IDs.csv', sep=',', index=False)\n",
"\n",
"common4_moved.to_csv('20191123_RelocalizedProteins_4of6repl.csv', sep=',')\n",
"common4_ID = common4_moved.loc[:, 'Gene':'ProteinInfo']\n",
"common4_ID.reset_index(inplace=True)\n",
"common4_ID.drop(['Gene', 'ProteinInfo'], axis=1, inplace=True)\n",
"common4_ID.to_csv('20191123_RelocalizedProteins_4of6repl_IDs.csv', sep=',', index=False)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### STRING Analyses for Moved Proteins\n",
"\n",
"[Proteins identical in 4+ replicates/condition](https://version-11-0.string-db.org/cgi/network.pl?networkId=ItKzBS0Uuj2q)<br>\n",
"[Proteins identical in 5+ replicates/condition](https://version-11-0.string-db.org/cgi/network.pl?networkId=qm0ZBoHpO6ED)<br>"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# Centroid Distance Analysis"
]
},
{
"cell_type": "code",
"execution_count": 17,
"metadata": {},
"outputs": [],
"source": [
"# Import packages\n",
"import os\n",
"import math\n",
"import pandas as pd\n",
"import numpy as np\n",
"import seaborn as sns\n",
"import matplotlib.pyplot as plt\n",
"from sklearn.covariance import MinCovDet\n",
"from scipy.stats import percentileofscore\n",
"from scipy.stats import rankdata\n",
"from scipy.stats import t\n",
"from scipy.spatial.distance import euclidean"
]
},
{
"cell_type": "code",
"execution_count": 18,
"metadata": {},
"outputs": [],
"source": [
"# Set working directory to dNef 1 folder\n",
"os.chdir('Path/To/Data')"
]
},
{
"cell_type": "code",
"execution_count": 19,
"metadata": {},
"outputs": [],
"source": [
"# Reading in full data for first dNef experiment\n",
"ua1 = pd.read_csv('UnA/20190612_UnA_RowSum.csv', index_col=0)\n",
"ub1 = pd.read_csv('UnB/20190612_UnB_RowSum.csv', index_col=0)\n",
"uc1 = pd.read_csv('UnC/20190612_UnC_RowSum.csv', index_col=0)\n",
"\n",
"ia1 = pd.read_csv('IndA/20190612_IndA_RowSum.csv', index_col=0)\n",
"ib1 = pd.read_csv('IndB/20190612_IndB_RowSum.csv', index_col=0)\n",
"ic1 = pd.read_csv('IndC/20190612_IndC_RowSum.csv', index_col=0)\n",
"\n",
"dNef1_moved = pd.read_csv('20190613_EuclideanDistance_FDR0.0125.csv', index_col=0)"
]
},
{
"cell_type": "code",
"execution_count": 20,
"metadata": {},
"outputs": [],
"source": [
"# Labeling replicate columns for ease of identification in joined dataframe\n",
"ua1.columns = ['Gene_ua1', 'Protein information_ua1', '3K_ua1', '5.4K_ua1', '12.2K_ua1', \n",
" '24K_ua1', '78.4K_ua1', '110K_ua1', '195.5K_ua1']\n",
"ub1.columns = ['Gene_ub1', 'Protein information_ub1', '3K_ub1', '5.4K_ub1', '12.2K_ub1', \n",
" '24K_ub1', '78.4K_ub1', '110K_ub1', '195.5K_ub1']\n",
"uc1.columns = ['Gene_uc1', 'Protein information_uc1', '3K_uc1', '5.4K_uc1', '12.2K_uc1', \n",
" '24K_uc1', '78.4K_uc1', '110K_uc1', '195.5K_uc1']\n",
"ia1.columns = ['Gene_ia1', 'Protein information_ia1', '3K_ia1', '5.4K_ia1', '12.2K_ia1', \n",
" '24K_ia1', '78.4K_ia1', '110K_ia1', '195.5K_ia1']\n",
"ib1.columns = ['Gene_ib1', 'Protein information_ib1', '3K_ib1', '5.4K_ib1', '12.2K_ib1', \n",
" '24K_ib1', '78.4K_ib1', '110K_ib1', '195.5K_ib1']\n",
"ic1.columns = ['Gene_ic1', 'Protein information_ic1', '3K_ic1', '5.4K_ic1', '12.2K_ic1', \n",
" '24K_ic1', '78.4K_ic1', '110K_ic1', '195.5K_ic1']"
]
},
{
"cell_type": "code",
"execution_count": 21,
"metadata": {},
"outputs": [],
"source": [
"# Set working directory to dNef 2 folder\n",
"os.chdir('Path/To/Data')"
]
},
{
"cell_type": "code",
"execution_count": 22,
"metadata": {},
"outputs": [],
"source": [
"# Reading in full data for second dNef experiment\n",
"ua2 = pd.read_csv('UnA/20191122_UnA_RowSum.csv', index_col=0)\n",
"ub2 = pd.read_csv('UnB/20191122_UnB_RowSum.csv', index_col=0)\n",
"uc2 = pd.read_csv('UnC/20191122_UnC_RowSum.csv', index_col=0)\n",
"\n",
"ia2 = pd.read_csv('IndA/20191122_IndA_RowSum.csv', index_col=0)\n",
"ib2 = pd.read_csv('IndB/20191122_IndB_RowSum.csv', index_col=0)\n",
"ic2 = pd.read_csv('IndC/20191122_IndC_RowSum.csv', index_col=0)\n",
"\n",
"dNef2_moved = pd.read_csv('20191122_EuclideanDistance_FDR0.025.csv', index_col=0)"
]
},
{
"cell_type": "code",
"execution_count": 23,
"metadata": {},
"outputs": [],
"source": [
"# Labeling replicate columns for ease of identification in joined dataframe\n",
"ua2.columns = ['Gene_ua2', 'Protein information_ua2', '3K_ua2', '5.4K_ua2', '12.2K_ua2', \n",
" '24K_ua2', '78.4K_ua2', '110K_ua2', '195.5K_ua2']\n",
"ub2.columns = ['Gene_ub2', 'Protein information_ub2', '3K_ub2', '5.4K_ub2', '12.2K_ub2', \n",
" '24K_ub2', '78.4K_ub2', '110K_ub2', '195.5K_ub2']\n",
"uc2.columns = ['Gene_uc2', 'Protein information_uc2', '3K_uc2', '5.4K_uc2', '12.2K_uc2', \n",
" '24K_uc2', '78.4K_uc2', '110K_uc2', '195.5K_uc2']\n",
"ia2.columns = ['Gene_ia2', 'Protein information_ia2', '3K_ia2', '5.4K_ia2', '12.2K_ia2', \n",
" '24K_ia2', '78.4K_ia2', '110K_ia2', '195.5K_ia2']\n",
"ib2.columns = ['Gene_ib2', 'Protein information_ib2', '3K_ib2', '5.4K_ib2', '12.2K_ib2', \n",
" '24K_ib2', '78.4K_ib2', '110K_ib2', '195.5K_ib2']\n",
"ic2.columns = ['Gene_ic2', 'Protein information_ic2', '3K_ic2', '5.4K_ic2', '12.2K_ic2', \n",
" '24K_ic2', '78.4K_ic2', '110K_ic2', '195.5K_ic2']"
]
},
{
"cell_type": "code",
"execution_count": 24,
"metadata": {},
"outputs": [],
"source": [
"# Set working directory to folder comparing biological replicates\n",
"os.chdir('C:/Users/ooma1/OneDrive - UC San Diego/Guatelli Lab/Cell Fractionation/20191122 Jurkat TREHIV dNef fract MS/ComparingBiolRepl/')"
]
},
{
"cell_type": "code",
"execution_count": 25,
"metadata": {},
"outputs": [],
"source": [
"# Pull out proteins common across all six replicates\n",
"dflist = [ua1, ub1, uc1, ia1, ib1, ic1, ua2, ub2, uc2, ia2, ib2, ic2]\n",
"common = pd.concat(dflist, axis=1, join='inner')"
]
},
{
"cell_type": "code",
"execution_count": 26,
"metadata": {},
"outputs": [],
"source": [
"# Redefine replicates with only common proteins\n",
"ua1 = common.loc[:, 'Gene_ua1':'195.5K_ua1']\n",
"ub1 = common.loc[:, 'Gene_ub1':'195.5K_ub1']\n",
"uc1 = common.loc[:, 'Gene_uc1':'195.5K_uc1']\n",
"ia1 = common.loc[:, 'Gene_ia1':'195.5K_ia1']\n",
"ib1 = common.loc[:, 'Gene_ib1':'195.5K_ib1']\n",
"ic1 = common.loc[:, 'Gene_ic1':'195.5K_ic1']\n",
"\n",
"ua2 = common.loc[:, 'Gene_ua2':'195.5K_ua2']\n",
"ub2 = common.loc[:, 'Gene_ub2':'195.5K_ub2']\n",
"uc2 = common.loc[:, 'Gene_uc2':'195.5K_uc2']\n",
"ia2 = common.loc[:, 'Gene_ia2':'195.5K_ia2']\n",
"ib2 = common.loc[:, 'Gene_ib2':'195.5K_ib2']\n",
"ic2 = common.loc[:, 'Gene_ic2':'195.5K_ic2']"
]
},
{
"cell_type": "code",
"execution_count": 27,
"metadata": {},
"outputs": [],
"source": [
"# Changing columns back to replicate invariant names\n",
"ua1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"ub1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"uc1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"ia1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"ib1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"ic1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"\n",
"ua2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"ub2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"uc2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"ia2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"ib2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"ic2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']"
]
},
{
"cell_type": "code",
"execution_count": 28,
"metadata": {},
"outputs": [],
"source": [
"# Calculate centroid arrays for uninduced and induced\n",
"un_cent = np.zeros((len(common.index), 7))\n",
"ind_cent = np.zeros((len(common.index), 7))\n",
"\n",
"fractlist = ['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
"\n",
"for protein, row in zip(common.index, range(0, len(common.index))):\n",
" for fraction, col in zip(fractlist, range(0, 7)):\n",
" tempa1 = ua1.at[protein, fraction]\n",
" tempb1 = ub1.at[protein, fraction]\n",
" tempc1 = uc1.at[protein, fraction]\n",
" tempa2 = ua1.at[protein, fraction]\n",
" tempb2 = ub1.at[protein, fraction]\n",
" tempc2 = uc1.at[protein, fraction]\n",
" un_cent[row, col] = np.mean([tempa1, tempb1, tempc1, tempa2, tempb2, tempc2])\n",
"\n",
"for protein, row in zip(common.index, range(0, len(common.index))):\n",
" for fraction, col in zip(fractlist, range(0, 7)):\n",
" tempa1 = ia2.at[protein, fraction]\n",
" tempb1 = ib2.at[protein, fraction]\n",
" tempc1 = ic2.at[protein, fraction]\n",
" tempa2 = ia2.at[protein, fraction]\n",
" tempb2 = ib2.at[protein, fraction]\n",
" tempc2 = ic2.at[protein, fraction]\n",
" ind_cent[row, col] = np.mean([tempa1, tempb1, tempc1, tempa2, tempb2, tempc2])"
]
},
{
"cell_type": "code",
"execution_count": 29,
"metadata": {},
"outputs": [],
"source": [
"# Calculate Euclidean distance from each replicate to its respective centroid\n",
"un_stats = np.zeros((len(common.index), 7))\n",
"ind_stats = np.zeros((len(common.index), 7))\n",
"\n",
"unlist = [ua1, ub1, uc1, ua2, ub2, uc2]\n",
"indlist = [ia1, ib1, ic1, ia2, ib2, ic2]\n",
"\n",
"for protein, row in zip(common.index, range(0, len(common.index))):\n",
" for df, col in zip(unlist, range(0,6)):\n",
" temp = df.loc[protein, '3K':'195.5K']\n",
" avg = un_cent[row]\n",
" un_stats[row, col] = euclidean(temp, avg)\n",
"\n",
"for protein, row in zip(common.index, range(0, len(common.index))):\n",
" for df, col in zip(indlist, range(0,6)):\n",
" temp = df.loc[protein, '3K':'195.5K']\n",
" avg = ind_cent[row]\n",
" ind_stats[row, col] = euclidean(temp, avg)"
]
},
{
"cell_type": "code",
"execution_count": 30,
"metadata": {},
"outputs": [],
"source": [
"# Calculate standard deviation of distances for each condition\n",
"un_stats[:, 6] = np.std(un_stats[:, 0:6], axis=1, ddof=1)\n",
"ind_stats[:, 6] = np.std(ind_stats[:, 0:6], axis=1, ddof=1)"
]
},
{
"cell_type": "code",
"execution_count": 31,
"metadata": {},
"outputs": [],
"source": [
"# Calculate t-statistic and p-value for each protein\n",
"# Using Benjamini-Hochberg correction for p-values\n",
"mvmt_stats = np.zeros((len(common.index), 5))\n",
"\n",
"# Column 0 will contain the Euclidean distance between the uninduced and induced centroids\n",
"for row in range(0, len(common.index)):\n",
" mvmt_stats[row, 0] = euclidean(un_cent[row], ind_cent[row])\n",
"\n",
"# Column 1 will contain the t-statistic\n",
"for row in range(0, len(common.index)):\n",
" unvar = (un_stats[row, 3])**2\n",
" indvar = (ind_stats[row, 3])**2\n",
" mvmt_stats[row, 1] = ((mvmt_stats[row, 0])/np.sqrt((unvar+indvar)/3))\n",
"\n",
"# Column 2 will contain the p-value\n",
"for row in range(0, len(common.index)):\n",
" tstat = mvmt_stats[row, 1]\n",
" unvar = (un_stats[row, 3])**2\n",
" indvar = (ind_stats[row, 3])**2\n",
" dof = (2*((unvar**2+(2*unvar*indvar)+indvar**2)/(unvar**2+indvar**2)))\n",
" mvmt_stats[row, 2] = t.sf(x=tstat, df=dof)\n",
"\n",
"# Column 3 will contain the rank value for each p-value\n",
"mvmt_stats[:, 3] = rankdata(mvmt_stats[:, 2])\n",
"\n",
"# Column 4 will contain the Benjamini-Hochberg critical values; using an FDR of 0.305 or 30.5%\n",
"mvmt_stats[:, 4] = (mvmt_stats[:, 3]/len(common.index))*0.305"
]
},
{
"cell_type": "code",
"execution_count": 32,
"metadata": {},
"outputs": [],
"source": [
"# Adding the p-value and critical value for each protein, then sorting by p-value; FDR of 30.5%\n",
"sigmvmt = common.loc[:, 'Gene_ua1':'Protein information_ua1']\n",
"sigmvmt['p-value'] = mvmt_stats[:, 2]\n",
"sigmvmt['CritValue'] = mvmt_stats[:, 4]\n",
"sigmvmt.sort_values(by='p-value', inplace=True)"
]
},
{
"cell_type": "code",
"execution_count": 33,
"metadata": {},
"outputs": [],
"source": [
"# Determining the maximum p-value < CritValue and subsetting out all p-values less than that cutoff; FDR of 30.5%\n",
"# Taking just the top 500\n",
"df = sigmvmt[sigmvmt['p-value'] < sigmvmt['CritValue']]\n",
"cutoff = df['p-value'].max()\n",
"moved = sigmvmt[sigmvmt['p-value'] <= cutoff]"
]
},
{
"cell_type": "code",
"execution_count": 34,
"metadata": {},
"outputs": [],
"source": [
"# Saving top 500 from combined movement analysis\n",
"moved500 = moved.iloc[0:500, :]\n",
"moved500.to_csv('20191123_CombinedMvmtAnalysis_top500_30.5FDR.csv', sep=',')\n",
"moved500.reset_index(inplace=True)\n",
"df = moved500.iloc[:, 0:1]\n",
"df.to_csv('20191123_CombinedMvmtAnalysis_top500_IDs.csv', sep=',', header=False, index=False)"
]
},
{
"cell_type": "code",
"execution_count": 35,
"metadata": {},
"outputs": [],
"source": [
"# Comparing the 2.5% FDR list of dNef2 and the 1.25% FDR list of dNef1, only looking at top 1,000 proteins from each\n",
"dNef1_top1000 = dNef1_moved.iloc[0:1000, :]\n",
"dNef2_top1000 = dNef2_moved.iloc[0:1000, :]\n",
"moved_common = pd.concat([dNef1_top1000, dNef2_top1000], join='inner', axis=1)\n",
"moved_common.to_csv('20191123_ComparativeMvmtAnalysis.csv', sep=',')"
]
},
{
"cell_type": "code",
"execution_count": 36,
"metadata": {},
"outputs": [],
"source": [
"# Making IDs only files for moved_common\n",
"moved_common.reset_index(inplace=True)\n",
"df = moved_common.iloc[:, 0:1]\n",
"df.to_csv('20191123_ComparativeMvmtAnalysis_IDs.csv', sep=',', header=False, index=False)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## STRING Analyses for Centroid Distance Analysis\n",
"\n",
"[Top 500 proteins from combined distance analysis](https://version-11-0.string-db.org/cgi/network.pl?networkId=KFMHH4dLyizq)<br>\n",
"[Common moved proteins between the biological replicates](https://version-11-0.string-db.org/cgi/network.pl?networkId=cmYXI9cw3biu)"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.7.4"
}
},
"nbformat": 4,
"nbformat_minor": 4
}