InducibleHIV-Fractionation / ComparingdNefrepl / Ntbk3_MvmtAnalysis.ipynb
Ntbk3_MvmtAnalysis.ipynb
Raw
{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": 1,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Inline graphing\n",
    "%matplotlib inline"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Import necessary packages\n",
    "import os\n",
    "import pandas as pd\n",
    "import numpy as np\n",
    "import seaborn as sns\n",
    "import matplotlib.pyplot as plt\n",
    "from matplotlib_venn import venn3, venn2\n",
    "from scipy import stats\n",
    "\n",
    "pd.set_option('mode.chained_assignment', None)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Set working directory to dNef 1 folder\n",
    "os.chdir('Path/To/Data')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 4,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Reading in full data for first dNef experiment\n",
    "ua1 = pd.read_csv('UnA/20190612_UnA_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "ua1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "ua1.columns = ['Gene_ua1', 'ProteinInfo_ua1', 'Compartment_ua1', 'Probability_ua1']\n",
    "\n",
    "ub1 = pd.read_csv('UnB/20190612_UnB_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "ub1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "ub1.columns = ['Gene_ub1', 'ProteinInfo_ub1', 'Compartment_ub1', 'Probability_ub1']\n",
    "\n",
    "uc1 = pd.read_csv('UnC/20190612_UnC_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "uc1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "uc1.columns = ['Gene_uc1', 'ProteinInfo_uc1', 'Compartment_uc1', 'Probability_uc1']\n",
    "\n",
    "ia1 = pd.read_csv('IndA/20190612_IndA_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "ia1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "ia1.columns = ['Gene_ia1', 'ProteinInfo_ia1', 'Compartment_ia1', 'Probability_ia1']\n",
    "\n",
    "ib1 = pd.read_csv('IndB/20190612_IndB_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "ib1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "ib1.columns = ['Gene_ib1', 'ProteinInfo_ib1', 'Compartment_ib1', 'Probability_ib1']\n",
    "\n",
    "ic1 = pd.read_csv('IndC/20190612_IndC_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "ic1.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "ic1.columns = ['Gene_ic1', 'ProteinInfo_ic1', 'Compartment_ic1', 'Probability_ic1']"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 5,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Set working directory to dNef 2 folder\n",
    "os.chdir('Path/To/Data')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 6,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Reading in full data for second dNef experiment\n",
    "ua2 = pd.read_csv('UnA/20191122_UnA_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "ua2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "ua2.columns = ['Gene_ua2', 'ProteinInfo_ua2', 'Compartment_ua2', 'Probability_ua2']\n",
    "\n",
    "ub2 = pd.read_csv('UnB/20191122_UnB_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "ub2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "ub2.columns = ['Gene_ub2', 'ProteinInfo_ub2', 'Compartment_ub2', 'Probability_ub2']\n",
    "\n",
    "uc2 = pd.read_csv('UnC/20191122_UnC_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "uc2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "uc2.columns = ['Gene_uc2', 'ProteinInfo_uc2', 'Compartment_uc2', 'Probability_uc2']\n",
    "\n",
    "ia2 = pd.read_csv('IndA/20191122_IndA_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "ia2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "ia2.columns = ['Gene_ia2', 'ProteinInfo_ia2', 'Compartment_ia2', 'Probability_ia2']\n",
    "\n",
    "ib2 = pd.read_csv('IndB/20191122_IndB_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "ib2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "ib2.columns = ['Gene_ib2', 'ProteinInfo_ib2', 'Compartment_ib2', 'Probability_ib2']\n",
    "\n",
    "ic2 = pd.read_csv('IndC/20191122_IndC_SVCidentified_postiter_threshold.csv', index_col=0)\n",
    "ic2.drop(columns=['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K'], inplace=True)\n",
    "ic2.columns = ['Gene_ic2', 'ProteinInfo_ic2', 'Compartment_ic2', 'Probability_ic2']"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 7,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Set working directory to folder comparing biological replicates\n",
    "os.chdir('C:/Users/ooma1/OneDrive - UC San Diego/Guatelli Lab/Cell Fractionation/20191122 Jurkat TREHIV dNef fract MS/ComparingBiolRepl/')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 8,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Defining commonly detected protein dataframes for uninduced\n",
    "unlist = [ua1, ub1, uc1, ua2, ub2, uc2]\n",
    "\n",
    "un = pd.concat(unlist, axis=1, join='outer', sort='True')\n",
    "\n",
    "for item in un.index:\n",
    "    if un.at[item, 'ProteinInfo_ua1'] is np.nan:\n",
    "        if un.at[item, 'ProteinInfo_ub1'] is not np.nan:\n",
    "            un.at[item, 'ProteinInfo_ua1'] = un.at[item, 'ProteinInfo_ub1']\n",
    "        elif un.at[item, 'ProteinInfo_uc1'] is not np.nan:\n",
    "            un.at[item, 'ProteinInfo_ua1'] = un.at[item, 'ProteinInfo_uc1']\n",
    "        elif un.at[item, 'ProteinInfo_ua2'] is not np.nan:\n",
    "            un.at[item, 'ProteinInfo_ua1'] = un.at[item, 'ProteinInfo_ua2']\n",
    "        elif un.at[item, 'ProteinInfo_ub2'] is not np.nan:\n",
    "            un.at[item, 'ProteinInfo_ua1'] = un.at[item, 'ProteinInfo_ub2']\n",
    "        elif un.at[item, 'ProteinInfo_uc2'] is not np.nan:\n",
    "            un.at[item, 'ProteinInfo_ua1'] = un.at[item, 'ProteinInfo_uc2']\n",
    "                \n",
    "for item in un.index:\n",
    "    if un.at[item, 'Gene_ua1'] is np.nan:\n",
    "        if un.at[item, 'Gene_ub1'] is not np.nan:\n",
    "            un.at[item, 'Gene_ua1'] = un.at[item, 'Gene_ub1']\n",
    "        elif un.at[item, 'Gene_uc1'] is not np.nan:\n",
    "            un.at[item, 'Gene_ua1'] = un.at[item, 'Gene_uc1']\n",
    "        elif un.at[item, 'Gene_ua2'] is not np.nan:\n",
    "            un.at[item, 'Gene_ua1'] = un.at[item, 'Gene_ua2']\n",
    "        elif un.at[item, 'Gene_ub2'] is not np.nan:\n",
    "            un.at[item, 'Gene_ua1'] = un.at[item, 'Gene_ub2']\n",
    "        elif un.at[item, 'Gene_uc2'] is not np.nan:\n",
    "            un.at[item, 'Gene_ua1'] = un.at[item, 'Gene_uc2']\n",
    "\n",
    "un.drop(columns=['ProteinInfo_ub1', 'ProteinInfo_uc1', 'ProteinInfo_ua2', 'ProteinInfo_ub2', 'ProteinInfo_uc2', \n",
    "                 'Gene_ub1', 'Gene_uc1', 'Gene_ua2', 'Gene_ub2', 'Gene_uc2'], inplace=True)\n",
    "un.rename({'Gene_ua1':'Gene', 'ProteinInfo_ua1':'ProteinInfo'}, axis='columns', inplace=True)\n",
    "un.loc[:, 'Compartment_ua1':'Probability_uc2'] = un.loc[:, 'Compartment_ua1':'Probability_uc2'].fillna('ND')\n",
    "ndset = ['ND']\n",
    "un['NDcount'] = (un.isin(ndset).sum(1))/2"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 9,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Defining commonly detected protein dataframes for induced\n",
    "indlist = [ia1, ib1, ic1, ia2, ib2, ic2]\n",
    "\n",
    "ind = pd.concat(indlist, axis=1, join='outer', sort='True')\n",
    "\n",
    "for item in ind.index:\n",
    "    if ind.at[item, 'ProteinInfo_ia1'] is np.nan:\n",
    "        if ind.at[item, 'ProteinInfo_ib1'] is not np.nan:\n",
    "            ind.at[item, 'ProteinInfo_ia1'] = ind.at[item, 'ProteinInfo_ib1']\n",
    "        elif ind.at[item, 'ProteinInfo_ic1'] is not np.nan:\n",
    "            ind.at[item, 'ProteinInfo_ia1'] = ind.at[item, 'ProteinInfo_ic1']\n",
    "        elif ind.at[item, 'ProteinInfo_ia2'] is not np.nan:\n",
    "            ind.at[item, 'ProteinInfo_ia1'] = ind.at[item, 'ProteinInfo_ia2']\n",
    "        elif ind.at[item, 'ProteinInfo_ib2'] is not np.nan:\n",
    "            ind.at[item, 'ProteinInfo_ia1'] = ind.at[item, 'ProteinInfo_ib2']\n",
    "        elif ind.at[item, 'ProteinInfo_ic2'] is not np.nan:\n",
    "            ind.at[item, 'ProteinInfo_ia1'] = ind.at[item, 'ProteinInfo_ic2']\n",
    "                \n",
    "for item in ind.index:\n",
    "    if ind.at[item, 'Gene_ia1'] is np.nan:\n",
    "        if ind.at[item, 'Gene_ib1'] is not np.nan:\n",
    "            ind.at[item, 'Gene_ia1'] = ind.at[item, 'Gene_ib1']\n",
    "        elif ind.at[item, 'Gene_ic1'] is not np.nan:\n",
    "            ind.at[item, 'Gene_ia1'] = ind.at[item, 'Gene_ic1']\n",
    "        elif ind.at[item, 'Gene_ia2'] is not np.nan:\n",
    "            ind.at[item, 'Gene_ia1'] = ind.at[item, 'Gene_ia2']\n",
    "        elif ind.at[item, 'Gene_ib2'] is not np.nan:\n",
    "            ind.at[item, 'Gene_ia1'] = ind.at[item, 'Gene_ib2']\n",
    "        elif ind.at[item, 'Gene_ic2'] is not np.nan:\n",
    "            ind.at[item, 'Gene_ia1'] = ind.at[item, 'Gene_ic2']\n",
    "\n",
    "ind.drop(columns=['ProteinInfo_ib1', 'ProteinInfo_ic1', 'ProteinInfo_ia2', 'ProteinInfo_ib2', 'ProteinInfo_ic2', \n",
    "                 'Gene_ib1', 'Gene_ic1', 'Gene_ia2', 'Gene_ib2', 'Gene_ic2'], inplace=True)\n",
    "ind.rename({'Gene_ia1':'Gene', 'ProteinInfo_ia1':'ProteinInfo'}, axis='columns', inplace=True)\n",
    "ind.loc[:, 'Compartment_ia1':'Probability_ic2'] = ind.loc[:, 'Compartment_ia1':'Probability_ic2'].fillna('ND')\n",
    "ndset = ['ND']\n",
    "ind['NDcount'] = (ind.isin(ndset).sum(1))/2"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 10,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Creating dataframes with only those proteins classified in all 6 replicates for a given condition (excludes proteins that were markers for\n",
    "# both dNef biological replicates)\n",
    "un_common = un[un['NDcount'] == 0.0]\n",
    "un_common = un_common[((un_common['Probability_ua1'] > 0) & (un_common['Probability_ua2'] > 0))]\n",
    "\n",
    "ind_common = ind[ind['NDcount'] == 0.0]\n",
    "ind_common = ind_common[((ind_common['Probability_ia1'] > 0) & (ind_common['Probability_ia2'] > 0))]"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 11,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Generating dataframes for finding row modes in compartment columns\n",
    "un_temp = un_common.drop(columns=['Probability_ua1', 'Probability_ub1', 'Probability_uc1', \n",
    "                                  'Probability_ua2', 'Probability_ub2', 'Probability_uc2'])\n",
    "un_temp = un_temp.loc[:, 'Compartment_ua1':'Compartment_uc2']\n",
    "\n",
    "ind_temp = ind_common.drop(columns=['Probability_ia1', 'Probability_ib1', 'Probability_ic1', \n",
    "                                  'Probability_ia2', 'Probability_ib2', 'Probability_ic2'])\n",
    "ind_temp = ind_temp.loc[:, 'Compartment_ia1':'Compartment_ic2']\n",
    "\n",
    "# Taking row mode and adding this data back onto the dataframes with proteins classified in all 6 replicates for a given condition\n",
    "a = un_temp.mode(axis=1).iloc[:, 0:3]\n",
    "a.columns = ['Mode1', 'Mode2', 'Mode3']\n",
    "un_common = pd.concat([un_common, a], axis=1, join='inner')\n",
    "\n",
    "b = ind_temp.mode(axis=1).iloc[:, 0:3]\n",
    "b.columns = ['Mode1', 'Mode2', 'Mode3']\n",
    "ind_common = pd.concat([ind_common, b], axis=1, join='inner')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 12,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Counting instances of row mode for each protein\n",
    "x, y = stats.mode(un_temp, axis=1)\n",
    "un_common['ModeCount'] = y\n",
    "\n",
    "x, y = stats.mode(ind_temp, axis=1)\n",
    "ind_common['ModeCount'] = y"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 13,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Generating dataframes for each condition with at least 4, 5, or 6 replicates classified the same\n",
    "un6 = un_common[un_common.ModeCount >= 6]\n",
    "un6.drop(columns=['NDcount', 'Mode2', 'Mode3'], inplace=True)\n",
    "un6.rename({'Mode1':'Mode_un'}, inplace=True, axis='columns')\n",
    "\n",
    "un5 = un_common[un_common.ModeCount >= 5]\n",
    "un5.drop(columns=['NDcount', 'Mode2', 'Mode3'], inplace=True)\n",
    "un5.rename({'Mode1':'Mode_un'}, inplace=True, axis='columns')\n",
    "\n",
    "un4 = un_common[un_common.ModeCount >= 4]\n",
    "un4.drop(columns=['NDcount', 'Mode2', 'Mode3'], inplace=True)\n",
    "un4.rename({'Mode1':'Mode_un'}, inplace=True, axis='columns')\n",
    "\n",
    "ind6 = ind_common[ind_common.ModeCount >= 6]\n",
    "ind6.drop(columns=['NDcount', 'Mode2', 'Mode3', 'Gene', 'ProteinInfo'], inplace=True)\n",
    "ind6.rename({'Mode1':'Mode_ind'}, inplace=True, axis='columns')\n",
    "\n",
    "ind5 = ind_common[ind_common.ModeCount >= 5]\n",
    "ind5.drop(columns=['NDcount', 'Mode2', 'Mode3', 'Gene', 'ProteinInfo'], inplace=True)\n",
    "ind5.rename({'Mode1':'Mode_ind'}, inplace=True, axis='columns')\n",
    "\n",
    "ind4 = ind_common[ind_common.ModeCount >= 4]\n",
    "ind4.drop(columns=['NDcount', 'Mode2', 'Mode3', 'Gene', 'ProteinInfo'], inplace=True)\n",
    "ind4.rename({'Mode1':'Mode_ind'}, inplace=True, axis='columns')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 14,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Concatenating uninduced and induced for each cutoff\n",
    "common6 = pd.concat([un6, ind6], join='inner', axis=1, sort=True)\n",
    "common5 = pd.concat([un5, ind5], join='inner', axis=1, sort=True)\n",
    "common4 = pd.concat([un4, ind4], join='inner', axis=1, sort=True)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 15,
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/html": [
       "<div>\n",
       "<style scoped>\n",
       "    .dataframe tbody tr th:only-of-type {\n",
       "        vertical-align: middle;\n",
       "    }\n",
       "\n",
       "    .dataframe tbody tr th {\n",
       "        vertical-align: top;\n",
       "    }\n",
       "\n",
       "    .dataframe thead th {\n",
       "        text-align: right;\n",
       "    }\n",
       "</style>\n",
       "<table border=\"1\" class=\"dataframe\">\n",
       "  <thead>\n",
       "    <tr style=\"text-align: right;\">\n",
       "      <th></th>\n",
       "      <th>Gene</th>\n",
       "      <th>ProteinInfo</th>\n",
       "      <th>Compartment_ua1</th>\n",
       "      <th>Probability_ua1</th>\n",
       "      <th>Compartment_ub1</th>\n",
       "      <th>Probability_ub1</th>\n",
       "      <th>Compartment_uc1</th>\n",
       "      <th>Probability_uc1</th>\n",
       "      <th>Compartment_ua2</th>\n",
       "      <th>Probability_ua2</th>\n",
       "      <th>...</th>\n",
       "      <th>Compartment_ic1</th>\n",
       "      <th>Probability_ic1</th>\n",
       "      <th>Compartment_ia2</th>\n",
       "      <th>Probability_ia2</th>\n",
       "      <th>Compartment_ib2</th>\n",
       "      <th>Probability_ib2</th>\n",
       "      <th>Compartment_ic2</th>\n",
       "      <th>Probability_ic2</th>\n",
       "      <th>Mode_ind</th>\n",
       "      <th>ModeCount</th>\n",
       "    </tr>\n",
       "  </thead>\n",
       "  <tbody>\n",
       "  </tbody>\n",
       "</table>\n",
       "<p>0 rows × 30 columns</p>\n",
       "</div>"
      ],
      "text/plain": [
       "Empty DataFrame\n",
       "Columns: [Gene, ProteinInfo, Compartment_ua1, Probability_ua1, Compartment_ub1, Probability_ub1, Compartment_uc1, Probability_uc1, Compartment_ua2, Probability_ua2, Compartment_ub2, Probability_ub2, Compartment_uc2, Probability_uc2, Mode_un, ModeCount, Compartment_ia1, Probability_ia1, Compartment_ib1, Probability_ib1, Compartment_ic1, Probability_ic1, Compartment_ia2, Probability_ia2, Compartment_ib2, Probability_ib2, Compartment_ic2, Probability_ic2, Mode_ind, ModeCount]\n",
       "Index: []\n",
       "\n",
       "[0 rows x 30 columns]"
      ]
     },
     "metadata": {},
     "output_type": "display_data"
    },
    {
     "data": {
      "text/html": [
       "<div>\n",
       "<style scoped>\n",
       "    .dataframe tbody tr th:only-of-type {\n",
       "        vertical-align: middle;\n",
       "    }\n",
       "\n",
       "    .dataframe tbody tr th {\n",
       "        vertical-align: top;\n",
       "    }\n",
       "\n",
       "    .dataframe thead th {\n",
       "        text-align: right;\n",
       "    }\n",
       "</style>\n",
       "<table border=\"1\" class=\"dataframe\">\n",
       "  <thead>\n",
       "    <tr style=\"text-align: right;\">\n",
       "      <th></th>\n",
       "      <th>Gene</th>\n",
       "      <th>ProteinInfo</th>\n",
       "      <th>Compartment_ua1</th>\n",
       "      <th>Probability_ua1</th>\n",
       "      <th>Compartment_ub1</th>\n",
       "      <th>Probability_ub1</th>\n",
       "      <th>Compartment_uc1</th>\n",
       "      <th>Probability_uc1</th>\n",
       "      <th>Compartment_ua2</th>\n",
       "      <th>Probability_ua2</th>\n",
       "      <th>...</th>\n",
       "      <th>Compartment_ic1</th>\n",
       "      <th>Probability_ic1</th>\n",
       "      <th>Compartment_ia2</th>\n",
       "      <th>Probability_ia2</th>\n",
       "      <th>Compartment_ib2</th>\n",
       "      <th>Probability_ib2</th>\n",
       "      <th>Compartment_ic2</th>\n",
       "      <th>Probability_ic2</th>\n",
       "      <th>Mode_ind</th>\n",
       "      <th>ModeCount</th>\n",
       "    </tr>\n",
       "  </thead>\n",
       "  <tbody>\n",
       "    <tr>\n",
       "      <td>P27448</td>\n",
       "      <td>MARK3</td>\n",
       "      <td>MAP/microtubule affinity-regulating kinase 3 (...</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.548</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>0.894</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>...</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>1</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.808</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.846</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.99</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>5</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>Q12959</td>\n",
       "      <td>DLG1</td>\n",
       "      <td>Disks large homolog 1 (Synapse-associated prot...</td>\n",
       "      <td>Peroxisome</td>\n",
       "      <td>0.838</td>\n",
       "      <td>Peroxisome</td>\n",
       "      <td>0.822</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.468</td>\n",
       "      <td>Peroxisome</td>\n",
       "      <td>0.832</td>\n",
       "      <td>...</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>0.734</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>0.902</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>0.948</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>0.804</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>5</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>Q9P2E9</td>\n",
       "      <td>RRBP1</td>\n",
       "      <td>Ribosome-binding protein 1 (180 kDa ribosome r...</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.446</td>\n",
       "      <td>ER</td>\n",
       "      <td>0.942</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.332</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.668</td>\n",
       "      <td>...</td>\n",
       "      <td>ER</td>\n",
       "      <td>0.926</td>\n",
       "      <td>ER</td>\n",
       "      <td>0.992</td>\n",
       "      <td>ER</td>\n",
       "      <td>0.958</td>\n",
       "      <td>ER</td>\n",
       "      <td>0.994</td>\n",
       "      <td>ER</td>\n",
       "      <td>5</td>\n",
       "    </tr>\n",
       "  </tbody>\n",
       "</table>\n",
       "<p>3 rows × 30 columns</p>\n",
       "</div>"
      ],
      "text/plain": [
       "         Gene                                        ProteinInfo  \\\n",
       "P27448  MARK3  MAP/microtubule affinity-regulating kinase 3 (...   \n",
       "Q12959   DLG1  Disks large homolog 1 (Synapse-associated prot...   \n",
       "Q9P2E9  RRBP1  Ribosome-binding protein 1 (180 kDa ribosome r...   \n",
       "\n",
       "       Compartment_ua1 Probability_ua1 Compartment_ub1 Probability_ub1  \\\n",
       "P27448    Unclassified           0.548        Endosome           0.894   \n",
       "Q12959      Peroxisome           0.838      Peroxisome           0.822   \n",
       "Q9P2E9    Unclassified           0.446              ER           0.942   \n",
       "\n",
       "       Compartment_uc1 Probability_uc1 Compartment_ua2 Probability_ua2  ...  \\\n",
       "P27448        Endosome               1        Endosome               1  ...   \n",
       "Q12959    Unclassified           0.468      Peroxisome           0.832  ...   \n",
       "Q9P2E9    Unclassified           0.332    Unclassified           0.668  ...   \n",
       "\n",
       "              Compartment_ic1 Probability_ic1        Compartment_ia2  \\\n",
       "P27448  Large Protein Complex               1  Large Protein Complex   \n",
       "Q12959        Plasma membrane           0.734        Plasma membrane   \n",
       "Q9P2E9                     ER           0.926                     ER   \n",
       "\n",
       "       Probability_ia2        Compartment_ib2  Probability_ib2  \\\n",
       "P27448           0.808  Large Protein Complex            0.846   \n",
       "Q12959           0.902        Plasma membrane            0.948   \n",
       "Q9P2E9           0.992                     ER            0.958   \n",
       "\n",
       "              Compartment_ic2 Probability_ic2               Mode_ind ModeCount  \n",
       "P27448  Large Protein Complex            0.99  Large Protein Complex         5  \n",
       "Q12959        Plasma membrane           0.804        Plasma membrane         5  \n",
       "Q9P2E9                     ER           0.994                     ER         5  \n",
       "\n",
       "[3 rows x 30 columns]"
      ]
     },
     "metadata": {},
     "output_type": "display_data"
    },
    {
     "data": {
      "text/html": [
       "<div>\n",
       "<style scoped>\n",
       "    .dataframe tbody tr th:only-of-type {\n",
       "        vertical-align: middle;\n",
       "    }\n",
       "\n",
       "    .dataframe tbody tr th {\n",
       "        vertical-align: top;\n",
       "    }\n",
       "\n",
       "    .dataframe thead th {\n",
       "        text-align: right;\n",
       "    }\n",
       "</style>\n",
       "<table border=\"1\" class=\"dataframe\">\n",
       "  <thead>\n",
       "    <tr style=\"text-align: right;\">\n",
       "      <th></th>\n",
       "      <th>Gene</th>\n",
       "      <th>ProteinInfo</th>\n",
       "      <th>Compartment_ua1</th>\n",
       "      <th>Probability_ua1</th>\n",
       "      <th>Compartment_ub1</th>\n",
       "      <th>Probability_ub1</th>\n",
       "      <th>Compartment_uc1</th>\n",
       "      <th>Probability_uc1</th>\n",
       "      <th>Compartment_ua2</th>\n",
       "      <th>Probability_ua2</th>\n",
       "      <th>...</th>\n",
       "      <th>Compartment_ic1</th>\n",
       "      <th>Probability_ic1</th>\n",
       "      <th>Compartment_ia2</th>\n",
       "      <th>Probability_ia2</th>\n",
       "      <th>Compartment_ib2</th>\n",
       "      <th>Probability_ib2</th>\n",
       "      <th>Compartment_ic2</th>\n",
       "      <th>Probability_ic2</th>\n",
       "      <th>Mode_ind</th>\n",
       "      <th>ModeCount</th>\n",
       "    </tr>\n",
       "  </thead>\n",
       "  <tbody>\n",
       "    <tr>\n",
       "      <td>B0I1T2</td>\n",
       "      <td>MYO1G</td>\n",
       "      <td>Unconventional myosin-Ig [Cleaved into: Minor ...</td>\n",
       "      <td>Peroxisome</td>\n",
       "      <td>0.772</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.51</td>\n",
       "      <td>Peroxisome</td>\n",
       "      <td>0.722</td>\n",
       "      <td>Peroxisome</td>\n",
       "      <td>0.818</td>\n",
       "      <td>...</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.584</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.536</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.386</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>0.956</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>5</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>HIVNL43_GAG</td>\n",
       "      <td>gag</td>\n",
       "      <td>Gag</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>1</td>\n",
       "      <td>Mitochondrion</td>\n",
       "      <td>1</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.7</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>1</td>\n",
       "      <td>...</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.476</td>\n",
       "      <td>Golgi</td>\n",
       "      <td>0.78</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.508</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.576</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>4</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>O00560</td>\n",
       "      <td>SDCB1</td>\n",
       "      <td>Syntenin-1 (Melanoma differentiation-associate...</td>\n",
       "      <td>ER</td>\n",
       "      <td>1</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.498</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.5</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.494</td>\n",
       "      <td>...</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.626</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>0.974</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.574</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>0.814</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>4</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>O00584</td>\n",
       "      <td>RNT2</td>\n",
       "      <td>Ribonuclease T2 (EC 3.1.27.-) (Ribonuclease 6)</td>\n",
       "      <td>Lysosome</td>\n",
       "      <td>0.806</td>\n",
       "      <td>Lysosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Lysosome</td>\n",
       "      <td>0.88</td>\n",
       "      <td>Lysosome</td>\n",
       "      <td>1</td>\n",
       "      <td>...</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.57</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>0.842</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.6</td>\n",
       "      <td>Lysosome</td>\n",
       "      <td>0.946</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>4</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>O14972</td>\n",
       "      <td>DSCR3</td>\n",
       "      <td>Vacuolar protein sorting-associated protein 26...</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.514</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>...</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.854</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.838</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.994</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>4</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "      <td>...</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>Q9UKK6</td>\n",
       "      <td>NXT1</td>\n",
       "      <td>NTF2-related export protein 1 (Protein p15)</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.796</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>...</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.676</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.916</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.982</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.998</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>4</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>Q9ULH0</td>\n",
       "      <td>KDIS</td>\n",
       "      <td>Kinase D-interacting substrate of 220 kDa (Ank...</td>\n",
       "      <td>Golgi</td>\n",
       "      <td>0.798</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.432</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.644</td>\n",
       "      <td>Golgi</td>\n",
       "      <td>0.714</td>\n",
       "      <td>...</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>0.812</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>0.872</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.632</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>1</td>\n",
       "      <td>Plasma membrane</td>\n",
       "      <td>4</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>Q9ULV4</td>\n",
       "      <td>COR1C</td>\n",
       "      <td>Coronin-1C (Coronin-3) (hCRNN4)</td>\n",
       "      <td>Golgi</td>\n",
       "      <td>0.778</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.67</td>\n",
       "      <td>Golgi</td>\n",
       "      <td>0.996</td>\n",
       "      <td>Golgi</td>\n",
       "      <td>0.85</td>\n",
       "      <td>...</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.508</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>0.818</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.572</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>0.95</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>4</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>Q9Y376</td>\n",
       "      <td>CAB39</td>\n",
       "      <td>Calcium-binding protein 39 (MO25alpha) (Protei...</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>0.72</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.854</td>\n",
       "      <td>...</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.556</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.654</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>0.542</td>\n",
       "      <td>Unclassified</td>\n",
       "      <td>4</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <td>Q9Y6W5</td>\n",
       "      <td>WASF2</td>\n",
       "      <td>Wiskott-Aldrich syndrome protein family member...</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>0.992</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>0.998</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.908</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>1</td>\n",
       "      <td>...</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>0.834</td>\n",
       "      <td>Endosome</td>\n",
       "      <td>1</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.642</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>0.686</td>\n",
       "      <td>Large Protein Complex</td>\n",
       "      <td>4</td>\n",
       "    </tr>\n",
       "  </tbody>\n",
       "</table>\n",
       "<p>67 rows × 30 columns</p>\n",
       "</div>"
      ],
      "text/plain": [
       "              Gene                                        ProteinInfo  \\\n",
       "B0I1T2       MYO1G  Unconventional myosin-Ig [Cleaved into: Minor ...   \n",
       "HIVNL43_GAG    gag                                                Gag   \n",
       "O00560       SDCB1  Syntenin-1 (Melanoma differentiation-associate...   \n",
       "O00584        RNT2     Ribonuclease T2 (EC 3.1.27.-) (Ribonuclease 6)   \n",
       "O14972       DSCR3  Vacuolar protein sorting-associated protein 26...   \n",
       "...            ...                                                ...   \n",
       "Q9UKK6        NXT1        NTF2-related export protein 1 (Protein p15)   \n",
       "Q9ULH0        KDIS  Kinase D-interacting substrate of 220 kDa (Ank...   \n",
       "Q9ULV4       COR1C                    Coronin-1C (Coronin-3) (hCRNN4)   \n",
       "Q9Y376       CAB39  Calcium-binding protein 39 (MO25alpha) (Protei...   \n",
       "Q9Y6W5       WASF2  Wiskott-Aldrich syndrome protein family member...   \n",
       "\n",
       "                   Compartment_ua1 Probability_ua1 Compartment_ub1  \\\n",
       "B0I1T2                  Peroxisome           0.772    Unclassified   \n",
       "HIVNL43_GAG  Large Protein Complex               1   Mitochondrion   \n",
       "O00560                          ER               1    Unclassified   \n",
       "O00584                    Lysosome           0.806        Lysosome   \n",
       "O14972                    Endosome               1        Endosome   \n",
       "...                            ...             ...             ...   \n",
       "Q9UKK6                    Endosome               1        Endosome   \n",
       "Q9ULH0                       Golgi           0.798    Unclassified   \n",
       "Q9ULV4                       Golgi           0.778    Unclassified   \n",
       "Q9Y376                    Endosome               1        Endosome   \n",
       "Q9Y6W5                    Endosome           0.992        Endosome   \n",
       "\n",
       "            Probability_ub1        Compartment_uc1 Probability_uc1  \\\n",
       "B0I1T2                 0.51             Peroxisome           0.722   \n",
       "HIVNL43_GAG               1           Unclassified             0.7   \n",
       "O00560                0.498           Unclassified             0.5   \n",
       "O00584                    1               Lysosome            0.88   \n",
       "O14972                    1           Unclassified           0.514   \n",
       "...                     ...                    ...             ...   \n",
       "Q9UKK6                    1  Large Protein Complex           0.796   \n",
       "Q9ULH0                0.432           Unclassified           0.644   \n",
       "Q9ULV4                 0.67                  Golgi           0.996   \n",
       "Q9Y376                    1               Endosome            0.72   \n",
       "Q9Y6W5                0.998  Large Protein Complex           0.908   \n",
       "\n",
       "                   Compartment_ua2 Probability_ua2  ...  \\\n",
       "B0I1T2                  Peroxisome           0.818  ...   \n",
       "HIVNL43_GAG  Large Protein Complex               1  ...   \n",
       "O00560                Unclassified           0.494  ...   \n",
       "O00584                    Lysosome               1  ...   \n",
       "O14972                    Endosome               1  ...   \n",
       "...                            ...             ...  ...   \n",
       "Q9UKK6                    Endosome               1  ...   \n",
       "Q9ULH0                       Golgi           0.714  ...   \n",
       "Q9ULV4                       Golgi            0.85  ...   \n",
       "Q9Y376       Large Protein Complex           0.854  ...   \n",
       "Q9Y6W5       Large Protein Complex               1  ...   \n",
       "\n",
       "                   Compartment_ic1 Probability_ic1        Compartment_ia2  \\\n",
       "B0I1T2                Unclassified           0.584           Unclassified   \n",
       "HIVNL43_GAG           Unclassified           0.476                  Golgi   \n",
       "O00560                Unclassified           0.626               Endosome   \n",
       "O00584                Unclassified            0.57        Plasma membrane   \n",
       "O14972       Large Protein Complex           0.854               Endosome   \n",
       "...                            ...             ...                    ...   \n",
       "Q9UKK6                Unclassified           0.676  Large Protein Complex   \n",
       "Q9ULH0             Plasma membrane           0.812        Plasma membrane   \n",
       "Q9ULV4                Unclassified           0.508               Endosome   \n",
       "Q9Y376                Unclassified           0.556               Endosome   \n",
       "Q9Y6W5                    Endosome           0.834               Endosome   \n",
       "\n",
       "            Probability_ia2        Compartment_ib2  Probability_ib2  \\\n",
       "B0I1T2                0.536           Unclassified            0.386   \n",
       "HIVNL43_GAG            0.78           Unclassified            0.508   \n",
       "O00560                0.974           Unclassified            0.574   \n",
       "O00584                0.842           Unclassified              0.6   \n",
       "O14972                    1  Large Protein Complex            0.838   \n",
       "...                     ...                    ...              ...   \n",
       "Q9UKK6                0.916  Large Protein Complex            0.982   \n",
       "Q9ULH0                0.872           Unclassified            0.632   \n",
       "Q9ULV4                0.818           Unclassified            0.572   \n",
       "Q9Y376                    1           Unclassified            0.654   \n",
       "Q9Y6W5                    1  Large Protein Complex            0.642   \n",
       "\n",
       "                   Compartment_ic2 Probability_ic2               Mode_ind  \\\n",
       "B0I1T2             Plasma membrane           0.956           Unclassified   \n",
       "HIVNL43_GAG           Unclassified           0.576           Unclassified   \n",
       "O00560                    Endosome           0.814               Endosome   \n",
       "O00584                    Lysosome           0.946           Unclassified   \n",
       "O14972       Large Protein Complex           0.994  Large Protein Complex   \n",
       "...                            ...             ...                    ...   \n",
       "Q9UKK6       Large Protein Complex           0.998  Large Protein Complex   \n",
       "Q9ULH0             Plasma membrane               1        Plasma membrane   \n",
       "Q9ULV4                    Endosome            0.95               Endosome   \n",
       "Q9Y376                Unclassified           0.542           Unclassified   \n",
       "Q9Y6W5       Large Protein Complex           0.686  Large Protein Complex   \n",
       "\n",
       "            ModeCount  \n",
       "B0I1T2              5  \n",
       "HIVNL43_GAG         4  \n",
       "O00560              4  \n",
       "O00584              4  \n",
       "O14972              4  \n",
       "...               ...  \n",
       "Q9UKK6              4  \n",
       "Q9ULH0              4  \n",
       "Q9ULV4              4  \n",
       "Q9Y376              4  \n",
       "Q9Y6W5              4  \n",
       "\n",
       "[67 rows x 30 columns]"
      ]
     },
     "metadata": {},
     "output_type": "display_data"
    }
   ],
   "source": [
    "# Subsetting out proteins that have moved\n",
    "common6_moved = common6[common6['Mode_un'] != common6['Mode_ind']]\n",
    "display(common6_moved)\n",
    "common5_moved = common5[common5['Mode_un'] != common5['Mode_ind']]\n",
    "display(common5_moved)\n",
    "common4_moved = common4[common4['Mode_un'] != common4['Mode_ind']]\n",
    "display(common4_moved)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 16,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Saving common5_moved and common4_moved\n",
    "common5_moved.to_csv('20191123_RelocalizedProteins_5of6repl.csv', sep=',')\n",
    "common5_ID = common5_moved.loc[:, 'Gene':'ProteinInfo']\n",
    "common5_ID.reset_index(inplace=True)\n",
    "common5_ID.drop(['Gene', 'ProteinInfo'], axis=1, inplace=True)\n",
    "common5_ID.to_csv('20191123_RelocalizedProteins_5of6repl_IDs.csv', sep=',', index=False)\n",
    "\n",
    "common4_moved.to_csv('20191123_RelocalizedProteins_4of6repl.csv', sep=',')\n",
    "common4_ID = common4_moved.loc[:, 'Gene':'ProteinInfo']\n",
    "common4_ID.reset_index(inplace=True)\n",
    "common4_ID.drop(['Gene', 'ProteinInfo'], axis=1, inplace=True)\n",
    "common4_ID.to_csv('20191123_RelocalizedProteins_4of6repl_IDs.csv', sep=',', index=False)"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "### STRING Analyses for Moved Proteins\n",
    "\n",
    "[Proteins identical in 4+ replicates/condition](https://version-11-0.string-db.org/cgi/network.pl?networkId=ItKzBS0Uuj2q)<br>\n",
    "[Proteins identical in 5+ replicates/condition](https://version-11-0.string-db.org/cgi/network.pl?networkId=qm0ZBoHpO6ED)<br>"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "# Centroid Distance Analysis"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 17,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Import packages\n",
    "import os\n",
    "import math\n",
    "import pandas as pd\n",
    "import numpy as np\n",
    "import seaborn as sns\n",
    "import matplotlib.pyplot as plt\n",
    "from sklearn.covariance import MinCovDet\n",
    "from scipy.stats import percentileofscore\n",
    "from scipy.stats import rankdata\n",
    "from scipy.stats import t\n",
    "from scipy.spatial.distance import euclidean"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 18,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Set working directory to dNef 1 folder\n",
    "os.chdir('Path/To/Data')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 19,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Reading in full data for first dNef experiment\n",
    "ua1 = pd.read_csv('UnA/20190612_UnA_RowSum.csv', index_col=0)\n",
    "ub1 = pd.read_csv('UnB/20190612_UnB_RowSum.csv', index_col=0)\n",
    "uc1 = pd.read_csv('UnC/20190612_UnC_RowSum.csv', index_col=0)\n",
    "\n",
    "ia1 = pd.read_csv('IndA/20190612_IndA_RowSum.csv', index_col=0)\n",
    "ib1 = pd.read_csv('IndB/20190612_IndB_RowSum.csv', index_col=0)\n",
    "ic1 = pd.read_csv('IndC/20190612_IndC_RowSum.csv', index_col=0)\n",
    "\n",
    "dNef1_moved = pd.read_csv('20190613_EuclideanDistance_FDR0.0125.csv', index_col=0)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 20,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Labeling replicate columns for ease of identification in joined dataframe\n",
    "ua1.columns = ['Gene_ua1', 'Protein information_ua1', '3K_ua1', '5.4K_ua1', '12.2K_ua1', \n",
    "              '24K_ua1', '78.4K_ua1', '110K_ua1', '195.5K_ua1']\n",
    "ub1.columns = ['Gene_ub1', 'Protein information_ub1', '3K_ub1', '5.4K_ub1', '12.2K_ub1', \n",
    "              '24K_ub1', '78.4K_ub1', '110K_ub1', '195.5K_ub1']\n",
    "uc1.columns = ['Gene_uc1', 'Protein information_uc1', '3K_uc1', '5.4K_uc1', '12.2K_uc1', \n",
    "              '24K_uc1', '78.4K_uc1', '110K_uc1', '195.5K_uc1']\n",
    "ia1.columns = ['Gene_ia1', 'Protein information_ia1', '3K_ia1', '5.4K_ia1', '12.2K_ia1', \n",
    "              '24K_ia1', '78.4K_ia1', '110K_ia1', '195.5K_ia1']\n",
    "ib1.columns = ['Gene_ib1', 'Protein information_ib1', '3K_ib1', '5.4K_ib1', '12.2K_ib1', \n",
    "              '24K_ib1', '78.4K_ib1', '110K_ib1', '195.5K_ib1']\n",
    "ic1.columns = ['Gene_ic1', 'Protein information_ic1', '3K_ic1', '5.4K_ic1', '12.2K_ic1', \n",
    "              '24K_ic1', '78.4K_ic1', '110K_ic1', '195.5K_ic1']"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 21,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Set working directory to dNef 2 folder\n",
    "os.chdir('Path/To/Data')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 22,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Reading in full data for second dNef experiment\n",
    "ua2 = pd.read_csv('UnA/20191122_UnA_RowSum.csv', index_col=0)\n",
    "ub2 = pd.read_csv('UnB/20191122_UnB_RowSum.csv', index_col=0)\n",
    "uc2 = pd.read_csv('UnC/20191122_UnC_RowSum.csv', index_col=0)\n",
    "\n",
    "ia2 = pd.read_csv('IndA/20191122_IndA_RowSum.csv', index_col=0)\n",
    "ib2 = pd.read_csv('IndB/20191122_IndB_RowSum.csv', index_col=0)\n",
    "ic2 = pd.read_csv('IndC/20191122_IndC_RowSum.csv', index_col=0)\n",
    "\n",
    "dNef2_moved = pd.read_csv('20191122_EuclideanDistance_FDR0.025.csv', index_col=0)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 23,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Labeling replicate columns for ease of identification in joined dataframe\n",
    "ua2.columns = ['Gene_ua2', 'Protein information_ua2', '3K_ua2', '5.4K_ua2', '12.2K_ua2', \n",
    "              '24K_ua2', '78.4K_ua2', '110K_ua2', '195.5K_ua2']\n",
    "ub2.columns = ['Gene_ub2', 'Protein information_ub2', '3K_ub2', '5.4K_ub2', '12.2K_ub2', \n",
    "              '24K_ub2', '78.4K_ub2', '110K_ub2', '195.5K_ub2']\n",
    "uc2.columns = ['Gene_uc2', 'Protein information_uc2', '3K_uc2', '5.4K_uc2', '12.2K_uc2', \n",
    "              '24K_uc2', '78.4K_uc2', '110K_uc2', '195.5K_uc2']\n",
    "ia2.columns = ['Gene_ia2', 'Protein information_ia2', '3K_ia2', '5.4K_ia2', '12.2K_ia2', \n",
    "              '24K_ia2', '78.4K_ia2', '110K_ia2', '195.5K_ia2']\n",
    "ib2.columns = ['Gene_ib2', 'Protein information_ib2', '3K_ib2', '5.4K_ib2', '12.2K_ib2', \n",
    "              '24K_ib2', '78.4K_ib2', '110K_ib2', '195.5K_ib2']\n",
    "ic2.columns = ['Gene_ic2', 'Protein information_ic2', '3K_ic2', '5.4K_ic2', '12.2K_ic2', \n",
    "              '24K_ic2', '78.4K_ic2', '110K_ic2', '195.5K_ic2']"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 24,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Set working directory to folder comparing biological replicates\n",
    "os.chdir('C:/Users/ooma1/OneDrive - UC San Diego/Guatelli Lab/Cell Fractionation/20191122 Jurkat TREHIV dNef fract MS/ComparingBiolRepl/')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 25,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Pull out proteins common across all six replicates\n",
    "dflist = [ua1, ub1, uc1, ia1, ib1, ic1, ua2, ub2, uc2, ia2, ib2, ic2]\n",
    "common = pd.concat(dflist, axis=1, join='inner')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 26,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Redefine replicates with only common proteins\n",
    "ua1 = common.loc[:, 'Gene_ua1':'195.5K_ua1']\n",
    "ub1 = common.loc[:, 'Gene_ub1':'195.5K_ub1']\n",
    "uc1 = common.loc[:, 'Gene_uc1':'195.5K_uc1']\n",
    "ia1 = common.loc[:, 'Gene_ia1':'195.5K_ia1']\n",
    "ib1 = common.loc[:, 'Gene_ib1':'195.5K_ib1']\n",
    "ic1 = common.loc[:, 'Gene_ic1':'195.5K_ic1']\n",
    "\n",
    "ua2 = common.loc[:, 'Gene_ua2':'195.5K_ua2']\n",
    "ub2 = common.loc[:, 'Gene_ub2':'195.5K_ub2']\n",
    "uc2 = common.loc[:, 'Gene_uc2':'195.5K_uc2']\n",
    "ia2 = common.loc[:, 'Gene_ia2':'195.5K_ia2']\n",
    "ib2 = common.loc[:, 'Gene_ib2':'195.5K_ib2']\n",
    "ic2 = common.loc[:, 'Gene_ic2':'195.5K_ic2']"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 27,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Changing columns back to replicate invariant names\n",
    "ua1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "ub1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "uc1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "ia1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "ib1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "ic1.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "\n",
    "ua2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "ub2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "uc2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "ia2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "ib2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "ic2.columns = ['Gene', 'Protein information', '3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 28,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Calculate centroid arrays for uninduced and induced\n",
    "un_cent = np.zeros((len(common.index), 7))\n",
    "ind_cent = np.zeros((len(common.index), 7))\n",
    "\n",
    "fractlist = ['3K', '5.4K', '12.2K', '24K', '78.4K', '110K', '195.5K']\n",
    "\n",
    "for protein, row in zip(common.index, range(0, len(common.index))):\n",
    "    for fraction, col in zip(fractlist, range(0, 7)):\n",
    "        tempa1 = ua1.at[protein, fraction]\n",
    "        tempb1 = ub1.at[protein, fraction]\n",
    "        tempc1 = uc1.at[protein, fraction]\n",
    "        tempa2 = ua1.at[protein, fraction]\n",
    "        tempb2 = ub1.at[protein, fraction]\n",
    "        tempc2 = uc1.at[protein, fraction]\n",
    "        un_cent[row, col] = np.mean([tempa1, tempb1, tempc1, tempa2, tempb2, tempc2])\n",
    "\n",
    "for protein, row in zip(common.index, range(0, len(common.index))):\n",
    "    for fraction, col in zip(fractlist, range(0, 7)):\n",
    "        tempa1 = ia2.at[protein, fraction]\n",
    "        tempb1 = ib2.at[protein, fraction]\n",
    "        tempc1 = ic2.at[protein, fraction]\n",
    "        tempa2 = ia2.at[protein, fraction]\n",
    "        tempb2 = ib2.at[protein, fraction]\n",
    "        tempc2 = ic2.at[protein, fraction]\n",
    "        ind_cent[row, col] = np.mean([tempa1, tempb1, tempc1, tempa2, tempb2, tempc2])"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 29,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Calculate Euclidean distance from each replicate to its respective centroid\n",
    "un_stats = np.zeros((len(common.index), 7))\n",
    "ind_stats = np.zeros((len(common.index), 7))\n",
    "\n",
    "unlist = [ua1, ub1, uc1, ua2, ub2, uc2]\n",
    "indlist = [ia1, ib1, ic1, ia2, ib2, ic2]\n",
    "\n",
    "for protein, row in zip(common.index, range(0, len(common.index))):\n",
    "    for df, col in zip(unlist, range(0,6)):\n",
    "        temp = df.loc[protein, '3K':'195.5K']\n",
    "        avg = un_cent[row]\n",
    "        un_stats[row, col] = euclidean(temp, avg)\n",
    "\n",
    "for protein, row in zip(common.index, range(0, len(common.index))):\n",
    "    for df, col in zip(indlist, range(0,6)):\n",
    "        temp = df.loc[protein, '3K':'195.5K']\n",
    "        avg = ind_cent[row]\n",
    "        ind_stats[row, col] = euclidean(temp, avg)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 30,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Calculate standard deviation of distances for each condition\n",
    "un_stats[:, 6] = np.std(un_stats[:, 0:6], axis=1, ddof=1)\n",
    "ind_stats[:, 6] = np.std(ind_stats[:, 0:6], axis=1, ddof=1)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 31,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Calculate t-statistic and p-value for each protein\n",
    "# Using Benjamini-Hochberg correction for p-values\n",
    "mvmt_stats = np.zeros((len(common.index), 5))\n",
    "\n",
    "# Column 0 will contain the Euclidean distance between the uninduced and induced centroids\n",
    "for row in range(0, len(common.index)):\n",
    "    mvmt_stats[row, 0] = euclidean(un_cent[row], ind_cent[row])\n",
    "\n",
    "# Column 1 will contain the t-statistic\n",
    "for row in range(0, len(common.index)):\n",
    "    unvar = (un_stats[row, 3])**2\n",
    "    indvar = (ind_stats[row, 3])**2\n",
    "    mvmt_stats[row, 1] = ((mvmt_stats[row, 0])/np.sqrt((unvar+indvar)/3))\n",
    "\n",
    "# Column 2 will contain the p-value\n",
    "for row in range(0, len(common.index)):\n",
    "    tstat = mvmt_stats[row, 1]\n",
    "    unvar = (un_stats[row, 3])**2\n",
    "    indvar = (ind_stats[row, 3])**2\n",
    "    dof = (2*((unvar**2+(2*unvar*indvar)+indvar**2)/(unvar**2+indvar**2)))\n",
    "    mvmt_stats[row, 2] = t.sf(x=tstat, df=dof)\n",
    "\n",
    "# Column 3 will contain the rank value for each p-value\n",
    "mvmt_stats[:, 3] = rankdata(mvmt_stats[:, 2])\n",
    "\n",
    "# Column 4 will contain the Benjamini-Hochberg critical values; using an FDR of 0.305 or 30.5%\n",
    "mvmt_stats[:, 4] = (mvmt_stats[:, 3]/len(common.index))*0.305"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 32,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Adding the p-value and critical value for each protein, then sorting by p-value; FDR of 30.5%\n",
    "sigmvmt = common.loc[:, 'Gene_ua1':'Protein information_ua1']\n",
    "sigmvmt['p-value'] = mvmt_stats[:, 2]\n",
    "sigmvmt['CritValue'] = mvmt_stats[:, 4]\n",
    "sigmvmt.sort_values(by='p-value', inplace=True)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 33,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Determining the maximum p-value < CritValue and subsetting out all p-values less than that cutoff; FDR of 30.5%\n",
    "# Taking just the top 500\n",
    "df = sigmvmt[sigmvmt['p-value'] < sigmvmt['CritValue']]\n",
    "cutoff = df['p-value'].max()\n",
    "moved = sigmvmt[sigmvmt['p-value'] <= cutoff]"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 34,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Saving top 500 from combined movement analysis\n",
    "moved500 = moved.iloc[0:500, :]\n",
    "moved500.to_csv('20191123_CombinedMvmtAnalysis_top500_30.5FDR.csv', sep=',')\n",
    "moved500.reset_index(inplace=True)\n",
    "df = moved500.iloc[:, 0:1]\n",
    "df.to_csv('20191123_CombinedMvmtAnalysis_top500_IDs.csv', sep=',', header=False, index=False)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 35,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Comparing the 2.5% FDR list of dNef2 and the 1.25% FDR list of dNef1, only looking at top 1,000 proteins from each\n",
    "dNef1_top1000 = dNef1_moved.iloc[0:1000, :]\n",
    "dNef2_top1000 = dNef2_moved.iloc[0:1000, :]\n",
    "moved_common = pd.concat([dNef1_top1000, dNef2_top1000], join='inner', axis=1)\n",
    "moved_common.to_csv('20191123_ComparativeMvmtAnalysis.csv', sep=',')"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 36,
   "metadata": {},
   "outputs": [],
   "source": [
    "# Making IDs only files for moved_common\n",
    "moved_common.reset_index(inplace=True)\n",
    "df = moved_common.iloc[:, 0:1]\n",
    "df.to_csv('20191123_ComparativeMvmtAnalysis_IDs.csv', sep=',', header=False, index=False)"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "## STRING Analyses for Centroid Distance Analysis\n",
    "\n",
    "[Top 500 proteins from combined distance analysis](https://version-11-0.string-db.org/cgi/network.pl?networkId=KFMHH4dLyizq)<br>\n",
    "[Common moved proteins between the biological replicates](https://version-11-0.string-db.org/cgi/network.pl?networkId=cmYXI9cw3biu)"
   ]
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python 3",
   "language": "python",
   "name": "python3"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.7.4"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 4
}