#!/usr/bin/env python
# coding: utf-8

# In[ ]:


import streamlit as st
import pandas as pd
from PIL import Image
import subprocess
import os
import base64
import pickle

# Logo image
image = Image.open('logo.png')

st.image(image, use_column_width=True)


st.markdown("""
# Vitamin D Receptor (VDR)-Pred

VDR-Pred is a biological activity prediction tool using a machine learning-based QSAR model. 

**Credits**
- App built in `Python` + `Streamlit`, by Centre for Research in Molecular Modeling (CERMM), Concordia University, Montreal, Quebec, Canada, H4B 1R6 
- Descriptor calculated using [PaDEL-Descriptor](http://www.yapcwsoft.com/dd/padeldescriptor/) [[Read the Paper]](https://doi.org/10.1002/jcc.21707).

""")


# In[ ]:

# Navigation bar
rad=st.sidebar.selectbox("Menu", ["Home", "Help", "Team", "Terms of Services"])

if rad =="Home":
  st.title ('Home')
   
  st.header ('About VDR-Pred')
  
  st.markdown("""
Vitamin D Receptor (VDR) is a drug target for many infectious and immunological diseases. VDR-Pred tool employs to predict VDR agonistic activity (i.e. pIC50) of drug-like molecules using Machine learning algorithm-based QSAR models. The screening results allow the user to use the predicted molecules to activate the vitamin D receptor enzyme. 
""")

  st.header ('Inhibitory Concentration-50 (IC50)')
  st.markdown("""
  The half-maximal inhibitory concentration (i.e. IC50) is defined as a quantitative measurement of how much amount of an actual inhibitory molecule (i.e. drug or chemical) is required to inhibit 50 % of targeted biological activities (i.e. cell growth, enzymatic action, and biological pathways). 
""")

  st.header ('pIC50')
  st.markdown("""

pIC50 is the negative (-) logarithmic value of IC50 while converting into molar (Ex. IC50 of 1µM is 10-6M, the pIC50 is 6.0). The advantage of pIC50 is to think about the potency of drug-like compounds, and it is essential for dealing with concentration-dependent relationships.

""")

if rad =="Help":

  st.title ('Help')
  
  st.header ('Steps to calculate the pIC50')
  
  col1, mid, col2 = st.columns([1,1,200])
  with col1:
    st.header ('Steps 1-2')
  with col2:
    st.image('Step 1.jpg', width= 500)
    
  st.markdown("""

**Step 1:** Upload a .txt file with “SMILES” of small molecules that are selected from CHEMBL, PubChem, or DrugBank or ZINC database. The uploaded .txt file must contain a “SMILES” and “ID” of selected small molecules. The SMILES and reference ID (Ex:`C=C1C[C@H](C[C@@H](C)' and ‘CHEMBL100’) must be written IUPAC standard format. User may avoid non-standard format and characters. 

**Step 2:** Click Submit button. The calculation may take few seconds to 15 minutes based on the number of compounds.
""")

  col1, mid, col2 = st.columns([1,1,200])
  with col1:
    st.header ('Steps 3-4')
  with col2:
    st.image('Step 2.jpg', width= 500)
  st.markdown("""
  **Step 3:** User can visualize the list of uploaded input molecules detail (i.e. SMILE and ID) in tabular form.
  
  **Step 4:** User can visualize the calculated molecular descriptors of the input molecules in tabular form.

""")

  col1, mid, col2 = st.columns([1,1,200])
  with col1:
    st.header ('Steps 5-7')
  with col2:
    st.image('Step 3.jpg', width= 500)
  st.markdown("""

""")
  st.markdown("""
**Step 5:** User can visualize Subset of molecular descriptors from previously build QSAR model in tabular form .

**Step 6:** The calculated molecular descriptors and pIC50 values displayed in tabular form.

**Step 7:** User can download the results in .csv format.
""")


if rad == "Team":

  st.title ('Team')
  st.markdown(
     "**Prof. Gilles H. Peslherbe, PhD**,  \n" 
     "Principal Investigator & Director of CERMM  \n"
     "CERMM, Concordia University, Montreal, Canada  \n"
     
  
     "**Dr. Gurudeeban Selvaraj, PhD**,  \n"
     "Research Associate & Developer  \n"
     "CERMM, Concordia University, Montreal, Canada  \n"
    
    
     "**Dr. Satyavani Kaliamurthi, PhD**,  \n"
     "Horizon Postdoctoral Fellow & Co-Developer  \n"
     "CERMM, Concordia University, Montreal, Canada  \n"
     
     
     "**Dr. Denise Koch, PhD**,  \n"
     "CERMM Manager & Co-Developer,  \n" 
     "CERMM, Concordia University, Montreal, Canada  \n"
     
     "**Prof. Dongqing Wei, PhD**,  \n"
     "Professor & International Collaborator,  \n" 
     "College of Life Sciences and Biotechnology,  \n" 
     "Shanghai Jiao Tong University, Shanghai 200240, China  "
     
     
     )

  st.title ('Contact')
  st.markdown(
  "CERMM,  \n" 	
  "Concordia University, 7141 Sherbrooke West Street.,  \n"
  "Montreal, Quebec, Canada H4B1R6  \n"
  "Phone:	XXXXXXXX  \n"
  "Email: XXXXXXXX"
  )
 
  st.title ('Acknowledgements')
  st.markdown(
    "Natural Sciences and Engineering Research Council (NSERC) Discovery Grant, Canada  \n"
    "Horizon Postdoctoral Fellowship, Concordia University, Montreal, Quebec, Canada"
 )

if rad == "Terms of Services":
  st.title ('Terms of Services')
  st.markdown('* This VDR-Pred is available to you subject to the Terms of Service described on this page. Your use of this site constitutes your agreement to these terms and conditions.  \n'
  '* The contents of the VDR-Pred are intended for educational or research purposes.  \n'
  '* All materials and services provided on the Web application are protected by copyright or other intellectual property rights, and owned or controlled by the CERMM, Concordia University.')

# Molecular descriptor calculator
def desc_calc():
    # Performs the descriptor calculation
    bashCommand = "java -Xms2G -Xmx2G -Djava.awt.headless=true -jar ./PaDEL-Descriptor/PaDEL-Descriptor.jar -removesalt -standardizenitro -fingerprints -descriptortypes ./PaDEL-Descriptor/PubchemFingerprinter.xml -dir ./ -file descriptors_output.csv"
    process = subprocess.Popen(bashCommand.split(), stdout=subprocess.PIPE)
    output, error = process.communicate()
    os.remove('molecule.smi')

# File download
def filedownload(df):
    csv = df.to_csv(index=False)
    b64 = base64.b64encode(csv.encode()).decode()  # strings <-> bytes conversions
    href = f'<a href="data:file/csv;base64,{b64}" download="prediction.csv">Download Predictions</a>'
    return href

# Model building
def build_model(input_data):
    # Reads in saved regression model
    load_model = pickle.load(open('vdr_model.pkl', 'rb'))
    # Apply model to make predictions
    prediction = load_model.predict(input_data)
    st.header('**Prediction output**')
    prediction_output = pd.Series(prediction, name='pIC50')
    molecule_name = pd.Series(load_data[1], name='molecule_name')
    df = pd.concat([molecule_name, prediction_output], axis=1)
    st.write(df)
    st.markdown(filedownload(df), unsafe_allow_html=True)

# Sidebar 
with st.sidebar.header('Upload your data'):
    uploaded_file = st.sidebar.file_uploader("Upload your input file", type=['txt'])
    st.sidebar.markdown("""
[Example input file](https://github.com/Gurudeeban-Selvaraj/advanced-therapeutics/blob/main/example.txt)
""")

if st.sidebar.button('Submit'):
    load_data = pd.read_table(uploaded_file, sep=' ', header=None)
    load_data.to_csv('molecule.smi', sep = '\t', header = False, index = False)

    st.header('**Original input data**')
    st.write(load_data)

    with st.spinner("Calculating descriptors..."):
        desc_calc()

    # Read in calculated descriptors and display the dataframe
    st.header('**Calculated molecular descriptors**')
    desc = pd.read_csv('descriptors_output.csv')
    st.write(desc)
    st.write(desc.shape)

    # Read descriptor list used in previously built model
    st.header('**Subset of descriptors from previously built models**')
    Xlist = list(pd.read_csv('descriptor_list.csv').columns)
    desc_subset = desc[Xlist]
    st.write(desc_subset)
    st.write(desc_subset.shape)

    # Apply trained model to make prediction on query compounds
    build_model(desc_subset)
    
else:
    st.info('Contact administrator!')

