|
|
|
|
|
|
| import requests
|
| import csv
|
| import pandas as pd
|
| from bs4 import BeautifulSoup
|
| import os
|
| import re |
| from _bootstrap import DATA_DIR, use_project_root |
|
|
| use_project_root() |
|
|
| def getAllRecords():
|
| print("Calling All structures page ... ")
|
| url = requests.get(baseUrl+allRecordsEndPoint)
|
| htmltext = url.text
|
| data = pd.read_html(htmltext)[0]
|
| print(str(data.count()["PDB"]) + " PDBs is loaded now, lets save them....")
|
| return data
|
|
|
| def writeDataToCSV(recordType, data):
|
| fvsNumber = int(data["basic_data"]["Number of Fvs"])
|
| empty_6_items = ["","","","","",""]
|
| empty_3_items = ["","",""]
|
| file_dir = str(DATA_DIR / recordType / (str(fvsNumber) + '.csv')) |
|
|
| if recordType == "nano":
|
| headers = ['pdb', 'Hchain', 'Hchain_sequence', 'Heavy subgroup',
|
| 'Species', 'In complex?','scFv?','Has constant domain?',
|
| 'CDRH1', 'CDRH2','CDRH3',
|
| 'Antigen chain_1', 'Antigen chain_2', 'Antigen chain_3',
|
| 'antigen_type','antigen_name','Antigen species',
|
| 'Antigen sequence_1', 'Antigen sequence_2','Antigen sequence_3',
|
| 'HL', 'HC1', 'HC2','LC1','LC2','dc','Method','Resolution','Has constant region'
|
| ]
|
| else:
|
| headers = ['pdb', 'Hchain', 'Lchain', 'Hchain_sequence','Lchain_sequence', 'Heavy subgroup',
|
| 'Light subgroup', 'Species', 'In complex?','scFv?','Has constant domain?',
|
| 'CDRH1', 'CDRH2','CDRH3','CDRL1', 'CDRL2','CDRL3',
|
| 'Antigen chain_1', 'Antigen chain_2', 'Antigen chain_3',
|
| 'antigen_type','antigen_name','Antigen species',
|
| 'Antigen sequence_1', 'Antigen sequence_2','Antigen sequence_3',
|
| 'HL', 'HC1', 'HC2','LC1','LC2','dc','Method','Resolution','Light chain type','Has constant region'
|
| ]
|
|
|
| with open(file_dir, 'a', encoding='UTF8', newline='') as b:
|
| wr = csv.writer(b)
|
| is_empty = os.stat(file_dir).st_size == 0
|
| if is_empty:
|
| wr.writerow(headers)
|
|
|
| for i in range(fvsNumber):
|
| cdrSec = data["fv_"+str(i)].get("CDR Sequences (chothia definition)")
|
| antigenDetailsSec = data["fv_"+str(i)].get("Antigen Details")
|
| orientationSec = data["fv_"+str(i)].get("Orientation Angles (from ABangle)")
|
| if recordType == "nano":
|
| rowData = [
|
| data["basic_data"].get("PDB"),
|
| data["fv_"+str(i)]["Fv Details"].get("Heavy chain"),
|
| data["fv_"+str(i)].get("heavyChainSeq"),
|
| data["fv_"+str(i)]["Fv Details"].get("Heavy subgroup"),
|
| data["fv_"+str(i)]["Fv Details"].get("Species"),
|
| data["fv_"+str(i)]["Fv Details"].get("In complex?"),
|
| data["fv_"+str(i)]["Fv Details"].get("scFv?"),
|
| data["fv_"+str(i)]["Fv Details"].get("Has constant domain?"),
|
| ]
|
| else:
|
| rowData = [
|
| data["basic_data"].get("PDB"),
|
| data["fv_"+str(i)]["Fv Details"].get("Heavy chain"),
|
| data["fv_"+str(i)]["Fv Details"].get("Light chain"),
|
| data["fv_"+str(i)].get("heavyChainSeq"),
|
| data["fv_"+str(i)].get("lightChainSeq"),
|
| data["fv_"+str(i)]["Fv Details"].get("Heavy subgroup"),
|
| data["fv_"+str(i)]["Fv Details"].get("Light subgroup"),
|
| data["fv_"+str(i)]["Fv Details"].get("Species"),
|
| data["fv_"+str(i)]["Fv Details"].get("In complex?"),
|
| data["fv_"+str(i)]["Fv Details"].get("scFv?"),
|
| data["fv_"+str(i)]["Fv Details"].get("Has constant domain?"),
|
| ]
|
| antiGenChain1 = antiGenChain2 = antiGenChain3 = ""
|
| antiGenSeq1 = antiGenSeq2 = antiGenSeq3 = ""
|
| if antigenDetailsSec != None:
|
| antiGenString = data["fv_"+str(i)]["Antigen Details"]["Antigen chains"]
|
| if antiGenString != None:
|
| antiGenChains = antiGenString.split(",")
|
| if int(len(antiGenChains)) >= 1:
|
| antiGenChain1 = antiGenChains[0]
|
| if int(len(antiGenChains)) >= 2:
|
| antiGenChain2 = antiGenChains[1]
|
| if int(len(antiGenChains)) >= 3:
|
| antiGenChain3 = antiGenChains[2]
|
|
|
| antiGenstring = data["fv_"+str(i)]["Antigen Details"]["Antigen sequence"]
|
| if antiGenstring != None:
|
| antiGenSeqs = antiGenstring.split("/")
|
| if int(len(antiGenSeqs)) >= 1:
|
| antiGenSeq1 = antiGenSeqs[0]
|
| if int(len(antiGenSeqs)) >= 2:
|
| antiGenSeq2 = antiGenSeqs[1]
|
| if int(len(antiGenSeqs)) >= 3:
|
| antiGenSeq3 = antiGenSeqs[2]
|
|
|
| if cdrSec != None and recordType != "nano":
|
| rowData.append(data["fv_"+str(i)]["CDR Sequences (chothia definition)"].get("CDRH1"))
|
| rowData.append(data["fv_"+str(i)]["CDR Sequences (chothia definition)"].get("CDRH2"))
|
| rowData.append(data["fv_"+str(i)]["CDR Sequences (chothia definition)"].get("CDRH3"))
|
| rowData.append(data["fv_"+str(i)]["CDR Sequences (chothia definition)"].get("CDRL1"))
|
| rowData.append(data["fv_"+str(i)]["CDR Sequences (chothia definition)"].get("CDRL2"))
|
| rowData.append(data["fv_"+str(i)]["CDR Sequences (chothia definition)"].get("CDRL3"))
|
| if cdrSec != None and recordType == "nano":
|
| rowData.append(data["fv_"+str(i)]["CDR Sequences (chothia definition)"].get("CDRH1"))
|
| rowData.append(data["fv_"+str(i)]["CDR Sequences (chothia definition)"].get("CDRH2"))
|
| rowData.append(data["fv_"+str(i)]["CDR Sequences (chothia definition)"].get("CDRH3"))
|
|
|
| if cdrSec == None and recordType == "nano":
|
| rowData.extend(empty_3_items)
|
| if cdrSec == None and recordType != "nano":
|
| rowData.extend(empty_6_items)
|
|
|
| rowData.append(antiGenChain1)
|
| rowData.append(antiGenChain2)
|
| rowData.append(antiGenChain3)
|
|
|
| if antigenDetailsSec != None:
|
| rowData.append(data["fv_"+str(i)]["Antigen Details"].get("Antigen type"))
|
| rowData.append(data["fv_"+str(i)]["Antigen Details"].get("Antigen name"))
|
| rowData.append(data["fv_"+str(i)]["Antigen Details"].get("Antigen species"))
|
| else:
|
| rowData.extend(empty_3_items)
|
|
|
| rowData.append(antiGenSeq1)
|
| rowData.append(antiGenSeq2)
|
| rowData.append(antiGenSeq3)
|
|
|
| if orientationSec != None:
|
| rowData.append(removeUnit(data["fv_"+str(i)]["Orientation Angles (from ABangle)"].get("HL")))
|
| rowData.append(removeUnit(data["fv_"+str(i)]["Orientation Angles (from ABangle)"].get("HC1")))
|
| rowData.append(removeUnit(data["fv_"+str(i)]["Orientation Angles (from ABangle)"].get("HC2")))
|
| rowData.append(removeUnit(data["fv_"+str(i)]["Orientation Angles (from ABangle)"].get("LC1")))
|
| rowData.append(removeUnit(data["fv_"+str(i)]["Orientation Angles (from ABangle)"].get("LC2")))
|
| rowData.append(removeUnit(data["fv_"+str(i)]["Orientation Angles (from ABangle)"].get("dc")))
|
| else:
|
| rowData.extend(empty_6_items)
|
|
|
| rowData.append(data["basic_data"].get("Method"))
|
|
|
| rowData.append(removeUnit(data["basic_data"].get("Resolution")))
|
|
|
| if recordType != "nano":
|
| rowData.append(data["basic_data"].get("Light chain type"))
|
|
|
| rowData.append(data["basic_data"].get("Has constant region"))
|
|
|
|
|
| with open(file_dir, 'a', encoding='UTF8', newline='') as body:
|
| writer = csv.writer(body)
|
| writer.writerow(rowData)
|
|
|
| def getFvData(pdbFvDiv):
|
|
|
| fvData = {}
|
| seqTableNumber = 0
|
| for table in pdbFvDiv.find_all('table', class_='table-results'):
|
| tableData = {}
|
| for i, row in enumerate(table.find_all('tr')):
|
| if i == 0:
|
| header = row.find('th').text.strip()
|
| else:
|
| tds = [row.findAll('td')]
|
| for td in tds :
|
| tableData[td[0].string] = td[1].string
|
| fvData[header] = tableData
|
|
|
| for table in pdbFvDiv.find_all('table', class_='table-alignment'):
|
| for i, row in enumerate(table.find_all('tr')):
|
| if i == 0:
|
|
|
| continue
|
| else:
|
| if seqTableNumber == 0:
|
| heavyChainSeq = row.find_all("td")
|
| heavyChainSeqData = ""
|
| for chain in heavyChainSeq:
|
| heavyChainSeqData += chain.text
|
| fvData["heavyChainSeq"] = heavyChainSeqData
|
| else:
|
| lightChainSeq = row.find_all("td")
|
| lightChainSeqData = ""
|
| for chain in lightChainSeq:
|
| lightChainSeqData += chain.text
|
| fvData["lightChainSeq"] = lightChainSeqData
|
| seqTableNumber += 1
|
| return fvData
|
|
|
| def getRecordType(data):
|
| heavyChainCounter = 0
|
| lightChainCounter = 0
|
|
|
|
|
|
|
| i = -1
|
| for fvs in data.items():
|
| i+=1
|
| if i == 0:
|
|
|
| continue
|
| for fv in fvs:
|
| if isinstance(fv, dict):
|
| if fv['Fv Details']['Heavy chain'] != None:
|
| heavyChainCounter+=1
|
| if fv['Fv Details']['Light chain'] != None:
|
| lightChainCounter+=1
|
|
|
| if heavyChainCounter == lightChainCounter == int(data["basic_data"]["Number of Fvs"]):
|
| global antiBodiesCounter
|
| antiBodiesCounter+=1
|
| return "anti"
|
| elif heavyChainCounter == int(data["basic_data"]["Number of Fvs"]) and lightChainCounter == 0:
|
| global nanoBodiesCounter
|
| nanoBodiesCounter+=1
|
| return "nano"
|
| else:
|
| global combinedBodiesCounter
|
| combinedBodiesCounter+=1
|
| return "combined"
|
|
|
| def removeUnit(value):
|
| if value == None:
|
| return value
|
| else:
|
| newValue = re.findall(r"[-+]?(?:\d*\.*\d+)", value)
|
| return newValue[0]
|
|
|
| os.makedirs('./conf/data/nano')
|
| os.makedirs('./conf/data/anti')
|
| os.makedirs('./conf/data/combined')
|
|
|
| baseUrl = "https://opig.stats.ox.ac.uk/"
|
| allRecordsEndPoint = "webapps/sabdab-sabpred/sabdab/nanobodies/?all=true"
|
| antiBodiesCounter = 0
|
| nanoBodiesCounter = 0
|
| combinedBodiesCounter = 0
|
| failedBodiesCounter = 0
|
| recordsCount = 0
|
| allRecords = getAllRecords()
|
| for index, row in allRecords.iterrows():
|
| try:
|
| recordsCount+=1
|
| print("Record number "+str(recordsCount) + ": Calling "+row['PDB']+" page to be saved ... ")
|
| pdbEndpoint = "webapps/sabdab-sabpred/sabdab/structureviewer/?pdb="+row['PDB']
|
|
|
| url = requests.get(baseUrl+pdbEndpoint)
|
| soup = BeautifulSoup(url.content, "html.parser")
|
|
|
| pdbDetailsTableDiv = soup.find(id="details")
|
| pdbDetailsTableRows = pdbDetailsTableDiv.find('table').find_all('tr')
|
| wholeData = {}
|
| tds = [row.findAll('td') for row in pdbDetailsTableRows]
|
| wholeData["basic_data"] = { td[0].string: td[1].string for td in tds }
|
|
|
| for fvIndex in range(int(wholeData["basic_data"]["Number of Fvs"])):
|
|
|
| collapseDivId = "collapse_"+str(fvIndex)
|
| pdbFvDiv = soup.find(id=collapseDivId)
|
|
|
| fvData = getFvData(pdbFvDiv)
|
| wholeData["fv_"+str(fvIndex)] = fvData
|
|
|
| recordType = getRecordType(wholeData)
|
| writeDataToCSV(recordType, wholeData)
|
| except Exception as e:
|
| with open('errors.txt', 'a') as log:
|
| log.write(row['PDB']+ " falied to be loaded > "+str(e))
|
| failedBodiesCounter+=1
|
| log.write('\n')
|
| print(row['PDB']+" Failed, Fail number: "+str(failedBodiesCounter)+ " till now")
|
| continue
|
| print("I am done with follwing scraps")
|
| print("Nanobodies counter is " + str(nanoBodiesCounter))
|
| print("Antibodies counter is " + str(antiBodiesCounter))
|
| print("Combined counter is " + str(combinedBodiesCounter))
|
| print("Total fails counter is " + str(failedBodiesCounter))
|
|
|
|
|