Not a member of Pastebin yet?
Sign Up,
it unlocks many cool features!
- import pandas as pd
- import math
- import os
- import numpy.random as rd
- class DNA():
- '''This is a class for DNA sequences and their properties. All that you needed is pandas and an excel or csv sheet
- formatted such that the first column is the sequence name and the second column is the sequence.
- Ex:
- Sequence 1, ATGCGCGATCGATCGATGGACTTAGCAAATCGCT
- Sequence 2, ACTACTACTCGCATGCATACGGACGACTGACTG
- ...
- Calculations are taken from http://biotools.nubic.northwestern.edu/OligoCalc.html#helpbasic'''
- def __init__(self,name,sequence):
- # Basic sequence information
- self.name = name
- self.sequence = sequence
- self.baselength = len(sequence)
- # Count the number of the respective nucleotides
- self.num_As = sequence.count('A')
- self.num_Ts = sequence.count('T')
- self.num_Gs = sequence.count('G')
- self.num_Cs = sequence.count('C')
- # Basic Salt information units = M
- self.sodium = 0.050
- self.magnesium = 0
- def GC_content(self):
- '''Simple percentage of G and C out of total sequence'''
- gc = ((self.num_Cs + self.num_Gs) / self.baselength)*100
- return int(gc)
- def AnHydro_MolWeight(self):
- '''Calculation: (Anhydrous Molecular Weight = (An x 313.21) + (Tn x 304.2) + (Cn x 289.18) + (Gn x 329.21) - 61.96)
- Taken from http://biotools.nubic.northwestern.edu/OligoCalc.html#helpbasic'''
- mol = (self.num_As * 313.21) + (self.num_Ts * 304.2) + (self.num_Cs * 289.18) + (self.num_Gs * 329.21) - 61.96
- return int(mol)
- def Melting_Temp(self):
- '''Calculation: (Tm= 100.5 + (41 * (yG+zC)/(wA+xT+yG+zC)) - (820/(wA+xT+yG+zC)) + 16.6*log10([Na+]))
- Taken from http://biotools.nubic.northwestern.edu/OligoCalc.html#helpbasic'''
- tm = Tm= 100.5 + (41 * (self.num_Gs+self.num_Cs)/(self.num_As+self.num_Ts+self.num_Gs+self.num_Cs)) - (820/(self.num_As+self.num_Ts+self.num_Gs+self.num_Cs)) + 16.6*math.log10(self.sodium)
- return int(tm)
- def Estimate_Ext_Coeff(self):
- '''Calculation: (E260 = ( (nA × 15.4) + (nC × 7.4) + (nG × 11.5) + (nT × 8.7) ) × 0.9 × 1000)
- Taken from https://www.atdbio.com/content/1/Ultraviolet-absorbance-of-oligonucleotides#Calculation-of-extinction-coefficient'''
- e = ((self.num_As * 15.4) + (self.num_Cs * 7.4) + (self.num_Gs * 11.5) + (self.num_Ts * 8.7)) * 0.9 * 1000
- return int(e)
- class Optical():
- '''Next is a class made for calculating concentration from absorbance'''
- def __init__(self,absorbance, extinction):
- self.Abs = absorbance
- self.path = 1 # path is either 1 or 0.3 cm from mm
- self.ext_coeff = extinction
- self.dilution = 100
- def Concentration(self):
- conc = ((((self.Abs)/(self.path*self.ext_coeff))*self.dilution)/1E-6)
- return conc
- # read the data into a DataFrame
- file = 'dna_seq.csv' #input('Please enter sequence file name:')
- df = pd.read_csv(str(file))
- Info_Table = pd.DataFrame()
- # Loop through the given sequences
- for i in range(0,len(df["Seq"])):
- # add some random absorbance data
- df.loc[i,'Absorbance'] = rd.random()
- # call the DNA class
- dna = DNA(df.loc[i,"Name"],df.loc[i,"Seq"])
- # Once called the salts can be changed to adjust the melting temp
- dna.sodium = 0.050
- # Here trying to calculate the concentration
- optics = Optical(df.loc[i,'Absorbance'],dna.Estimate_Ext_Coeff())
- # print the whatever dna properties...
- Info_Table.loc[i,'Sequence Name'] = dna.name
- Info_Table.loc[i,'Sequence'] = dna.sequence
- Info_Table.loc[i, "Molecular Weight (g/mol)"] = dna.AnHydro_MolWeight()
- Info_Table.loc[i,"Melting Temp (C)"] = dna.Melting_Temp()
- Info_Table.loc[i,"GC Content (%)"] = dna.GC_content()
- Info_Table.loc[i,"Length"] = dna.baselength
- Info_Table.loc[i,'Estimate Exctinction (M-1 cm-1)'] = dna.Estimate_Ext_Coeff()
- Info_Table.loc[i,'Concentration (mM)'] = round(optics.Concentration(),3)
- Info_Table.loc[i,'Absorbance'] = round(optics.Abs,3)
- # pick an export file
- export = 'data.csv'
- # Export to file, make sure the method and the filetype match
- Info_Table.to_csv(export)
- # Print the file name and the PATH
- print('--Results saved as:(' + export + ') --In path:' + os.getcwd())
Advertisement
Add Comment
Please, Sign In to add comment