hello.py Code
name = input("What is your name? ")
print("hello,", name)
dna.py Code
import csv
import sys
def main():
# Check for command-line usage
if len(sys.argv) != 3:
print("Usage: dna.py databases/[filename] sequences/[filename]")
sys.exit(1)
# Read database file into a variable
database_list = list()
database_file = open(sys.argv[1], "r")
# csv file saved as dictionary, each with column name being the key (header)
datadb = csv.DictReader(database_file)
# delievers a list like ['AGATC', 'AATG', 'TATC'...], given the fieldnames in the csv (skips the name column)
STRs = datadb.fieldnames[1:]
for d in datadb:
database_list.append(d)
database_file.close()
# Read DNA sequence file into a variable
dna_file = open(sys.argv[2], "r")
datadna = dna_file.read()
# Find longest match of each STR in DNA sequence
STR_dict = dict()
for STR in STRs:
# print(f"STR key {STR} its value:", longest_match(datadna, STR))
STR_dict[STR] = longest_match(datadna, STR) # storing key's STR with its value
dna_file.close()
# Check database for matching profiles
for data in database_list: # getting the data of each dictionary
personSTR = [] # making an empty list for later comparsions / clearing the list
tmplist = list(data.values())[1:] # slicing the name header
for p in tmplist:
personSTR.append(int(p)) # converting each value to int and adding this value to the personSTR list
# print(f"{personSTR} (personSTR) == {list(STR_dict.values())} (STR_dict.values()) ")
if personSTR == list(STR_dict.values()): #comparing the values for match detections
print(data["name"])
break
else:
print("No match")
return
def longest_match(sequence, subsequence):
"""Returns length of longest run of subsequence in sequence."""
# Initialize variables
longest_run = 0
subsequence_length = len(subsequence)
sequence_length = len(sequence)
# Check each character in sequence for most consecutive runs of subsequence
for i in range(sequence_length):
# Initialize count of consecutive runs
count = 0
# Check for a subsequence match in a "substring" (a subset of characters) within sequence
# If a match, move substring to next potential match in sequence
# Continue moving substring and checking for matches until out of consecutive matches
while True:
# Adjust substring start and end
start = i + count * subsequence_length
end = start + subsequence_length
# If there is a match in the substring
if sequence[start:end] == subsequence:
count += 1
# If there is no match in the substring
else:
break
# Update most consecutive matches found
longest_run = max(longest_run, count)
# After checking for runs at each character in sequence, return longest run found
return longest_run
main()