hello.py Code

name = input("What is your name? ")
print("hello,", name)

dna.py Code

import csv
import sys

def main():

    # Check for command-line usage
    if len(sys.argv) != 3:
        print("Usage: dna.py databases/[filename] sequences/[filename]")
        sys.exit(1)

    # Read database file into a variable
    database_list = list()
    database_file = open(sys.argv[1], "r")
    # csv file saved as dictionary, each with column name being the key (header)
    datadb = csv.DictReader(database_file)
    # delievers a list like ['AGATC', 'AATG', 'TATC'...], given the fieldnames in the csv (skips the name column)
    STRs = datadb.fieldnames[1:]
    for d in datadb:
        database_list.append(d)

    database_file.close()

    # Read DNA sequence file into a variable
    dna_file = open(sys.argv[2], "r")
    datadna = dna_file.read()

    # Find longest match of each STR in DNA sequence
    STR_dict = dict()
    for STR in STRs:
        # print(f"STR key {STR} its value:", longest_match(datadna, STR))
        STR_dict[STR] = longest_match(datadna, STR)  # storing key's STR with its value

    dna_file.close()

    # Check database for matching profiles
    for data in database_list:  # getting the data of each dictionary
        personSTR = []         # making an empty list for later comparsions / clearing the list
        tmplist = list(data.values())[1:]  # slicing the name header
        for p in tmplist:
            personSTR.append(int(p))  # converting each value to int and adding this value to the personSTR list
        # print(f"{personSTR} (personSTR) == {list(STR_dict.values())} (STR_dict.values()) ")
        if personSTR == list(STR_dict.values()): #comparing the values for match detections
            print(data["name"])
            break
    else:
        print("No match")

    return

def longest_match(sequence, subsequence):
    """Returns length of longest run of subsequence in sequence."""

    # Initialize variables
    longest_run = 0
    subsequence_length = len(subsequence)
    sequence_length = len(sequence)

    # Check each character in sequence for most consecutive runs of subsequence
    for i in range(sequence_length):

        # Initialize count of consecutive runs
        count = 0

        # Check for a subsequence match in a "substring" (a subset of characters) within sequence
        # If a match, move substring to next potential match in sequence
        # Continue moving substring and checking for matches until out of consecutive matches
        while True:

            # Adjust substring start and end
            start = i + count * subsequence_length
            end = start + subsequence_length

            # If there is a match in the substring
            if sequence[start:end] == subsequence:
                count += 1

            # If there is no match in the substring
            else:
                break

        # Update most consecutive matches found
        longest_run = max(longest_run, count)

    # After checking for runs at each character in sequence, return longest run found
    return longest_run

main()