diff --git a/config.json b/config.json index 3ee61ae7183..261627b9380 100644 --- a/config.json +++ b/config.json @@ -2294,9 +2294,27 @@ "prerequisites": [], "difficulty": 4, "status": "deprecated" + }, + { + "slug": "nucleotide-count", + "name": "Nucleotide Count", + "uuid": "105f25ec-7ce2-4797-893e-05e3792ebd91", + "practices": [], + "prerequisites": [], + "difficulty": 2, + "status": "deprecated" + }, + { + "slug": "parallel-letter-frequency", + "name": "Parallel Letter Frequency", + "uuid": "da03fca4-4606-48d8-9137-6e40396f7759", + "practices": [], + "prerequisites": [], + "difficulty": 3, + "status": "deprecated" } ], - "foregone": ["lens-person", "nucleotide-count", "parallel-letter-frequency"] + "foregone": ["lens-person"] }, "concepts": [ { diff --git a/exercises/practice/nucleotide-count/.docs/instructions.md b/exercises/practice/nucleotide-count/.docs/instructions.md new file mode 100644 index 00000000000..548d9ba5a5e --- /dev/null +++ b/exercises/practice/nucleotide-count/.docs/instructions.md @@ -0,0 +1,23 @@ +# Instructions + +Each of us inherits from our biological parents a set of chemical instructions known as DNA that influence how our bodies are constructed. +All known life depends on DNA! + +> Note: You do not need to understand anything about nucleotides or DNA to complete this exercise. + +DNA is a long chain of other chemicals and the most important are the four nucleotides, adenine, cytosine, guanine and thymine. +A single DNA chain can contain billions of these four nucleotides and the order in which they occur is important! +We call the order of these nucleotides in a bit of DNA a "DNA sequence". + +We represent a DNA sequence as an ordered collection of these four nucleotides and a common way to do that is with a string of characters such as "ATTACG" for a DNA sequence of 6 nucleotides. +'A' for adenine, 'C' for cytosine, 'G' for guanine, and 'T' for thymine. + +Given a string representing a DNA sequence, count how many of each nucleotide is present. +If the string contains characters that aren't A, C, G, or T then it is invalid and you should signal an error. + +For example: + +```text +"GATTACA" -> 'A': 3, 'C': 1, 'G': 1, 'T': 2 +"INVALID" -> error +``` diff --git a/exercises/practice/nucleotide-count/.meta/config.json b/exercises/practice/nucleotide-count/.meta/config.json new file mode 100644 index 00000000000..e0c108f7b88 --- /dev/null +++ b/exercises/practice/nucleotide-count/.meta/config.json @@ -0,0 +1,32 @@ +{ + "blurb": "Given a DNA string, compute how many times each nucleotide occurs in the string.", + "authors": [], + "contributors": [ + "behrtam", + "cmccandless", + "Dog", + "ikhadykin", + "kytrinyx", + "lowks", + "mostlybadfly", + "N-Parsons", + "Oniwa", + "orozcoadrian", + "pheanex", + "sjakobi", + "tqa236" + ], + "files": { + "solution": [ + "nucleotide_count.py" + ], + "test": [ + "nucleotide_count_test.py" + ], + "example": [ + ".meta/example.py" + ] + }, + "source": "The Calculating DNA Nucleotides_problem at Rosalind", + "source_url": "https://rosalind.info/problems/dna/" +} diff --git a/exercises/practice/nucleotide-count/.meta/example.py b/exercises/practice/nucleotide-count/.meta/example.py new file mode 100644 index 00000000000..e79a6a7ec7d --- /dev/null +++ b/exercises/practice/nucleotide-count/.meta/example.py @@ -0,0 +1,18 @@ +NUCLEOTIDES = 'ATCG' + + +def count(strand, abbreviation): + _validate(abbreviation) + return strand.count(abbreviation) + + +def nucleotide_counts(strand): + return { + abbr: strand.count(abbr) + for abbr in NUCLEOTIDES + } + + +def _validate(abbreviation): + if abbreviation not in NUCLEOTIDES: + raise ValueError(f'{abbreviation} is not a nucleotide.') diff --git a/exercises/practice/nucleotide-count/.meta/tests.toml b/exercises/practice/nucleotide-count/.meta/tests.toml new file mode 100644 index 00000000000..79b22f7a855 --- /dev/null +++ b/exercises/practice/nucleotide-count/.meta/tests.toml @@ -0,0 +1,18 @@ +# This is an auto-generated file. Regular comments will be removed when this +# file is regenerated. Regenerating will not touch any manually added keys, +# so comments can be added in a "comment" key. + +[3e5c30a8-87e2-4845-a815-a49671ade970] +description = "empty strand" + +[a0ea42a6-06d9-4ac6-828c-7ccaccf98fec] +description = "can count one nucleotide in single-character input" + +[eca0d565-ed8c-43e7-9033-6cefbf5115b5] +description = "strand with repeated nucleotide" + +[40a45eac-c83f-4740-901a-20b22d15a39f] +description = "strand with multiple nucleotides" + +[b4c47851-ee9e-4b0a-be70-a86e343bd851] +description = "strand with invalid nucleotides" diff --git a/exercises/practice/nucleotide-count/nucleotide_count.py b/exercises/practice/nucleotide-count/nucleotide_count.py new file mode 100644 index 00000000000..7f794acbfb3 --- /dev/null +++ b/exercises/practice/nucleotide-count/nucleotide_count.py @@ -0,0 +1,6 @@ +def count(strand, nucleotide): + pass + + +def nucleotide_counts(strand): + pass diff --git a/exercises/practice/nucleotide-count/nucleotide_count_test.py b/exercises/practice/nucleotide-count/nucleotide_count_test.py new file mode 100644 index 00000000000..bd5b5bddcb7 --- /dev/null +++ b/exercises/practice/nucleotide-count/nucleotide_count_test.py @@ -0,0 +1,46 @@ +"""Tests for the nucleotide-count exercise + +Implementation note: +The count function must raise a ValueError with a meaningful error message +in case of a bad argument. +""" +import unittest + +from nucleotide_count import count, nucleotide_counts + + +class NucleotideCountTest(unittest.TestCase): + def test_empty_dna_string_has_no_adenosine(self): + self.assertEqual(count('', 'A'), 0) + + def test_empty_dna_string_has_no_nucleotides(self): + expected = {'A': 0, 'T': 0, 'C': 0, 'G': 0} + self.assertEqual(nucleotide_counts(""), expected) + + def test_repetitive_cytidine_gets_counted(self): + self.assertEqual(count('CCCCC', 'C'), 5) + + def test_repetitive_sequence_has_only_guanosine(self): + expected = {'A': 0, 'T': 0, 'C': 0, 'G': 8} + self.assertEqual(nucleotide_counts('GGGGGGGG'), expected) + + def test_counts_only_thymidine(self): + self.assertEqual(count('GGGGGTAACCCGG', 'T'), 1) + + def test_validates_nucleotides(self): + with self.assertRaisesWithMessage(ValueError): + count("GACT", 'X') + + def test_counts_all_nucleotides(self): + dna = ('AGCTTTTCATTCTGACTGCAACGGGCAATATGTCT' + 'CTGTGTGGATTAAAAAAAGAGTGTCTGATAGCAGC') + expected = {'A': 20, 'T': 21, 'G': 17, 'C': 12} + self.assertEqual(nucleotide_counts(dna), expected) + + # Utility functions + def assertRaisesWithMessage(self, exception): + return self.assertRaisesRegex(exception, r".+") + + +if __name__ == '__main__': + unittest.main() diff --git a/exercises/practice/parallel-letter-frequency/.docs/instructions.md b/exercises/practice/parallel-letter-frequency/.docs/instructions.md new file mode 100644 index 00000000000..85abcf86a42 --- /dev/null +++ b/exercises/practice/parallel-letter-frequency/.docs/instructions.md @@ -0,0 +1,7 @@ +# Instructions + +Count the frequency of letters in texts using parallel computation. + +Parallelism is about doing things in parallel that can also be done sequentially. +A common example is counting the frequency of letters. +Create a function that returns the total frequency of each letter in a list of texts and that employs parallelism. diff --git a/exercises/practice/parallel-letter-frequency/.meta/config.json b/exercises/practice/parallel-letter-frequency/.meta/config.json new file mode 100644 index 00000000000..3945e3f2324 --- /dev/null +++ b/exercises/practice/parallel-letter-frequency/.meta/config.json @@ -0,0 +1,25 @@ +{ + "blurb": "Count the frequency of letters in texts using parallel computation.", + "authors": [ + "forgeRW" + ], + "contributors": [ + "behrtam", + "cmccandless", + "Dog", + "kytrinyx", + "N-Parsons", + "tqa236" + ], + "files": { + "solution": [ + "parallel_letter_frequency.py" + ], + "test": [ + "parallel_letter_frequency_test.py" + ], + "example": [ + ".meta/example.py" + ] + } +} diff --git a/exercises/practice/parallel-letter-frequency/.meta/example.py b/exercises/practice/parallel-letter-frequency/.meta/example.py new file mode 100644 index 00000000000..5a16fb31c17 --- /dev/null +++ b/exercises/practice/parallel-letter-frequency/.meta/example.py @@ -0,0 +1,51 @@ +# -*- coding: utf-8 -*- +from collections import Counter +from threading import Lock, Thread +from time import sleep +from queue import Queue + + +TOTAL_WORKERS = 3 # Maximum number of threads chosen arbitrarily + +class LetterCounter: + + def __init__(self): + self.lock = Lock() + self.value = Counter() + + def add_counter(self, counter_to_add): + self.lock.acquire() + try: + self.value = self.value + counter_to_add + finally: + self.lock.release() + + +def count_letters(queue_of_texts, letter_to_frequency, worker_id): + while not queue_of_texts.empty(): + sleep(worker_id + 1) + line_input = queue_of_texts.get() + if line_input is not None: + letters_in_line = Counter(idx for idx in line_input.lower() if idx.isalpha()) + letter_to_frequency.add_counter(letters_in_line) + queue_of_texts.task_done() + if line_input is None: + break + + +def calculate(list_of_texts): + queue_of_texts = Queue() + for line in list_of_texts: + queue_of_texts.put(line) + letter_to_frequency = LetterCounter() + threads = [] + for idx in range(TOTAL_WORKERS): + worker = Thread(target=count_letters, args=(queue_of_texts, letter_to_frequency, idx)) + worker.start() + threads.append(worker) + queue_of_texts.join() + for _ in range(TOTAL_WORKERS): + queue_of_texts.put(None) + for thread in threads: + thread.join() + return letter_to_frequency.value diff --git a/exercises/practice/parallel-letter-frequency/.meta/tests.toml b/exercises/practice/parallel-letter-frequency/.meta/tests.toml new file mode 100644 index 00000000000..6cf36e6fd2d --- /dev/null +++ b/exercises/practice/parallel-letter-frequency/.meta/tests.toml @@ -0,0 +1,62 @@ +# This is an auto-generated file. +# +# Regenerating this file via `configlet sync` will: +# - Recreate every `description` key/value pair +# - Recreate every `reimplements` key/value pair, where they exist in problem-specifications +# - Remove any `include = true` key/value pair (an omitted `include` key implies inclusion) +# - Preserve any other key/value pair +# +# As user-added comments (using the # character) will be removed when this file +# is regenerated, comments can be added via a `comment` key. + +[c054d642-c1fa-4234-8007-9339f2337886] +description = "no texts" +include = false + +[818031be-49dc-4675-b2f9-c4047f638a2a] +description = "one text with one letter" +include = false + +[c0b81d1b-940d-4cea-9f49-8445c69c17ae] +description = "one text with multiple letters" +include = false + +[708ff1e0-f14a-43fd-adb5-e76750dcf108] +description = "two texts with one letter" +include = false + +[1b5c28bb-4619-4c9d-8db9-a4bb9c3bdca0] +description = "two texts with multiple letters" +include = false + +[6366e2b8-b84c-4334-a047-03a00a656d63] +description = "ignore letter casing" +include = false + +[92ebcbb0-9181-4421-a784-f6f5aa79f75b] +description = "ignore whitespace" +include = false + +[bc5f4203-00ce-4acc-a5fa-f7b865376fd9] +description = "ignore punctuation" +include = false + +[68032b8b-346b-4389-a380-e397618f6831] +description = "ignore numbers" +include = false + +[aa9f97ac-3961-4af1-88e7-6efed1bfddfd] +description = "Unicode letters" +include = false + +[7b1da046-701b-41fc-813e-dcfb5ee51813] +description = "combination of lower- and uppercase letters, punctuation and white space" +include = false + +[4727f020-df62-4dcf-99b2-a6e58319cb4f] +description = "large texts" +include = false + +[adf8e57b-8e54-4483-b6b8-8b32c115884c] +description = "many small texts" +include = false diff --git a/exercises/practice/parallel-letter-frequency/parallel_letter_frequency.py b/exercises/practice/parallel-letter-frequency/parallel_letter_frequency.py new file mode 100644 index 00000000000..1b9e1f7f9cf --- /dev/null +++ b/exercises/practice/parallel-letter-frequency/parallel_letter_frequency.py @@ -0,0 +1,2 @@ +def calculate(text_input): + pass diff --git a/exercises/practice/parallel-letter-frequency/parallel_letter_frequency_test.py b/exercises/practice/parallel-letter-frequency/parallel_letter_frequency_test.py new file mode 100644 index 00000000000..23860c7f47b --- /dev/null +++ b/exercises/practice/parallel-letter-frequency/parallel_letter_frequency_test.py @@ -0,0 +1,61 @@ +# -*- coding: utf-8 -*- +from collections import Counter +import unittest + +from parallel_letter_frequency import calculate + + +class ParallelLetterFrequencyTest(unittest.TestCase): + def test_one_letter(self): + actual = calculate(['a']) + expected = {'a': 1} + self.assertDictEqual(actual, expected) + + def test_case_insensitivity(self): + actual = calculate(['aA']) + expected = {'a': 2} + self.assertDictEqual(actual, expected) + + def test_numbers(self): + actual = calculate(['012', '345', '6789']) + expected = {} + self.assertDictEqual(actual, expected) + + def test_punctuations(self): + actual = calculate([r'[]\;,', './{}|', ':"<>?']) + expected = {} + self.assertDictEqual(actual, expected) + + def test_whitespaces(self): + actual = calculate([' ', '\t ', '\n\n']) + expected = {} + self.assertDictEqual(actual, expected) + + def test_repeated_string_with_known_frequencies(self): + letter_frequency = 3 + text_input = 'abc\n' * letter_frequency + actual = calculate(text_input.split('\n')) + expected = {'a': letter_frequency, 'b': letter_frequency, + 'c': letter_frequency} + self.assertDictEqual(actual, expected) + + def test_multiline_text(self): + text_input = "3 Quotes from Excerism Homepage:\n" + \ + "\tOne moment you feel like you're\n" + \ + "getting it. The next moment you're\n" + \ + "stuck.\n" + \ + "\tYou know what it’s like to be fluent.\n" + \ + "Suddenly you’re feeling incompetent\n" + \ + "and clumsy.\n" + \ + "\tHaphazard, convoluted code is\n" + \ + "infuriating, not to mention costly. That\n" + \ + "slapdash explosion of complexity is an\n" + \ + "expensive yak shave waiting to\n" + \ + "happen." + actual = calculate(text_input.split('\n')) + expected = Counter([x for x in text_input.lower() if x.isalpha()]) + self.assertDictEqual(actual, expected) + + +if __name__ == '__main__': + unittest.main()