Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 19 additions & 1 deletion config.json
Original file line number Diff line number Diff line change
Expand Up @@ -2294,9 +2294,27 @@
"prerequisites": [],
"difficulty": 4,
"status": "deprecated"
},
{
"slug": "nucleotide-count",
"name": "Nucleotide Count",
"uuid": "105f25ec-7ce2-4797-893e-05e3792ebd91",
"practices": [],
"prerequisites": [],
"difficulty": 2,
"status": "deprecated"
},
{
"slug": "parallel-letter-frequency",
"name": "Parallel Letter Frequency",
"uuid": "da03fca4-4606-48d8-9137-6e40396f7759",
"practices": [],
"prerequisites": [],
"difficulty": 3,
"status": "deprecated"
}
],
"foregone": ["lens-person", "nucleotide-count", "parallel-letter-frequency"]
"foregone": ["lens-person"]
},
"concepts": [
{
Expand Down
23 changes: 23 additions & 0 deletions exercises/practice/nucleotide-count/.docs/instructions.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
# Instructions

Each of us inherits from our biological parents a set of chemical instructions known as DNA that influence how our bodies are constructed.
All known life depends on DNA!

> Note: You do not need to understand anything about nucleotides or DNA to complete this exercise.

DNA is a long chain of other chemicals and the most important are the four nucleotides, adenine, cytosine, guanine and thymine.
A single DNA chain can contain billions of these four nucleotides and the order in which they occur is important!
We call the order of these nucleotides in a bit of DNA a "DNA sequence".

We represent a DNA sequence as an ordered collection of these four nucleotides and a common way to do that is with a string of characters such as "ATTACG" for a DNA sequence of 6 nucleotides.
'A' for adenine, 'C' for cytosine, 'G' for guanine, and 'T' for thymine.

Given a string representing a DNA sequence, count how many of each nucleotide is present.
If the string contains characters that aren't A, C, G, or T then it is invalid and you should signal an error.

For example:

```text
"GATTACA" -> 'A': 3, 'C': 1, 'G': 1, 'T': 2
"INVALID" -> error
```
32 changes: 32 additions & 0 deletions exercises/practice/nucleotide-count/.meta/config.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,32 @@
{
"blurb": "Given a DNA string, compute how many times each nucleotide occurs in the string.",
"authors": [],
"contributors": [
"behrtam",
"cmccandless",
"Dog",
"ikhadykin",
"kytrinyx",
"lowks",
"mostlybadfly",
"N-Parsons",
"Oniwa",
"orozcoadrian",
"pheanex",
"sjakobi",
"tqa236"
],
"files": {
"solution": [
"nucleotide_count.py"
],
"test": [
"nucleotide_count_test.py"
],
"example": [
".meta/example.py"
]
},
"source": "The Calculating DNA Nucleotides_problem at Rosalind",
"source_url": "https://rosalind.info/problems/dna/"
}
18 changes: 18 additions & 0 deletions exercises/practice/nucleotide-count/.meta/example.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
NUCLEOTIDES = 'ATCG'


def count(strand, abbreviation):
_validate(abbreviation)
return strand.count(abbreviation)


def nucleotide_counts(strand):
return {
abbr: strand.count(abbr)
for abbr in NUCLEOTIDES
}


def _validate(abbreviation):
if abbreviation not in NUCLEOTIDES:
raise ValueError(f'{abbreviation} is not a nucleotide.')
18 changes: 18 additions & 0 deletions exercises/practice/nucleotide-count/.meta/tests.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
# This is an auto-generated file. Regular comments will be removed when this
# file is regenerated. Regenerating will not touch any manually added keys,
# so comments can be added in a "comment" key.

[3e5c30a8-87e2-4845-a815-a49671ade970]
description = "empty strand"

[a0ea42a6-06d9-4ac6-828c-7ccaccf98fec]
description = "can count one nucleotide in single-character input"

[eca0d565-ed8c-43e7-9033-6cefbf5115b5]
description = "strand with repeated nucleotide"

[40a45eac-c83f-4740-901a-20b22d15a39f]
description = "strand with multiple nucleotides"

[b4c47851-ee9e-4b0a-be70-a86e343bd851]
description = "strand with invalid nucleotides"
6 changes: 6 additions & 0 deletions exercises/practice/nucleotide-count/nucleotide_count.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
def count(strand, nucleotide):
pass


def nucleotide_counts(strand):
pass
46 changes: 46 additions & 0 deletions exercises/practice/nucleotide-count/nucleotide_count_test.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
"""Tests for the nucleotide-count exercise

Implementation note:
The count function must raise a ValueError with a meaningful error message
in case of a bad argument.
"""
import unittest

from nucleotide_count import count, nucleotide_counts


class NucleotideCountTest(unittest.TestCase):
def test_empty_dna_string_has_no_adenosine(self):
self.assertEqual(count('', 'A'), 0)

def test_empty_dna_string_has_no_nucleotides(self):
expected = {'A': 0, 'T': 0, 'C': 0, 'G': 0}
self.assertEqual(nucleotide_counts(""), expected)

def test_repetitive_cytidine_gets_counted(self):
self.assertEqual(count('CCCCC', 'C'), 5)

def test_repetitive_sequence_has_only_guanosine(self):
expected = {'A': 0, 'T': 0, 'C': 0, 'G': 8}
self.assertEqual(nucleotide_counts('GGGGGGGG'), expected)

def test_counts_only_thymidine(self):
self.assertEqual(count('GGGGGTAACCCGG', 'T'), 1)

def test_validates_nucleotides(self):
with self.assertRaisesWithMessage(ValueError):
count("GACT", 'X')

def test_counts_all_nucleotides(self):
dna = ('AGCTTTTCATTCTGACTGCAACGGGCAATATGTCT'
'CTGTGTGGATTAAAAAAAGAGTGTCTGATAGCAGC')
expected = {'A': 20, 'T': 21, 'G': 17, 'C': 12}
self.assertEqual(nucleotide_counts(dna), expected)

# Utility functions
def assertRaisesWithMessage(self, exception):
return self.assertRaisesRegex(exception, r".+")


if __name__ == '__main__':
unittest.main()
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
# Instructions

Count the frequency of letters in texts using parallel computation.

Parallelism is about doing things in parallel that can also be done sequentially.
A common example is counting the frequency of letters.
Create a function that returns the total frequency of each letter in a list of texts and that employs parallelism.
25 changes: 25 additions & 0 deletions exercises/practice/parallel-letter-frequency/.meta/config.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
{
"blurb": "Count the frequency of letters in texts using parallel computation.",
"authors": [
"forgeRW"
],
"contributors": [
"behrtam",
"cmccandless",
"Dog",
"kytrinyx",
"N-Parsons",
"tqa236"
],
"files": {
"solution": [
"parallel_letter_frequency.py"
],
"test": [
"parallel_letter_frequency_test.py"
],
"example": [
".meta/example.py"
]
}
}
51 changes: 51 additions & 0 deletions exercises/practice/parallel-letter-frequency/.meta/example.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
# -*- coding: utf-8 -*-
from collections import Counter
from threading import Lock, Thread
from time import sleep
from queue import Queue


TOTAL_WORKERS = 3 # Maximum number of threads chosen arbitrarily

class LetterCounter:

def __init__(self):
self.lock = Lock()
self.value = Counter()

def add_counter(self, counter_to_add):
self.lock.acquire()
try:
self.value = self.value + counter_to_add
finally:
self.lock.release()


def count_letters(queue_of_texts, letter_to_frequency, worker_id):
while not queue_of_texts.empty():
sleep(worker_id + 1)
line_input = queue_of_texts.get()
if line_input is not None:
letters_in_line = Counter(idx for idx in line_input.lower() if idx.isalpha())
letter_to_frequency.add_counter(letters_in_line)
queue_of_texts.task_done()
if line_input is None:
break


def calculate(list_of_texts):
queue_of_texts = Queue()
for line in list_of_texts:
queue_of_texts.put(line)
letter_to_frequency = LetterCounter()
threads = []
for idx in range(TOTAL_WORKERS):
worker = Thread(target=count_letters, args=(queue_of_texts, letter_to_frequency, idx))
worker.start()
threads.append(worker)
queue_of_texts.join()
for _ in range(TOTAL_WORKERS):
queue_of_texts.put(None)
for thread in threads:
thread.join()
return letter_to_frequency.value
62 changes: 62 additions & 0 deletions exercises/practice/parallel-letter-frequency/.meta/tests.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,62 @@
# This is an auto-generated file.
#
# Regenerating this file via `configlet sync` will:
# - Recreate every `description` key/value pair
# - Recreate every `reimplements` key/value pair, where they exist in problem-specifications
# - Remove any `include = true` key/value pair (an omitted `include` key implies inclusion)
# - Preserve any other key/value pair
#
# As user-added comments (using the # character) will be removed when this file
# is regenerated, comments can be added via a `comment` key.

[c054d642-c1fa-4234-8007-9339f2337886]
description = "no texts"
include = false

[818031be-49dc-4675-b2f9-c4047f638a2a]
description = "one text with one letter"
include = false

[c0b81d1b-940d-4cea-9f49-8445c69c17ae]
description = "one text with multiple letters"
include = false

[708ff1e0-f14a-43fd-adb5-e76750dcf108]
description = "two texts with one letter"
include = false

[1b5c28bb-4619-4c9d-8db9-a4bb9c3bdca0]
description = "two texts with multiple letters"
include = false

[6366e2b8-b84c-4334-a047-03a00a656d63]
description = "ignore letter casing"
include = false

[92ebcbb0-9181-4421-a784-f6f5aa79f75b]
description = "ignore whitespace"
include = false

[bc5f4203-00ce-4acc-a5fa-f7b865376fd9]
description = "ignore punctuation"
include = false

[68032b8b-346b-4389-a380-e397618f6831]
description = "ignore numbers"
include = false

[aa9f97ac-3961-4af1-88e7-6efed1bfddfd]
description = "Unicode letters"
include = false

[7b1da046-701b-41fc-813e-dcfb5ee51813]
description = "combination of lower- and uppercase letters, punctuation and white space"
include = false

[4727f020-df62-4dcf-99b2-a6e58319cb4f]
description = "large texts"
include = false

[adf8e57b-8e54-4483-b6b8-8b32c115884c]
description = "many small texts"
include = false
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
def calculate(text_input):
pass
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
# -*- coding: utf-8 -*-
from collections import Counter
import unittest

from parallel_letter_frequency import calculate


class ParallelLetterFrequencyTest(unittest.TestCase):
def test_one_letter(self):
actual = calculate(['a'])
expected = {'a': 1}
self.assertDictEqual(actual, expected)

def test_case_insensitivity(self):
actual = calculate(['aA'])
expected = {'a': 2}
self.assertDictEqual(actual, expected)

def test_numbers(self):
actual = calculate(['012', '345', '6789'])
expected = {}
self.assertDictEqual(actual, expected)

def test_punctuations(self):
actual = calculate([r'[]\;,', './{}|', ':"<>?'])
expected = {}
self.assertDictEqual(actual, expected)

def test_whitespaces(self):
actual = calculate([' ', '\t ', '\n\n'])
expected = {}
self.assertDictEqual(actual, expected)

def test_repeated_string_with_known_frequencies(self):
letter_frequency = 3
text_input = 'abc\n' * letter_frequency
actual = calculate(text_input.split('\n'))
expected = {'a': letter_frequency, 'b': letter_frequency,
'c': letter_frequency}
self.assertDictEqual(actual, expected)

def test_multiline_text(self):
text_input = "3 Quotes from Excerism Homepage:\n" + \
"\tOne moment you feel like you're\n" + \
"getting it. The next moment you're\n" + \
"stuck.\n" + \
"\tYou know what it鈥檚 like to be fluent.\n" + \
"Suddenly you鈥檙e feeling incompetent\n" + \
"and clumsy.\n" + \
"\tHaphazard, convoluted code is\n" + \
"infuriating, not to mention costly. That\n" + \
"slapdash explosion of complexity is an\n" + \
"expensive yak shave waiting to\n" + \
"happen."
actual = calculate(text_input.split('\n'))
expected = Counter([x for x in text_input.lower() if x.isalpha()])
self.assertDictEqual(actual, expected)


if __name__ == '__main__':
unittest.main()
Loading