Data commit
This commit is contained in:
parent
7387c8f97b
commit
cb5bb5e222
199093 changed files with 3378972 additions and 0 deletions
2
Task/Bioinformatics-Sequence-mutation/00-META.yaml
Normal file
2
Task/Bioinformatics-Sequence-mutation/00-META.yaml
Normal file
|
|
@ -0,0 +1,2 @@
|
|||
---
|
||||
from: http://rosettacode.org/wiki/Bioinformatics/Sequence_mutation
|
||||
15
Task/Bioinformatics-Sequence-mutation/00-TASK.txt
Normal file
15
Task/Bioinformatics-Sequence-mutation/00-TASK.txt
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
;Task:
|
||||
Given a string of characters A, C, G, and T representing a DNA sequence write a routine to mutate the sequence, (string) by:
|
||||
# Choosing a random base position in the sequence.
|
||||
# Mutate the sequence by doing one of either:
|
||||
## '''S'''wap the base at that position by changing it to one of A, C, G, or T. (which has a chance of swapping the base for the same base)
|
||||
## '''D'''elete the chosen base at the position.
|
||||
## '''I'''nsert another base randomly chosen from A,C, G, or T into the sequence at that position.
|
||||
# Randomly generate a test DNA sequence of at least 200 bases
|
||||
# "Pretty print" the sequence and a count of its size, and the count of each base in the sequence
|
||||
# Mutate the sequence ten times.
|
||||
# "Pretty print" the sequence after all mutations, and a count of its size, and the count of each base in the sequence.
|
||||
|
||||
;Extra credit:
|
||||
* Give more information on the individual mutations applied.
|
||||
* Allow mutations to be weighted and/or chosen.
|
||||
|
|
@ -0,0 +1,51 @@
|
|||
UInt32 seed = 0
|
||||
F nonrandom(n)
|
||||
:seed = 1664525 * :seed + 1013904223
|
||||
R Int(:seed >> 16) % n
|
||||
F nonrandom_choice(lst)
|
||||
R lst[nonrandom(lst.len)]
|
||||
|
||||
F basecount(dna)
|
||||
DefaultDict[Char, Int] d
|
||||
L(c) dna
|
||||
d[c]++
|
||||
R sorted(d.items())
|
||||
|
||||
F seq_split(dna, n = 50)
|
||||
R (0 .< dna.len).step(n).map(i -> @dna[i .< i + @n])
|
||||
|
||||
F seq_pp(dna, n = 50)
|
||||
L(part) seq_split(dna, n)
|
||||
print(‘#5: #.’.format(L.index * n, part))
|
||||
print("\n BASECOUNT:")
|
||||
V tot = 0
|
||||
L(base, count) basecount(dna)
|
||||
print(‘ #3: #.’.format(base, count))
|
||||
tot += count
|
||||
V (base, count) = (‘TOT’, tot)
|
||||
print(‘ #3= #.’.format(base, count))
|
||||
|
||||
F seq_mutate(String =dna; count = 1, kinds = ‘IDSSSS’, choice = ‘ATCG’)
|
||||
[(String, Int)] mutation
|
||||
V k2txt = [‘I’ = ‘Insert’, ‘D’ = ‘Delete’, ‘S’ = ‘Substitute’]
|
||||
L 0 .< count
|
||||
V kind = nonrandom_choice(kinds)
|
||||
V index = nonrandom(dna.len + 1)
|
||||
I kind == ‘I’
|
||||
dna = dna[0 .< index]‘’nonrandom_choice(choice)‘’dna[index..]
|
||||
E I kind == ‘D’ & !dna.empty
|
||||
dna = dna[0 .< index]‘’dna[index+1..]
|
||||
E I kind == ‘S’ & !dna.empty
|
||||
dna = dna[0 .< index]‘’nonrandom_choice(choice)‘’dna[index+1..]
|
||||
mutation.append((k2txt[kind], index))
|
||||
R (dna, mutation)
|
||||
|
||||
print(‘SEQUENCE:’)
|
||||
V sequence = ‘TCAATCATTAATCGATTAATACATTCAATTTGAACATCTCCAGGAGAAGGCAGGGTAATCTCGTGTAGCCGTGCTTGGGGCCTCCGATATGGCCGGGGAATTTCAAAGTATAGTGTGCATCCCCTCATAATACATAGATCTATAGGTAAGTATATGGGTTGACGTTGTTAGATGCGATACACGTGCACACTTTATGAATTTTACGTTCCTCTGCCTAGAGTGCCAAGTTTCAATTTGCTACGGTTCCTCA’
|
||||
seq_pp(sequence)
|
||||
print("\n\nMUTATIONS:")
|
||||
V (mseq, m) = seq_mutate(sequence, 10)
|
||||
L(kind, index) m
|
||||
print(‘ #10 @#.’.format(kind, index))
|
||||
print()
|
||||
seq_pp(mseq)
|
||||
|
|
@ -0,0 +1,117 @@
|
|||
with Ada.Containers.Vectors;
|
||||
with Ada.Numerics.Discrete_Random;
|
||||
with Ada.Text_Io;
|
||||
|
||||
procedure Mutations is
|
||||
|
||||
Width : constant := 60;
|
||||
|
||||
type Nucleotide_Type is (A, C, G, T);
|
||||
type Operation_Type is (Delete, Insert, Swap);
|
||||
type Position_Type is new Natural;
|
||||
|
||||
package Position_Io is new Ada.Text_Io.Integer_Io (Position_Type);
|
||||
package Nucleotide_Io is new Ada.Text_Io.Enumeration_Io (Nucleotide_Type);
|
||||
package Operation_Io is new Ada.Text_Io.Enumeration_Io (Operation_Type);
|
||||
|
||||
use Ada.Text_Io, Position_Io, Nucleotide_Io, Operation_Io;
|
||||
|
||||
package Sequence_Vectors is new Ada.Containers.Vectors (Index_Type => Position_Type,
|
||||
Element_Type => Nucleotide_Type);
|
||||
package Nucleotide_Generators is new Ada.Numerics.Discrete_Random (Result_Subtype => Nucleotide_Type);
|
||||
package Operation_Generators is new Ada.Numerics.Discrete_Random (Result_Subtype => Operation_Type);
|
||||
|
||||
procedure Pretty_Print (Sequence : Sequence_Vectors.Vector) is
|
||||
First : Position_Type := Sequence.First_Index;
|
||||
Last : Position_Type;
|
||||
Count : array (Nucleotide_Type) of Natural := (others => 0);
|
||||
begin
|
||||
Last := Position_Type'Min (First + Width - 1,
|
||||
Sequence.Last_Index);
|
||||
loop
|
||||
Position_Io.Put (First, Width => 4);
|
||||
Put (": ");
|
||||
for N in First .. Last loop
|
||||
declare
|
||||
Nucleotide : Nucleotide_Type renames Sequence (N);
|
||||
begin
|
||||
Put (Nucleotide);
|
||||
Count (Nucleotide) := Count (Nucleotide) + 1;
|
||||
end;
|
||||
end loop;
|
||||
New_Line;
|
||||
exit when Last = Sequence.Last_Index;
|
||||
First := Last + 1;
|
||||
Last := Position_Type'Min (First + Width - 1,
|
||||
Sequence.Last_Index);
|
||||
end loop;
|
||||
|
||||
for N in Count'Range loop
|
||||
Put ("Count of "); Put (N); Put (" is "); Put (Natural'Image (Count (N))); New_Line;
|
||||
end loop;
|
||||
|
||||
end Pretty_Print;
|
||||
|
||||
function Random_Position (First, Last : Position_Type) return Position_Type is
|
||||
subtype Position_Range is Position_Type range First .. Last;
|
||||
package Position_Generators is new Ada.Numerics.Discrete_Random (Result_Subtype => Position_Range);
|
||||
Generator : Position_Generators.Generator;
|
||||
begin
|
||||
Position_Generators.Reset (Generator);
|
||||
return Position_Generators.Random (Generator);
|
||||
end Random_Position;
|
||||
|
||||
Nucleotide_Generator : Nucleotide_Generators.Generator;
|
||||
Operation_Generator : Operation_Generators.Generator;
|
||||
|
||||
Sequence : Sequence_Vectors.Vector;
|
||||
Position : Position_Type;
|
||||
Nucleotide : Nucleotide_Type;
|
||||
Operation : Operation_Type;
|
||||
begin
|
||||
Nucleotide_Generators.Reset (Nucleotide_Generator);
|
||||
Operation_Generators.Reset (Operation_Generator);
|
||||
|
||||
for A in 1 .. 200 loop
|
||||
Sequence.Append (Nucleotide_Generators.Random (Nucleotide_Generator));
|
||||
end loop;
|
||||
|
||||
Put_Line ("Initial sequence:");
|
||||
Pretty_Print (Sequence);
|
||||
New_Line;
|
||||
|
||||
Put_Line ("Mutations:");
|
||||
for Mutate in 1 .. 10 loop
|
||||
|
||||
Operation := Operation_Generators.Random (Operation_Generator);
|
||||
case Operation is
|
||||
|
||||
when Delete =>
|
||||
Position := Random_Position (Sequence.First_Index, Sequence.Last_Index);
|
||||
Sequence.Delete (Index => Position);
|
||||
Put (Operation); Put (" at position "); Put (Position, Width => 0); New_Line;
|
||||
|
||||
when Insert =>
|
||||
Position := Random_Position (Sequence.First_Index, Sequence.Last_Index + 1);
|
||||
Nucleotide := Nucleotide_Generators.Random (Nucleotide_Generator);
|
||||
Sequence.Insert (Before => Position,
|
||||
New_Item => Nucleotide);
|
||||
Put (Operation); Put (" "); Put (Nucleotide); Put (" at position ");
|
||||
Put (Position, Width => 0); New_Line;
|
||||
|
||||
when Swap =>
|
||||
Position := Random_Position (Sequence.First_Index, Sequence.Last_Index);
|
||||
Nucleotide := Nucleotide_Generators.Random (Nucleotide_Generator);
|
||||
Sequence.Replace_Element (Index => Position,
|
||||
New_Item => Nucleotide);
|
||||
Put (Operation); Put (" at position "); Put (Position, Width => 0);
|
||||
Put (" to "); Put (Nucleotide); New_Line;
|
||||
|
||||
end case;
|
||||
end loop;
|
||||
|
||||
New_Line;
|
||||
Put_Line ("Mutated sequence:");
|
||||
Pretty_Print (Sequence);
|
||||
|
||||
end Mutations;
|
||||
|
|
@ -0,0 +1,68 @@
|
|||
bases: ["A" "T" "G" "C"]
|
||||
dna: map 1..200 => [sample bases]
|
||||
|
||||
prettyPrint: function [in][
|
||||
count: #[ A: 0, T: 0, G: 0, C: 0 ]
|
||||
|
||||
loop.with:'i split.every:50 in 'line [
|
||||
prints [pad to :string i*50 3 ":"]
|
||||
print map split.every:10 line => join
|
||||
|
||||
loop split line 'ch [
|
||||
case [ch=]
|
||||
when? -> "A" -> count\A: count\A + 1
|
||||
when? -> "T" -> count\T: count\T + 1
|
||||
when? -> "G" -> count\G: count\G + 1
|
||||
when? -> "C" -> count\C: count\C + 1
|
||||
else []
|
||||
]
|
||||
]
|
||||
print ["Total count => A:" count\A, "T:" count\T "G:" count\G "C:" count\C]
|
||||
]
|
||||
|
||||
performRandomModifications: function [seq,times][
|
||||
result: new seq
|
||||
|
||||
loop times [x][
|
||||
what: random 1 3
|
||||
case [what=]
|
||||
when? -> 1 [
|
||||
ind: random 0 (size result)
|
||||
previous: get result ind
|
||||
next: sample bases
|
||||
set result ind next
|
||||
print ["changing base at position" ind "from" previous "to" next]
|
||||
]
|
||||
when? -> 2 [
|
||||
ind: random 0 (size result)
|
||||
next: sample bases
|
||||
result: insert result ind next
|
||||
print ["inserting base" next "at position" ind]
|
||||
]
|
||||
else [
|
||||
ind: random 0 (size result)
|
||||
previous: get result ind
|
||||
result: remove result .index ind
|
||||
print ["deleting base" previous "at position" ind]
|
||||
]
|
||||
]
|
||||
return result
|
||||
]
|
||||
|
||||
print "------------------------------"
|
||||
print " Initial sequence"
|
||||
print "------------------------------"
|
||||
prettyPrint dna
|
||||
print ""
|
||||
|
||||
print "------------------------------"
|
||||
print " Modifying sequence"
|
||||
print "------------------------------"
|
||||
dna: performRandomModifications dna 10
|
||||
print ""
|
||||
|
||||
print "------------------------------"
|
||||
print " Final sequence"
|
||||
print "------------------------------"
|
||||
prettyPrint dna
|
||||
print ""
|
||||
|
|
@ -0,0 +1,124 @@
|
|||
#include <array>
|
||||
#include <iomanip>
|
||||
#include <iostream>
|
||||
#include <random>
|
||||
#include <string>
|
||||
|
||||
class sequence_generator {
|
||||
public:
|
||||
sequence_generator();
|
||||
std::string generate_sequence(size_t length);
|
||||
void mutate_sequence(std::string&);
|
||||
static void print_sequence(std::ostream&, const std::string&);
|
||||
enum class operation { change, erase, insert };
|
||||
void set_weight(operation, unsigned int);
|
||||
private:
|
||||
char get_random_base() {
|
||||
return bases_[base_dist_(engine_)];
|
||||
}
|
||||
operation get_random_operation();
|
||||
static const std::array<char, 4> bases_;
|
||||
std::mt19937 engine_;
|
||||
std::uniform_int_distribution<size_t> base_dist_;
|
||||
std::array<unsigned int, 3> operation_weight_;
|
||||
unsigned int total_weight_;
|
||||
};
|
||||
|
||||
const std::array<char, 4> sequence_generator::bases_{ 'A', 'C', 'G', 'T' };
|
||||
|
||||
sequence_generator::sequence_generator() : engine_(std::random_device()()),
|
||||
base_dist_(0, bases_.size() - 1),
|
||||
total_weight_(operation_weight_.size()) {
|
||||
operation_weight_.fill(1);
|
||||
}
|
||||
|
||||
sequence_generator::operation sequence_generator::get_random_operation() {
|
||||
std::uniform_int_distribution<unsigned int> op_dist(0, total_weight_ - 1);
|
||||
unsigned int n = op_dist(engine_), op = 0, weight = 0;
|
||||
for (; op < operation_weight_.size(); ++op) {
|
||||
weight += operation_weight_[op];
|
||||
if (n < weight)
|
||||
break;
|
||||
}
|
||||
return static_cast<operation>(op);
|
||||
}
|
||||
|
||||
void sequence_generator::set_weight(operation op, unsigned int weight) {
|
||||
total_weight_ -= operation_weight_[static_cast<size_t>(op)];
|
||||
operation_weight_[static_cast<size_t>(op)] = weight;
|
||||
total_weight_ += weight;
|
||||
}
|
||||
|
||||
std::string sequence_generator::generate_sequence(size_t length) {
|
||||
std::string sequence;
|
||||
sequence.reserve(length);
|
||||
for (size_t i = 0; i < length; ++i)
|
||||
sequence += get_random_base();
|
||||
return sequence;
|
||||
}
|
||||
|
||||
void sequence_generator::mutate_sequence(std::string& sequence) {
|
||||
std::uniform_int_distribution<size_t> dist(0, sequence.length() - 1);
|
||||
size_t pos = dist(engine_);
|
||||
char b;
|
||||
switch (get_random_operation()) {
|
||||
case operation::change:
|
||||
b = get_random_base();
|
||||
std::cout << "Change base at position " << pos << " from "
|
||||
<< sequence[pos] << " to " << b << '\n';
|
||||
sequence[pos] = b;
|
||||
break;
|
||||
case operation::erase:
|
||||
std::cout << "Erase base " << sequence[pos] << " at position "
|
||||
<< pos << '\n';
|
||||
sequence.erase(pos, 1);
|
||||
break;
|
||||
case operation::insert:
|
||||
b = get_random_base();
|
||||
std::cout << "Insert base " << b << " at position "
|
||||
<< pos << '\n';
|
||||
sequence.insert(pos, 1, b);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
void sequence_generator::print_sequence(std::ostream& out, const std::string& sequence) {
|
||||
constexpr size_t base_count = bases_.size();
|
||||
std::array<size_t, base_count> count = { 0 };
|
||||
for (size_t i = 0, n = sequence.length(); i < n; ++i) {
|
||||
if (i % 50 == 0) {
|
||||
if (i != 0)
|
||||
out << '\n';
|
||||
out << std::setw(3) << i << ": ";
|
||||
}
|
||||
out << sequence[i];
|
||||
for (size_t j = 0; j < base_count; ++j) {
|
||||
if (bases_[j] == sequence[i]) {
|
||||
++count[j];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
out << '\n';
|
||||
out << "Base counts:\n";
|
||||
size_t total = 0;
|
||||
for (size_t j = 0; j < base_count; ++j) {
|
||||
total += count[j];
|
||||
out << bases_[j] << ": " << count[j] << ", ";
|
||||
}
|
||||
out << "Total: " << total << '\n';
|
||||
}
|
||||
|
||||
int main() {
|
||||
sequence_generator gen;
|
||||
gen.set_weight(sequence_generator::operation::change, 2);
|
||||
std::string sequence = gen.generate_sequence(250);
|
||||
std::cout << "Initial sequence:\n";
|
||||
sequence_generator::print_sequence(std::cout, sequence);
|
||||
constexpr int count = 10;
|
||||
for (int i = 0; i < count; ++i)
|
||||
gen.mutate_sequence(sequence);
|
||||
std::cout << "After " << count << " mutations:\n";
|
||||
sequence_generator::print_sequence(std::cout, sequence);
|
||||
return 0;
|
||||
}
|
||||
|
|
@ -0,0 +1,211 @@
|
|||
#include<stdlib.h>
|
||||
#include<stdio.h>
|
||||
#include<time.h>
|
||||
|
||||
typedef struct genome{
|
||||
char base;
|
||||
struct genome *next;
|
||||
}genome;
|
||||
|
||||
typedef struct{
|
||||
char mutation;
|
||||
int position;
|
||||
}genomeChange;
|
||||
|
||||
typedef struct{
|
||||
int adenineCount,thymineCount,cytosineCount,guanineCount;
|
||||
}baseCounts;
|
||||
|
||||
genome *strand;
|
||||
baseCounts baseData;
|
||||
int genomeLength = 100, lineLength = 50;
|
||||
|
||||
int numDigits(int num){
|
||||
int len = 1;
|
||||
|
||||
while(num>10){
|
||||
num /= 10;
|
||||
len++;
|
||||
}
|
||||
return len;
|
||||
}
|
||||
|
||||
void generateStrand(){
|
||||
|
||||
int baseChoice = rand()%4, i;
|
||||
genome *strandIterator, *newStrand;
|
||||
|
||||
baseData.adenineCount = 0;
|
||||
baseData.thymineCount = 0;
|
||||
baseData.cytosineCount = 0;
|
||||
baseData.guanineCount = 0;
|
||||
|
||||
strand = (genome*)malloc(sizeof(genome));
|
||||
strand->base = baseChoice==0?'A':(baseChoice==1?'T':(baseChoice==2?'C':'G'));
|
||||
baseChoice==0?baseData.adenineCount++:(baseChoice==1?baseData.thymineCount++:(baseChoice==2?baseData.cytosineCount++:baseData.guanineCount++));
|
||||
strand->next = NULL;
|
||||
|
||||
strandIterator = strand;
|
||||
|
||||
for(i=1;i<genomeLength;i++){
|
||||
baseChoice = rand()%4;
|
||||
|
||||
newStrand = (genome*)malloc(sizeof(genome));
|
||||
newStrand->base = baseChoice==0?'A':(baseChoice==1?'T':(baseChoice==2?'C':'G'));
|
||||
baseChoice==0?baseData.adenineCount++:(baseChoice==1?baseData.thymineCount++:(baseChoice==2?baseData.cytosineCount++:baseData.guanineCount++));
|
||||
newStrand->next = NULL;
|
||||
|
||||
strandIterator->next = newStrand;
|
||||
strandIterator = newStrand;
|
||||
}
|
||||
}
|
||||
|
||||
genomeChange generateMutation(int swapWeight, int insertionWeight, int deletionWeight){
|
||||
int mutationChoice = rand()%(swapWeight + insertionWeight + deletionWeight);
|
||||
|
||||
genomeChange mutationCommand;
|
||||
|
||||
mutationCommand.mutation = mutationChoice<swapWeight?'S':((mutationChoice>=swapWeight && mutationChoice<swapWeight+insertionWeight)?'I':'D');
|
||||
mutationCommand.position = rand()%genomeLength;
|
||||
|
||||
return mutationCommand;
|
||||
}
|
||||
|
||||
void printGenome(){
|
||||
int rows, width = numDigits(genomeLength), len = 0,i,j;
|
||||
lineLength = (genomeLength<lineLength)?genomeLength:lineLength;
|
||||
|
||||
rows = genomeLength/lineLength + (genomeLength%lineLength!=0);
|
||||
|
||||
genome* strandIterator = strand;
|
||||
|
||||
printf("\n\nGenome : \n--------\n");
|
||||
|
||||
for(i=0;i<rows;i++){
|
||||
printf("\n%*d%3s",width,len,":");
|
||||
|
||||
for(j=0;j<lineLength && strandIterator!=NULL;j++){
|
||||
printf("%c",strandIterator->base);
|
||||
strandIterator = strandIterator->next;
|
||||
}
|
||||
len += lineLength;
|
||||
}
|
||||
|
||||
while(strandIterator!=NULL){
|
||||
printf("%c",strandIterator->base);
|
||||
strandIterator = strandIterator->next;
|
||||
}
|
||||
|
||||
printf("\n\nBase Counts\n-----------");
|
||||
|
||||
printf("\n%*c%3s%*d",width,'A',":",width,baseData.adenineCount);
|
||||
printf("\n%*c%3s%*d",width,'T',":",width,baseData.thymineCount);
|
||||
printf("\n%*c%3s%*d",width,'C',":",width,baseData.cytosineCount);
|
||||
printf("\n%*c%3s%*d",width,'G',":",width,baseData.guanineCount);
|
||||
|
||||
printf("\n\nTotal:%*d",width,baseData.adenineCount + baseData.thymineCount + baseData.cytosineCount + baseData.guanineCount);
|
||||
|
||||
printf("\n");
|
||||
}
|
||||
|
||||
void mutateStrand(int numMutations, int swapWeight, int insertionWeight, int deletionWeight){
|
||||
int i,j,width,baseChoice;
|
||||
genomeChange newMutation;
|
||||
genome *strandIterator, *strandFollower, *newStrand;
|
||||
|
||||
for(i=0;i<numMutations;i++){
|
||||
strandIterator = strand;
|
||||
strandFollower = strand;
|
||||
newMutation = generateMutation(swapWeight,insertionWeight,deletionWeight);
|
||||
width = numDigits(genomeLength);
|
||||
|
||||
for(j=0;j<newMutation.position;j++){
|
||||
strandFollower = strandIterator;
|
||||
strandIterator = strandIterator->next;
|
||||
}
|
||||
|
||||
if(newMutation.mutation=='S'){
|
||||
if(strandIterator->base=='A'){
|
||||
strandIterator->base='T';
|
||||
printf("\nSwapping A at position : %*d with T",width,newMutation.position);
|
||||
}
|
||||
else if(strandIterator->base=='A'){
|
||||
strandIterator->base='T';
|
||||
printf("\nSwapping A at position : %*d with T",width,newMutation.position);
|
||||
}
|
||||
else if(strandIterator->base=='C'){
|
||||
strandIterator->base='G';
|
||||
printf("\nSwapping C at position : %*d with G",width,newMutation.position);
|
||||
}
|
||||
else{
|
||||
strandIterator->base='C';
|
||||
printf("\nSwapping G at position : %*d with C",width,newMutation.position);
|
||||
}
|
||||
}
|
||||
|
||||
else if(newMutation.mutation=='I'){
|
||||
baseChoice = rand()%4;
|
||||
|
||||
newStrand = (genome*)malloc(sizeof(genome));
|
||||
newStrand->base = baseChoice==0?'A':(baseChoice==1?'T':(baseChoice==2?'C':'G'));
|
||||
printf("\nInserting %c at position : %*d",newStrand->base,width,newMutation.position);
|
||||
baseChoice==0?baseData.adenineCount++:(baseChoice==1?baseData.thymineCount++:(baseChoice==2?baseData.cytosineCount++:baseData.guanineCount++));
|
||||
newStrand->next = strandIterator;
|
||||
strandFollower->next = newStrand;
|
||||
genomeLength++;
|
||||
}
|
||||
|
||||
else{
|
||||
strandFollower->next = strandIterator->next;
|
||||
strandIterator->next = NULL;
|
||||
printf("\nDeleting %c at position : %*d",strandIterator->base,width,newMutation.position);
|
||||
free(strandIterator);
|
||||
genomeLength--;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int main(int argc,char* argv[])
|
||||
{
|
||||
int numMutations = 10, swapWeight = 10, insertWeight = 10, deleteWeight = 10;
|
||||
|
||||
if(argc==1||argc>6){
|
||||
printf("Usage : %s <Genome Length> <Optional number of mutations> <Optional Swapping weight> <Optional Insertion weight> <Optional Deletion weight>\n",argv[0]);
|
||||
return 0;
|
||||
}
|
||||
|
||||
switch(argc){
|
||||
case 2: genomeLength = atoi(argv[1]);
|
||||
break;
|
||||
case 3: genomeLength = atoi(argv[1]);
|
||||
numMutations = atoi(argv[2]);
|
||||
break;
|
||||
case 4: genomeLength = atoi(argv[1]);
|
||||
numMutations = atoi(argv[2]);
|
||||
swapWeight = atoi(argv[3]);
|
||||
break;
|
||||
case 5: genomeLength = atoi(argv[1]);
|
||||
numMutations = atoi(argv[2]);
|
||||
swapWeight = atoi(argv[3]);
|
||||
insertWeight = atoi(argv[4]);
|
||||
break;
|
||||
case 6: genomeLength = atoi(argv[1]);
|
||||
numMutations = atoi(argv[2]);
|
||||
swapWeight = atoi(argv[3]);
|
||||
insertWeight = atoi(argv[4]);
|
||||
deleteWeight = atoi(argv[5]);
|
||||
break;
|
||||
};
|
||||
|
||||
srand(time(NULL));
|
||||
generateStrand();
|
||||
|
||||
printf("\nOriginal:");
|
||||
printGenome();
|
||||
mutateStrand(numMutations,swapWeight,insertWeight,deleteWeight);
|
||||
|
||||
printf("\n\nMutated:");
|
||||
printGenome();
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
|
@ -0,0 +1,73 @@
|
|||
(defun random_base ()
|
||||
(random 4))
|
||||
|
||||
(defun basechar (base)
|
||||
(char "ACTG" base))
|
||||
|
||||
(defun generate_genome (genome_length)
|
||||
(let (genome '())
|
||||
(loop for i below genome_length do
|
||||
(push (random_base) genome))
|
||||
(return-from generate_genome genome)))
|
||||
|
||||
(defun map_genome (genome)
|
||||
(let (seq '())
|
||||
(loop for n from (1- (length genome)) downto 0 do
|
||||
(push (position (char genome n) "ACTG") seq))
|
||||
seq))
|
||||
|
||||
(defun output_genome_info (genome &optional (genome_name "ORIGINAL"))
|
||||
(let ((ac 0) (tc 0) (cc 0) (gc 0))
|
||||
(format t "~% ---- ~a ----" genome_name)
|
||||
(do ((n 0 (1+ n)))
|
||||
((= n (length genome)))
|
||||
(when (= 0 (mod n 50)) (format t "~& ~4d: " (1+ n)))
|
||||
(case (nth n genome)
|
||||
(0 (incf ac))
|
||||
(1 (incf tc))
|
||||
(2 (incf cc))
|
||||
(3 (incf gc)))
|
||||
(format t "~c" (basechar (nth n genome))))
|
||||
(format t "~2%- Total : ~3d~%A : ~d C : ~d~%T : ~d G : ~d~2%" (length genome) ac tc cc gc)))
|
||||
|
||||
(defun insert_base (genome)
|
||||
(let ((place (random (length genome)))
|
||||
(base (random_base)))
|
||||
(format t "Insert + ~c at ~3d~%"
|
||||
(basechar base) (+ 1 place))
|
||||
(if (= 0 place)
|
||||
(push base genome)
|
||||
(push base (cdr (nthcdr (1- place) genome))))
|
||||
(return-from insert_base genome)))
|
||||
|
||||
(defun swap_base (genome)
|
||||
(let ((place (random (length genome)))
|
||||
(base (random_base)))
|
||||
(format t "Swap ~c -> ~c at ~3d~%"
|
||||
(basechar (nth place genome)) (basechar base) (+ 1 place))
|
||||
(setf (nth place genome) base)
|
||||
(return-from swap_base genome)))
|
||||
|
||||
(defun delete_base (genome)
|
||||
(let ((place (random (length genome))))
|
||||
(format t "Delete - ~c at ~3d~%"
|
||||
(basechar (nth place genome)) (+ 1 place))
|
||||
(if (= 0 place) (pop genome)
|
||||
(pop (cdr (nthcdr (1- place) genome))))
|
||||
(return-from delete_base genome)))
|
||||
|
||||
(defun mutate (genome_length n_mutations
|
||||
&key (ins_w 10) (swp_w 10) (del_w 10)
|
||||
(genome (generate_genome genome_length) has_genome))
|
||||
(if has_genome (setf genome (map_genome genome)))
|
||||
(output_genome_info genome)
|
||||
(format t " ---- MUTATION SEQUENCE ----~%")
|
||||
(do ((n 0 (1+ n)))
|
||||
((= n n_mutations))
|
||||
(setf mutation_type (random (+ ins_w swp_w del_w)))
|
||||
(format t "~3d : " (1+ n))
|
||||
(setf genome
|
||||
(cond ((< mutation_type ins_w) (insert_base genome))
|
||||
((< mutation_type (+ ins_w swp_w)) (swap_base genome))
|
||||
(t (delete_base genome)))))
|
||||
(output_genome_info genome "MUTATED"))
|
||||
|
|
@ -0,0 +1,74 @@
|
|||
USING: assocs combinators.random formatting grouping io kernel
|
||||
macros math math.statistics namespaces prettyprint quotations
|
||||
random sequences sorting ;
|
||||
IN: sequence-mutation
|
||||
|
||||
SYMBOL: verbose? ! Turn on to show mutation details.
|
||||
! Off by default.
|
||||
|
||||
! Return a random base as a character.
|
||||
: rand-base ( -- n ) "ACGT" random ;
|
||||
|
||||
! Generate a random dna sequence of length n.
|
||||
: <dna> ( n -- seq ) [ rand-base ] "" replicate-as ;
|
||||
|
||||
! Prettyprint a dna sequence in blocks of n.
|
||||
: .dna ( seq n -- )
|
||||
"SEQUENCE:" print [ group ] keep
|
||||
[ * swap " %3d: %s\n" printf ] curry each-index ;
|
||||
|
||||
! Show a histogram of bases in a dna sequence and their total.
|
||||
: show-counts ( seq -- )
|
||||
"BASE COUNTS:" print histogram >alist [ first ] sort-with
|
||||
[ [ " %c: %3d\n" printf ] assoc-each ]
|
||||
[ "TOTAL: " write [ second ] [ + ] map-reduce . ] bi ;
|
||||
|
||||
! Prettyprint the overall state of a dna sequence.
|
||||
: show-dna ( seq -- ) [ 50 .dna nl ] [ show-counts nl ] bi ;
|
||||
|
||||
! Call a quotation only if verbose? is on.
|
||||
: log ( quot -- ) verbose? get [ call ] [ drop ] if ; inline
|
||||
|
||||
! Set index n to a random base.
|
||||
: bswap ( n seq -- seq' )
|
||||
[ rand-base ] 2dip 3dup [ nth ] keepd spin
|
||||
[ " index %3d: swapping %c with %c\n" printf ] 3curry log
|
||||
[ set-nth ] keep ;
|
||||
|
||||
! Remove the base at index n.
|
||||
: bdelete ( n seq -- seq' )
|
||||
2dup dupd nth [ " index %3d: deleting %c\n" printf ]
|
||||
2curry log remove-nth ;
|
||||
|
||||
! Insert a random base at index n.
|
||||
: binsert ( n seq -- seq' )
|
||||
[ rand-base ] 2dip over reach
|
||||
[ " index %3d: inserting %c\n" printf ] 2curry log
|
||||
insert-nth ;
|
||||
|
||||
! Allow "passing" probabilities to casep. This is necessary
|
||||
! because casep is a macro.
|
||||
MACRO: build-casep-seq ( seq -- quot )
|
||||
{ [ bswap ] [ bdelete ] [ binsert ] } zip 1quotation ;
|
||||
|
||||
! Mutate a dna sequence according to some weights.
|
||||
! For example,
|
||||
! "ACGT" { 0.1 0.3 0.6 } mutate
|
||||
! means swap with 0.1 probability, delete with 0.3 probability,
|
||||
! and insert with 0.6 probability.
|
||||
: mutate ( dna-seq weights-seq -- dna-seq' )
|
||||
[ [ length random ] keep ] [ build-casep-seq ] bi* casep ;
|
||||
inline
|
||||
|
||||
! Prettyprint a sequence of weights.
|
||||
: show-weights ( seq -- )
|
||||
"MUTATION PROBABILITIES:" print
|
||||
" swap: %.2f\n delete: %.2f\n insert: %.2f\n\n" vprintf
|
||||
;
|
||||
|
||||
: main ( -- )
|
||||
verbose? on "ORIGINAL " write 200 <dna> dup show-dna 10
|
||||
{ 0.2 0.2 0.6 } dup show-weights "MUTATION LOG:" print
|
||||
[ mutate ] curry times nl "MUTATED " write show-dna ;
|
||||
|
||||
MAIN: main
|
||||
|
|
@ -0,0 +1,62 @@
|
|||
'' Rosetta Code problem: https://rosettacode.org/wiki/Bioinformatics/Sequence_mutation
|
||||
'' by Jjuanhdez, 05/2023
|
||||
|
||||
Randomize Timer
|
||||
|
||||
Dim As Integer r, i
|
||||
r = Int(Rnd * (300))
|
||||
|
||||
Dim Shared As String dnaS
|
||||
For i = 1 To 200 + r : dnaS += Mid("ACGT", Int(Rnd * (4))+1, 1) : Next
|
||||
|
||||
Sub show()
|
||||
Dim As Integer acgt(4), i, j, x, total
|
||||
|
||||
For i = 1 To Len(dnaS)
|
||||
x = Instr("ACGT", Mid(dnaS, i, 1))
|
||||
acgt(x) += 1
|
||||
Next
|
||||
|
||||
For i = 1 To 4 : total += acgt(i) : Next
|
||||
|
||||
For i = 1 To Len(dnaS) Step 50
|
||||
Print i; ":"; !"\t";
|
||||
For j = 0 To 49 Step 10
|
||||
Print Mid(dnaS, i+j, 10); " ";
|
||||
Next
|
||||
Print
|
||||
Next
|
||||
Print !"\nBase counts: A:"; acgt(1); ", C:"; acgt(2); ", G:"; acgt(3); ", T:"; acgt(4); ", total:"; total
|
||||
End Sub
|
||||
|
||||
|
||||
Sub mutate()
|
||||
Dim As Integer i, p
|
||||
Dim As String sdiS, repS, wasS
|
||||
|
||||
Print
|
||||
For i = 1 To 10
|
||||
p = Int(Rnd * (Len(dnaS))) + 1
|
||||
sdiS = Mid("SDI", Int(Rnd * (3)) + 1, 1)
|
||||
repS = Mid("ACGT", Int(Rnd * (4)) + 1, 1)
|
||||
wasS = Mid(dnaS, p, 1)
|
||||
Select Case sdiS
|
||||
Case "S"
|
||||
Mid(dnaS, p, 1) = repS
|
||||
Print "swapped "; wasS; " at "; p; " for "; repS
|
||||
Case "D"
|
||||
dnaS = Left(dnaS, p - 1) + Right(dnaS, Len(dnaS) - p)
|
||||
Print "deleted "; wasS; " at "; p
|
||||
Case "I"
|
||||
dnaS = Left(dnaS, p - 1) + repS + Right(dnaS, (Len(dnaS) - p + 1))
|
||||
Print "inserted "; repS; " at "; p; ", before "; wasS
|
||||
End Select
|
||||
Next
|
||||
Print
|
||||
End Sub
|
||||
|
||||
show()
|
||||
mutate()
|
||||
show()
|
||||
|
||||
Sleep
|
||||
|
|
@ -0,0 +1,99 @@
|
|||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math/rand"
|
||||
"sort"
|
||||
"time"
|
||||
)
|
||||
|
||||
const bases = "ACGT"
|
||||
|
||||
// 'w' contains the weights out of 300 for each
|
||||
// of swap, delete or insert in that order.
|
||||
func mutate(dna string, w [3]int) string {
|
||||
le := len(dna)
|
||||
// get a random position in the dna to mutate
|
||||
p := rand.Intn(le)
|
||||
// get a random number between 0 and 299 inclusive
|
||||
r := rand.Intn(300)
|
||||
bytes := []byte(dna)
|
||||
switch {
|
||||
case r < w[0]: // swap
|
||||
base := bases[rand.Intn(4)]
|
||||
fmt.Printf(" Change @%3d %q to %q\n", p, bytes[p], base)
|
||||
bytes[p] = base
|
||||
case r < w[0]+w[1]: // delete
|
||||
fmt.Printf(" Delete @%3d %q\n", p, bytes[p])
|
||||
copy(bytes[p:], bytes[p+1:])
|
||||
bytes = bytes[0 : le-1]
|
||||
default: // insert
|
||||
base := bases[rand.Intn(4)]
|
||||
bytes = append(bytes, 0)
|
||||
copy(bytes[p+1:], bytes[p:])
|
||||
fmt.Printf(" Insert @%3d %q\n", p, base)
|
||||
bytes[p] = base
|
||||
}
|
||||
return string(bytes)
|
||||
}
|
||||
|
||||
// Generate a random dna sequence of given length.
|
||||
func generate(le int) string {
|
||||
bytes := make([]byte, le)
|
||||
for i := 0; i < le; i++ {
|
||||
bytes[i] = bases[rand.Intn(4)]
|
||||
}
|
||||
return string(bytes)
|
||||
}
|
||||
|
||||
// Pretty print dna and stats.
|
||||
func prettyPrint(dna string, rowLen int) {
|
||||
fmt.Println("SEQUENCE:")
|
||||
le := len(dna)
|
||||
for i := 0; i < le; i += rowLen {
|
||||
k := i + rowLen
|
||||
if k > le {
|
||||
k = le
|
||||
}
|
||||
fmt.Printf("%5d: %s\n", i, dna[i:k])
|
||||
}
|
||||
baseMap := make(map[byte]int) // allows for 'any' base
|
||||
for i := 0; i < le; i++ {
|
||||
baseMap[dna[i]]++
|
||||
}
|
||||
var bases []byte
|
||||
for k := range baseMap {
|
||||
bases = append(bases, k)
|
||||
}
|
||||
sort.Slice(bases, func(i, j int) bool { // get bases into alphabetic order
|
||||
return bases[i] < bases[j]
|
||||
})
|
||||
|
||||
fmt.Println("\nBASE COUNT:")
|
||||
for _, base := range bases {
|
||||
fmt.Printf(" %c: %3d\n", base, baseMap[base])
|
||||
}
|
||||
fmt.Println(" ------")
|
||||
fmt.Println(" Σ:", le)
|
||||
fmt.Println(" ======\n")
|
||||
}
|
||||
|
||||
// Express weights as a string.
|
||||
func wstring(w [3]int) string {
|
||||
return fmt.Sprintf(" Change: %d\n Delete: %d\n Insert: %d\n", w[0], w[1], w[2])
|
||||
}
|
||||
|
||||
func main() {
|
||||
rand.Seed(time.Now().UnixNano())
|
||||
dna := generate(250)
|
||||
prettyPrint(dna, 50)
|
||||
muts := 10
|
||||
w := [3]int{100, 100, 100} // use e.g. {0, 300, 0} to choose only deletions
|
||||
fmt.Printf("WEIGHTS (ex 300):\n%s\n", wstring(w))
|
||||
fmt.Printf("MUTATIONS (%d):\n", muts)
|
||||
for i := 0; i < muts; i++ {
|
||||
dna = mutate(dna, w)
|
||||
}
|
||||
fmt.Println()
|
||||
prettyPrint(dna, 50)
|
||||
}
|
||||
|
|
@ -0,0 +1,80 @@
|
|||
import Data.List (group, sort)
|
||||
import Data.List.Split (chunksOf)
|
||||
import System.Random (Random, randomR, random, newStdGen, randoms, getStdRandom)
|
||||
import Text.Printf (PrintfArg(..), fmtChar, fmtPrecision, formatString, IsChar(..), printf)
|
||||
|
||||
data Mutation = Swap | Delete | Insert deriving (Show, Eq, Ord, Enum, Bounded)
|
||||
data DNABase = A | C | G | T deriving (Show, Read, Eq, Ord, Enum, Bounded)
|
||||
type DNASequence = [DNABase]
|
||||
|
||||
data Result = Swapped Mutation Int (DNABase, DNABase)
|
||||
| InsertDeleted Mutation Int DNABase
|
||||
|
||||
instance Random DNABase where
|
||||
randomR (a, b) g = case randomR (fromEnum a, fromEnum b) g of (x, y) -> (toEnum x, y)
|
||||
random = randomR (minBound, maxBound)
|
||||
|
||||
instance Random Mutation where
|
||||
randomR (a, b) g = case randomR (fromEnum a, fromEnum b) g of (x, y) -> (toEnum x, y)
|
||||
random = randomR (minBound, maxBound)
|
||||
|
||||
instance PrintfArg DNABase where
|
||||
formatArg x fmt = formatString (show x) (fmt { fmtChar = 's', fmtPrecision = Nothing })
|
||||
|
||||
instance PrintfArg Mutation where
|
||||
formatArg x fmt = formatString (show x) (fmt { fmtChar = 's', fmtPrecision = Nothing })
|
||||
|
||||
instance IsChar DNABase where
|
||||
toChar = head . show
|
||||
fromChar = read . pure
|
||||
|
||||
chunkedDNASequence :: DNASequence -> [(Int, [DNABase])]
|
||||
chunkedDNASequence = zip [50,100..] . chunksOf 50
|
||||
|
||||
baseCounts :: DNASequence -> [(DNABase, Int)]
|
||||
baseCounts = fmap ((,) . head <*> length) . group . sort
|
||||
|
||||
newSequence :: Int -> IO DNASequence
|
||||
newSequence n = take n . randoms <$> newStdGen
|
||||
|
||||
mutateSequence :: DNASequence -> IO (Result, DNASequence)
|
||||
mutateSequence [] = fail "empty dna sequence"
|
||||
mutateSequence ds = randomMutation >>= mutate ds
|
||||
where
|
||||
randomMutation = head . randoms <$> newStdGen
|
||||
mutate xs m = do
|
||||
i <- randomIndex (length xs)
|
||||
case m of
|
||||
Swap -> randomDNA >>= \d -> pure (Swapped Swap i (xs !! pred i, d), swapElement i d xs)
|
||||
Insert -> randomDNA >>= \d -> pure (InsertDeleted Insert i d, insertElement i d xs)
|
||||
Delete -> pure (InsertDeleted Delete i (xs !! pred i), dropElement i xs)
|
||||
where
|
||||
dropElement i xs = take (pred i) xs <> drop i xs
|
||||
insertElement i e xs = take i xs <> [e] <> drop i xs
|
||||
swapElement i a xs = take (pred i) xs <> [a] <> drop i xs
|
||||
randomIndex n = getStdRandom (randomR (1, n))
|
||||
randomDNA = head . randoms <$> newStdGen
|
||||
|
||||
mutate :: Int -> DNASequence -> IO DNASequence
|
||||
mutate 0 s = pure s
|
||||
mutate n s = do
|
||||
(r, ms) <- mutateSequence s
|
||||
case r of
|
||||
Swapped m i (a, b) -> printf "%6s @ %-3d : %s -> %s \n" m i a b
|
||||
InsertDeleted m i a -> printf "%6s @ %-3d : %s\n" m i a
|
||||
mutate (pred n) ms
|
||||
|
||||
main :: IO ()
|
||||
main = do
|
||||
ds <- newSequence 200
|
||||
putStrLn "\nInitial Sequence:" >> showSequence ds
|
||||
putStrLn "\nBase Counts:" >> showBaseCounts ds
|
||||
showSumBaseCounts ds
|
||||
ms <- mutate 10 ds
|
||||
putStrLn "\nMutated Sequence:" >> showSequence ms
|
||||
putStrLn "\nBase Counts:" >> showBaseCounts ms
|
||||
showSumBaseCounts ms
|
||||
where
|
||||
showSequence = mapM_ (uncurry (printf "%3d: %s\n")) . chunkedDNASequence
|
||||
showBaseCounts = mapM_ (uncurry (printf "%s: %3d\n")) . baseCounts
|
||||
showSumBaseCounts xs = putStrLn (replicate 6 '-') >> printf "Σ: %d\n\n" (length xs)
|
||||
|
|
@ -0,0 +1,36 @@
|
|||
ACGT=: 'ACGT'
|
||||
MUTS=: ;: 'del ins mut'
|
||||
|
||||
NB. generate sequence of size y of uniformly selected nucleotides.
|
||||
NB. represent sequences as ints in range i.4 pretty printed. nuc
|
||||
NB. defined separately to avoid fixing value inside mutation
|
||||
NB. functions.
|
||||
nuc=: monad : '?4'
|
||||
dna=: nuc"0 @ i.
|
||||
|
||||
NB. randomly mutate nucleotide at a random index by deletion insertion
|
||||
NB. or mutation of a nucleotide.
|
||||
del=: {.,[:}.}.
|
||||
ins=: {.,nuc@],}.
|
||||
mut=: {.,nuc@],[:}.}.
|
||||
|
||||
NB. pretty print nucleotides in rows of 50 with numbering
|
||||
seq=: [: (;~ [: (4&":"0) 50*i.@#) _50]\{&ACGT
|
||||
|
||||
sim=: monad define
|
||||
'n k ws'=. y NB. initial size, mutations, and weights for mutations
|
||||
ws=. (% +/) ws NB. normalize weights
|
||||
A=.0$]D0=.D=. dna n NB. initial dna and history of actions
|
||||
|
||||
NB. k times do a random action according to weights and record it
|
||||
for. i.k do.
|
||||
D=.". action=. (":?#D),' ',(":MUTS{::~(+/\ws)I.?0),' D'
|
||||
A=. action ; A
|
||||
end.
|
||||
|
||||
echo 'actions';,. A-.a:
|
||||
echo ('mutation';'probability') , MUTS ,. <"0 ws
|
||||
('start';'end'),.(seq D0) ,: seq D
|
||||
)
|
||||
|
||||
simulate=: (sim@(1 1 1&; &. |. ))`sim@.(3=#)
|
||||
|
|
@ -0,0 +1,104 @@
|
|||
import java.io.BufferedReader;
|
||||
import java.io.IOException;
|
||||
import java.io.StringReader;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Random;
|
||||
|
||||
class Program {
|
||||
List<Character> sequence;
|
||||
Random random;
|
||||
|
||||
SequenceMutation() {
|
||||
sequence = new ArrayList<>();
|
||||
random = new Random();
|
||||
}
|
||||
|
||||
void generate(int amount) {
|
||||
for (int count = 0; count < amount; count++)
|
||||
sequence.add(randomBase());
|
||||
}
|
||||
|
||||
void mutate(int amount) {
|
||||
int index;
|
||||
for (int count = 0; count < amount; count++) {
|
||||
index = random.nextInt(0, sequence.size());
|
||||
switch (random.nextInt(0, 3)) {
|
||||
case 0 -> sequence.set(index, randomBase());
|
||||
case 1 -> sequence.remove(index);
|
||||
case 2 -> sequence.add(index, randomBase());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private char randomBase() {
|
||||
return switch (random.nextInt(0, 4)) {
|
||||
case 0 -> 'A';
|
||||
case 1 -> 'C';
|
||||
case 2 -> 'G';
|
||||
case 3 -> 'T';
|
||||
default -> '?';
|
||||
};
|
||||
}
|
||||
|
||||
private Base count(String string) {
|
||||
int a = 0, c = 0, g = 0, t = 0;
|
||||
for (char base : string.toCharArray()) {
|
||||
switch (base) {
|
||||
case 'A' -> a++;
|
||||
case 'C' -> c++;
|
||||
case 'G' -> g++;
|
||||
case 'T' -> t++;
|
||||
}
|
||||
}
|
||||
return new Base(a, c, g, t);
|
||||
}
|
||||
|
||||
/* used exclusively for count totals */
|
||||
private record Base(int a, int c, int g, int t) {
|
||||
int total() {
|
||||
return a + c + g + t;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
return "[A %2d, C %2d, G %2d, T %2d]".formatted(a, c, g, t);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
StringBuilder string = new StringBuilder();
|
||||
StringBuilder stringB = new StringBuilder();
|
||||
String newline = System.lineSeparator();
|
||||
for (int index = 0; index < sequence.size(); index++) {
|
||||
if (index != 0 && index % 50 == 0)
|
||||
string.append(newline);
|
||||
string.append(sequence.get(index));
|
||||
stringB.append(sequence.get(index));
|
||||
}
|
||||
try {
|
||||
BufferedReader reader = new BufferedReader(new StringReader(string.toString()));
|
||||
string = new StringBuilder();
|
||||
int count = 0;
|
||||
String line;
|
||||
while ((line = reader.readLine()) != null) {
|
||||
string.append(count++);
|
||||
string.append(" %-50s ".formatted(line));
|
||||
string.append(count(line));
|
||||
string.append(newline);
|
||||
}
|
||||
} catch (IOException exception) {
|
||||
/* ignore */
|
||||
}
|
||||
string.append(newline);
|
||||
Base bases = count(stringB.toString());
|
||||
int total = bases.total();
|
||||
string.append("Total of %d bases%n".formatted(total));
|
||||
string.append("A %3d (%.2f%%)%n".formatted(bases.a, ((double) bases.a / total) * 100));
|
||||
string.append("C %3d (%.2f%%)%n".formatted(bases.c, ((double) bases.c / total) * 100));
|
||||
string.append("G %3d (%.2f%%)%n".formatted(bases.g, ((double) bases.g / total) * 100));
|
||||
string.append("T %3d (%.2f%%)%n".formatted(bases.t, ((double) bases.t / total) * 100));
|
||||
return string.toString();
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,113 @@
|
|||
import java.util.Arrays;
|
||||
import java.util.Random;
|
||||
|
||||
public class SequenceMutation {
|
||||
public static void main(String[] args) {
|
||||
SequenceMutation sm = new SequenceMutation();
|
||||
sm.setWeight(OP_CHANGE, 3);
|
||||
String sequence = sm.generateSequence(250);
|
||||
System.out.println("Initial sequence:");
|
||||
printSequence(sequence);
|
||||
int count = 10;
|
||||
for (int i = 0; i < count; ++i)
|
||||
sequence = sm.mutateSequence(sequence);
|
||||
System.out.println("After " + count + " mutations:");
|
||||
printSequence(sequence);
|
||||
}
|
||||
|
||||
public SequenceMutation() {
|
||||
totalWeight_ = OP_COUNT;
|
||||
Arrays.fill(operationWeight_, 1);
|
||||
}
|
||||
|
||||
public String generateSequence(int length) {
|
||||
char[] ch = new char[length];
|
||||
for (int i = 0; i < length; ++i)
|
||||
ch[i] = getRandomBase();
|
||||
return new String(ch);
|
||||
}
|
||||
|
||||
public void setWeight(int operation, int weight) {
|
||||
totalWeight_ -= operationWeight_[operation];
|
||||
operationWeight_[operation] = weight;
|
||||
totalWeight_ += weight;
|
||||
}
|
||||
|
||||
public String mutateSequence(String sequence) {
|
||||
char[] ch = sequence.toCharArray();
|
||||
int pos = random_.nextInt(ch.length);
|
||||
int operation = getRandomOperation();
|
||||
if (operation == OP_CHANGE) {
|
||||
char b = getRandomBase();
|
||||
System.out.println("Change base at position " + pos + " from "
|
||||
+ ch[pos] + " to " + b);
|
||||
ch[pos] = b;
|
||||
} else if (operation == OP_ERASE) {
|
||||
System.out.println("Erase base " + ch[pos] + " at position " + pos);
|
||||
char[] newCh = new char[ch.length - 1];
|
||||
System.arraycopy(ch, 0, newCh, 0, pos);
|
||||
System.arraycopy(ch, pos + 1, newCh, pos, ch.length - pos - 1);
|
||||
ch = newCh;
|
||||
} else if (operation == OP_INSERT) {
|
||||
char b = getRandomBase();
|
||||
System.out.println("Insert base " + b + " at position " + pos);
|
||||
char[] newCh = new char[ch.length + 1];
|
||||
System.arraycopy(ch, 0, newCh, 0, pos);
|
||||
System.arraycopy(ch, pos, newCh, pos + 1, ch.length - pos);
|
||||
newCh[pos] = b;
|
||||
ch = newCh;
|
||||
}
|
||||
return new String(ch);
|
||||
}
|
||||
|
||||
public static void printSequence(String sequence) {
|
||||
int[] count = new int[BASES.length];
|
||||
for (int i = 0, n = sequence.length(); i < n; ++i) {
|
||||
if (i % 50 == 0) {
|
||||
if (i != 0)
|
||||
System.out.println();
|
||||
System.out.printf("%3d: ", i);
|
||||
}
|
||||
char ch = sequence.charAt(i);
|
||||
System.out.print(ch);
|
||||
for (int j = 0; j < BASES.length; ++j) {
|
||||
if (BASES[j] == ch) {
|
||||
++count[j];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
System.out.println();
|
||||
System.out.println("Base counts:");
|
||||
int total = 0;
|
||||
for (int j = 0; j < BASES.length; ++j) {
|
||||
total += count[j];
|
||||
System.out.print(BASES[j] + ": " + count[j] + ", ");
|
||||
}
|
||||
System.out.println("Total: " + total);
|
||||
}
|
||||
|
||||
private char getRandomBase() {
|
||||
return BASES[random_.nextInt(BASES.length)];
|
||||
}
|
||||
|
||||
private int getRandomOperation() {
|
||||
int n = random_.nextInt(totalWeight_), op = 0;
|
||||
for (int weight = 0; op < OP_COUNT; ++op) {
|
||||
weight += operationWeight_[op];
|
||||
if (n < weight)
|
||||
break;
|
||||
}
|
||||
return op;
|
||||
}
|
||||
|
||||
private final Random random_ = new Random();
|
||||
private int[] operationWeight_ = new int[OP_COUNT];
|
||||
private int totalWeight_ = 0;
|
||||
|
||||
private static final int OP_CHANGE = 0;
|
||||
private static final int OP_ERASE = 1;
|
||||
private static final int OP_INSERT = 2;
|
||||
private static final int OP_COUNT = 3;
|
||||
private static final char[] BASES = {'A', 'C', 'G', 'T'};
|
||||
}
|
||||
|
|
@ -0,0 +1,157 @@
|
|||
// Basic set-up
|
||||
const numBases = 250
|
||||
const numMutations = 30
|
||||
const bases = ['A', 'C', 'G', 'T'];
|
||||
|
||||
// Utility functions
|
||||
/**
|
||||
* Return a shallow copy of an array
|
||||
* @param {Array<*>} arr
|
||||
* @returns {*[]}
|
||||
*/
|
||||
const copy = arr => [...arr];
|
||||
|
||||
/**
|
||||
* Get a random int up to but excluding the the given number
|
||||
* @param {number} max
|
||||
* @returns {number}
|
||||
*/
|
||||
const randTo = max => (Math.random() * max) | 0;
|
||||
|
||||
/**
|
||||
* Given an array return a random element and the index of that element from
|
||||
* the array.
|
||||
* @param {Array<*>} arr
|
||||
* @returns {[*[], number]}
|
||||
*/
|
||||
const randSelect = arr => {
|
||||
const at = randTo(arr.length);
|
||||
return [arr[at], at];
|
||||
};
|
||||
|
||||
/**
|
||||
* Given a number or string, return a left padded string
|
||||
* @param {string|number} v
|
||||
* @returns {string}
|
||||
*/
|
||||
const pad = v => ('' + v).padStart(4, ' ');
|
||||
|
||||
/**
|
||||
* Count the number of elements that match the given value in an array
|
||||
* @param {Array<string>} arr
|
||||
* @returns {function(string): number}
|
||||
*/
|
||||
const filterCount = arr => s => arr.filter(e => e === s).length;
|
||||
|
||||
/**
|
||||
* Utility logging function
|
||||
* @param {string|number} v
|
||||
* @param {string|number} n
|
||||
*/
|
||||
const print = (v, n) => console.log(`${pad(v)}:\t${n}`)
|
||||
|
||||
/**
|
||||
* Utility function to randomly select a new base, and an index in the given
|
||||
* sequence.
|
||||
* @param {Array<string>} seq
|
||||
* @param {Array<string>} bases
|
||||
* @returns {[string, string, number]}
|
||||
*/
|
||||
const getVars = (seq, bases) => {
|
||||
const [newBase, _] = randSelect(bases);
|
||||
const [extBase, randPos] = randSelect(seq);
|
||||
return [newBase, extBase, randPos];
|
||||
};
|
||||
|
||||
// Bias the operations
|
||||
/**
|
||||
* Given a map of function to ratio, return an array of those functions
|
||||
* appearing ratio number of times in the array.
|
||||
* @param weightMap
|
||||
* @returns {Array<function>}
|
||||
*/
|
||||
const weightedOps = weightMap => {
|
||||
return [...weightMap.entries()].reduce((p, [op, weight]) =>
|
||||
[...p, ...(Array(weight).fill(op))], []);
|
||||
};
|
||||
|
||||
// Pretty Print functions
|
||||
const prettyPrint = seq => {
|
||||
let idx = 0;
|
||||
const rem = seq.reduce((p, c) => {
|
||||
const s = p + c;
|
||||
if (s.length === 50) {
|
||||
print(idx, s);
|
||||
idx = idx + 50;
|
||||
return '';
|
||||
}
|
||||
return s;
|
||||
}, '');
|
||||
if (rem !== '') {
|
||||
print(idx, rem);
|
||||
}
|
||||
}
|
||||
|
||||
const printBases = seq => {
|
||||
const filterSeq = filterCount(seq);
|
||||
let tot = 0;
|
||||
[...bases].forEach(e => {
|
||||
const cnt = filterSeq(e);
|
||||
print(e, cnt);
|
||||
tot = tot + cnt;
|
||||
})
|
||||
print('Σ', tot);
|
||||
}
|
||||
|
||||
// Mutation definitions
|
||||
const swap = ([hist, seq]) => {
|
||||
const arr = copy(seq);
|
||||
const [newBase, extBase, randPos] = getVars(arr, bases);
|
||||
arr.splice(randPos, 1, newBase);
|
||||
return [[...hist, `Swapped ${extBase} for ${newBase} at ${randPos}`], arr];
|
||||
};
|
||||
|
||||
const del = ([hist, seq]) => {
|
||||
const arr = copy(seq);
|
||||
const [newBase, extBase, randPos] = getVars(arr, bases);
|
||||
arr.splice(randPos, 1);
|
||||
return [[...hist, `Deleted ${extBase} at ${randPos}`], arr];
|
||||
}
|
||||
|
||||
const insert = ([hist, seq]) => {
|
||||
const arr = copy(seq);
|
||||
const [newBase, extBase, randPos] = getVars(arr, bases);
|
||||
arr.splice(randPos, 0, newBase);
|
||||
return [[...hist, `Inserted ${newBase} at ${randPos}`], arr];
|
||||
}
|
||||
|
||||
// Create the starting sequence
|
||||
const seq = Array(numBases).fill(undefined).map(
|
||||
() => randSelect(bases)[0]);
|
||||
|
||||
// Create a weighted set of mutations
|
||||
const weightMap = new Map()
|
||||
.set(swap, 1)
|
||||
.set(del, 1)
|
||||
.set(insert, 1);
|
||||
const operations = weightedOps(weightMap);
|
||||
const mutations = Array(numMutations).fill(undefined).map(
|
||||
() => randSelect(operations)[0]);
|
||||
|
||||
// Mutate the sequence
|
||||
const [hist, mut] = mutations.reduce((p, c) => c(p), [[], seq]);
|
||||
|
||||
console.log('ORIGINAL SEQUENCE:')
|
||||
prettyPrint(seq);
|
||||
|
||||
console.log('\nBASE COUNTS:')
|
||||
printBases(seq);
|
||||
|
||||
console.log('\nMUTATION LOG:')
|
||||
hist.forEach((e, i) => console.log(`${i}:\t${e}`));
|
||||
|
||||
console.log('\nMUTATED SEQUENCE:')
|
||||
prettyPrint(mut);
|
||||
|
||||
console.log('\nMUTATED BASE COUNTS:')
|
||||
printBases(mut);
|
||||
|
|
@ -0,0 +1,53 @@
|
|||
dnabases = ['A', 'C', 'G', 'T']
|
||||
randpos(seq) = rand(1:length(seq)) # 1
|
||||
mutateat(pos, seq) = (s = seq[:]; s[pos] = rand(dnabases); s) # 2-1
|
||||
deleteat(pos, seq) = [seq[1:pos-1]; seq[pos+1:end]] # 2-2
|
||||
randinsertat(pos, seq) = [seq[1:pos]; rand(dnabases); seq[pos+1:end]] # 2-3
|
||||
|
||||
function weightedmutation(seq, pos, weights=[1, 1, 1], verbose=true) # Extra credit
|
||||
p, r = weights ./ sum(weights), rand()
|
||||
f = (r <= p[1]) ? mutateat : (r < p[1] + p[2]) ? deleteat : randinsertat
|
||||
verbose && print("Mutate by ", f == mutateat ? "swap" :
|
||||
f == deleteat ? "delete" : "insert")
|
||||
return f(pos, seq)
|
||||
end
|
||||
|
||||
function weightedrandomsitemutation(seq, weights=[1, 1, 1], verbose=true)
|
||||
position = randpos(seq)
|
||||
newseq = weightedmutation(seq, position, weights, verbose)
|
||||
verbose && println(" at position $position")
|
||||
return newseq
|
||||
end
|
||||
|
||||
randdnasequence(n) = rand(dnabases, n) # 3
|
||||
|
||||
function dnasequenceprettyprint(seq, colsize=50) # 4
|
||||
println(length(seq), "nt DNA sequence:\n")
|
||||
rows = [seq[i:min(length(seq), i + colsize - 1)] for i in 1:colsize:length(seq)]
|
||||
for (i, r) in enumerate(rows)
|
||||
println(lpad(colsize * (i - 1), 5), " ", String(r))
|
||||
end
|
||||
bases = [[c, 0] for c in dnabases]
|
||||
for c in seq, base in bases
|
||||
if c == base[1]
|
||||
base[2] += 1
|
||||
end
|
||||
end
|
||||
println("\nNucleotide counts:\n")
|
||||
for base in bases
|
||||
println(lpad(base[1], 10), lpad(string(base[2]), 12))
|
||||
end
|
||||
println(lpad("Other", 10), lpad(string(length(seq) - sum(x[2] for x in bases)), 12))
|
||||
println(" _________________\n", lpad("Total", 10), lpad(string(length(seq)), 12))
|
||||
end
|
||||
|
||||
function testbioseq()
|
||||
sequence = randdnasequence(500)
|
||||
dnasequenceprettyprint(sequence)
|
||||
for _ in 1:10 # 5
|
||||
sequence = weightedrandomsitemutation(sequence)
|
||||
end
|
||||
println("\n Mutated:"); dnasequenceprettyprint(sequence) # 6
|
||||
end
|
||||
|
||||
testbioseq()
|
||||
|
|
@ -0,0 +1,29 @@
|
|||
math.randomseed(os.time())
|
||||
bases = {"A","C","T","G"}
|
||||
function randbase() return bases[math.random(#bases)] end
|
||||
|
||||
function mutate(seq)
|
||||
local i,h = math.random(#seq), "%-6s %3s at %3d"
|
||||
local old,new = seq:sub(i,i), randbase()
|
||||
local ops = {
|
||||
function(s) h=h:format("Swap", old..">"..new, i) return s:sub(1,i-1)..new..s:sub(i+1) end,
|
||||
function(s) h=h:format("Delete", " -"..old, i) return s:sub(1,i-1)..s:sub(i+1) end,
|
||||
function(s) h=h:format("Insert", " +"..new, i) return s:sub(1,i-1)..new..s:sub(i) end,
|
||||
}
|
||||
local weighted = { 1,1,2,3 }
|
||||
local n = weighted[math.random(#weighted)]
|
||||
return ops[n](seq), h
|
||||
end
|
||||
|
||||
local seq,hist="",{} for i = 1, 200 do seq=seq..randbase() end
|
||||
print("ORIGINAL:")
|
||||
prettyprint(seq)
|
||||
print()
|
||||
|
||||
for i = 1, 10 do seq,h=mutate(seq) hist[#hist+1]=h end
|
||||
print("MUTATIONS:")
|
||||
for i,h in ipairs(hist) do print(" "..h) end
|
||||
print()
|
||||
|
||||
print("MUTATED:")
|
||||
prettyprint(seq)
|
||||
|
|
@ -0,0 +1,35 @@
|
|||
SeedRandom[13122345];
|
||||
seq = BioSequence["DNA", "ATAAACGTACGTTTTTAGGCT"];
|
||||
randompos = RandomInteger[seq["SequenceLength"]];
|
||||
seq = StringReplacePart[seq, RandomChoice[{"A", "T", "C", "G"}], {randompos, randompos}];
|
||||
randompos = RandomInteger[seq["SequenceLength"]];
|
||||
seq = StringReplacePart[seq, "", {randompos, randompos}];
|
||||
randompos = RandomInteger[seq["SequenceLength"]];
|
||||
seq = StringInsert[seq, RandomChoice[{"A", "T", "C", "G"}], randompos];
|
||||
seq = BioSequence["DNA", StringJoin@RandomChoice[{"A", "T", "C", "G"}, 250]];
|
||||
size = 50;
|
||||
parts = StringPartition[seq["SequenceString"], UpTo[size]];
|
||||
begins = Most[Accumulate[Prepend[StringLength /@ parts, 1]]];
|
||||
ends = Rest[Accumulate[Prepend[StringLength /@ parts, 0]]];
|
||||
StringRiffle[MapThread[ToString[#1] <> "-" <> ToString[#2] <> ": " <> #3 &, {begins, ends, parts}], "\n"]
|
||||
Tally[Characters[seq["SequenceString"]]]
|
||||
Do[
|
||||
type = RandomChoice[{1, 2, 3}];
|
||||
Switch[type, 1,
|
||||
randompos = RandomInteger[seq["SequenceLength"]];
|
||||
seq = StringReplacePart[seq, RandomChoice[{"A", "T", "C", "G"}], {randompos, randompos}];
|
||||
, 2,
|
||||
randompos = RandomInteger[seq["SequenceLength"]];
|
||||
seq = StringReplacePart[seq, "", {randompos, randompos}];
|
||||
, 3,
|
||||
randompos = RandomInteger[seq["SequenceLength"]];
|
||||
seq = StringInsert[seq, RandomChoice[{"A", "T", "C", "G"}], randompos];
|
||||
]
|
||||
,
|
||||
{10}
|
||||
]
|
||||
parts = StringPartition[seq["SequenceString"], UpTo[size]];
|
||||
begins = Most[Accumulate[Prepend[StringLength /@ parts, 1]]];
|
||||
ends = Rest[Accumulate[Prepend[StringLength /@ parts, 0]]];
|
||||
StringRiffle[MapThread[ToString[#1] <> "-" <> ToString[#2] <> ": " <> #3 &, {begins, ends, parts}], "\n"]
|
||||
Tally[Characters[seq["SequenceString"]]]
|
||||
|
|
@ -0,0 +1,117 @@
|
|||
import random
|
||||
import strformat
|
||||
import strutils
|
||||
|
||||
type
|
||||
|
||||
# Enumeration type for bases.
|
||||
Base {.pure.} = enum A, C, G, T, Other = "other"
|
||||
|
||||
# Sequence of bases.
|
||||
DnaSequence = string
|
||||
|
||||
# Kind of mutation.
|
||||
Mutation = enum mutSwap, mutDelete, mutInsert
|
||||
|
||||
const MaxBaseVal = ord(Base.high) - 1 # Maximum base value.
|
||||
|
||||
#---------------------------------------------------------------------------------------------------
|
||||
|
||||
template toChar(base: Base): char = ($base)[0]
|
||||
|
||||
#---------------------------------------------------------------------------------------------------
|
||||
|
||||
proc newDnaSeq(length: Natural): DnaSequence =
|
||||
## Create a DNA sequence of given length.
|
||||
|
||||
result = newStringOfCap(length)
|
||||
for _ in 1..length:
|
||||
result.add($Base(rand(MaxBaseVal)))
|
||||
|
||||
#---------------------------------------------------------------------------------------------------
|
||||
|
||||
proc mutate(dnaSeq: var DnaSequence) =
|
||||
## Mutate a sequence (it is changed in place).
|
||||
|
||||
# Choose randomly the position of mutation.
|
||||
let idx = rand(dnaSeq.high)
|
||||
|
||||
# Choose randomly the kind of mutation.
|
||||
let mut = Mutation(rand(ord(Mutation.high)))
|
||||
|
||||
# Apply the mutation.
|
||||
case mut
|
||||
|
||||
of mutSwap:
|
||||
let newBase = Base(rand(MaxBaseVal))
|
||||
echo fmt"Changing base at position {idx + 1} from {dnaSeq[idx]} to {newBase}"
|
||||
dnaSeq[idx] = newBase.toChar
|
||||
|
||||
of mutDelete:
|
||||
echo fmt"Deleting base {dnaSeq[idx]} at position {idx + 1}"
|
||||
dnaSeq.delete(idx, idx)
|
||||
|
||||
of mutInsert:
|
||||
let newBase = Base(rand(MaxBaseVal))
|
||||
echo fmt"Inserting base {newBase} at position {idx + 1}"
|
||||
dnaSeq.insert($newBase, idx)
|
||||
|
||||
#---------------------------------------------------------------------------------------------------
|
||||
|
||||
proc display(dnaSeq: DnaSequence) =
|
||||
## Display a DNA sequence using EMBL format.
|
||||
|
||||
var counts: array[Base, Natural] # Count of bases.
|
||||
for c in dnaSeq:
|
||||
inc counts[parseEnum[Base]($c, Other)] # Use Other as default value.
|
||||
|
||||
# Display the SQ line.
|
||||
var sqline = fmt"SQ {dnaSeq.len} BP; "
|
||||
for (base, count) in counts.pairs:
|
||||
sqline &= fmt"{count} {base}; "
|
||||
echo sqline
|
||||
|
||||
# Display the sequence.
|
||||
var idx = 0
|
||||
var row = newStringOfCap(80)
|
||||
var remaining = dnaSeq.len
|
||||
|
||||
while remaining > 0:
|
||||
row.setLen(0)
|
||||
row.add(" ")
|
||||
|
||||
# Add groups of 10 bases.
|
||||
for group in 1..6:
|
||||
let nextIdx = idx + min(10, remaining)
|
||||
for i in idx..<nextIdx:
|
||||
row.add($dnaSeq[i])
|
||||
row.add(' ')
|
||||
dec remaining, nextIdx - idx
|
||||
idx = nextIdx
|
||||
if remaining == 0:
|
||||
break
|
||||
|
||||
# Append the number of the last base in the row.
|
||||
row.add(spaces(72 - row.len))
|
||||
row.add(fmt"{idx:>8}")
|
||||
echo row
|
||||
|
||||
# Add termination.
|
||||
echo "//"
|
||||
|
||||
#———————————————————————————————————————————————————————————————————————————————————————————————————
|
||||
|
||||
randomize()
|
||||
var dnaSeq = newDnaSeq(200)
|
||||
echo "Initial sequence"
|
||||
echo "———————————————\n"
|
||||
dnaSeq.display()
|
||||
|
||||
echo "\nMutations"
|
||||
echo "—————————\n"
|
||||
for _ in 1..10:
|
||||
dnaSeq.mutate()
|
||||
|
||||
echo "\nMutated sequence"
|
||||
echo "————————————————\n"
|
||||
dnaSeq.display()
|
||||
|
|
@ -0,0 +1,44 @@
|
|||
use strict;
|
||||
use warnings;
|
||||
use feature 'say';
|
||||
|
||||
my @bases = <A C G T>;
|
||||
|
||||
my $dna;
|
||||
$dna .= $bases[int rand 4] for 1..200;
|
||||
|
||||
my %cnt;
|
||||
$cnt{$_}++ for split //, $dna;
|
||||
|
||||
sub pretty {
|
||||
my($string) = @_;
|
||||
my $chunk = 10;
|
||||
my $wrap = 5 * ($chunk+1);
|
||||
($string =~ s/(.{$chunk})/$1 /gr) =~ s/(.{$wrap})/$1\n/gr;
|
||||
}
|
||||
|
||||
sub mutate {
|
||||
my($dna,$count) = @_;
|
||||
my $orig = $dna;
|
||||
substr($dna,rand length $dna,1) = $bases[int rand 4] while $count > diff($orig, $dna) =~ tr/acgt//;
|
||||
$dna
|
||||
}
|
||||
|
||||
sub diff {
|
||||
my($orig, $repl) = @_;
|
||||
for my $i (0 .. -1+length $orig) {
|
||||
substr($repl,$i,1, lc substr $repl,$i,1) if substr($orig,$i,1) ne substr($repl,$i,1);
|
||||
}
|
||||
$repl;
|
||||
}
|
||||
|
||||
say "Original DNA strand:\n" . pretty($dna);
|
||||
say "Total bases: ". length $dna;
|
||||
say "$_: $cnt{$_}" for @bases;
|
||||
|
||||
my $mutate = mutate($dna, 10);
|
||||
%cnt = ();
|
||||
$cnt{$_}++ for split //, $mutate;
|
||||
say "\nMutated DNA strand:\n" . pretty diff $dna, $mutate;
|
||||
say "Total bases: ". length $mutate;
|
||||
say "$_: $cnt{$_}" for @bases;
|
||||
|
|
@ -0,0 +1,37 @@
|
|||
(phixonline)-->
|
||||
<span style="color: #004080;">string</span> <span style="color: #000000;">dna</span> <span style="color: #0000FF;">=</span> <span style="color: #7060A8;">repeat</span><span style="color: #0000FF;">(</span><span style="color: #008000;">' '</span><span style="color: #0000FF;">,</span><span style="color: #000000;">200</span><span style="color: #0000FF;">+</span><span style="color: #7060A8;">rand</span><span style="color: #0000FF;">(</span><span style="color: #000000;">300</span><span style="color: #0000FF;">))</span>
|
||||
<span style="color: #008080;">for</span> <span style="color: #000000;">i</span><span style="color: #0000FF;">=</span><span style="color: #000000;">1</span> <span style="color: #008080;">to</span> <span style="color: #7060A8;">length</span><span style="color: #0000FF;">(</span><span style="color: #000000;">dna</span><span style="color: #0000FF;">)</span> <span style="color: #008080;">do</span> <span style="color: #000000;">dna</span><span style="color: #0000FF;">[</span><span style="color: #000000;">i</span><span style="color: #0000FF;">]</span> <span style="color: #0000FF;">=</span> <span style="color: #008000;">"ACGT"</span><span style="color: #0000FF;">[</span><span style="color: #7060A8;">rand</span><span style="color: #0000FF;">(</span><span style="color: #000000;">4</span><span style="color: #0000FF;">)]</span> <span style="color: #008080;">end</span> <span style="color: #008080;">for</span>
|
||||
|
||||
<span style="color: #008080;">procedure</span> <span style="color: #000000;">show</span><span style="color: #0000FF;">()</span>
|
||||
<span style="color: #004080;">sequence</span> <span style="color: #000000;">acgt</span> <span style="color: #0000FF;">=</span> <span style="color: #7060A8;">repeat</span><span style="color: #0000FF;">(</span><span style="color: #000000;">0</span><span style="color: #0000FF;">,</span><span style="color: #000000;">5</span><span style="color: #0000FF;">)</span>
|
||||
<span style="color: #008080;">for</span> <span style="color: #000000;">i</span><span style="color: #0000FF;">=</span><span style="color: #000000;">1</span> <span style="color: #008080;">to</span> <span style="color: #7060A8;">length</span><span style="color: #0000FF;">(</span><span style="color: #000000;">dna</span><span style="color: #0000FF;">)</span> <span style="color: #008080;">do</span>
|
||||
<span style="color: #000000;">acgt</span><span style="color: #0000FF;">[</span><span style="color: #7060A8;">find</span><span style="color: #0000FF;">(</span><span style="color: #000000;">dna</span><span style="color: #0000FF;">[</span><span style="color: #000000;">i</span><span style="color: #0000FF;">],</span><span style="color: #008000;">"ACGT"</span><span style="color: #0000FF;">)]</span> <span style="color: #0000FF;">+=</span> <span style="color: #000000;">1</span>
|
||||
<span style="color: #008080;">end</span> <span style="color: #008080;">for</span>
|
||||
<span style="color: #000000;">acgt</span><span style="color: #0000FF;">[$]</span> <span style="color: #0000FF;">=</span> <span style="color: #7060A8;">sum</span><span style="color: #0000FF;">(</span><span style="color: #000000;">acgt</span><span style="color: #0000FF;">)</span>
|
||||
<span style="color: #004080;">sequence</span> <span style="color: #000000;">s</span> <span style="color: #0000FF;">=</span> <span style="color: #7060A8;">split</span><span style="color: #0000FF;">(</span><span style="color: #7060A8;">trim</span><span style="color: #0000FF;">(</span><span style="color: #7060A8;">join_by</span><span style="color: #0000FF;">(</span><span style="color: #7060A8;">split</span><span style="color: #0000FF;">(</span><span style="color: #7060A8;">join_by</span><span style="color: #0000FF;">(</span><span style="color: #000000;">dna</span><span style="color: #0000FF;">,</span><span style="color: #000000;">1</span><span style="color: #0000FF;">,</span><span style="color: #000000;">10</span><span style="color: #0000FF;">,</span><span style="color: #008000;">""</span><span style="color: #0000FF;">),</span><span style="color: #008000;">"\n"</span><span style="color: #0000FF;">),</span><span style="color: #000000;">1</span><span style="color: #0000FF;">,</span><span style="color: #000000;">5</span><span style="color: #0000FF;">,</span><span style="color: #008000;">" "</span><span style="color: #0000FF;">)),</span><span style="color: #008000;">"\n"</span><span style="color: #0000FF;">)</span>
|
||||
<span style="color: #008080;">for</span> <span style="color: #000000;">i</span><span style="color: #0000FF;">=</span><span style="color: #000000;">1</span> <span style="color: #008080;">to</span> <span style="color: #7060A8;">length</span><span style="color: #0000FF;">(</span><span style="color: #000000;">s</span><span style="color: #0000FF;">)</span> <span style="color: #008080;">do</span>
|
||||
<span style="color: #7060A8;">printf</span><span style="color: #0000FF;">(</span><span style="color: #000000;">1</span><span style="color: #0000FF;">,</span><span style="color: #008000;">"%3d: %s\n"</span><span style="color: #0000FF;">,{(</span><span style="color: #000000;">i</span><span style="color: #0000FF;">-</span><span style="color: #000000;">1</span><span style="color: #0000FF;">)*</span><span style="color: #000000;">50</span><span style="color: #0000FF;">+</span><span style="color: #000000;">1</span><span style="color: #0000FF;">,</span><span style="color: #000000;">s</span><span style="color: #0000FF;">[</span><span style="color: #000000;">i</span><span style="color: #0000FF;">]})</span>
|
||||
<span style="color: #008080;">end</span> <span style="color: #008080;">for</span>
|
||||
<span style="color: #7060A8;">printf</span><span style="color: #0000FF;">(</span><span style="color: #000000;">1</span><span style="color: #0000FF;">,</span><span style="color: #008000;">"\nBase counts: A:%d, C:%d, G:%d, T:%d, total:%d\n"</span><span style="color: #0000FF;">,</span><span style="color: #000000;">acgt</span><span style="color: #0000FF;">)</span>
|
||||
<span style="color: #008080;">end</span> <span style="color: #008080;">procedure</span>
|
||||
|
||||
<span style="color: #008080;">procedure</span> <span style="color: #000000;">mutate</span><span style="color: #0000FF;">()</span>
|
||||
<span style="color: #7060A8;">printf</span><span style="color: #0000FF;">(</span><span style="color: #000000;">1</span><span style="color: #0000FF;">,</span><span style="color: #008000;">"\n"</span><span style="color: #0000FF;">)</span>
|
||||
<span style="color: #008080;">for</span> <span style="color: #000000;">i</span><span style="color: #0000FF;">=</span><span style="color: #000000;">1</span> <span style="color: #008080;">to</span> <span style="color: #000000;">10</span> <span style="color: #008080;">do</span>
|
||||
<span style="color: #004080;">integer</span> <span style="color: #000000;">p</span> <span style="color: #0000FF;">=</span> <span style="color: #7060A8;">rand</span><span style="color: #0000FF;">(</span><span style="color: #7060A8;">length</span><span style="color: #0000FF;">(</span><span style="color: #000000;">dna</span><span style="color: #0000FF;">)),</span>
|
||||
<span style="color: #000000;">sdi</span> <span style="color: #0000FF;">=</span> <span style="color: #008000;">"SDI"</span><span style="color: #0000FF;">[</span><span style="color: #7060A8;">rand</span><span style="color: #0000FF;">(</span><span style="color: #000000;">3</span><span style="color: #0000FF;">)],</span>
|
||||
<span style="color: #000000;">rep</span> <span style="color: #0000FF;">=</span> <span style="color: #008000;">"ACGT"</span><span style="color: #0000FF;">[</span><span style="color: #7060A8;">rand</span><span style="color: #0000FF;">(</span><span style="color: #000000;">4</span><span style="color: #0000FF;">)],</span>
|
||||
<span style="color: #000000;">was</span> <span style="color: #0000FF;">=</span> <span style="color: #000000;">dna</span><span style="color: #0000FF;">[</span><span style="color: #000000;">p</span><span style="color: #0000FF;">]</span>
|
||||
<span style="color: #008080;">switch</span> <span style="color: #000000;">sdi</span> <span style="color: #008080;">do</span>
|
||||
<span style="color: #008080;">case</span> <span style="color: #008000;">'S'</span><span style="color: #0000FF;">:</span><span style="color: #000000;">dna</span><span style="color: #0000FF;">[</span><span style="color: #000000;">p</span><span style="color: #0000FF;">]</span> <span style="color: #0000FF;">=</span> <span style="color: #000000;">rep</span> <span style="color: #7060A8;">printf</span><span style="color: #0000FF;">(</span><span style="color: #000000;">1</span><span style="color: #0000FF;">,</span><span style="color: #008000;">"swapped %c at %d for %c\n"</span><span style="color: #0000FF;">,{</span><span style="color: #000000;">was</span><span style="color: #0000FF;">,</span><span style="color: #000000;">p</span><span style="color: #0000FF;">,</span><span style="color: #000000;">rep</span><span style="color: #0000FF;">})</span>
|
||||
<span style="color: #008080;">case</span> <span style="color: #008000;">'D'</span><span style="color: #0000FF;">:</span><span style="color: #000000;">dna</span><span style="color: #0000FF;">[</span><span style="color: #000000;">p</span><span style="color: #0000FF;">..</span><span style="color: #000000;">p</span><span style="color: #0000FF;">]</span> <span style="color: #0000FF;">=</span> <span style="color: #008000;">""</span> <span style="color: #7060A8;">printf</span><span style="color: #0000FF;">(</span><span style="color: #000000;">1</span><span style="color: #0000FF;">,</span><span style="color: #008000;">"deleted %c at %d\n"</span><span style="color: #0000FF;">,{</span><span style="color: #000000;">was</span><span style="color: #0000FF;">,</span><span style="color: #000000;">p</span><span style="color: #0000FF;">})</span>
|
||||
<span style="color: #008080;">case</span> <span style="color: #008000;">'I'</span><span style="color: #0000FF;">:</span><span style="color: #000000;">dna</span><span style="color: #0000FF;">[</span><span style="color: #000000;">p</span><span style="color: #0000FF;">..</span><span style="color: #000000;">p</span><span style="color: #0000FF;">-</span><span style="color: #000000;">1</span><span style="color: #0000FF;">]</span> <span style="color: #0000FF;">=</span> <span style="color: #008000;">""</span><span style="color: #0000FF;">&</span><span style="color: #000000;">rep</span> <span style="color: #7060A8;">printf</span><span style="color: #0000FF;">(</span><span style="color: #000000;">1</span><span style="color: #0000FF;">,</span><span style="color: #008000;">"inserted %c at %d, before %c\n"</span><span style="color: #0000FF;">,{</span><span style="color: #000000;">rep</span><span style="color: #0000FF;">,</span><span style="color: #000000;">p</span><span style="color: #0000FF;">,</span><span style="color: #000000;">was</span><span style="color: #0000FF;">})</span>
|
||||
<span style="color: #008080;">end</span> <span style="color: #008080;">switch</span>
|
||||
<span style="color: #008080;">end</span> <span style="color: #008080;">for</span>
|
||||
<span style="color: #7060A8;">printf</span><span style="color: #0000FF;">(</span><span style="color: #000000;">1</span><span style="color: #0000FF;">,</span><span style="color: #008000;">"\n"</span><span style="color: #0000FF;">)</span>
|
||||
<span style="color: #008080;">end</span> <span style="color: #008080;">procedure</span>
|
||||
|
||||
<span style="color: #000000;">show</span><span style="color: #0000FF;">()</span>
|
||||
<span style="color: #000000;">mutate</span><span style="color: #0000FF;">()</span>
|
||||
<span style="color: #000000;">show</span><span style="color: #0000FF;">()</span>
|
||||
<!--
|
||||
|
|
@ -0,0 +1,58 @@
|
|||
#BASE$="ACGT"
|
||||
#SEQLEN=200
|
||||
#PROTOCOL=#True
|
||||
|
||||
Global dna.s
|
||||
Define i.i
|
||||
|
||||
Procedure pprint()
|
||||
Define p.i, cnt.i, sum.i
|
||||
|
||||
For p=1 To Len(dna) Step 50
|
||||
Print(RSet(Str(p-1)+": ",5))
|
||||
PrintN(Mid(dna,p,50))
|
||||
Next
|
||||
PrintN("Base counts:")
|
||||
For p=1 To 4
|
||||
cnt=CountString(dna,Mid(#BASE$,p,1)) : sum+cnt
|
||||
Print(Mid(#BASE$,p,1)+": "+Str(cnt)+", ")
|
||||
Next
|
||||
PrintN("Total: "+Str(sum))
|
||||
EndProcedure
|
||||
|
||||
Procedure InsertAtPos(basenr.i,position.i)
|
||||
If #PROTOCOL : PrintN("Insert base "+Mid(#BASE$,basenr,1)+" at position "+Str(position)) : EndIf
|
||||
dna=InsertString(dna,Mid(#BASE$,basenr,1),position)
|
||||
EndProcedure
|
||||
|
||||
Procedure EraseAtPos(position.i)
|
||||
If #PROTOCOL : PrintN("Erase base "+Mid(dna,position,1)+" at position "+Str(position)) : EndIf
|
||||
If position>0 And position<=Len(dna)
|
||||
dna=Left(dna,position-1)+Right(dna,Len(dna)-position)
|
||||
EndIf
|
||||
EndProcedure
|
||||
|
||||
Procedure OverwriteAtPos(basenr.i,position.i)
|
||||
If #PROTOCOL : PrintN("Change base at position "+Str(position)+" from "+Mid(dna,position,1)+" to "+Mid(#BASE$,basenr,1)) : EndIf
|
||||
If position>0 And position<=Len(dna)
|
||||
position-1
|
||||
PokeS(@dna+2*position,Mid(#BASE$,basenr,1),-1,#PB_String_NoZero)
|
||||
EndIf
|
||||
EndProcedure
|
||||
|
||||
If OpenConsole()=0 : End 1 : EndIf
|
||||
For i=1 To #SEQLEN : dna+Mid(#BASE$,Random(4,1),1) : Next
|
||||
PrintN("Initial sequence:")
|
||||
pprint()
|
||||
|
||||
For i=1 To 10
|
||||
Select Random(2)
|
||||
Case 0 : InsertAtPos(Random(4,1),Random(Len(dna),1))
|
||||
Case 1 : EraseAtPos(Random(Len(dna),1))
|
||||
Case 2 : OverwriteAtPos(Random(4,1),Random(Len(dna),1))
|
||||
EndSelect
|
||||
Next
|
||||
|
||||
PrintN("After 10 mutations:")
|
||||
pprint()
|
||||
Input()
|
||||
|
|
@ -0,0 +1,46 @@
|
|||
import random
|
||||
from collections import Counter
|
||||
|
||||
def basecount(dna):
|
||||
return sorted(Counter(dna).items())
|
||||
|
||||
def seq_split(dna, n=50):
|
||||
return [dna[i: i+n] for i in range(0, len(dna), n)]
|
||||
|
||||
def seq_pp(dna, n=50):
|
||||
for i, part in enumerate(seq_split(dna, n)):
|
||||
print(f"{i*n:>5}: {part}")
|
||||
print("\n BASECOUNT:")
|
||||
tot = 0
|
||||
for base, count in basecount(dna):
|
||||
print(f" {base:>3}: {count}")
|
||||
tot += count
|
||||
base, count = 'TOT', tot
|
||||
print(f" {base:>3}= {count}")
|
||||
|
||||
def seq_mutate(dna, count=1, kinds="IDSSSS", choice="ATCG" ):
|
||||
mutation = []
|
||||
k2txt = dict(I='Insert', D='Delete', S='Substitute')
|
||||
for _ in range(count):
|
||||
kind = random.choice(kinds)
|
||||
index = random.randint(0, len(dna))
|
||||
if kind == 'I': # Insert
|
||||
dna = dna[:index] + random.choice(choice) + dna[index:]
|
||||
elif kind == 'D' and dna: # Delete
|
||||
dna = dna[:index] + dna[index+1:]
|
||||
elif kind == 'S' and dna: # Substitute
|
||||
dna = dna[:index] + random.choice(choice) + dna[index+1:]
|
||||
mutation.append((k2txt[kind], index))
|
||||
return dna, mutation
|
||||
|
||||
if __name__ == '__main__':
|
||||
length = 250
|
||||
print("SEQUENCE:")
|
||||
sequence = ''.join(random.choices('ACGT', weights=(1, 0.8, .9, 1.1), k=length))
|
||||
seq_pp(sequence)
|
||||
print("\n\nMUTATIONS:")
|
||||
mseq, m = seq_mutate(sequence, 10)
|
||||
for kind, index in m:
|
||||
print(f" {kind:>10} @{index}")
|
||||
print()
|
||||
seq_pp(mseq)
|
||||
|
|
@ -0,0 +1,18 @@
|
|||
[ $ "ACGT" 4 random peek ] is randomgene ( --> c )
|
||||
|
||||
[ $ "" swap times
|
||||
[ randomgene join ] ] is randomsequence ( n --> $ )
|
||||
|
||||
[ dup size random
|
||||
3 random
|
||||
[ table
|
||||
[ pluck drop ]
|
||||
[ randomgene unrot stuff ]
|
||||
[ randomgene unrot poke ] ]
|
||||
do ] is mutate ( $ --> $ )
|
||||
|
||||
200 randomsequence
|
||||
dup prettyprint cr cr dup tallybases
|
||||
cr cr say "Mutating..." cr
|
||||
10 times mutate
|
||||
dup prettyprint cr cr tallybases
|
||||
|
|
@ -0,0 +1,78 @@
|
|||
#lang racket
|
||||
|
||||
(define current-S-weight (make-parameter 1))
|
||||
(define current-D-weight (make-parameter 1))
|
||||
(define current-I-weight (make-parameter 1))
|
||||
|
||||
(define bases '(#\A #\C #\G #\T))
|
||||
|
||||
(define (fold-sequence seq kons #:finalise (finalise (λ x (apply values x))) . k0s)
|
||||
(define (recur seq . ks)
|
||||
(if (null? seq)
|
||||
(call-with-values (λ () (apply finalise ks)) (λ vs (apply values vs)))
|
||||
(call-with-values (λ () (apply kons (car seq) ks)) (λ ks+ (apply recur (cdr seq) ks+)))))
|
||||
(apply recur (if (string? seq) (string->list (regexp-replace* #px"[^ACGT]" seq "")) seq) k0s))
|
||||
|
||||
(define (sequence->pretty-printed-string seq)
|
||||
(define (fmt idx cs-rev) (format "~a: ~a" (~a idx #:width 3 #:align 'right) (list->string (reverse cs-rev))))
|
||||
(fold-sequence
|
||||
seq
|
||||
(λ (b n start-idx lns-rev cs-rev)
|
||||
(if (zero? (modulo n 50))
|
||||
(values (+ n 1) n (if (pair? cs-rev) (cons (fmt start-idx cs-rev) lns-rev) lns-rev) (cons b null))
|
||||
(values (+ n 1) start-idx lns-rev (cons b cs-rev))))
|
||||
0 0 null null
|
||||
#:finalise (λ (n idx lns-rev cs-rev)
|
||||
(string-join (reverse (if (null? cs-rev) lns-rev (cons (fmt idx cs-rev) lns-rev))) "\n"))))
|
||||
|
||||
(define (count-bases b as cs gs ts n)
|
||||
(values (+ as (if (eq? b #\A) 1 0)) (+ cs (if (eq? b #\C) 1 0)) (+ gs (if (eq? b #\T) 1 0)) (+ ts (if (eq? b #\G) 1 0)) (add1 n)))
|
||||
|
||||
(define (report-sequence s)
|
||||
(define-values (as cs gs ts n) (fold-sequence s count-bases 0 0 0 0 0))
|
||||
(printf "SEQUENCE:~%~a~%" (sequence->pretty-printed-string s))
|
||||
(printf "BASE COUNT:~%-----------~%~a~%"
|
||||
(string-join (map (λ (c n) (format " ~a :~a" c (~a #:width 4 #:align 'right n))) bases (list as ts cs gs)) "\n"))
|
||||
(printf "TOTAL: ~a~%" n))
|
||||
|
||||
|
||||
(define (make-random-sequence-string l)
|
||||
(list->string (for/list ((_ l)) (list-ref bases (random 4)))))
|
||||
|
||||
(define (weighted-random-call weights-and-functions . args)
|
||||
(let loop ((r (random)) (wfs weights-and-functions))
|
||||
(if (<= r (car wfs)) (apply (cadr wfs) args) (loop (- r (car wfs)) (cddr wfs)))))
|
||||
|
||||
(define (mutate-S s)
|
||||
(let ((r (random (string-length s))) (i (string (list-ref bases (random 4)))))
|
||||
(printf "Mutate at ~a -> ~a~%" r i)
|
||||
(string-append (substring s 0 r) i (substring s (add1 r)))))
|
||||
|
||||
(define (mutate-D s)
|
||||
(let ((r (random (string-length s))))
|
||||
(printf "Delete at ~a~%" r)
|
||||
(string-append (substring s 0 r) (substring s (add1 r)))))
|
||||
|
||||
(define (mutate-I s)
|
||||
(let ((r (random (string-length s))) (i (string (list-ref bases (random 4)))))
|
||||
(printf "Insert at ~a -> ~a~%" r i)
|
||||
(string-append (substring s 0 r) i (substring s r))))
|
||||
|
||||
(define (mutate s)
|
||||
(define W (+ (current-S-weight) (current-D-weight) (current-I-weight)))
|
||||
(weighted-random-call
|
||||
(list (/ (current-S-weight) W) mutate-S (/ (current-D-weight) W) mutate-D (/ (current-I-weight) W) mutate-I)
|
||||
s))
|
||||
|
||||
(module+
|
||||
main
|
||||
(define initial-sequence (make-random-sequence-string 200))
|
||||
(report-sequence initial-sequence)
|
||||
(newline)
|
||||
(define s+ (for/fold ((s initial-sequence)) ((_ 10)) (mutate s)))
|
||||
(newline)
|
||||
(report-sequence s+)
|
||||
(newline)
|
||||
(define s+d (parameterize ((current-D-weight 5)) (for/fold ((s initial-sequence)) ((_ 10)) (mutate s))))
|
||||
(newline)
|
||||
(report-sequence s+d))
|
||||
|
|
@ -0,0 +1,32 @@
|
|||
my @bases = <A C G T>;
|
||||
|
||||
# The DNA strand
|
||||
my $dna = @bases.roll(200).join;
|
||||
|
||||
|
||||
# The Task
|
||||
put "ORIGINAL DNA STRAND:";
|
||||
put pretty $dna;
|
||||
put "\nTotal bases: ", +my $bases = $dna.comb.Bag;
|
||||
put $bases.sort( ~*.key ).join: "\n";
|
||||
|
||||
put "\nMUTATED DNA STRAND:";
|
||||
my $mutate = $dna.&mutate(10);
|
||||
put pretty diff $dna, $mutate;
|
||||
put "\nTotal bases: ", +my $mutated = $mutate.comb.Bag;
|
||||
put $mutated.sort( ~*.key ).join: "\n";
|
||||
|
||||
|
||||
# Helper subs
|
||||
sub pretty ($string, $wrap = 50) {
|
||||
$string.comb($wrap).map( { sprintf "%8d: %s", $++ * $wrap, $_ } ).join: "\n"
|
||||
}
|
||||
|
||||
sub mutate ($dna is copy, $count = 1) {
|
||||
$dna.substr-rw((^$dna.chars).roll, 1) = @bases.roll for ^$count;
|
||||
$dna
|
||||
}
|
||||
|
||||
sub diff ($orig, $repl) {
|
||||
($orig.comb Z $repl.comb).map( -> ($o, $r) { $o eq $r ?? $o !! $r.lc }).join
|
||||
}
|
||||
|
|
@ -0,0 +1,112 @@
|
|||
row = 0
|
||||
dnaList = []
|
||||
base = ["A","C","G","T"]
|
||||
long = 20
|
||||
see "Initial sequence:" + nl
|
||||
see " 12345678901234567890" + nl
|
||||
see " " + long + ": "
|
||||
|
||||
for nr = 1 to 200
|
||||
row = row + 1
|
||||
rnd = random(3)+1
|
||||
baseStr = base[rnd]
|
||||
see baseStr # + " "
|
||||
if (row%20) = 0 and long < 200
|
||||
long = long + 20
|
||||
see nl
|
||||
if long < 100
|
||||
see " " + long + ": "
|
||||
else
|
||||
see "" + long + ": "
|
||||
ok
|
||||
ok
|
||||
add(dnaList,baseStr)
|
||||
next
|
||||
see nl+ " 12345678901234567890" + nl
|
||||
|
||||
baseCount(dnaList)
|
||||
|
||||
for n = 1 to 10
|
||||
rnd = random(2)+1
|
||||
switch rnd
|
||||
on 1
|
||||
baseSwap(dnaList)
|
||||
on 2
|
||||
baseDelete(dnaList)
|
||||
on 3
|
||||
baseInsert(dnaList)
|
||||
off
|
||||
next
|
||||
showDna(dnaList)
|
||||
baseCount(dnaList)
|
||||
|
||||
func baseInsert(dnaList)
|
||||
rnd1 = random(len(dnaList)-1)+1
|
||||
rnd2 = random(len(base)-1)+1
|
||||
insert(dnaList,rnd1,base[rnd2])
|
||||
see "Insert base " + base[rnd2] + " at position " + rnd1 + nl
|
||||
return dnaList
|
||||
|
||||
func baseDelete(dnaList)
|
||||
rnd = random(len(dnaList)-1)+1
|
||||
del(dnaList,rnd)
|
||||
see "Erase base " + dnaList[rnd] + " at position " + rnd + nl
|
||||
return dnaList
|
||||
|
||||
func baseSwap(dnaList)
|
||||
rnd1 = random(len(dnaList))
|
||||
rnd2 = random(3)+1
|
||||
see "Change base at position " + rnd1 + " from " + dnaList[rnd1] + " to " + base[rnd2] + nl
|
||||
dnaList[rnd1] = base[rnd2]
|
||||
|
||||
func showDna(dnaList)
|
||||
long = 20
|
||||
see nl + "After 10 mutations:" + nl
|
||||
see " 12345678901234567890" + nl
|
||||
see " " + long + ": "
|
||||
for nr = 1 to len(dnaList)
|
||||
row = row + 1
|
||||
see dnaList[nr]
|
||||
if (row%20) = 0 and long < 200
|
||||
long = long + 20
|
||||
see nl
|
||||
if long < 100
|
||||
see " " + long + ": "
|
||||
else
|
||||
see "" + long + ": "
|
||||
ok
|
||||
ok
|
||||
next
|
||||
see nl+ " 12345678901234567890" + nl
|
||||
|
||||
func baseCount(dnaList)
|
||||
dnaBase = [:A=0, :C=0, :G=0, :T=0]
|
||||
lenDna = len(dnaList)
|
||||
for n = 1 to lenDna
|
||||
dnaStr = dnaList[n]
|
||||
switch dnaStr
|
||||
on "A"
|
||||
strA = dnaBase["A"]
|
||||
strA++
|
||||
dnaBase["A"] = strA
|
||||
on "C"
|
||||
strC = dnaBase["C"]
|
||||
strC++
|
||||
dnaBase["C"] = strC
|
||||
on "G"
|
||||
strG = dnaBase["G"]
|
||||
strG++
|
||||
dnaBase["G"] = strG
|
||||
on "T"
|
||||
strT = dnaBase["T"]
|
||||
strT++
|
||||
dnaBase["T"] = strT
|
||||
off
|
||||
next
|
||||
see nl
|
||||
see "A: " + dnaBase["A"] + ", "
|
||||
see "T: " + dnaBase["T"] + ", "
|
||||
see "C: " + dnaBase["C"] + ", "
|
||||
see "G: " + dnaBase["G"] + ", "
|
||||
total = dnaBase["A"] + dnaBase["T"] + dnaBase["C"] + dnaBase["G"]
|
||||
see "Total: " + total+ nl + nl
|
||||
|
|
@ -0,0 +1,34 @@
|
|||
class DNA_Seq
|
||||
attr_accessor :seq
|
||||
|
||||
def initialize(bases: %i[A C G T] , size: 0)
|
||||
@bases = bases
|
||||
@seq = Array.new(size){ bases.sample }
|
||||
end
|
||||
|
||||
def mutate(n = 10)
|
||||
n.times{|n| method([:s, :d, :i].sample).call}
|
||||
end
|
||||
|
||||
def to_s(n = 50)
|
||||
just_size = @seq.size / n
|
||||
(0...@seq.size).step(n).map{|from| "#{from.to_s.rjust(just_size)} " + @seq[from, n].join}.join("\n") +
|
||||
"\nTotal #{seq.size}: #{@seq.tally.sort.to_h.inspect}\n\n"
|
||||
end
|
||||
|
||||
def s = @seq[rand_index]= @bases.sample
|
||||
def d = @seq.delete_at(rand_index)
|
||||
def i = @seq.insert(rand_index, @bases.sample )
|
||||
alias :swap :s
|
||||
alias :delete :d
|
||||
alias :insert :i
|
||||
|
||||
private
|
||||
def rand_index = rand( @seq.size )
|
||||
end
|
||||
|
||||
puts test = DNA_Seq.new(size: 200)
|
||||
test.mutate
|
||||
puts test
|
||||
test.delete
|
||||
puts test
|
||||
|
|
@ -0,0 +1,96 @@
|
|||
use rand::prelude::*;
|
||||
use std::collections::HashMap;
|
||||
use std::fmt::{Display, Formatter, Error};
|
||||
|
||||
pub struct Seq<'a> {
|
||||
alphabet: Vec<&'a str>,
|
||||
distr: rand::distributions::Uniform<usize>,
|
||||
pos_distr: rand::distributions::Uniform<usize>,
|
||||
seq: Vec<&'a str>,
|
||||
}
|
||||
|
||||
impl Display for Seq<'_> {
|
||||
fn fmt(&self, f: &mut Formatter) -> Result<(), Error> {
|
||||
|
||||
let pretty: String = self.seq
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, nt)| if (i + 1) % 60 == 0 { format!("{}\n", nt) } else { nt.to_string() })
|
||||
.collect();
|
||||
|
||||
let counts_hm = self.seq
|
||||
.iter()
|
||||
.fold(HashMap::<&str, usize>::new(), |mut m, nt| {
|
||||
*m.entry(nt).or_default() += 1;
|
||||
m
|
||||
});
|
||||
|
||||
let mut counts_vec: Vec<(&str, usize)> = counts_hm.into_iter().collect();
|
||||
counts_vec.sort_by(|a, b| a.0.cmp(&b.0));
|
||||
let counts_string = counts_vec
|
||||
.iter()
|
||||
.fold(String::new(), |mut counts_string, (nt, count)| {
|
||||
counts_string += &format!("{} = {}\n", nt, count);
|
||||
counts_string
|
||||
});
|
||||
|
||||
write!(f, "Seq:\n{}\n\nLength: {}\n\nCounts:\n{}", pretty, self.seq.len(), counts_string)
|
||||
}
|
||||
}
|
||||
|
||||
impl Seq<'_> {
|
||||
pub fn new(alphabet: Vec<&str>, len: usize) -> Seq {
|
||||
let distr = rand::distributions::Uniform::new_inclusive(0, alphabet.len() - 1);
|
||||
let pos_distr = rand::distributions::Uniform::new_inclusive(0, len - 1);
|
||||
|
||||
let seq: Vec<&str> = (0..len)
|
||||
.map(|_| {
|
||||
alphabet[thread_rng().sample(distr)]
|
||||
})
|
||||
.collect();
|
||||
Seq { alphabet, distr, pos_distr, seq }
|
||||
}
|
||||
|
||||
pub fn insert(&mut self) {
|
||||
let pos = thread_rng().sample(self.pos_distr);
|
||||
let nt = self.alphabet[thread_rng().sample(self.distr)];
|
||||
println!("Inserting {} at position {}", nt, pos);
|
||||
self.seq.insert(pos, nt);
|
||||
}
|
||||
|
||||
pub fn delete(&mut self) {
|
||||
let pos = thread_rng().sample(self.pos_distr);
|
||||
println!("Deleting {} at position {}", self.seq[pos], pos);
|
||||
self.seq.remove(pos);
|
||||
}
|
||||
|
||||
pub fn swap(&mut self) {
|
||||
let pos = thread_rng().sample(self.pos_distr);
|
||||
let cur_nt = self.seq[pos];
|
||||
let new_nt = self.alphabet[thread_rng().sample(self.distr)];
|
||||
println!("Replacing {} at position {} with {}", cur_nt, pos, new_nt);
|
||||
self.seq[pos] = new_nt;
|
||||
}
|
||||
}
|
||||
|
||||
fn main() {
|
||||
|
||||
let mut seq = Seq::new(vec!["A", "C", "T", "G"], 200);
|
||||
println!("Initial sequnce:\n{}", seq);
|
||||
|
||||
let mut_distr = rand::distributions::Uniform::new_inclusive(0, 2);
|
||||
|
||||
for _ in 0..10 {
|
||||
let mutation = thread_rng().sample(mut_distr);
|
||||
|
||||
if mutation == 0 {
|
||||
seq.insert()
|
||||
} else if mutation == 1 {
|
||||
seq.delete()
|
||||
} else {
|
||||
seq.swap()
|
||||
}
|
||||
}
|
||||
|
||||
println!("\nMutated sequence:\n{}", seq);
|
||||
}
|
||||
|
|
@ -0,0 +1,57 @@
|
|||
let bases: [Character] = ["A", "C", "G", "T"]
|
||||
|
||||
enum Action: CaseIterable {
|
||||
case swap, delete, insert
|
||||
}
|
||||
|
||||
@discardableResult
|
||||
func mutate(dna: inout String) -> Action {
|
||||
guard let i = dna.indices.shuffled().first(where: { $0 != dna.endIndex }) else {
|
||||
fatalError()
|
||||
}
|
||||
|
||||
let action = Action.allCases.randomElement()!
|
||||
|
||||
switch action {
|
||||
case .swap:
|
||||
dna.replaceSubrange(i..<i, with: [bases.randomElement()!])
|
||||
case .delete:
|
||||
dna.remove(at: i)
|
||||
case .insert:
|
||||
dna.insert(bases.randomElement()!, at: i)
|
||||
}
|
||||
|
||||
return action
|
||||
}
|
||||
|
||||
var d = ""
|
||||
|
||||
for _ in 0..<200 {
|
||||
d.append(bases.randomElement()!)
|
||||
}
|
||||
|
||||
func printSeq(_ dna: String) {
|
||||
for startI in stride(from: 0, to: dna.count, by: 50) {
|
||||
print("\(startI): \(dna.dropFirst(startI).prefix(50))")
|
||||
}
|
||||
|
||||
print()
|
||||
print("Size: \(dna.count)")
|
||||
print()
|
||||
|
||||
let counts = dna.reduce(into: [:], { $0[$1, default: 0] += 1 })
|
||||
|
||||
for (char, count) in counts.sorted(by: { $0.key < $1.key }) {
|
||||
print("\(char): \(count)")
|
||||
}
|
||||
}
|
||||
|
||||
printSeq(d)
|
||||
|
||||
print()
|
||||
|
||||
for _ in 0..<20 {
|
||||
mutate(dna: &d)
|
||||
}
|
||||
|
||||
printSeq(d)
|
||||
|
|
@ -0,0 +1,93 @@
|
|||
import rand
|
||||
import rand.seed
|
||||
|
||||
const bases = "ACGT"
|
||||
|
||||
// 'w' contains the weights out of 300 for each
|
||||
// of swap, delete or insert in that order.
|
||||
fn mutate(dna string, w [3]int) string {
|
||||
le := dna.len
|
||||
// get a random position in the dna to mutate
|
||||
p := rand.intn(le) or {0}
|
||||
// get a random number between 0 and 299 inclusive
|
||||
r := rand.intn(300) or {0}
|
||||
mut bytes := dna.bytes()
|
||||
match true {
|
||||
r < w[0] { // swap
|
||||
base := bases[rand.intn(4) or {0}]
|
||||
println(" Change @${p:3} ${[bytes[p]].bytestr()} to ${[base].bytestr()}")
|
||||
bytes[p] = base
|
||||
}
|
||||
r < w[0]+w[1] { // delete
|
||||
println(" Delete @${p:3} ${bytes[p]}")
|
||||
bytes.delete(p)
|
||||
//copy(bytes[p:], bytes[p+1:])
|
||||
bytes = bytes[0..le-1]
|
||||
}
|
||||
else { // insert
|
||||
base := bases[rand.intn(4) or {0}]
|
||||
bytes << 0
|
||||
bytes.insert(p,bytes[p])
|
||||
//copy(bytes[p+1:], bytes[p:])
|
||||
println(" Insert @${p:3} $base")
|
||||
bytes[p] = base
|
||||
}
|
||||
}
|
||||
return bytes.bytestr()
|
||||
}
|
||||
|
||||
// Generate a random dna sequence of given length.
|
||||
fn generate(le int) string {
|
||||
mut bytes := []u8{len:le}
|
||||
for i := 0; i < le; i++ {
|
||||
bytes[i] = bases[rand.intn(4) or {0}]
|
||||
}
|
||||
return bytes.bytestr()
|
||||
}
|
||||
|
||||
// Pretty print dna and stats.
|
||||
fn pretty_print(dna string, rowLen int) {
|
||||
println("SEQUENCE:")
|
||||
le := dna.len
|
||||
for i := 0; i < le; i += rowLen {
|
||||
mut k := i + rowLen
|
||||
if k > le {
|
||||
k = le
|
||||
}
|
||||
println("${i:5}: ${dna[i..k]}")
|
||||
}
|
||||
mut base_map := map[byte]int{} // allows for 'any' base
|
||||
for i in 0..le {
|
||||
base_map[dna[i]]++
|
||||
}
|
||||
mut bb := base_map.keys()
|
||||
bb.sort()
|
||||
|
||||
println("\nBASE COUNT:")
|
||||
for base in bb {
|
||||
println(" $base: ${base_map[base]:3}")
|
||||
}
|
||||
println(" ------")
|
||||
println(" Σ: $le")
|
||||
println(" ======\n")
|
||||
}
|
||||
|
||||
// Express weights as a string.
|
||||
fn wstring(w [3]int) string {
|
||||
return " Change: ${w[0]}\n Delete: ${w[1]}\n Insert: ${w[2]}\n"
|
||||
}
|
||||
|
||||
fn main() {
|
||||
rand.seed(seed.time_seed_array(2))
|
||||
mut dna := generate(250)
|
||||
pretty_print(dna, 50)
|
||||
muts := 10
|
||||
w := [100, 100, 100]! // use e.g. {0, 300, 0} to choose only deletions
|
||||
println("WEIGHTS (ex 300):\n${wstring(w)}")
|
||||
println("MUTATIONS ($muts):")
|
||||
for _ in 0..muts {
|
||||
dna = mutate(dna, w)
|
||||
}
|
||||
println('')
|
||||
pretty_print(dna, 50)
|
||||
}
|
||||
|
|
@ -0,0 +1,78 @@
|
|||
import "random" for Random
|
||||
import "/fmt" for Fmt
|
||||
import "/sort" for Sort
|
||||
|
||||
var rand = Random.new()
|
||||
var bases = "ACGT"
|
||||
|
||||
// 'w' contains the weights out of 300 for each
|
||||
// of swap, delete or insert in that order.
|
||||
var mutate = Fn.new { |dna, w|
|
||||
var le = dna.count
|
||||
// get a random position in the dna to mutate
|
||||
var p = rand.int(le)
|
||||
// get a random number between 0 and 299 inclusive
|
||||
var r = rand.int(300)
|
||||
var chars = dna.toList
|
||||
if (r < w[0]) { // swap
|
||||
var base = bases[rand.int(4)]
|
||||
Fmt.print(" Change @$3d $q to $q", p, chars[p], base)
|
||||
chars[p] = base
|
||||
} else if (r < w[0] + w[1]) { // delete
|
||||
Fmt.print(" Delete @$3d $q", p, chars[p])
|
||||
chars.removeAt(p)
|
||||
} else { // insert
|
||||
var base = bases[rand.int(4)]
|
||||
Fmt.print(" Insert @$3d $q", p, base)
|
||||
chars.insert(p, base)
|
||||
}
|
||||
return chars.join()
|
||||
}
|
||||
|
||||
// Generate a random dna sequence of given length.
|
||||
var generate = Fn.new { |le|
|
||||
var chars = [""] * le
|
||||
for (i in 0...le) chars[i] = bases[rand.int(4)]
|
||||
return chars.join()
|
||||
}
|
||||
|
||||
// Pretty print dna and stats.
|
||||
var prettyPrint = Fn.new { |dna, rowLen|
|
||||
System.print("SEQUENCE:")
|
||||
var le = dna.count
|
||||
var i = 0
|
||||
while (i < le) {
|
||||
var k = i + rowLen
|
||||
if (k > le) k = le
|
||||
Fmt.print("$5d: $s", i, dna[i...k])
|
||||
i = i + rowLen
|
||||
}
|
||||
var baseMap = {}
|
||||
for (i in 0...le) {
|
||||
var v = baseMap[dna[i]]
|
||||
baseMap[dna[i]] = (v) ? v + 1 : 1
|
||||
}
|
||||
var bases = []
|
||||
for (k in baseMap.keys) bases.add(k)
|
||||
Sort.insertion(bases) // get bases into alphabetic order
|
||||
System.print("\nBASE COUNT:")
|
||||
for (base in bases) Fmt.print(" $s: $3d", base, baseMap[base])
|
||||
System.print(" ------")
|
||||
System.print(" Σ: %(le)")
|
||||
System.print(" ======\n")
|
||||
}
|
||||
|
||||
// Express weights as a string.
|
||||
var wstring = Fn.new { |w|
|
||||
return Fmt.swrite(" Change: $d\n Delete: $d\n Insert: $d\n", w[0], w[1], w[2])
|
||||
}
|
||||
|
||||
var dna = generate.call(250)
|
||||
prettyPrint.call(dna, 50)
|
||||
var muts = 10
|
||||
var w = [100, 100, 100] // use e.g. {0, 300, 0} to choose only deletions
|
||||
Fmt.print("WEIGHTS (ex 300):\n$s", wstring.call(w))
|
||||
Fmt.print("MUTATIONS ($d):", muts)
|
||||
for (i in 0...muts) dna = mutate.call(dna, w)
|
||||
System.print()
|
||||
prettyPrint.call(dna, 50)
|
||||
|
|
@ -0,0 +1,53 @@
|
|||
// Rosetta Code problem: http://rosettacode.org/wiki/Sequence_mutation
|
||||
// by Galileo, 07/2022
|
||||
|
||||
r = int(ran(300))
|
||||
|
||||
for i = 1 to 200 + r : dna$ = dna$ + mid$("ACGT", int(ran(4))+1, 1) : next
|
||||
|
||||
sub show()
|
||||
local acgt(4), i, j, x, total
|
||||
|
||||
for i = 1 to len(dna$)
|
||||
x = instr("ACGT", mid$(dna$, i, 1))
|
||||
acgt(x) = acgt(x) + 1
|
||||
next
|
||||
|
||||
for i = 1 to 4 : total = total + acgt(i) : next
|
||||
|
||||
for i = 1 to len(dna$) step 50
|
||||
print i, ":\t";
|
||||
for j = 0 to 49 step 10
|
||||
print mid$(dna$, i+j, 10), " ";
|
||||
next
|
||||
print
|
||||
next
|
||||
print "\nBase counts: A: ", acgt(1), ", C: ", acgt(2), ", G: ", acgt(3), ", T: ", acgt(4), ", total: ", total
|
||||
end sub
|
||||
|
||||
|
||||
sub mutate()
|
||||
local i, p, sdi$, rep$, was$
|
||||
|
||||
print
|
||||
|
||||
for i = 1 to 10
|
||||
p = int(ran(len(dna$))) + 1
|
||||
sdi$ = mid$("SDI", int(ran(3)) + 1, 1)
|
||||
rep$ = mid$("ACGT", int(ran(4)) + 1, 1)
|
||||
was$ = mid$(dna$, p, 1)
|
||||
switch sdi$
|
||||
case "S": mid$(dna$, p, 1) = rep$
|
||||
print "swapped ", was$, " at ", p, " for ", rep$ : break
|
||||
case "D": dna$ = left$(dna$, p - 1) + right$(dna$, len(dna$) - p)
|
||||
print "deleted ", was$, " at ", p : break
|
||||
case "I": dna$ = left$(dna$, p - 1) + rep$ + right$(dna$, (len(dna$) - p + 1))
|
||||
print "inserted ", rep$, " at ", p, ", before ", was$ : break
|
||||
end switch
|
||||
next
|
||||
print
|
||||
end sub
|
||||
|
||||
show()
|
||||
mutate()
|
||||
show()
|
||||
|
|
@ -0,0 +1,33 @@
|
|||
var [const] bases="ACGT", lbases=bases.toLower();
|
||||
dna:=(190).pump(Data().howza(3),(0).random.fp(0,4),bases.get); // bucket of bytes
|
||||
|
||||
foreach s,m in (T("Original","Mutated").zip(T(True,False))){
|
||||
println("\n",s," DNA strand:"); dnaPP(dna);
|
||||
println("Base Counts: ", dna.len()," : ",
|
||||
dna.text.toUpper().counts() // ("A",5, "C",10, ...)
|
||||
.pump(String,Void.Read,"%s(%d) ".fmt));
|
||||
if(m) mutate(dna,10,True);
|
||||
}
|
||||
|
||||
fcn mutate(dna,count=1,verbose=False){
|
||||
if(verbose) println("Mutating:");
|
||||
do(count){
|
||||
n,rb := (0).random(dna.len()), lbases[(0).random(4)];
|
||||
switch( (0).random(3) ){
|
||||
case(0){ if(verbose) println("Change[%d] '%s' to '%s'".fmt(n,dna.charAt(n),rb));
|
||||
dna[n]=rb;
|
||||
}
|
||||
case(1){ if(verbose) println("Delete[%d] '%s'".fmt(n,dna.charAt(n)));
|
||||
dna.del(n);
|
||||
}
|
||||
else{ if(verbose) println("Insert[%d] '%s'".fmt(n,rb));
|
||||
dna.insert(n,rb);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fcn dnaPP(dna,N=50){
|
||||
[0..*,50].zipWith(fcn(n,bases){ println("%6d: %s".fmt(n,bases.concat())) },
|
||||
dna.walker().walk.fp(50)).pump(Void); // .pump forces the iterator
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue