September 2017 Update
This commit is contained in:
parent
bba7bfd280
commit
ba8067c3b7
14570 changed files with 153136 additions and 63871 deletions
|
|
@ -10,249 +10,249 @@
|
|||
/* -------- aux stuff ---------- */
|
||||
void* mem_alloc(size_t item_size, size_t n_item)
|
||||
{
|
||||
size_t *x = calloc(1, sizeof(size_t)*2 + n_item * item_size);
|
||||
x[0] = item_size;
|
||||
x[1] = n_item;
|
||||
return x + 2;
|
||||
size_t *x = calloc(1, sizeof(size_t)*2 + n_item * item_size);
|
||||
x[0] = item_size;
|
||||
x[1] = n_item;
|
||||
return x + 2;
|
||||
}
|
||||
|
||||
void* mem_extend(void *m, size_t new_n)
|
||||
{
|
||||
size_t *x = (size_t*)m - 2;
|
||||
x = realloc(x, sizeof(size_t) * 2 + *x * new_n);
|
||||
if (new_n > x[1])
|
||||
memset((char*)(x + 2) + x[0] * x[1], 0, x[0] * (new_n - x[1]));
|
||||
x[1] = new_n;
|
||||
return x + 2;
|
||||
size_t *x = (size_t*)m - 2;
|
||||
x = realloc(x, sizeof(size_t) * 2 + *x * new_n);
|
||||
if (new_n > x[1])
|
||||
memset((char*)(x + 2) + x[0] * x[1], 0, x[0] * (new_n - x[1]));
|
||||
x[1] = new_n;
|
||||
return x + 2;
|
||||
}
|
||||
|
||||
inline void _clear(void *m)
|
||||
{
|
||||
size_t *x = (size_t*)m - 2;
|
||||
memset(m, 0, x[0] * x[1]);
|
||||
size_t *x = (size_t*)m - 2;
|
||||
memset(m, 0, x[0] * x[1]);
|
||||
}
|
||||
|
||||
#define _new(type, n) mem_alloc(sizeof(type), n)
|
||||
#define _del(m) { free((size_t*)(m) - 2); m = 0; }
|
||||
#define _len(m) *((size_t*)m - 1)
|
||||
#define _setsize(m, n) m = mem_extend(m, n)
|
||||
#define _extend(m) m = mem_extend(m, _len(m) * 2)
|
||||
#define _new(type, n) mem_alloc(sizeof(type), n)
|
||||
#define _del(m) { free((size_t*)(m) - 2); m = 0; }
|
||||
#define _len(m) *((size_t*)m - 1)
|
||||
#define _setsize(m, n) m = mem_extend(m, n)
|
||||
#define _extend(m) m = mem_extend(m, _len(m) * 2)
|
||||
|
||||
|
||||
/* ----------- LZW stuff -------------- */
|
||||
typedef uint8_t byte;
|
||||
typedef uint16_t ushort;
|
||||
|
||||
#define M_CLR 256 /* clear table marker */
|
||||
#define M_EOD 257 /* end-of-data marker */
|
||||
#define M_NEW 258 /* new code index */
|
||||
#define M_CLR 256 /* clear table marker */
|
||||
#define M_EOD 257 /* end-of-data marker */
|
||||
#define M_NEW 258 /* new code index */
|
||||
|
||||
/* encode and decode dictionary structures.
|
||||
for encoding, entry at code index is a list of indices that follow current one,
|
||||
i.e. if code 97 is 'a', code 387 is 'ab', and code 1022 is 'abc',
|
||||
then dict[97].next['b'] = 387, dict[387].next['c'] = 1022, etc. */
|
||||
typedef struct {
|
||||
ushort next[256];
|
||||
ushort next[256];
|
||||
} lzw_enc_t;
|
||||
|
||||
/* for decoding, dictionary contains index of whatever prefix index plus trailing
|
||||
byte. i.e. like previous example,
|
||||
dict[1022] = { c: 'c', prev: 387 },
|
||||
dict[387] = { c: 'b', prev: 97 },
|
||||
dict[97] = { c: 'a', prev: 0 }
|
||||
dict[1022] = { c: 'c', prev: 387 },
|
||||
dict[387] = { c: 'b', prev: 97 },
|
||||
dict[97] = { c: 'a', prev: 0 }
|
||||
the "back" element is used for temporarily chaining indices when resolving
|
||||
a code to bytes
|
||||
*/
|
||||
typedef struct {
|
||||
ushort prev, back;
|
||||
byte c;
|
||||
ushort prev, back;
|
||||
byte c;
|
||||
} lzw_dec_t;
|
||||
|
||||
byte* lzw_encode(byte *in, int max_bits)
|
||||
{
|
||||
int len = _len(in), bits = 9, next_shift = 512;
|
||||
ushort code, c, nc, next_code = M_NEW;
|
||||
lzw_enc_t *d = _new(lzw_enc_t, 512);
|
||||
int len = _len(in), bits = 9, next_shift = 512;
|
||||
ushort code, c, nc, next_code = M_NEW;
|
||||
lzw_enc_t *d = _new(lzw_enc_t, 512);
|
||||
|
||||
if (max_bits > 15) max_bits = 15;
|
||||
if (max_bits < 9 ) max_bits = 12;
|
||||
if (max_bits > 15) max_bits = 15;
|
||||
if (max_bits < 9 ) max_bits = 12;
|
||||
|
||||
byte *out = _new(ushort, 4);
|
||||
int out_len = 0, o_bits = 0;
|
||||
uint32_t tmp = 0;
|
||||
byte *out = _new(ushort, 4);
|
||||
int out_len = 0, o_bits = 0;
|
||||
uint32_t tmp = 0;
|
||||
|
||||
inline void write_bits(ushort x) {
|
||||
tmp = (tmp << bits) | x;
|
||||
o_bits += bits;
|
||||
if (_len(out) <= out_len) _extend(out);
|
||||
while (o_bits >= 8) {
|
||||
o_bits -= 8;
|
||||
out[out_len++] = tmp >> o_bits;
|
||||
tmp &= (1 << o_bits) - 1;
|
||||
}
|
||||
}
|
||||
inline void write_bits(ushort x) {
|
||||
tmp = (tmp << bits) | x;
|
||||
o_bits += bits;
|
||||
if (_len(out) <= out_len) _extend(out);
|
||||
while (o_bits >= 8) {
|
||||
o_bits -= 8;
|
||||
out[out_len++] = tmp >> o_bits;
|
||||
tmp &= (1 << o_bits) - 1;
|
||||
}
|
||||
}
|
||||
|
||||
//write_bits(M_CLR);
|
||||
for (code = *(in++); --len; ) {
|
||||
c = *(in++);
|
||||
if ((nc = d[code].next[c]))
|
||||
code = nc;
|
||||
else {
|
||||
write_bits(code);
|
||||
nc = d[code].next[c] = next_code++;
|
||||
code = c;
|
||||
}
|
||||
//write_bits(M_CLR);
|
||||
for (code = *(in++); --len; ) {
|
||||
c = *(in++);
|
||||
if ((nc = d[code].next[c]))
|
||||
code = nc;
|
||||
else {
|
||||
write_bits(code);
|
||||
nc = d[code].next[c] = next_code++;
|
||||
code = c;
|
||||
}
|
||||
|
||||
/* next new code would be too long for current table */
|
||||
if (next_code == next_shift) {
|
||||
/* either reset table back to 9 bits */
|
||||
if (++bits > max_bits) {
|
||||
/* table clear marker must occur before bit reset */
|
||||
write_bits(M_CLR);
|
||||
/* next new code would be too long for current table */
|
||||
if (next_code == next_shift) {
|
||||
/* either reset table back to 9 bits */
|
||||
if (++bits > max_bits) {
|
||||
/* table clear marker must occur before bit reset */
|
||||
write_bits(M_CLR);
|
||||
|
||||
bits = 9;
|
||||
next_shift = 512;
|
||||
next_code = M_NEW;
|
||||
_clear(d);
|
||||
} else /* or extend table */
|
||||
_setsize(d, next_shift *= 2);
|
||||
}
|
||||
}
|
||||
bits = 9;
|
||||
next_shift = 512;
|
||||
next_code = M_NEW;
|
||||
_clear(d);
|
||||
} else /* or extend table */
|
||||
_setsize(d, next_shift *= 2);
|
||||
}
|
||||
}
|
||||
|
||||
write_bits(code);
|
||||
write_bits(M_EOD);
|
||||
if (tmp) write_bits(tmp);
|
||||
write_bits(code);
|
||||
write_bits(M_EOD);
|
||||
if (tmp) write_bits(tmp);
|
||||
|
||||
_del(d);
|
||||
_del(d);
|
||||
|
||||
_setsize(out, out_len);
|
||||
return out;
|
||||
_setsize(out, out_len);
|
||||
return out;
|
||||
}
|
||||
|
||||
byte* lzw_decode(byte *in)
|
||||
{
|
||||
byte *out = _new(byte, 4);
|
||||
int out_len = 0;
|
||||
byte *out = _new(byte, 4);
|
||||
int out_len = 0;
|
||||
|
||||
inline void write_out(byte c)
|
||||
{
|
||||
while (out_len >= _len(out)) _extend(out);
|
||||
out[out_len++] = c;
|
||||
}
|
||||
inline void write_out(byte c)
|
||||
{
|
||||
while (out_len >= _len(out)) _extend(out);
|
||||
out[out_len++] = c;
|
||||
}
|
||||
|
||||
lzw_dec_t *d = _new(lzw_dec_t, 512);
|
||||
int len, j, next_shift = 512, bits = 9, n_bits = 0;
|
||||
ushort code, c, t, next_code = M_NEW;
|
||||
lzw_dec_t *d = _new(lzw_dec_t, 512);
|
||||
int len, j, next_shift = 512, bits = 9, n_bits = 0;
|
||||
ushort code, c, t, next_code = M_NEW;
|
||||
|
||||
uint32_t tmp = 0;
|
||||
inline void get_code() {
|
||||
while(n_bits < bits) {
|
||||
if (len > 0) {
|
||||
len --;
|
||||
tmp = (tmp << 8) | *(in++);
|
||||
n_bits += 8;
|
||||
} else {
|
||||
tmp = tmp << (bits - n_bits);
|
||||
n_bits = bits;
|
||||
}
|
||||
}
|
||||
n_bits -= bits;
|
||||
code = tmp >> n_bits;
|
||||
tmp &= (1 << n_bits) - 1;
|
||||
}
|
||||
uint32_t tmp = 0;
|
||||
inline void get_code() {
|
||||
while(n_bits < bits) {
|
||||
if (len > 0) {
|
||||
len --;
|
||||
tmp = (tmp << 8) | *(in++);
|
||||
n_bits += 8;
|
||||
} else {
|
||||
tmp = tmp << (bits - n_bits);
|
||||
n_bits = bits;
|
||||
}
|
||||
}
|
||||
n_bits -= bits;
|
||||
code = tmp >> n_bits;
|
||||
tmp &= (1 << n_bits) - 1;
|
||||
}
|
||||
|
||||
inline void clear_table() {
|
||||
_clear(d);
|
||||
for (j = 0; j < 256; j++) d[j].c = j;
|
||||
next_code = M_NEW;
|
||||
next_shift = 512;
|
||||
bits = 9;
|
||||
};
|
||||
inline void clear_table() {
|
||||
_clear(d);
|
||||
for (j = 0; j < 256; j++) d[j].c = j;
|
||||
next_code = M_NEW;
|
||||
next_shift = 512;
|
||||
bits = 9;
|
||||
};
|
||||
|
||||
clear_table(); /* in case encoded bits didn't start with M_CLR */
|
||||
for (len = _len(in); len;) {
|
||||
get_code();
|
||||
if (code == M_EOD) break;
|
||||
if (code == M_CLR) {
|
||||
clear_table();
|
||||
continue;
|
||||
}
|
||||
clear_table(); /* in case encoded bits didn't start with M_CLR */
|
||||
for (len = _len(in); len;) {
|
||||
get_code();
|
||||
if (code == M_EOD) break;
|
||||
if (code == M_CLR) {
|
||||
clear_table();
|
||||
continue;
|
||||
}
|
||||
|
||||
if (code >= next_code) {
|
||||
fprintf(stderr, "Bad sequence\n");
|
||||
_del(out);
|
||||
goto bail;
|
||||
}
|
||||
if (code >= next_code) {
|
||||
fprintf(stderr, "Bad sequence\n");
|
||||
_del(out);
|
||||
goto bail;
|
||||
}
|
||||
|
||||
d[next_code].prev = c = code;
|
||||
while (c > 255) {
|
||||
t = d[c].prev; d[t].back = c; c = t;
|
||||
}
|
||||
d[next_code].prev = c = code;
|
||||
while (c > 255) {
|
||||
t = d[c].prev; d[t].back = c; c = t;
|
||||
}
|
||||
|
||||
d[next_code - 1].c = c;
|
||||
d[next_code - 1].c = c;
|
||||
|
||||
while (d[c].back) {
|
||||
write_out(d[c].c);
|
||||
t = d[c].back; d[c].back = 0; c = t;
|
||||
}
|
||||
write_out(d[c].c);
|
||||
while (d[c].back) {
|
||||
write_out(d[c].c);
|
||||
t = d[c].back; d[c].back = 0; c = t;
|
||||
}
|
||||
write_out(d[c].c);
|
||||
|
||||
if (++next_code >= next_shift) {
|
||||
if (++bits > 16) {
|
||||
/* if input was correct, we'd have hit M_CLR before this */
|
||||
fprintf(stderr, "Too many bits\n");
|
||||
_del(out);
|
||||
goto bail;
|
||||
}
|
||||
_setsize(d, next_shift *= 2);
|
||||
}
|
||||
}
|
||||
if (++next_code >= next_shift) {
|
||||
if (++bits > 16) {
|
||||
/* if input was correct, we'd have hit M_CLR before this */
|
||||
fprintf(stderr, "Too many bits\n");
|
||||
_del(out);
|
||||
goto bail;
|
||||
}
|
||||
_setsize(d, next_shift *= 2);
|
||||
}
|
||||
}
|
||||
|
||||
/* might be ok, so just whine, don't be drastic */
|
||||
if (code != M_EOD) fputs("Bits did not end in EOD\n", stderr);
|
||||
/* might be ok, so just whine, don't be drastic */
|
||||
if (code != M_EOD) fputs("Bits did not end in EOD\n", stderr);
|
||||
|
||||
_setsize(out, out_len);
|
||||
bail: _del(d);
|
||||
return out;
|
||||
_setsize(out, out_len);
|
||||
bail: _del(d);
|
||||
return out;
|
||||
}
|
||||
|
||||
int main()
|
||||
{
|
||||
int i, fd = open("unixdict.txt", O_RDONLY);
|
||||
int i, fd = open("unixdict.txt", O_RDONLY);
|
||||
|
||||
if (fd == -1) {
|
||||
fprintf(stderr, "Can't read file\n");
|
||||
return 1;
|
||||
};
|
||||
if (fd == -1) {
|
||||
fprintf(stderr, "Can't read file\n");
|
||||
return 1;
|
||||
};
|
||||
|
||||
struct stat st;
|
||||
fstat(fd, &st);
|
||||
struct stat st;
|
||||
fstat(fd, &st);
|
||||
|
||||
byte *in = _new(char, st.st_size);
|
||||
read(fd, in, st.st_size);
|
||||
_setsize(in, st.st_size);
|
||||
close(fd);
|
||||
byte *in = _new(char, st.st_size);
|
||||
read(fd, in, st.st_size);
|
||||
_setsize(in, st.st_size);
|
||||
close(fd);
|
||||
|
||||
printf("input size: %d\n", _len(in));
|
||||
printf("input size: %d\n", _len(in));
|
||||
|
||||
byte *enc = lzw_encode(in, 9);
|
||||
printf("encoded size: %d\n", _len(enc));
|
||||
byte *enc = lzw_encode(in, 9);
|
||||
printf("encoded size: %d\n", _len(enc));
|
||||
|
||||
byte *dec = lzw_decode(enc);
|
||||
printf("decoded size: %d\n", _len(dec));
|
||||
byte *dec = lzw_decode(enc);
|
||||
printf("decoded size: %d\n", _len(dec));
|
||||
|
||||
for (i = 0; i < _len(dec); i++)
|
||||
if (dec[i] != in[i]) {
|
||||
printf("bad decode at %d\n", i);
|
||||
break;
|
||||
}
|
||||
for (i = 0; i < _len(dec); i++)
|
||||
if (dec[i] != in[i]) {
|
||||
printf("bad decode at %d\n", i);
|
||||
break;
|
||||
}
|
||||
|
||||
if (i == _len(dec)) printf("Decoded ok\n");
|
||||
if (i == _len(dec)) printf("Decoded ok\n");
|
||||
|
||||
|
||||
_del(in);
|
||||
_del(enc);
|
||||
_del(dec);
|
||||
_del(in);
|
||||
_del(enc);
|
||||
_del(dec);
|
||||
|
||||
return 0;
|
||||
return 0;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,36 +0,0 @@
|
|||
lzw = (s) ->
|
||||
dct = {} # map substrings to codes between 256 and 4096
|
||||
stream = [] # array of compression results
|
||||
|
||||
# initialize basic ASCII characters
|
||||
for code_num in [0..255]
|
||||
c = String.fromCharCode(code_num)
|
||||
dct[c] = code_num
|
||||
code_num = 256
|
||||
|
||||
i = 0
|
||||
while i < s.length
|
||||
# Find word and new_word
|
||||
# word = longest substr already encountered, or next character
|
||||
# new_word = word plus next character, a new substr to encode
|
||||
word = ''
|
||||
j = i
|
||||
while j < s.length
|
||||
new_word = word + s[j]
|
||||
break if !dct[new_word]
|
||||
word = new_word
|
||||
j += 1
|
||||
|
||||
# stream out the code for the substring
|
||||
stream.push dct[word]
|
||||
|
||||
# build up our encoding dictionary
|
||||
if code_num < 4096
|
||||
dct[new_word] = code_num
|
||||
code_num += 1
|
||||
|
||||
# advance thru the string
|
||||
i += word.length
|
||||
stream
|
||||
|
||||
console.log lzw "TOBEORNOTTOBEORTOBEORNOT"
|
||||
|
|
@ -1,17 +0,0 @@
|
|||
> coffee lzw.coffee
|
||||
[ 84,
|
||||
79,
|
||||
66,
|
||||
69,
|
||||
79,
|
||||
82,
|
||||
78,
|
||||
79,
|
||||
84,
|
||||
256,
|
||||
258,
|
||||
260,
|
||||
265,
|
||||
259,
|
||||
261,
|
||||
263 ]
|
||||
|
|
@ -1,112 +1,112 @@
|
|||
class
|
||||
APPLICATION
|
||||
APPLICATION
|
||||
|
||||
create
|
||||
make
|
||||
make
|
||||
|
||||
feature {NONE}
|
||||
|
||||
make
|
||||
local
|
||||
test: LINKED_LIST [INTEGER]
|
||||
do
|
||||
create test.make
|
||||
test := compress ("TOBEORNOTTOBEORTOBEORNOT")
|
||||
across
|
||||
test as t
|
||||
loop
|
||||
io.put_string (t.item.out + " ")
|
||||
end
|
||||
io.new_line
|
||||
io.put_string (decompress (test))
|
||||
end
|
||||
make
|
||||
local
|
||||
test: LINKED_LIST [INTEGER]
|
||||
do
|
||||
create test.make
|
||||
test := compress ("TOBEORNOTTOBEORTOBEORNOT")
|
||||
across
|
||||
test as t
|
||||
loop
|
||||
io.put_string (t.item.out + " ")
|
||||
end
|
||||
io.new_line
|
||||
io.put_string (decompress (test))
|
||||
end
|
||||
|
||||
decompress (compressed: LINKED_LIST [INTEGER]): STRING
|
||||
--Decompressed version of 'compressed'.
|
||||
local
|
||||
dictsize, i, k: INTEGER
|
||||
dictionary: HASH_TABLE [STRING, INTEGER]
|
||||
w, entry: STRING
|
||||
char: CHARACTER_8
|
||||
do
|
||||
dictsize := 256
|
||||
create dictionary.make (300)
|
||||
create entry.make_empty
|
||||
create Result.make_empty
|
||||
from
|
||||
i := 0
|
||||
until
|
||||
i > 256
|
||||
loop
|
||||
char := i.to_character_8
|
||||
dictionary.put (char.out, i)
|
||||
i := i + 1
|
||||
end
|
||||
w := compressed.first.to_character_8.out
|
||||
compressed.go_i_th (1)
|
||||
compressed.remove
|
||||
Result := w
|
||||
from
|
||||
k := 1
|
||||
until
|
||||
k > compressed.count
|
||||
loop
|
||||
if attached dictionary.at (compressed [k]) as ata then
|
||||
entry := ata
|
||||
elseif compressed [k] = dictsize then
|
||||
entry := w + w.at (1).out
|
||||
else
|
||||
io.put_string ("EXEPTION")
|
||||
end
|
||||
Result := Result + entry
|
||||
dictsize := dictsize + 1
|
||||
dictionary.put (w + entry.at (1).out, dictsize)
|
||||
w := entry
|
||||
k := k + 1
|
||||
end
|
||||
end
|
||||
decompress (compressed: LINKED_LIST [INTEGER]): STRING
|
||||
--Decompressed version of 'compressed'.
|
||||
local
|
||||
dictsize, i, k: INTEGER
|
||||
dictionary: HASH_TABLE [STRING, INTEGER]
|
||||
w, entry: STRING
|
||||
char: CHARACTER_8
|
||||
do
|
||||
dictsize := 256
|
||||
create dictionary.make (300)
|
||||
create entry.make_empty
|
||||
create Result.make_empty
|
||||
from
|
||||
i := 0
|
||||
until
|
||||
i > 256
|
||||
loop
|
||||
char := i.to_character_8
|
||||
dictionary.put (char.out, i)
|
||||
i := i + 1
|
||||
end
|
||||
w := compressed.first.to_character_8.out
|
||||
compressed.go_i_th (1)
|
||||
compressed.remove
|
||||
Result := w
|
||||
from
|
||||
k := 1
|
||||
until
|
||||
k > compressed.count
|
||||
loop
|
||||
if attached dictionary.at (compressed [k]) as ata then
|
||||
entry := ata
|
||||
elseif compressed [k] = dictsize then
|
||||
entry := w + w.at (1).out
|
||||
else
|
||||
io.put_string ("EXEPTION")
|
||||
end
|
||||
Result := Result + entry
|
||||
dictsize := dictsize + 1
|
||||
dictionary.put (w + entry.at (1).out, dictsize)
|
||||
w := entry
|
||||
k := k + 1
|
||||
end
|
||||
end
|
||||
|
||||
compress (uncompressed: STRING): LINKED_LIST [INTEGER]
|
||||
-- Compressed version of 'uncompressed'.
|
||||
local
|
||||
dictsize: INTEGER
|
||||
dictionary: HASH_TABLE [INTEGER, STRING]
|
||||
i: INTEGER
|
||||
w, wc: STRING
|
||||
char: CHARACTER_8
|
||||
do
|
||||
dictsize := 256
|
||||
create dictionary.make (256)
|
||||
create w.make_empty
|
||||
from
|
||||
i := 0
|
||||
until
|
||||
i > 256
|
||||
loop
|
||||
char := i.to_character_8
|
||||
dictionary.put (i, char.out)
|
||||
i := i + 1
|
||||
end
|
||||
create Result.make
|
||||
from
|
||||
i := 1
|
||||
until
|
||||
i > uncompressed.count
|
||||
loop
|
||||
wc := w + uncompressed [i].out
|
||||
if dictionary.has (wc) then
|
||||
w := wc
|
||||
else
|
||||
Result.extend (dictionary.at (w))
|
||||
dictSize := dictSize + 1
|
||||
dictionary.put (dictSize, wc)
|
||||
w := "" + uncompressed [i].out
|
||||
end
|
||||
i := i + 1
|
||||
end
|
||||
if w.count > 0 then
|
||||
Result.extend (dictionary.at (w))
|
||||
end
|
||||
end
|
||||
compress (uncompressed: STRING): LINKED_LIST [INTEGER]
|
||||
-- Compressed version of 'uncompressed'.
|
||||
local
|
||||
dictsize: INTEGER
|
||||
dictionary: HASH_TABLE [INTEGER, STRING]
|
||||
i: INTEGER
|
||||
w, wc: STRING
|
||||
char: CHARACTER_8
|
||||
do
|
||||
dictsize := 256
|
||||
create dictionary.make (256)
|
||||
create w.make_empty
|
||||
from
|
||||
i := 0
|
||||
until
|
||||
i > 256
|
||||
loop
|
||||
char := i.to_character_8
|
||||
dictionary.put (i, char.out)
|
||||
i := i + 1
|
||||
end
|
||||
create Result.make
|
||||
from
|
||||
i := 1
|
||||
until
|
||||
i > uncompressed.count
|
||||
loop
|
||||
wc := w + uncompressed [i].out
|
||||
if dictionary.has (wc) then
|
||||
w := wc
|
||||
else
|
||||
Result.extend (dictionary.at (w))
|
||||
dictSize := dictSize + 1
|
||||
dictionary.put (dictSize, wc)
|
||||
w := "" + uncompressed [i].out
|
||||
end
|
||||
i := i + 1
|
||||
end
|
||||
if w.count > 0 then
|
||||
Result.extend (dictionary.at (w))
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@
|
|||
test() ->
|
||||
Str = "TOBEORNOTTOBEORTOBEORNOT",
|
||||
[84,79,66,69,79,82,78,79,84,256,258,260,265,259,261,263] =
|
||||
encode(Str),
|
||||
encode(Str),
|
||||
Str = decode(encode(Str)),
|
||||
ok.
|
||||
|
||||
|
|
@ -24,11 +24,11 @@ encode([H|T], D, Free, Out) ->
|
|||
|
||||
find_match([H|T], L, LastVal, D, Free, Out) ->
|
||||
case dict:find([H|L], D) of
|
||||
{ok, Val} ->
|
||||
find_match(T, [H|L], Val, D, Free, Out);
|
||||
error ->
|
||||
D1 = dict:store([H|L], Free, D),
|
||||
encode([H|T], D1, Free+1, [LastVal|Out])
|
||||
{ok, Val} ->
|
||||
find_match(T, [H|L], Val, D, Free, Out);
|
||||
error ->
|
||||
D1 = dict:store([H|L], Free, D),
|
||||
encode([H|T], D1, Free+1, [LastVal|Out])
|
||||
end;
|
||||
find_match([], _, LastVal, _, _, Out) ->
|
||||
reverse([LastVal|Out]).
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ create w 256 allot \ counted string
|
|||
|
||||
: free-dict forth-wordlist set-current ;
|
||||
|
||||
: in-dict? ( key len -- ? ) \ can assume len > 1
|
||||
: in-dict? ( key len -- ? ) \ can assume len > 1
|
||||
dict search-wordlist dup if nip then ;
|
||||
|
||||
: lookup-dict ( key len -- symbol )
|
||||
|
|
@ -83,7 +83,7 @@ create entry 256 allot
|
|||
else
|
||||
abort" bad symbol!"
|
||||
then then
|
||||
entry count type \ output
|
||||
entry count type \ output
|
||||
entry 1+ c@ w+c
|
||||
w count add-symbol
|
||||
entry count w place
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
def compress = { text ->
|
||||
def dictionary = (1..255).inject([:]) { map, ch -> map."${(char)ch}" = ch; map }
|
||||
def dictionary = (0..<256).inject([:]) { map, ch -> map."${(char)ch}" = ch; map }
|
||||
def w = '', compressed = []
|
||||
text.each { ch ->
|
||||
def wc = "$w$ch"
|
||||
|
|
@ -7,7 +7,7 @@ def compress = { text ->
|
|||
w = wc
|
||||
} else {
|
||||
compressed << dictionary[w]
|
||||
dictionary[wc] = dictionary.size() + 1
|
||||
dictionary[wc] = dictionary.size()
|
||||
w = "$ch"
|
||||
}
|
||||
}
|
||||
|
|
@ -16,18 +16,22 @@ def compress = { text ->
|
|||
}
|
||||
|
||||
def decompress = { compressed ->
|
||||
def dictionary = (1..255).inject([:]) { map, ch -> map[ch] = "${(char)ch}"; map }
|
||||
String w = "${(char)compressed.remove(0)}"
|
||||
def dictionary = (0..<256).inject([:]) { map, ch -> map[ch] = "${(char)ch}"; map }
|
||||
int dictSize = 128;
|
||||
String w = "${(char)compressed[0]}"
|
||||
StringBuffer result = new StringBuffer(w)
|
||||
compressed.each { k ->
|
||||
|
||||
compressed.drop(1).each { k ->
|
||||
String entry = dictionary[k]
|
||||
if (!entry) {
|
||||
if (k != dictionary.size()) throw new IllegalArgumentException("Bad compressed k $k")
|
||||
entry = "$w${w[0]}"
|
||||
}
|
||||
result << entry
|
||||
dictionary[dictionary.size() + 1] = "$w${entry[0]}"
|
||||
|
||||
dictionary[dictionary.size()] = "$w${entry[0]}"
|
||||
w = entry
|
||||
}
|
||||
|
||||
result.toString()
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,81 +0,0 @@
|
|||
//LZW Compression/Decompression for Strings
|
||||
var LZW = {
|
||||
compress: function (uncompressed) {
|
||||
"use strict";
|
||||
// Build the dictionary.
|
||||
var i,
|
||||
dictionary = {},
|
||||
c,
|
||||
wc,
|
||||
w = "",
|
||||
result = [],
|
||||
dictSize = 256;
|
||||
for (i = 0; i < 256; i += 1) {
|
||||
dictionary[String.fromCharCode(i)] = i;
|
||||
}
|
||||
|
||||
for (i = 0; i < uncompressed.length; i += 1) {
|
||||
c = uncompressed.charAt(i);
|
||||
wc = w + c;
|
||||
//Do not use dictionary[wc] because javascript arrays
|
||||
//will return values for array['pop'], array['push'] etc
|
||||
// if (dictionary[wc]) {
|
||||
if (dictionary.hasOwnProperty(wc)) {
|
||||
w = wc;
|
||||
} else {
|
||||
result.push(dictionary[w]);
|
||||
// Add wc to the dictionary.
|
||||
dictionary[wc] = dictSize++;
|
||||
w = String(c);
|
||||
}
|
||||
}
|
||||
|
||||
// Output the code for w.
|
||||
if (w !== "") {
|
||||
result.push(dictionary[w]);
|
||||
}
|
||||
return result;
|
||||
},
|
||||
|
||||
|
||||
decompress: function (compressed) {
|
||||
"use strict";
|
||||
// Build the dictionary.
|
||||
var i,
|
||||
dictionary = [],
|
||||
w,
|
||||
result,
|
||||
k,
|
||||
entry = "",
|
||||
dictSize = 256;
|
||||
for (i = 0; i < 256; i += 1) {
|
||||
dictionary[i] = String.fromCharCode(i);
|
||||
}
|
||||
|
||||
w = String.fromCharCode(compressed[0]);
|
||||
result = w;
|
||||
for (i = 1; i < compressed.length; i += 1) {
|
||||
k = compressed[i];
|
||||
if (dictionary[k]) {
|
||||
entry = dictionary[k];
|
||||
} else {
|
||||
if (k === dictSize) {
|
||||
entry = w + w.charAt(0);
|
||||
} else {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
result += entry;
|
||||
|
||||
// Add w+entry[0] to the dictionary.
|
||||
dictionary[dictSize++] = w + entry.charAt(0);
|
||||
|
||||
w = entry;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
}, // For Test Purposes
|
||||
comp = LZW.compress("TOBEORNOTTOBEORTOBEORNOT"),
|
||||
decomp = LZW.decompress(comp);
|
||||
document.write(comp + '<br>' + decomp);
|
||||
|
|
@ -1,2 +0,0 @@
|
|||
84,79,66,69,79,82,78,79,84,256,258,260,265,259,261,263
|
||||
TOBEORNOTTOBEORTOBEORNOT
|
||||
62
Task/LZW-compression/Kotlin/lzw-compression.kotlin
Normal file
62
Task/LZW-compression/Kotlin/lzw-compression.kotlin
Normal file
|
|
@ -0,0 +1,62 @@
|
|||
// version 1.1.2
|
||||
|
||||
object Lzw {
|
||||
/** Compress a string to a list of output symbols. */
|
||||
fun compress(uncompressed: String): MutableList<Int> {
|
||||
// Build the dictionary.
|
||||
var dictSize = 256
|
||||
val dictionary = mutableMapOf<String, Int>()
|
||||
(0 until dictSize).forEach { dictionary.put(it.toChar().toString(), it)}
|
||||
|
||||
var w = ""
|
||||
val result = mutableListOf<Int>()
|
||||
for (c in uncompressed) {
|
||||
val wc = w + c
|
||||
if (dictionary.containsKey(wc))
|
||||
w = wc
|
||||
else {
|
||||
result.add(dictionary[w]!!)
|
||||
// Add wc to the dictionary.
|
||||
dictionary.put(wc, dictSize++)
|
||||
w = c.toString()
|
||||
}
|
||||
}
|
||||
|
||||
// Output the code for w
|
||||
if (!w.isEmpty()) result.add(dictionary[w]!!)
|
||||
return result
|
||||
}
|
||||
|
||||
/** Decompress a list of output symbols to a string. */
|
||||
fun decompress(compressed: MutableList<Int>): String {
|
||||
// Build the dictionary.
|
||||
var dictSize = 256
|
||||
val dictionary = mutableMapOf<Int, String>()
|
||||
(0 until dictSize).forEach { dictionary.put(it, it.toChar().toString())}
|
||||
|
||||
var w = compressed.removeAt(0).toChar().toString()
|
||||
val result = StringBuilder(w)
|
||||
for (k in compressed) {
|
||||
var entry: String
|
||||
if (dictionary.containsKey(k))
|
||||
entry = dictionary[k]!!
|
||||
else if (k == dictSize)
|
||||
entry = w + w[0]
|
||||
else
|
||||
throw IllegalArgumentException("Bad compressed k: $k")
|
||||
result.append(entry)
|
||||
|
||||
// Add w + entry[0] to the dictionary.
|
||||
dictionary.put(dictSize++, w + entry[0])
|
||||
w = entry
|
||||
}
|
||||
return result.toString()
|
||||
}
|
||||
}
|
||||
|
||||
fun main(args: Array<String>) {
|
||||
val compressed = Lzw.compress("TOBEORNOTTOBEORTOBEORNOT")
|
||||
println(compressed)
|
||||
val decompressed = Lzw.decompress(compressed)
|
||||
println(decompressed)
|
||||
}
|
||||
|
|
@ -49,7 +49,7 @@ class LZW
|
|||
}
|
||||
}
|
||||
$result .= $entry;
|
||||
$dictionary[$dictSize++] = $w + $entry[0];
|
||||
$dictionary[$dictSize++] = $w . $entry[0];
|
||||
$w = $entry;
|
||||
}
|
||||
return $result;
|
||||
|
|
|
|||
|
|
@ -4,17 +4,17 @@ sub compress(Str $uncompressed --> Seq) {
|
|||
|
||||
my $w = "";
|
||||
gather {
|
||||
for $uncompressed.comb -> $c {
|
||||
my $wc = $w ~ $c;
|
||||
if %dictionary{$wc}:exists { $w = $wc }
|
||||
else {
|
||||
take %dictionary{$w};
|
||||
%dictionary{$wc} = +%dictionary;
|
||||
$w = $c;
|
||||
}
|
||||
}
|
||||
for $uncompressed.comb -> $c {
|
||||
my $wc = $w ~ $c;
|
||||
if %dictionary{$wc}:exists { $w = $wc }
|
||||
else {
|
||||
take %dictionary{$w};
|
||||
%dictionary{$wc} = +%dictionary;
|
||||
$w = $c;
|
||||
}
|
||||
}
|
||||
|
||||
take %dictionary{$w} if $w.chars;
|
||||
take %dictionary{$w} if $w.chars;
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -24,16 +24,16 @@ sub decompress(@compressed --> Str) {
|
|||
|
||||
my $w = shift @compressed;
|
||||
join '', gather {
|
||||
take $w;
|
||||
for @compressed -> $k {
|
||||
my $entry;
|
||||
if %dictionary{$k}:exists { take $entry = %dictionary{$k} }
|
||||
elsif $k == $dict-size { take $entry = $w ~ $w.substr(0,1) }
|
||||
else { die "Bad compressed k: $k" }
|
||||
take $w;
|
||||
for @compressed -> $k {
|
||||
my $entry;
|
||||
if %dictionary{$k}:exists { take $entry = %dictionary{$k} }
|
||||
elsif $k == $dict-size { take $entry = $w ~ $w.substr(0,1) }
|
||||
else { die "Bad compressed k: $k" }
|
||||
|
||||
%dictionary{$dict-size++} = $w ~ $entry.substr(0,1);
|
||||
$w = $entry;
|
||||
}
|
||||
%dictionary{$dict-size++} = $w ~ $entry.substr(0,1);
|
||||
$w = $entry;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
57
Task/LZW-compression/Phix/lzw-compression.phix
Normal file
57
Task/LZW-compression/Phix/lzw-compression.phix
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
function compress(string uncompressed)
|
||||
integer dict = new_dict()
|
||||
sequence result = {}
|
||||
integer dictSize = 255, c
|
||||
string word = ""
|
||||
for i=0 to 255 do
|
||||
setd(""&i,i,dict)
|
||||
end for
|
||||
for i=1 to length(uncompressed) do
|
||||
c = uncompressed[i]
|
||||
if getd_index(word&c,dict) then
|
||||
word &= c
|
||||
else
|
||||
result &= getd(word,dict)
|
||||
dictSize += 1
|
||||
setd(word&c,dictSize,dict)
|
||||
word = ""&c
|
||||
end if
|
||||
end for
|
||||
if word!="" then
|
||||
result &= getd(word,dict)
|
||||
end if
|
||||
destroy_dict(dict)
|
||||
return result
|
||||
end function
|
||||
|
||||
function decompress(sequence compressed)
|
||||
integer dict = new_dict()
|
||||
integer dictSize = 255, k, ki
|
||||
string dent = "", result = "", word = ""
|
||||
for i=0 to 255 do
|
||||
setd(i,""&i,dict)
|
||||
end for
|
||||
for i=1 to length(compressed) do
|
||||
k = compressed[i]
|
||||
ki = getd_index(k,dict)
|
||||
if ki then
|
||||
dent = getd_by_index(ki,dict)
|
||||
elsif k=dictSize then
|
||||
dent = word&word[1]
|
||||
else
|
||||
return {NULL,i}
|
||||
end if
|
||||
result &= dent
|
||||
setd(dictSize,word&dent[1],dict)
|
||||
dictSize += 1
|
||||
word = dent
|
||||
end for
|
||||
destroy_dict(dict)
|
||||
return result
|
||||
end function
|
||||
|
||||
constant example = "TOBEORNOTTOBEORTOBEORNOT"
|
||||
sequence com = compress(example)
|
||||
--?com
|
||||
pp(com)
|
||||
?decompress(com)
|
||||
|
|
@ -31,13 +31,13 @@ def decompress(compressed):
|
|||
|
||||
# Build the dictionary.
|
||||
dict_size = 256
|
||||
dictionary = dict((chr(i), chr(i)) for i in xrange(dict_size))
|
||||
# in Python 3: dictionary = {chr(i): chr(i) for i in range(dict_size)}
|
||||
dictionary = dict((i, chr(i)) for i in xrange(dict_size))
|
||||
# in Python 3: dictionary = {i: chr(i) for i in range(dict_size)}
|
||||
|
||||
# use StringIO, otherwise this becomes O(N^2)
|
||||
# due to string concatenation in a loop
|
||||
result = StringIO()
|
||||
w = compressed.pop(0)
|
||||
w = chr(compressed.pop(0))
|
||||
result.write(w)
|
||||
for k in compressed:
|
||||
if k in dictionary:
|
||||
|
|
|
|||
105
Task/LZW-compression/REALbasic/lzw-compression.realbasic
Normal file
105
Task/LZW-compression/REALbasic/lzw-compression.realbasic
Normal file
|
|
@ -0,0 +1,105 @@
|
|||
Function compress(str as String) As String
|
||||
Dim i as integer
|
||||
Dim w as String = ""
|
||||
Dim c as String
|
||||
Dim strLen as Integer
|
||||
Dim wc as string
|
||||
Dim dic() as string
|
||||
Dim result() as string
|
||||
Dim lookup as integer
|
||||
Dim startPos as integer = 0
|
||||
Dim combined as String
|
||||
|
||||
strLen = len(str)
|
||||
|
||||
dic = populateDictionary
|
||||
|
||||
for i = 1 to strLen
|
||||
c = str.mid(i, 1)
|
||||
wc = w + c
|
||||
|
||||
startPos = getStartPos(wc)
|
||||
|
||||
lookup = findArrayPos(dic, wc, startPos)
|
||||
if (lookup <> -1) then
|
||||
w = wc
|
||||
Else
|
||||
startPos = getStartPos(w)
|
||||
lookup = findArrayPos(dic, w, startPos)
|
||||
if lookup <> -1 then
|
||||
result.Append(lookup.ToText)
|
||||
end if
|
||||
dic.append(wc)
|
||||
w = c
|
||||
end if
|
||||
next i
|
||||
|
||||
if (w <> "") then
|
||||
startPos = getStartPos(w)
|
||||
lookup = findArrayPos(dic, w, startPos)
|
||||
result.Append(lookup.ToText)
|
||||
end if
|
||||
|
||||
return join(result, ",")
|
||||
End Function
|
||||
|
||||
Function decompress(str as string) As string
|
||||
dim comStr() as string
|
||||
dim w as string
|
||||
dim result as string
|
||||
dim comStrLen as integer
|
||||
dim entry as string
|
||||
dim dic() as string
|
||||
dim i as integer
|
||||
|
||||
comStr = str.Split(",")
|
||||
comStrLen = comStr.Ubound
|
||||
dic = populateDictionary
|
||||
|
||||
w = chr(val(comStr(0)))
|
||||
result = w
|
||||
for i = 1 to comStrLen
|
||||
entry = dic(val(comStr(i)))
|
||||
result = result + entry
|
||||
dic.append(w + entry.mid(1,1))
|
||||
w = entry
|
||||
next i
|
||||
|
||||
return result
|
||||
End Function
|
||||
|
||||
Private Function findArrayPos(arr() as String, search as String, start as integer) As Integer
|
||||
dim arraySize as Integer
|
||||
dim arrayPosition as Integer = -1
|
||||
dim i as Integer
|
||||
|
||||
arraySize = UBound(arr)
|
||||
|
||||
for i = start to arraySize
|
||||
if (strcomp(arr(i), search, 0) = 0) then
|
||||
arrayPosition = i
|
||||
exit
|
||||
end if
|
||||
next i
|
||||
|
||||
return arrayPosition
|
||||
End Function
|
||||
|
||||
Private Function getStartPos(str as String) As integer
|
||||
if (len(str) = 1) then
|
||||
return 0
|
||||
else
|
||||
return 255
|
||||
end if
|
||||
End Function
|
||||
|
||||
Private Function populateDictionary() As string()
|
||||
dim dic() as string
|
||||
dim i as integer
|
||||
|
||||
for i = 0 to 255
|
||||
dic.append(Chr(i))
|
||||
next i
|
||||
|
||||
return dic
|
||||
End Function
|
||||
|
|
@ -69,8 +69,8 @@
|
|||
(set! dictionary
|
||||
(append dictionary
|
||||
(list (string-append
|
||||
(list-ref dictionary k)
|
||||
(string (string-ref (list-ref dictionary kn) 0)))))))))
|
||||
(list-ref dictionary k)
|
||||
(string (string-ref (list-ref dictionary kn) 0)))))))))
|
||||
|
||||
;; Build the resulting string
|
||||
(set! result (string-append result (list-ref dictionary k)))
|
||||
|
|
|
|||
|
|
@ -1,14 +1,14 @@
|
|||
# Compress a string to a list of output symbols.
|
||||
func compress(String uncompressed) -> Array {
|
||||
|
||||
|
||||
# Build the dictionary.
|
||||
var dict_size = 256
|
||||
var dictionary = Hash()
|
||||
|
||||
|
||||
for i in range(dict_size) {
|
||||
dictionary{i.chr} = i.chr
|
||||
}
|
||||
|
||||
|
||||
var w = ''
|
||||
var result = []
|
||||
uncompressed.each { |c|
|
||||
|
|
@ -23,26 +23,26 @@ func compress(String uncompressed) -> Array {
|
|||
w = c
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
# Output the code for w.
|
||||
if (w != '') {
|
||||
if (w != '') {
|
||||
result << dictionary{w}
|
||||
}
|
||||
|
||||
|
||||
return result
|
||||
}
|
||||
|
||||
|
||||
# Decompress a list of output ks to a string.
|
||||
func decompress(Array compressed) -> String {
|
||||
|
||||
|
||||
# Build the dictionary.
|
||||
var dict_size = 256
|
||||
var dictionary = Hash()
|
||||
|
||||
|
||||
for i in range(dict_size) {
|
||||
dictionary{i.chr} = i.chr
|
||||
}
|
||||
|
||||
|
||||
var w = compressed.shift
|
||||
var result = w
|
||||
compressed.each { |k|
|
||||
|
|
@ -55,16 +55,16 @@ func decompress(Array compressed) -> String {
|
|||
die "Bad compressed k: #{k}"
|
||||
}
|
||||
result += entry
|
||||
|
||||
|
||||
# Add w+entry[0] to the dictionary.
|
||||
dictionary{dict_size} = w+entry.first
|
||||
dict_size++
|
||||
|
||||
|
||||
w = entry
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
|
||||
# How to use:
|
||||
var compressed = compress('TOBEORNOTTOBEORTOBEORNOT')
|
||||
say compressed.join(' ')
|
||||
|
|
|
|||
28
Task/LZW-compression/Zkl/lzw-compression-1.zkl
Normal file
28
Task/LZW-compression/Zkl/lzw-compression-1.zkl
Normal file
|
|
@ -0,0 +1,28 @@
|
|||
fcn lzwCompress(uncompressed){ // text-->list of 12 bit ints
|
||||
dictionary:=(256).pump(Dictionary(),fcn(n){ return(n.toChar(),n) });
|
||||
w,compressed:="",List();
|
||||
foreach c in (uncompressed){
|
||||
wc:=w+c;
|
||||
if(dictionary.holds(wc)) w=wc;
|
||||
else{
|
||||
compressed.append(dictionary[w]); // 12 bits
|
||||
dictionary[wc]=dictionary.len();
|
||||
w=c;
|
||||
}
|
||||
}
|
||||
if(w) compressed.append(dictionary[w]);
|
||||
compressed
|
||||
}
|
||||
fcn lzwUncompress(compressed){ // compressed data-->text
|
||||
dictionary:=(256).pump(Dictionary(),fcn(n){ return(n,n.toChar()) });
|
||||
w,decommpressed:=dictionary[compressed[0]],Data(Void,w);
|
||||
foreach k in (compressed[1,*]){
|
||||
if(dictionary.holds(k)) entry:=dictionary[k];
|
||||
else if(k==dictionary.len()) entry:=w+w[0];
|
||||
else throw(Exception.ValueError("Invalid compressed data"));
|
||||
decommpressed.append(entry);
|
||||
dictionary.add(dictionary.len(),w+entry[0]);
|
||||
w=entry;
|
||||
}
|
||||
decommpressed.text
|
||||
}
|
||||
4
Task/LZW-compression/Zkl/lzw-compression-2.zkl
Normal file
4
Task/LZW-compression/Zkl/lzw-compression-2.zkl
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
compressed:=lzwCompress("TOBEORNOTTOBEORTOBEORNOT");
|
||||
compressed.toString(*).println();
|
||||
|
||||
lzwUncompress(compressed).println();
|
||||
Loading…
Add table
Add a link
Reference in a new issue