Data commit
This commit is contained in:
parent
7387c8f97b
commit
cb5bb5e222
199093 changed files with 3378972 additions and 0 deletions
41
Task/Text-processing-2/Python/text-processing-2-1.py
Normal file
41
Task/Text-processing-2/Python/text-processing-2-1.py
Normal file
|
|
@ -0,0 +1,41 @@
|
|||
import re
|
||||
import zipfile
|
||||
import StringIO
|
||||
|
||||
def munge2(readings):
|
||||
|
||||
datePat = re.compile(r'\d{4}-\d{2}-\d{2}')
|
||||
valuPat = re.compile(r'[-+]?\d+\.\d+')
|
||||
statPat = re.compile(r'-?\d+')
|
||||
allOk, totalLines = 0, 0
|
||||
datestamps = set([])
|
||||
for line in readings:
|
||||
totalLines += 1
|
||||
fields = line.split('\t')
|
||||
date = fields[0]
|
||||
pairs = [(fields[i],fields[i+1]) for i in range(1,len(fields),2)]
|
||||
|
||||
lineFormatOk = datePat.match(date) and \
|
||||
all( valuPat.match(p[0]) for p in pairs ) and \
|
||||
all( statPat.match(p[1]) for p in pairs )
|
||||
if not lineFormatOk:
|
||||
print 'Bad formatting', line
|
||||
continue
|
||||
|
||||
if len(pairs)!=24 or any( int(p[1]) < 1 for p in pairs ):
|
||||
print 'Missing values', line
|
||||
continue
|
||||
|
||||
if date in datestamps:
|
||||
print 'Duplicate datestamp', line
|
||||
continue
|
||||
datestamps.add(date)
|
||||
allOk += 1
|
||||
|
||||
print 'Lines with all readings: ', allOk
|
||||
print 'Total records: ', totalLines
|
||||
|
||||
#zfs = zipfile.ZipFile('readings.zip','r')
|
||||
#readings = StringIO.StringIO(zfs.read('readings.txt'))
|
||||
readings = open('readings.txt','r')
|
||||
munge2(readings)
|
||||
45
Task/Text-processing-2/Python/text-processing-2-2.py
Normal file
45
Task/Text-processing-2/Python/text-processing-2-2.py
Normal file
|
|
@ -0,0 +1,45 @@
|
|||
import re
|
||||
import zipfile
|
||||
import StringIO
|
||||
|
||||
def munge2(readings, debug=False):
|
||||
|
||||
datePat = re.compile(r'\d{4}-\d{2}-\d{2}')
|
||||
valuPat = re.compile(r'[-+]?\d+\.\d+')
|
||||
statPat = re.compile(r'-?\d+')
|
||||
totalLines = 0
|
||||
dupdate, badform, badlen, badreading = set(), set(), set(), 0
|
||||
datestamps = set([])
|
||||
for line in readings:
|
||||
totalLines += 1
|
||||
fields = line.split('\t')
|
||||
date = fields[0]
|
||||
pairs = [(fields[i],fields[i+1]) for i in range(1,len(fields),2)]
|
||||
|
||||
lineFormatOk = datePat.match(date) and \
|
||||
all( valuPat.match(p[0]) for p in pairs ) and \
|
||||
all( statPat.match(p[1]) for p in pairs )
|
||||
if not lineFormatOk:
|
||||
if debug: print 'Bad formatting', line
|
||||
badform.add(date)
|
||||
|
||||
if len(pairs)!=24 or any( int(p[1]) < 1 for p in pairs ):
|
||||
if debug: print 'Missing values', line
|
||||
if len(pairs)!=24: badlen.add(date)
|
||||
if any( int(p[1]) < 1 for p in pairs ): badreading += 1
|
||||
|
||||
if date in datestamps:
|
||||
if debug: print 'Duplicate datestamp', line
|
||||
dupdate.add(date)
|
||||
|
||||
datestamps.add(date)
|
||||
|
||||
print 'Duplicate dates:\n ', '\n '.join(sorted(dupdate))
|
||||
print 'Bad format:\n ', '\n '.join(sorted(badform))
|
||||
print 'Bad number of fields:\n ', '\n '.join(sorted(badlen))
|
||||
print 'Records with good readings: %i = %5.2f%%\n' % (
|
||||
totalLines-badreading, (totalLines-badreading)/float(totalLines)*100 )
|
||||
print 'Total records: ', totalLines
|
||||
|
||||
readings = open('readings.txt','r')
|
||||
munge2(readings)
|
||||
Loading…
Add table
Add a link
Reference in a new issue