Add various metrics to score the quality of a parse

Add various metrics to score the quality of a parse
This commit is contained in:
Vinayak Mehta
2016-08-30 14:52:49 +05:30
committed by GitHub
parent 43a009dab4
commit 552f9cf422
11 changed files with 1027 additions and 472 deletions
+310 -57
View File
@@ -4,8 +4,12 @@ import os
import sys
import time
import logging
import warnings
import numpy as np
from docopt import docopt
from collections import Counter
import matplotlib.pyplot as plt
from PyPDF2 import PdfFileReader
from camelot.pdf import Pdf
@@ -22,12 +26,23 @@ usage:
options:
-h, --help Show this screen.
-v, --version Show version.
-V, --verbose Verbose.
-p, --pages <pageno> Comma-separated list of page numbers.
Example: -p 1,3-6,10 [default: 1]
-P, --parallel Parallelize the parsing process.
-f, --format <format> Output format. (csv,tsv,html,json,xlsx) [default: csv]
-l, --log Print log to file.
-V, --verbose Verbose.
-l, --log Log to file.
-o, --output <directory> Output directory.
-M, --cmargin <cmargin> Char margin. Chars closer than cmargin are
grouped together to form a word. [default: 2.0]
-L, --lmargin <lmargin> Line margin. Lines closer than lmargin are
grouped together to form a textbox. [default: 0.5]
-W, --wmargin <wmargin> Word margin. Insert blank spaces between chars
if distance between words is greater than word
margin. [default: 0.1]
-S, --save-info Save parsing info for each page to a file.
-X, --plot <dist> Plot distributions. (page,all,rc)
-Z, --summary Summarize metrics.
camelot methods:
lattice Looks for lines between data.
@@ -47,12 +62,12 @@ options:
cells. Example: -F h, -F v, -F hv
-s, --scale <scale> Scaling factor. Large scaling factor leads to
smaller lines being detected. [default: 15]
-i, --invert Invert pdf image to make sure that lines are
in foreground.
-j, --jtol <jtol> Tolerance to account for when comparing joint
and line coordinates. [default: 2]
-m, --mtol <mtol> Tolerance to account for when merging lines
which are very close. [default: 2]
-i, --invert Invert pdf image to make sure that lines are
in foreground.
-d, --debug <debug> Debug by visualizing pdf geometry.
(contour,line,joint,table) Example: -d table
"""
@@ -69,17 +84,159 @@ options:
Example: -c 10.1,20.2,30.3
-y, --ytol <ytol> Tolerance to account for when grouping rows
together. [default: 2]
-M, --cmargin <cmargin> Char margin. Chars closer than cmargin are
grouped together to form a word. [default: 2.0]
-L, --lmargin <lmargin> Line margin. Lines closer than lmargin are
grouped together to form a textbox. [default: 0.5]
-W, --wmargin <wmargin> Word margin. Insert blank spaces between chars
if distance between words is greater than word
margin. [default: 0.1]
-m, --mtol <mtol> Tolerance to account for when merging columns
together. [default: 2]
-d, --debug Debug by visualizing textboxes.
"""
def plot_table_barchart(r, c, p, pno, tno):
row_idx = [i + 1 for i, row in enumerate(r)]
col_idx = [i + 1 for i, col in enumerate(c)]
r_index = np.arange(len(r))
c_index = np.arange(len(c))
width = 0.7
plt.figure(figsize=(8, 6))
plt.subplot(2, 1, 1)
plt.title('Percentage of empty cells in table: {0:.2f}'.format(p))
plt.xlabel('row index')
plt.ylabel('number of non-empty cells in row')
plt.bar(r_index, r)
plt.xticks(r_index + width * 0.5, row_idx)
plt.ylim(0, len(c))
plt.subplot(2, 1, 2)
plt.xlabel('column index')
plt.ylabel('number of non-empty cells in column')
plt.bar(c_index, c)
plt.xticks(c_index + width * 0.5, col_idx)
plt.ylim(0, len(r))
plt.savefig(''.join([pno, '_', tno, '.png']), dpi=300)
def plot_all_barchart(data, output):
r_empty_cells = []
for page_number in data.keys():
page = data[page_number]
for table_number in page.keys():
table = page[table_number]
r_empty_cells.extend([r / float(table['ncols']) for r in table['r_nempty_cells']])
c = Counter(r_empty_cells)
if 0.0 not in c:
c.update({0.0: 0})
if 1.0 not in c:
c.update({1.0: 0})
plt.figure(figsize=(8, 6))
plt.xlabel('percentage of non-empty cells in a row')
plt.ylabel('percentage of rows processed')
row_p = [count / float(sum(c.values())) for count in c.values()]
plt.bar(c.keys(), row_p, align='center', width=0.05)
plt.ylim(0, 1.0)
plt.savefig(''.join([output, '_all.png']), dpi=300)
def plot_rc_piechart(data, output):
from matplotlib import cm
tables = 0
rows, cols = [], []
for page_number in data.keys():
page = data[page_number]
for table_number in page.keys():
table = page[table_number]
tables += 1
rows.append(table['nrows'])
cols.append(table['ncols'])
r = Counter(rows)
c = Counter(cols)
plt.figure(figsize=(8, 6))
cs1 = cm.Set1(np.arange(len(r)) / float(len(r)))
ax1 = plt.subplot(211, aspect='equal')
ax1.pie(r.values(), colors=cs1, labels=r.keys(), startangle=90)
ax1.set_title('row distribution across tables')
cs2 = cm.Set1(np.arange(len(c)) / float(len(c)))
ax2 = plt.subplot(212, aspect='equal')
ax2.pie(c.values(), colors=cs2, labels=c.keys(), startangle=90)
ax2.set_title('column distribution across tables')
plt.savefig(''.join([output, '_rc.png']), dpi=300)
def summary(data, p_time):
from operator import itemgetter
from itertools import groupby
scores = []
continuous_tables = []
total_tables = 0
for page_number in data.keys():
page = data[page_number]
total_tables += len(page.keys())
for table_number in page.keys():
table = page[table_number]
continuous_tables.append((page_number, table_number, table['ncols']))
scores.append(table['score'])
avg_score = np.mean(scores)
ct_pages = []
header_string = ""
if len(continuous_tables) > 1:
tables = sorted(continuous_tables, key=lambda x: (int(x[0][5:]), int(x[1][6:])))
for k, g in groupby(tables, key=itemgetter(2)):
g = list(g)
tables_same_ncols = set([int(t[0][5:]) for t in g])
tables_same_ncols = sorted(list(tables_same_ncols))
for K, G in groupby(enumerate(tables_same_ncols), key=lambda (i, x): i - x):
G = list(G)
ct_pages.append((str(G[0][1]), str(G[-1][1])))
result_headers = []
for ct in ct_pages:
header_idx = {}
possible_headers = []
ncols = 0
for page_number in range(int(ct[0]), int(ct[1]) + 1):
page = data['page-{0}'.format(page_number)]
for table_number in page.keys():
table = page[table_number]
ncols = table['ncols']
for i, row in enumerate(table['data']):
try:
header_idx[tuple(row)].append(i)
except KeyError:
header_idx[tuple(row)] = [i]
possible_headers = sorted(header_idx, key=lambda k: len(header_idx[k]), reverse=True)[:10]
possible_headers = filter(lambda z: len(z) == ncols,
[filter(lambda x: x != '', p_h) for p_h in possible_headers])
modes = []
for p_h in possible_headers:
try:
modes.append((p_h, max(set(header_idx[p_h]), key=header_idx[p_h].count)))
except KeyError:
pass
header = modes[modes.index(min(modes, key=lambda x: x[1]))][0]
result_headers.append(header)
header_string = "Multi-page table headers*:\n"
header_string = ''.join([header_string, '\n'.join(['pages {0} -> {1}{2}{3}'.format(
'-'.join([cr[0][0], cr[0][1]]), '"', '","'.join(cr[1]), '"') for cr in zip(
ct_pages, result_headers)])])
avg_time = "Time taken per page: {0:.2f} seconds\n".format(
p_time / float(len(data))) if len(data) != 1 else ""
equal_ncols = "\nMulti-page tables on*: {0}\n".format(
', '.join(['-'.join(ct) for ct in ct_pages])) if len(data) != 1 else ""
stats = [len(data), p_time, avg_time, total_tables, avg_score, equal_ncols]
stat_string = ("Pages processed: {0}\nTime taken: {1:.2f} seconds\n"
"{2}Tables found: {3}\nAverage score: {4:.2f}{5}".format(*stats))
print(''.join([stat_string, header_string]))
def convert_to_html(table):
html = ''
html = ''.join([html, '<table border="1">\n'])
@@ -99,23 +256,23 @@ def write_to_disk(data, f='csv', output=None, filename=None):
if f in ['csv', 'tsv']:
import csv
delimiter = ',' if f == 'csv' else '\t'
for page in sorted(data):
for table in range(len(data[page])):
dsvname = '{0}_table_{1}.{2}'.format(page, table + 1, f)
for page_number in sorted(data.keys()):
for table_number in sorted(data[page_number].keys()):
dsvname = '{0}.{1}'.format(''.join([page_number, '_', table_number]), f)
with open(os.path.join(output, dsvname), 'w') as outfile:
writer = csv.writer(
outfile, delimiter=delimiter, quoting=csv.QUOTE_ALL)
for row in data[page][table]:
for row in data[page_number][table_number]['data']:
writer.writerow(row)
elif f == 'html':
htmlname = '{}.html'.format(froot)
for page in sorted(data):
for table in range(len(data[page])):
htmlname = '{0}.html'.format(froot)
for page_number in sorted(data.keys()):
for table_number in sorted(data[page_number].keys()):
with open(os.path.join(output, htmlname), 'a') as htmlfile:
htmlfile.write(convert_to_html(data[page][table]))
htmlfile.write(convert_to_html(data[page_number][table_number]['data']))
elif f == 'json':
import json
with open(os.path.join(output, '{}.json'.format(froot)), 'w') \
with open(os.path.join(output, '{0}.json'.format(froot)), 'w') \
as jsonfile:
json.dump(data, jsonfile)
elif f == 'xlsx':
@@ -123,12 +280,12 @@ def write_to_disk(data, f='csv', output=None, filename=None):
from pyexcel_xlsx import save_data
from collections import OrderedDict
xlsx_data = OrderedDict()
for page in sorted(data):
for table in range(len(data[page])):
sheet_name = '{0}_table_{1}'.format(page, table + 1)
for page_number in sorted(data.keys(), key=lambda x: int(x[5:])):
for table_number in sorted(data[page_number].keys(), key=lambda x: int(x[6:])):
sheet_name = ''.join([page_number, '_', table_number])
xlsx_data.update({sheet_name:
[row for row in data[page][table]]})
save_data(os.path.join(output, '{}.xlsx'.format(froot)), xlsx_data)
[row for row in data[page_number][table_number]['data']]})
save_data(os.path.join(output, '{0}.xlsx'.format(froot)), xlsx_data)
except ImportError:
print("link to install docs")
@@ -147,16 +304,17 @@ if __name__ == '__main__':
filename = args['<file>']
filedir = os.path.dirname(args['<file>'])
logname, __ = os.path.splitext(filename)
logname += '.log'
logname = ''.join([logname, '.log'])
scorename, __ = os.path.splitext(filename)
scorename = ''.join([scorename, '_info.csv'])
pngname, __ = os.path.splitext(filename)
if args['--log']:
FORMAT = '%(asctime)s - %(levelname)s - %(message)s'
if args['--output']:
logname = os.path.join(args['--output'], os.path.basename(logname))
logging.basicConfig(
filename=logname, filemode='w', level=logging.DEBUG)
else:
logging.basicConfig(
filename=logname, filemode='w', level=logging.DEBUG)
logging.basicConfig(
filename=logname, filemode='w', format=FORMAT, level=logging.DEBUG)
p = []
if args['--pages'] == '1':
@@ -173,47 +331,142 @@ if __name__ == '__main__':
else:
p.append({'start': int(r), 'end': int(r)})
margin_tuple = (float(args['--cmargin']), float(args['--lmargin']),
float(args['--wmargin']))
if args['<method>'] == 'lattice':
try:
extractor = Lattice(Pdf(filename, pagenos=p, clean=True),
fill=args['--fill'],
scale=int(args['--scale']),
jtol=int(args['--jtol']),
mtol=int(args['--mtol']),
invert=args['--invert'],
debug=args['--debug'],
verbose=args['--verbose'])
data = extractor.get_tables()
manager = Pdf(Lattice(
fill=args['--fill'],
scale=int(args['--scale']),
invert=args['--invert'],
jtol=int(args['--jtol']),
mtol=int(args['--mtol']),
pdf_margin=margin_tuple,
debug=args['--debug']),
filename,
pagenos=p,
parallel=args['--parallel'],
clean=True)
data = manager.extract()
processing_time = time.time() - start_time
vprint("Finished processing in", processing_time, "seconds")
logging.info("Finished processing in " + str(processing_time) + " seconds")
if args['--plot']:
if args['--output']:
pngname = os.path.join(args['--output'], os.path.basename(pngname))
plot_type = args['--plot'].split(',')
if 'page' in plot_type:
for page_number in sorted(data.keys(), key=lambda x: int(x[5:])):
page = data[page_number]
for table_number in sorted(page.keys(), key=lambda x: int(x[6:])):
table = page[table_number]
plot_table_barchart(table['r_nempty_cells'],
table['c_nempty_cells'],
table['empty_p'],
page_number,
table_number)
if 'all' in plot_type:
plot_all_barchart(data, pngname)
if 'rc' in plot_type:
plot_rc_piechart(data, pngname)
if args['--summary']:
summary(data, processing_time)
if args['--save-info']:
if args['--output']:
scorename = os.path.join(args['--output'], os.path.basename(scorename))
with open(scorename, 'w') as score_file:
score_file.write('table,nrows,ncols,empty_p,line_p,text_p,score\n')
for page_number in sorted(data.keys(), key=lambda x: int(x[5:])):
page = data[page_number]
for table_number in sorted(page.keys(), key=lambda x: int(x[6:])):
table = page[table_number]
score_file.write('{0},{1},{2},{3},{4},{5},{6}\n'.format(
''.join([page_number, '_', table_number]),
table['nrows'],
table['ncols'],
table['empty_p'],
table['line_p'],
table['text_p'],
table['score']))
if args['--debug']:
extractor.plot_geometry(args['--debug'])
manager.debug_plot()
except Exception as e:
logging.exception(e.message, exc_info=True)
sys.exit()
elif args['<method>'] == 'stream':
try:
extractor = Stream(Pdf(filename, pagenos=p,
char_margin=float(args['--cmargin']),
line_margin=float(args['--lmargin']),
word_margin=float(args['--wmargin']),
clean=True),
ncolumns=int(args['--ncols']),
columns=args['--columns'],
ytol=int(args['--ytol']),
debug=args['--debug'],
verbose=args['--verbose'])
data = extractor.get_tables()
manager = Pdf(Stream(
ncolumns=int(args['--ncols']),
columns=args['--columns'],
ytol=int(args['--ytol']),
mtol=int(args['--mtol']),
pdf_margin=margin_tuple,
debug=args['--debug']),
filename,
pagenos=p,
parallel=args['--parallel'],
clean=True)
data = manager.extract()
processing_time = time.time() - start_time
vprint("Finished processing in", processing_time, "seconds")
logging.info("Finished processing in " + str(processing_time) + " seconds")
if args['--plot']:
if args['--output']:
pngname = os.path.join(args['--output'], os.path.basename(pngname))
plot_type = args['--plot'].split(',')
if 'page' in plot_type:
for page_number in sorted(data.keys(), key=lambda x: int(x[5:])):
page = data[page_number]
for table_number in sorted(page.keys(), key=lambda x: int(x[6:])):
table = page[table_number]
plot_table_barchart(table['r_nempty_cells'],
table['c_nempty_cells'],
table['empty_p'],
page_number,
table_number)
if 'all' in plot_type:
plot_all_barchart(data, pngname)
if 'rc' in plot_type:
plot_rc_piechart(data, pngname)
if args['--summary']:
summary(data, processing_time)
if args['--save-info']:
if args['--output']:
scorename = os.path.join(args['--output'], os.path.basename(scorename))
with open(scorename, 'w') as score_file:
score_file.write('table,nrows,ncols,empty_p,,score\n')
for page_number in sorted(data.keys(), key=lambda x: int(x[5:])):
page = data[page_number]
for table_number in sorted(page.keys(), key=lambda x: int(x[6:])):
table = page[table_number]
score_file.write('{0},{1},{2},{3},{4}\n'.format(
''.join([page_number, '_', table_number]),
table['nrows'],
table['ncols'],
table['empty_p'],
table['score']))
if args['--debug']:
extractor.plot_text()
manager.debug_plot()
except Exception as e:
logging.exception(e.message, exc_info=True)
sys.exit()
if data is None:
if args['--debug']:
print("See 'camelot <method> -h' for various parameters you can tweak.")
else:
output = filedir if args['--output'] is None else args['--output']
write_to_disk(data, f=args['--format'],
output=output, filename=filename)
vprint("finished in", time.time() - start_time, "seconds")
logging.info("Time taken: " + str(time.time() - start_time) + " seconds")