forked from MarcelRobeer/VisualNarrator
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathrun.py
More file actions
300 lines (244 loc) · 11.4 KB
/
Copy pathrun.py
File metadata and controls
300 lines (244 loc) · 11.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
#!/usr/bin/env python
import sys
import string
import os.path
import timeit
import pkg_resources
from argparse import ArgumentParser
from spacy.en import English
from jinja2 import FileSystemLoader, Environment, PackageLoader
from app.io import Reader, Writer
from app.miner import StoryMiner
from app.matrix import Matrix
from app.userstory import UserStory
from app.utility import Utility, Printer
from app.pattern import Constructor
from app.statistics import Statistics, Counter
def main(filename, systemname, print_us, print_ont, statistics, link, prolog, per_role, threshold, base, weights):
"""General class to run the entire program
"""
# Initialize spaCy just once (this takes most of the time...)
print("Initializing Natural Language Processor . . .")
start_nlp_time = timeit.default_timer()
nlp = English()
nlp_time = timeit.default_timer() - start_nlp_time
start_parse_time = timeit.default_timer()
miner = StoryMiner()
# Read the input file
set = Reader.parse(filename)
us_id = 1
# Keep track of all errors
success = 0
fail = 0
list_of_fails = []
errors = ""
c = Counter()
# Keeps track of all succesfully created User Stories objects
us_instances = []
failed_stories = []
success_stories = []
# Parse every user story (remove punctuation and mine)
for s in set:
try:
user_story = parse(s, us_id, systemname, nlp, miner)
user_story = c.count(user_story)
success = success + 1
us_instances.append(user_story)
success_stories.append(s)
except ValueError as err:
failed_stories.append([us_id, s, err.args])
errors += "\n[User Story " + str(us_id) + " ERROR] " + str(err.args[0]) + "! (\"" + " ".join(str.split(s)) + "\")"
fail = fail + 1
us_id = us_id + 1
# Print errors (if found)
if errors:
Printer.print_head("PARSING ERRORS")
print(errors)
parse_time = timeit.default_timer() - start_parse_time
# Generate the term-by-user story matrix (m), and additional data in two other matrices
start_matr_time = timeit.default_timer()
matrix = Matrix(base, weights)
matrices = matrix.generate(us_instances, ' '.join(success_stories), nlp)
m = matrices[0]
count_matrix = matrices[1]
stories_list = matrices[2]
rme = matrices[3]
matr_time = timeit.default_timer() - start_matr_time
# Print details per user story, if argument '-u'/'--print_us' is chosen
if print_us:
print("Details:\n")
for us in us_instances:
Printer.print_us_data(us)
# Generate the ontology
start_gen_time = timeit.default_timer()
patterns = Constructor(nlp, us_instances, m)
out = patterns.make(systemname, threshold, link)
output_ontology = out[0]
output_prolog = out[1]
output_ontobj = out[2]
output_prologobj = out[3]
onto_per_role = out[4]
# Print out the ontology in the terminal, if argument '-o'/'--print_ont' is chosen
if print_ont:
Printer.print_head("MANCHESTER OWL")
print(output_ontology)
gen_time = timeit.default_timer() - start_gen_time
# Gather statistics and print the results
stats_time = 0
if statistics:
start_stats_time = timeit.default_timer()
statsarr = Statistics.to_stats_array(us_instances)
Printer.print_head("USER STORY STATISTICS")
Printer.print_stats(statsarr[0], True)
Printer.print_stats(statsarr[1], True)
Printer.print_subhead("Term - by - User Story Matrix ( Terms w/ total weight 0 hidden )")
hide_zero = m[(m['sum'] > 0)]
print(hide_zero)
stats_time = timeit.default_timer() - start_stats_time
# Write output files
w = Writer()
folder = "output/" + str(systemname)
reports_folder = folder + "/reports"
stats_folder = reports_folder + "/stats"
outputfile = w.make_file(folder + "/ontology", str(systemname), "omn", output_ontology)
files = [["Manchester Ontology", outputfile]]
outputcsv = ""
sent_outputcsv = ""
matrixcsv = ""
if statistics:
outputcsv = w.make_file(stats_folder, str(systemname), "csv", statsarr[0])
matrixcsv = w.make_file(stats_folder, str(systemname) + "-term_by_US_matrix", "csv", m)
sent_outputcsv = w.make_file(stats_folder, str(systemname) + "-sentences", "csv", statsarr[1])
files.append(["General statistics", outputcsv])
files.append(["Term-by-User Story matrix", matrixcsv])
files.append(["Sentence statistics", sent_outputcsv])
if prolog:
outputpl = w.make_file(folder + "/prolog", str(systemname), "pl", output_prolog)
files.append(["Prolog", outputpl])
if per_role:
for o in onto_per_role:
name = str(systemname) + "-" + str(o[0])
pont = w.make_file(folder + "/ontology", name, "omn", o[1])
files.append(["Individual Ontology for '" + str(o[0]) + "'", pont])
# Print the used ontology generation settings
Printer.print_gen_settings(matrix, base, threshold)
# Print details of the generation
Printer.print_details(fail, success, nlp_time, parse_time, matr_time, gen_time, stats_time)
report_dict = {
"stories": us_instances,
"failed_stories": failed_stories,
"systemname": systemname,
"us_success": success,
"us_fail": fail,
"times": [["Initializing Natural Language Processor (<em>spaCy</em> v" + pkg_resources.get_distribution("spacy").version + ")" , nlp_time], ["Mining User Stories", parse_time], ["Creating Factor Matrix", matr_time], ["Generating Manchester Ontology", gen_time], ["Gathering statistics", stats_time]],
"dir": os.path.dirname(os.path.realpath(__file__)),
"inputfile": filename,
"inputfile_lines": len(set),
"outputfiles": files,
"threshold": threshold,
"base": base,
"matrix": matrix,
"weights": m['sum'].copy().reset_index().sort_values(['sum'], ascending=False).values.tolist(),
"counts": count_matrix.reset_index().values.tolist(),
"classes": output_ontobj.classes,
"relationships": output_prologobj.relationships,
"types": list(count_matrix.columns.values),
"ontology": Utility.multiline(output_ontology)
}
# Finally, generate a report
report = w.make_file(reports_folder, str(systemname) + "_REPORT", "html", generate_report(report_dict))
files.append(["Report", report])
# Print the location and name of all output files
for file in files:
if str(file[1]) != "":
print(str(file[0]) + " file succesfully created at: \"" + str(file[1]) + "\"")
# Return objects so that they can be used as input for other tools
return {'us_instances': us_instances, 'output_ontobj': output_ontobj, 'output_prologobj': output_prologobj, 'matrix': m}
def parse(text, id, systemname, nlp, miner):
"""Create a new user story object and mines it to map all data in the user story text to a predefined model
:param text: The user story text
:param id: The user story ID, which can later be used to identify the user story
:param systemname: Name of the system this user story belongs to
:param nlp: Natural Language Processor (spaCy)
:param miner: instance of class Miner
:returns: A new user story object
"""
no_punct = Utility.remove_punct(text)
no_double_space = ' '.join(no_punct.split())
doc = nlp(no_double_space)
user_story = UserStory(id, text, no_double_space)
user_story.system.main = nlp(systemname)[0]
user_story.data = doc
#Printer.print_dependencies(user_story)
#Printer.print_noun_phrases(user_story)
miner.structure(user_story)
user_story.old_data = user_story.data
user_story.data = nlp(user_story.sentence)
miner.mine(user_story, nlp)
return user_story
def generate_report(report_dict):
"""Generates a report using Jinja2
:param report_dict: Dictionary containing all variables used in the report
:returns: HTML page
"""
CURR_DIR = os.path.dirname(os.path.abspath(__file__))
loader = FileSystemLoader( searchpath=str(CURR_DIR) + "/templates/" )
env = Environment( loader=loader, trim_blocks=True, lstrip_blocks=True )
env.globals['text'] = Utility.t
env.globals['is_i'] = Utility.is_i
env.globals['apply_tab'] = Utility.tab
env.globals['is_comment'] = Utility.is_comment
env.globals['occurence_list'] = Utility.occurence_list
env.tests['is_us'] = Utility.is_us
template = env.get_template("report.html")
return template.render(report_dict)
def program(*args):
p = ArgumentParser(
usage='''run.py <INPUT FILE> [<args>]
///////////////////////////////////////////
// PROGRAM_NAME //
///////////////////////////////////////////
This program has multiple functionalities:
(1) Mine user story information
(2) Generate an ontology from a user story set
(3) Generate Prolog from a user story set (including links to 'role', 'means' and 'ends')
(4) Get statistics for a user story set
''',
epilog='''{*} Utrecht University.
M.J. Robeer, 2015-2017''')
p.add_argument("filename",
help="input file with user stories", metavar="INPUT FILE",
type=lambda x: is_valid_file(p, x))
p.add_argument('--version', action='version', version='PROGRAM_NAME v0.9 BETA by M.J. Robeer')
g_p = p.add_argument_group("general arguments (optional)")
g_p.add_argument("-n", "--name", dest="system_name", help="your system name, as used in ontology and output file(s) generation", required=False)
g_p.add_argument("-u", "--print_us", dest="print_us", help="print data per user story in the console", action="store_true", default=False)
g_p.add_argument("-o", "--print_ont", dest="print_ont", help="print ontology in the console", action="store_true", default=False)
g_p.add_argument("-l", "--link", dest="link", help="link ontology classes to user story they originate from", action="store_true", default=False)
g_p.add_argument("--prolog", dest="prolog", help="generate prolog output (.pl)", action="store_true", default=False)
s_p = p.add_argument_group("statistics arguments (optional)")
s_p.add_argument("-s", "--statistics", dest="statistics", help="show user story set statistics and output these to a .csv file", action="store_true", default=False)
w_p = p.add_argument_group("conceptual model generation tuning (optional)")
w_p.add_argument("-p", "--per_role", dest="per_role", help="create an additional conceptual model per role", action="store_true", default=False)
w_p.add_argument("-t", dest="threshold", help="set threshold for conceptual model generation (INT, default = 1.0)", type=float, default=1.0)
w_p.add_argument("-b", dest="base_weight", help="set the base weight (INT, default = 1)", type=int, default=1)
w_p.add_argument("-wfr", dest="weight_func_role", help="weight of functional role (FLOAT, default = 1.0)", type=float, default=1)
w_p.add_argument("-wdo", dest="weight_main_obj", help="weight of main object (FLOAT, default = 1.0)", type=float, default=1)
w_p.add_argument("-wffm", dest="weight_ff_means", help="weight of noun in free form means (FLOAT, default = 0.7)", type=float, default=0.7)
w_p.add_argument("-wffe", dest="weight_ff_ends", help="weight of noun in free form ends (FLOAT, default = 0.5)", type=float, default=0.5)
w_p.add_argument("-wcompound", dest="weight_compound", help="weight of nouns in compound compared to head (FLOAT, default = 0.66)", type=float, default=0.66)
if (len(args) < 1):
args = p.parse_args()
else:
args = p.parse_args(args)
weights = [args.weight_func_role, args.weight_main_obj, args.weight_ff_means, args.weight_ff_ends, args.weight_compound]
if not args.system_name or args.system_name == '':
args.system_name = "System"
return main(args.filename, args.system_name, args.print_us, args.print_ont, args.statistics, args.link, args.prolog, args.per_role, args.threshold, args.base_weight, weights)
def is_valid_file(parser, arg):
if not os.path.exists(arg):
parser.error("Could not find file " + str(arg) + "!")
else:
return open(arg, 'r')
if __name__ == "__main__":
program()