Mercurial > repos > stevecassidy > nltktools
annotate g_pos.py @ 0:e991d4e60c17 draft
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
author | stevecassidy |
---|---|
date | Wed, 12 Oct 2016 22:17:53 -0400 |
parents | |
children | fb617586f4b2 |
rev | line source |
---|---|
0
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
1 import nltk |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
2 import argparse |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
3 import json |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
4 |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
5 def arguments(): |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
6 parser = argparse.ArgumentParser(description="tokenize a text") |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
7 parser.add_argument('--input', required=True, action="store", type=str, help="input text file") |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
8 parser.add_argument('--output', required=True, action="store", type=str, help="output file path") |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
9 args = parser.parse_args() |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
10 return args |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
11 |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
12 |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
13 def postag(in_file, out_file): |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
14 """Input: a text file with one token per line |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
15 Output: a version of the text with Part of Speech tags written as word/TAG |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
16 """ |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
17 text = unicode(open(in_file, 'r').read(), errors='ignore') |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
18 sentences = nltk.sent_tokenize(text) |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
19 output = open(out_file, 'w') |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
20 for sentence in sentences: |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
21 tokens = nltk.word_tokenize(sentence) |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
22 postags = nltk.pos_tag(tokens) |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
23 for postag in postags: |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
24 # print postag |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
25 output.write("%s/%s " % postag) |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
26 output.write('\n') |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
27 output.close() |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
28 |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
29 |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
30 if __name__ == '__main__': |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
31 args = arguments() |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
32 postag(args.input, args.output) |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
33 |
e991d4e60c17
planemo upload commit 0203cb3a0b40d9348674b2b098af805e2986abca-dirty
stevecassidy
parents:
diff
changeset
|
34 |