· 11 years ago · Aug 03, 2015, 06:23 PM
1# -*- coding: utf-8 -*-
2# Natural Language Toolkit: WordNet
3#
4# Copyright (C) 2001-2015 NLTK Project
5# Author: Steven Bethard <Steven.Bethard@colorado.edu>
6# Steven Bird <stevenbird1@gmail.com>
7# Edward Loper <edloper@gmail.com>
8# Nitin Madnani <nmadnani@ets.org>
9# URL: <http://nltk.org/>
10# For license information, see LICENSE.TXT
11
12"""
13An NLTK interface for WordNet
14
15WordNet is a lexical database of English.
16Using synsets, helps find conceptual relationships between words
17such as hypernyms, hyponyms, synonyms, antonyms etc.
18
19For details about WordNet see:
20http://wordnet.princeton.edu/
21"""
22
23from __future__ import print_function, unicode_literals
24
25import math
26import re
27from itertools import islice, chain
28from operator import itemgetter, attrgetter
29from collections import defaultdict, deque
30
31from nltk.corpus.reader import CorpusReader
32from nltk.util import binary_search_file as _binary_search_file
33from nltk.probability import FreqDist
34from nltk.compat import (iteritems, python_2_unicode_compatible,
35 total_ordering, xrange)
36
37######################################################################
38## Table of Contents
39######################################################################
40## - Constants
41## - Data Classes
42## - WordNetError
43## - Lemma
44## - Synset
45## - WordNet Corpus Reader
46## - WordNet Information Content Corpus Reader
47## - Similarity Metrics
48## - Demo
49
50######################################################################
51## Constants
52######################################################################
53
54#: Positive infinity (for similarity functions)
55_INF = 1e300
56
57#{ Part-of-speech constants
58ADJ, ADJ_SAT, ADV, NOUN, VERB = 'a', 's', 'r', 'n', 'v'
59#}
60
61POS_LIST = [NOUN, VERB, ADJ, ADV]
62
63#: A table of strings that are used to express verb frames.
64VERB_FRAME_STRINGS = (
65 None,
66 "Something %s",
67 "Somebody %s",
68 "It is %sing",
69 "Something is %sing PP",
70 "Something %s something Adjective/Noun",
71 "Something %s Adjective/Noun",
72 "Somebody %s Adjective",
73 "Somebody %s something",
74 "Somebody %s somebody",
75 "Something %s somebody",
76 "Something %s something",
77 "Something %s to somebody",
78 "Somebody %s on something",
79 "Somebody %s somebody something",
80 "Somebody %s something to somebody",
81 "Somebody %s something from somebody",
82 "Somebody %s somebody with something",
83 "Somebody %s somebody of something",
84 "Somebody %s something on somebody",
85 "Somebody %s somebody PP",
86 "Somebody %s something PP",
87 "Somebody %s PP",
88 "Somebody's (body part) %s",
89 "Somebody %s somebody to INFINITIVE",
90 "Somebody %s somebody INFINITIVE",
91 "Somebody %s that CLAUSE",
92 "Somebody %s to somebody",
93 "Somebody %s to INFINITIVE",
94 "Somebody %s whether INFINITIVE",
95 "Somebody %s somebody into V-ing something",
96 "Somebody %s something with something",
97 "Somebody %s INFINITIVE",
98 "Somebody %s VERB-ing",
99 "It %s that CLAUSE",
100 "Something %s INFINITIVE")
101
102SENSENUM_RE = re.compile(r'\.\d\d\.')
103
104######################################################################
105## Data Classes
106######################################################################
107
108[docs]class WordNetError(Exception):
109 """An exception class for wordnet-related errors."""
110
111
112@total_ordering
113class _WordNetObject(object):
114 """A common base class for lemmas and synsets."""
115
116 def hypernyms(self):
117 return self._related('@')
118
119 def _hypernyms(self):
120 return self._related('@', sort=False)
121
122 def instance_hypernyms(self):
123 return self._related('@i')
124
125 def _instance_hypernyms(self):
126 return self._related('@i', sort=False)
127
128 def hyponyms(self):
129 return self._related('~')
130
131 def instance_hyponyms(self):
132 return self._related('~i')
133
134 def member_holonyms(self):
135 return self._related('#m')
136
137 def substance_holonyms(self):
138 return self._related('#s')
139
140 def part_holonyms(self):
141 return self._related('#p')
142
143 def member_meronyms(self):
144 return self._related('%m')
145
146 def substance_meronyms(self):
147 return self._related('%s')
148
149 def part_meronyms(self):
150 return self._related('%p')
151
152 def topic_domains(self):
153 return self._related(';c')
154
155 def region_domains(self):
156 return self._related(';r')
157
158 def usage_domains(self):
159 return self._related(';u')
160
161 def attributes(self):
162 return self._related('=')
163
164 def entailments(self):
165 return self._related('*')
166
167 def causes(self):
168 return self._related('>')
169
170 def also_sees(self):
171 return self._related('^')
172
173 def verb_groups(self):
174 return self._related('$')
175
176 def similar_tos(self):
177 return self._related('&')
178
179 def __hash__(self):
180 return hash(self._name)
181
182 def __eq__(self, other):
183 return self._name == other._name
184
185 def __ne__(self, other):
186 return self._name != other._name
187
188 def __lt__(self, other):
189 return self._name < other._name
190
191
192@python_2_unicode_compatible
193[docs]class Lemma(_WordNetObject):
194 """
195 The lexical entry for a single morphological form of a
196 sense-disambiguated word.
197
198 Create a Lemma from a "<word>.<pos>.<number>.<lemma>" string where:
199 <word> is the morphological stem identifying the synset
200 <pos> is one of the module attributes ADJ, ADJ_SAT, ADV, NOUN or VERB
201 <number> is the sense number, counting from 0.
202 <lemma> is the morphological form of interest
203
204 Note that <word> and <lemma> can be different, e.g. the Synset
205 'salt.n.03' has the Lemmas 'salt.n.03.salt', 'salt.n.03.saltiness' and
206 'salt.n.03.salinity'.
207
208 Lemma attributes, accessible via methods with the same name::
209
210 - name: The canonical name of this lemma.
211 - synset: The synset that this lemma belongs to.
212 - syntactic_marker: For adjectives, the WordNet string identifying the
213 syntactic position relative modified noun. See:
214 http://wordnet.princeton.edu/man/wninput.5WN.html#sect10
215 For all other parts of speech, this attribute is None.
216 - count: The frequency of this lemma in wordnet.
217
218 Lemma methods:
219
220 Lemmas have the following methods for retrieving related Lemmas. They
221 correspond to the names for the pointer symbols defined here:
222 http://wordnet.princeton.edu/man/wninput.5WN.html#sect3
223 These methods all return lists of Lemmas:
224
225 - antonyms
226 - hypernyms, instance_hypernyms
227 - hyponyms, instance_hyponyms
228 - member_holonyms, substance_holonyms, part_holonyms
229 - member_meronyms, substance_meronyms, part_meronyms
230 - topic_domains, region_domains, usage_domains
231 - attributes
232 - derivationally_related_forms
233 - entailments
234 - causes
235 - also_sees
236 - verb_groups
237 - similar_tos
238 - pertainyms
239 """
240
241 __slots__ = ['_wordnet_corpus_reader', '_name', '_syntactic_marker',
242 '_synset', '_frame_strings', '_frame_ids',
243 '_lexname_index', '_lex_id', '_lang', '_key']
244
245 def __init__(self, wordnet_corpus_reader, synset, name,
246 lexname_index, lex_id, syntactic_marker):
247 self._wordnet_corpus_reader = wordnet_corpus_reader
248 self._name = name
249 self._syntactic_marker = syntactic_marker
250 self._synset = synset
251 self._frame_strings = []
252 self._frame_ids = []
253 self._lexname_index = lexname_index
254 self._lex_id = lex_id
255 self._lang = "en"
256
257 self._key = None # gets set later.
258
259[docs] def name(self):
260 return self._name
261
262[docs] def syntactic_marker(self):
263 return self._syntactic_marker
264
265[docs] def synset(self):
266 return self._synset
267
268[docs] def frame_strings(self):
269 return self._frame_strings
270
271[docs] def frame_ids(self):
272 return self._frame_ids
273
274[docs] def lang(self):
275 return self._lang
276
277[docs] def key(self):
278 return self._key
279
280 def __repr__(self):
281 tup = type(self).__name__, self._synset._name, self._name
282 return "%s('%s.%s')" % tup
283
284 def _related(self, relation_symbol):
285 get_synset = self._wordnet_corpus_reader._synset_from_pos_and_offset
286 return sorted([get_synset(pos, offset)._lemmas[lemma_index]
287 for pos, offset, lemma_index
288 in self._synset._lemma_pointers[self._name, relation_symbol]])
289
290[docs] def count(self):
291 """Return the frequency count for this Lemma"""
292 return self._wordnet_corpus_reader.lemma_count(self)
293
294[docs] def antonyms(self):
295 return self._related('!')
296
297[docs] def derivationally_related_forms(self):
298 return self._related('+')
299
300[docs] def pertainyms(self):
301 return self._related('\\')
302
303
304@python_2_unicode_compatible
305[docs]class Synset(_WordNetObject):
306 """Create a Synset from a "<lemma>.<pos>.<number>" string where:
307 <lemma> is the word's morphological stem
308 <pos> is one of the module attributes ADJ, ADJ_SAT, ADV, NOUN or VERB
309 <number> is the sense number, counting from 0.
310
311 Synset attributes, accessible via methods with the same name:
312
313 - name: The canonical name of this synset, formed using the first lemma
314 of this synset. Note that this may be different from the name
315 passed to the constructor if that string used a different lemma to
316 identify the synset.
317 - pos: The synset's part of speech, matching one of the module level
318 attributes ADJ, ADJ_SAT, ADV, NOUN or VERB.
319 - lemmas: A list of the Lemma objects for this synset.
320 - definition: The definition for this synset.
321 - examples: A list of example strings for this synset.
322 - offset: The offset in the WordNet dict file of this synset.
323 - lexname: The name of the lexicographer file containing this synset.
324
325 Synset methods:
326
327 Synsets have the following methods for retrieving related Synsets.
328 They correspond to the names for the pointer symbols defined here:
329 http://wordnet.princeton.edu/man/wninput.5WN.html#sect3
330 These methods all return lists of Synsets.
331
332 - hypernyms, instance_hypernyms
333 - hyponyms, instance_hyponyms
334 - member_holonyms, substance_holonyms, part_holonyms
335 - member_meronyms, substance_meronyms, part_meronyms
336 - attributes
337 - entailments
338 - causes
339 - also_sees
340 - verb_groups
341 - similar_tos
342
343 Additionally, Synsets support the following methods specific to the
344 hypernym relation:
345
346 - root_hypernyms
347 - common_hypernyms
348 - lowest_common_hypernyms
349
350 Note that Synsets do not support the following relations because
351 these are defined by WordNet as lexical relations:
352
353 - antonyms
354 - derivationally_related_forms
355 - pertainyms
356 """
357
358 __slots__ = ['_pos', '_offset', '_name', '_frame_ids',
359 '_lemmas', '_lemma_names',
360 '_definition', '_examples', '_lexname',
361 '_pointers', '_lemma_pointers', '_max_depth',
362 '_min_depth']
363
364 def __init__(self, wordnet_corpus_reader):
365 self._wordnet_corpus_reader = wordnet_corpus_reader
366 # All of these attributes get initialized by
367 # WordNetCorpusReader._synset_from_pos_and_line()
368
369 self._pos = None
370 self._offset = None
371 self._name = None
372 self._frame_ids = []
373 self._lemmas = []
374 self._lemma_names = []
375 self._definition = None
376 self._examples = []
377 self._lexname = None # lexicographer name
378 self._all_hypernyms = None
379
380 self._pointers = defaultdict(set)
381 self._lemma_pointers = defaultdict(set)
382
383[docs] def pos(self):
384 return self._pos
385
386[docs] def offset(self):
387 return self._offset
388
389[docs] def name(self):
390 return self._name
391
392[docs] def frame_ids(self):
393 return self._frame_ids
394
395[docs] def definition(self):
396 return self._definition
397
398[docs] def examples(self):
399 return self._examples
400
401[docs] def lexname(self):
402 return self._lexname
403
404 def _needs_root(self):
405 if self._pos == NOUN:
406 if self._wordnet_corpus_reader.get_version() == '1.6':
407 return True
408 else:
409 return False
410 elif self._pos == VERB:
411 return True
412
413[docs] def lemma_names(self, lang='en'):
414 '''Return all the lemma_names associated with the synset'''
415 if lang=='en':
416 return self._lemma_names
417 else:
418 self._wordnet_corpus_reader._load_lang_data(lang)
419
420 i = self._wordnet_corpus_reader.ss2of(self)
421 for x in self._wordnet_corpus_reader._lang_data[lang][0].keys():
422 if x == i:
423 return self._wordnet_corpus_reader._lang_data[lang][0][x]
424
425
426[docs] def lemmas(self, lang='en'):
427 '''Return all the lemma objects associated with the synset'''
428 if lang=='en':
429 return self._lemmas
430 else:
431 self._wordnet_corpus_reader._load_lang_data(lang)
432 lemmark = []
433 lemmy = self.lemma_names(lang)
434 for lem in lemmy:
435 temp= Lemma(self._wordnet_corpus_reader, self, lem, self._wordnet_corpus_reader._lexnames.index(self.lexname()), 0, None)
436 temp._lang=lang
437 lemmark.append(temp)
438 return lemmark
439
440
441[docs] def root_hypernyms(self):
442 """Get the topmost hypernyms of this synset in WordNet."""
443
444 result = []
445 seen = set()
446 todo = [self]
447 while todo:
448 next_synset = todo.pop()
449 if next_synset not in seen:
450 seen.add(next_synset)
451 next_hypernyms = next_synset.hypernyms() + \
452 next_synset.instance_hypernyms()
453 if not next_hypernyms:
454 result.append(next_synset)
455 else:
456 todo.extend(next_hypernyms)
457 return result
458
459# Simpler implementation which makes incorrect assumption that
460# hypernym hierarchy is acyclic:
461#
462# if not self.hypernyms():
463# return [self]
464# else:
465# return list(set(root for h in self.hypernyms()
466# for root in h.root_hypernyms()))
467
468[docs] def max_depth(self):
469 """
470 :return: The length of the longest hypernym path from this
471 synset to the root.
472 """
473
474 if "_max_depth" not in self.__dict__:
475 hypernyms = self.hypernyms() + self.instance_hypernyms()
476 if not hypernyms:
477 self._max_depth = 0
478 else:
479 self._max_depth = 1 + max(h.max_depth() for h in hypernyms)
480 return self._max_depth
481
482[docs] def min_depth(self):
483 """
484 :return: The length of the shortest hypernym path from this
485 synset to the root.
486 """
487
488 if "_min_depth" not in self.__dict__:
489 hypernyms = self.hypernyms() + self.instance_hypernyms()
490 if not hypernyms:
491 self._min_depth = 0
492 else:
493 self._min_depth = 1 + min(h.min_depth() for h in hypernyms)
494 return self._min_depth
495
496[docs] def closure(self, rel, depth=-1):
497 """Return the transitive closure of source under the rel
498 relationship, breadth-first
499
500 >>> from nltk.corpus import wordnet as wn
501 >>> dog = wn.synset('dog.n.01')
502 >>> hyp = lambda s:s.hypernyms()
503 >>> list(dog.closure(hyp))
504 [Synset('canine.n.02'), Synset('domestic_animal.n.01'),
505 Synset('carnivore.n.01'), Synset('animal.n.01'),
506 Synset('placental.n.01'), Synset('organism.n.01'),
507 Synset('mammal.n.01'), Synset('living_thing.n.01'),
508 Synset('vertebrate.n.01'), Synset('whole.n.02'),
509 Synset('chordate.n.01'), Synset('object.n.01'),
510 Synset('physical_entity.n.01'), Synset('entity.n.01')]
511
512 """
513 from nltk.util import breadth_first
514 synset_offsets = []
515 for synset in breadth_first(self, rel, depth):
516 if synset._offset != self._offset:
517 if synset._offset not in synset_offsets:
518 synset_offsets.append(synset._offset)
519 yield synset
520
521[docs] def hypernym_paths(self):
522 """
523 Get the path(s) from this synset to the root, where each path is a
524 list of the synset nodes traversed on the way to the root.
525
526 :return: A list of lists, where each list gives the node sequence
527 connecting the initial ``Synset`` node and a root node.
528 """
529 paths = []
530
531 hypernyms = self.hypernyms() + self.instance_hypernyms()
532 if len(hypernyms) == 0:
533 paths = [[self]]
534
535 for hypernym in hypernyms:
536 for ancestor_list in hypernym.hypernym_paths():
537 ancestor_list.append(self)
538 paths.append(ancestor_list)
539 return paths
540
541[docs] def common_hypernyms(self, other):
542 """
543 Find all synsets that are hypernyms of this synset and the
544 other synset.
545
546 :type other: Synset
547 :param other: other input synset.
548 :return: The synsets that are hypernyms of both synsets.
549 """
550 if not self._all_hypernyms:
551 self._all_hypernyms = set(self_synset
552 for self_synsets in self._iter_hypernym_lists()
553 for self_synset in self_synsets)
554 if not other._all_hypernyms:
555 other._all_hypernyms = set(other_synset
556 for other_synsets in other._iter_hypernym_lists()
557 for other_synset in other_synsets)
558 return list(self._all_hypernyms.intersection(other._all_hypernyms))
559
560[docs] def lowest_common_hypernyms(self, other, simulate_root=False, use_min_depth=False):
561 """
562 Get a list of lowest synset(s) that both synsets have as a hypernym.
563 When `use_min_depth == False` this means that the synset which appears as a
564 hypernym of both `self` and `other` with the lowest maximum depth is returned
565 or if there are multiple such synsets at the same depth they are all returned
566
567 However, if `use_min_depth == True` then the synset(s) which has/have the lowest
568 minimum depth and appear(s) in both paths is/are returned.
569
570 By setting the use_min_depth flag to True, the behavior of NLTK2 can be preserved.
571 This was changed in NLTK3 to give more accurate results in a small set of cases,
572 generally with synsets concerning people. (eg: 'chef.n.01', 'fireman.n.01', etc.)
573
574 This method is an implementation of Ted Pedersen's "Lowest Common Subsumer" method
575 from the Perl Wordnet module. It can return either "self" or "other" if they are a
576 hypernym of the other.
577
578 :type other: Synset
579 :param other: other input synset
580 :type simulate_root: bool
581 :param simulate_root: The various verb taxonomies do not
582 share a single root which disallows this metric from working for
583 synsets that are not connected. This flag (False by default)
584 creates a fake root that connects all the taxonomies. Set it
585 to True to enable this behavior. For the noun taxonomy,
586 there is usually a default root except for WordNet version 1.6.
587 If you are using wordnet 1.6, a fake root will need to be added
588 for nouns as well.
589 :type use_min_depth: bool
590 :param use_min_depth: This setting mimics older (v2) behavior of NLTK wordnet
591 If True, will use the min_depth function to calculate the lowest common
592 hypernyms. This is known to give strange results for some synset pairs
593 (eg: 'chef.n.01', 'fireman.n.01') but is retained for backwards compatibility
594 :return: The synsets that are the lowest common hypernyms of both synsets
595 """
596 synsets = self.common_hypernyms(other)
597 if simulate_root:
598 fake_synset = Synset(None)
599 fake_synset._name = '*ROOT*'
600 fake_synset.hypernyms = lambda: []
601 fake_synset.instance_hypernyms = lambda: []
602 synsets.append(fake_synset)
603
604 try:
605 if use_min_depth:
606 max_depth = max(s.min_depth() for s in synsets)
607 unsorted_lch = [s for s in synsets if s.min_depth() == max_depth]
608 else:
609 max_depth = max(s.max_depth() for s in synsets)
610 unsorted_lch = [s for s in synsets if s.max_depth() == max_depth]
611 return sorted(unsorted_lch)
612 except ValueError:
613 return []
614
615[docs] def hypernym_distances(self, distance=0, simulate_root=False):
616 """
617 Get the path(s) from this synset to the root, counting the distance
618 of each node from the initial node on the way. A set of
619 (synset, distance) tuples is returned.
620
621 :type distance: int
622 :param distance: the distance (number of edges) from this hypernym to
623 the original hypernym ``Synset`` on which this method was called.
624 :return: A set of ``(Synset, int)`` tuples where each ``Synset`` is
625 a hypernym of the first ``Synset``.
626 """
627 distances = set([(self, distance)])
628 for hypernym in self._hypernyms() + self._instance_hypernyms():
629 distances |= hypernym.hypernym_distances(distance+1, simulate_root=False)
630 if simulate_root:
631 fake_synset = Synset(None)
632 fake_synset._name = '*ROOT*'
633 fake_synset_distance = max(distances, key=itemgetter(1))[1]
634 distances.add((fake_synset, fake_synset_distance+1))
635 return distances
636
637 def _shortest_hypernym_paths(self, simulate_root):
638 if self._name == '*ROOT*':
639 return {self: 0}
640
641 queue = deque([(self, 0)])
642 path = {}
643
644 while queue:
645 s, depth = queue.popleft()
646 if s in path:
647 continue
648 path[s] = depth
649
650 depth += 1
651 queue.extend((hyp, depth) for hyp in s._hypernyms())
652 queue.extend((hyp, depth) for hyp in s._instance_hypernyms())
653
654 if simulate_root:
655 fake_synset = Synset(None)
656 fake_synset._name = '*ROOT*'
657 path[fake_synset] = max(path.values()) + 1
658
659 return path
660
661[docs] def shortest_path_distance(self, other, simulate_root=False):
662 """
663 Returns the distance of the shortest path linking the two synsets (if
664 one exists). For each synset, all the ancestor nodes and their
665 distances are recorded and compared. The ancestor node common to both
666 synsets that can be reached with the minimum number of traversals is
667 used. If no ancestor nodes are common, None is returned. If a node is
668 compared with itself 0 is returned.
669
670 :type other: Synset
671 :param other: The Synset to which the shortest path will be found.
672 :return: The number of edges in the shortest path connecting the two
673 nodes, or None if no path exists.
674 """
675
676 if self == other:
677 return 0
678
679 dist_dict1 = self._shortest_hypernym_paths(simulate_root)
680 dist_dict2 = other._shortest_hypernym_paths(simulate_root)
681
682 # For each ancestor synset common to both subject synsets, find the
683 # connecting path length. Return the shortest of these.
684
685 inf = float('inf')
686 path_distance = inf
687 for synset, d1 in iteritems(dist_dict1):
688 d2 = dist_dict2.get(synset, inf)
689 path_distance = min(path_distance, d1 + d2)
690
691 return None if math.isinf(path_distance) else path_distance
692
693[docs] def tree(self, rel, depth=-1, cut_mark=None):
694 """
695 >>> from nltk.corpus import wordnet as wn
696 >>> dog = wn.synset('dog.n.01')
697 >>> hyp = lambda s:s.hypernyms()
698 >>> from pprint import pprint
699 >>> pprint(dog.tree(hyp))
700 [Synset('dog.n.01'),
701 [Synset('canine.n.02'),
702 [Synset('carnivore.n.01'),
703 [Synset('placental.n.01'),
704 [Synset('mammal.n.01'),
705 [Synset('vertebrate.n.01'),
706 [Synset('chordate.n.01'),
707 [Synset('animal.n.01'),
708 [Synset('organism.n.01'),
709 [Synset('living_thing.n.01'),
710 [Synset('whole.n.02'),
711 [Synset('object.n.01'),
712 [Synset('physical_entity.n.01'),
713 [Synset('entity.n.01')]]]]]]]]]]]]],
714 [Synset('domestic_animal.n.01'),
715 [Synset('animal.n.01'),
716 [Synset('organism.n.01'),
717 [Synset('living_thing.n.01'),
718 [Synset('whole.n.02'),
719 [Synset('object.n.01'),
720 [Synset('physical_entity.n.01'), [Synset('entity.n.01')]]]]]]]]]
721 """
722
723 tree = [self]
724 if depth != 0:
725 tree += [x.tree(rel, depth-1, cut_mark) for x in rel(self)]
726 elif cut_mark:
727 tree += [cut_mark]
728 return tree
729
730 # interface to similarity methods
731
732[docs] def path_similarity(self, other, verbose=False, simulate_root=True):
733 """
734 Path Distance Similarity:
735 Return a score denoting how similar two word senses are, based on the
736 shortest path that connects the senses in the is-a (hypernym/hypnoym)
737 taxonomy. The score is in the range 0 to 1, except in those cases where
738 a path cannot be found (will only be true for verbs as there are many
739 distinct verb taxonomies), in which case None is returned. A score of
740 1 represents identity i.e. comparing a sense with itself will return 1.
741
742 :type other: Synset
743 :param other: The ``Synset`` that this ``Synset`` is being compared to.
744 :type simulate_root: bool
745 :param simulate_root: The various verb taxonomies do not
746 share a single root which disallows this metric from working for
747 synsets that are not connected. This flag (True by default)
748 creates a fake root that connects all the taxonomies. Set it
749 to false to disable this behavior. For the noun taxonomy,
750 there is usually a default root except for WordNet version 1.6.
751 If you are using wordnet 1.6, a fake root will be added for nouns
752 as well.
753 :return: A score denoting the similarity of the two ``Synset`` objects,
754 normally between 0 and 1. None is returned if no connecting path
755 could be found. 1 is returned if a ``Synset`` is compared with
756 itself.
757 """
758
759 distance = self.shortest_path_distance(other, simulate_root=simulate_root and self._needs_root())
760 if distance is None or distance < 0:
761 return None
762 return 1.0 / (distance + 1)
763
764[docs] def lch_similarity(self, other, verbose=False, simulate_root=True):
765 """
766 Leacock Chodorow Similarity:
767 Return a score denoting how similar two word senses are, based on the
768 shortest path that connects the senses (as above) and the maximum depth
769 of the taxonomy in which the senses occur. The relationship is given as
770 -log(p/2d) where p is the shortest path length and d is the taxonomy
771 depth.
772
773 :type other: Synset
774 :param other: The ``Synset`` that this ``Synset`` is being compared to.
775 :type simulate_root: bool
776 :param simulate_root: The various verb taxonomies do not
777 share a single root which disallows this metric from working for
778 synsets that are not connected. This flag (True by default)
779 creates a fake root that connects all the taxonomies. Set it
780 to false to disable this behavior. For the noun taxonomy,
781 there is usually a default root except for WordNet version 1.6.
782 If you are using wordnet 1.6, a fake root will be added for nouns
783 as well.
784 :return: A score denoting the similarity of the two ``Synset`` objects,
785 normally greater than 0. None is returned if no connecting path
786 could be found. If a ``Synset`` is compared with itself, the
787 maximum score is returned, which varies depending on the taxonomy
788 depth.
789 """
790
791 if self._pos != other._pos:
792 raise WordNetError('Computing the lch similarity requires ' + \
793 '%s and %s to have the same part of speech.' % \
794 (self, other))
795
796 need_root = self._needs_root()
797
798 if self._pos not in self._wordnet_corpus_reader._max_depth:
799 self._wordnet_corpus_reader._compute_max_depth(self._pos, need_root)
800
801 depth = self._wordnet_corpus_reader._max_depth[self._pos]
802
803 distance = self.shortest_path_distance(other, simulate_root=simulate_root and need_root)
804
805 if distance is None or distance < 0 or depth == 0:
806 return None
807 return -math.log((distance + 1) / (2.0 * depth))
808
809[docs] def wup_similarity(self, other, verbose=False, simulate_root=True):
810 """
811 Wu-Palmer Similarity:
812 Return a score denoting how similar two word senses are, based on the
813 depth of the two senses in the taxonomy and that of their Least Common
814 Subsumer (most specific ancestor node). Previously, the scores computed
815 by this implementation did _not_ always agree with those given by
816 Pedersen's Perl implementation of WordNet Similarity. However, with
817 the addition of the simulate_root flag (see below), the score for
818 verbs now almost always agree but not always for nouns.
819
820 The LCS does not necessarily feature in the shortest path connecting
821 the two senses, as it is by definition the common ancestor deepest in
822 the taxonomy, not closest to the two senses. Typically, however, it
823 will so feature. Where multiple candidates for the LCS exist, that
824 whose shortest path to the root node is the longest will be selected.
825 Where the LCS has multiple paths to the root, the longer path is used
826 for the purposes of the calculation.
827
828 :type other: Synset
829 :param other: The ``Synset`` that this ``Synset`` is being compared to.
830 :type simulate_root: bool
831 :param simulate_root: The various verb taxonomies do not
832 share a single root which disallows this metric from working for
833 synsets that are not connected. This flag (True by default)
834 creates a fake root that connects all the taxonomies. Set it
835 to false to disable this behavior. For the noun taxonomy,
836 there is usually a default root except for WordNet version 1.6.
837 If you are using wordnet 1.6, a fake root will be added for nouns
838 as well.
839 :return: A float score denoting the similarity of the two ``Synset`` objects,
840 normally greater than zero. If no connecting path between the two
841 senses can be found, None is returned.
842
843 """
844
845 need_root = self._needs_root()
846 # Note that to preserve behavior from NLTK2 we set use_min_depth=True
847 # It is possible that more accurate results could be obtained by
848 # removing this setting and it should be tested later on
849 subsumers = self.lowest_common_hypernyms(other, simulate_root=simulate_root and need_root, use_min_depth=True)
850
851 # If no LCS was found return None
852 if len(subsumers) == 0:
853 return None
854
855 subsumer = subsumers[0]
856
857 # Get the longest path from the LCS to the root,
858 # including a correction:
859 # - add one because the calculations include both the start and end
860 # nodes
861 depth = subsumer.max_depth() + 1
862
863 # Note: No need for an additional add-one correction for non-nouns
864 # to account for an imaginary root node because that is now automatically
865 # handled by simulate_root
866 # if subsumer._pos != NOUN:
867 # depth += 1
868
869 # Get the shortest path from the LCS to each of the synsets it is
870 # subsuming. Add this to the LCS path length to get the path
871 # length from each synset to the root.
872 len1 = self.shortest_path_distance(subsumer, simulate_root=simulate_root and need_root)
873 len2 = other.shortest_path_distance(subsumer, simulate_root=simulate_root and need_root)
874 if len1 is None or len2 is None:
875 return None
876 len1 += depth
877 len2 += depth
878 return (2.0 * depth) / (len1 + len2)
879
880[docs] def res_similarity(self, other, ic, verbose=False):
881 """
882 Resnik Similarity:
883 Return a score denoting how similar two word senses are, based on the
884 Information Content (IC) of the Least Common Subsumer (most specific
885 ancestor node).
886
887 :type other: Synset
888 :param other: The ``Synset`` that this ``Synset`` is being compared to.
889 :type ic: dict
890 :param ic: an information content object (as returned by ``nltk.corpus.wordnet_ic.ic()``).
891 :return: A float score denoting the similarity of the two ``Synset`` objects.
892 Synsets whose LCS is the root node of the taxonomy will have a
893 score of 0 (e.g. N['dog'][0] and N['table'][0]).
894 """
895
896 ic1, ic2, lcs_ic = _lcs_ic(self, other, ic)
897 return lcs_ic
898
899[docs] def jcn_similarity(self, other, ic, verbose=False):
900 """
901 Jiang-Conrath Similarity:
902 Return a score denoting how similar two word senses are, based on the
903 Information Content (IC) of the Least Common Subsumer (most specific
904 ancestor node) and that of the two input Synsets. The relationship is
905 given by the equation 1 / (IC(s1) + IC(s2) - 2 * IC(lcs)).
906
907 :type other: Synset
908 :param other: The ``Synset`` that this ``Synset`` is being compared to.
909 :type ic: dict
910 :param ic: an information content object (as returned by ``nltk.corpus.wordnet_ic.ic()``).
911 :return: A float score denoting the similarity of the two ``Synset`` objects.
912 """
913
914 if self == other:
915 return _INF
916
917 ic1, ic2, lcs_ic = _lcs_ic(self, other, ic)
918
919 # If either of the input synsets are the root synset, or have a
920 # frequency of 0 (sparse data problem), return 0.
921 if ic1 == 0 or ic2 == 0:
922 return 0
923
924 ic_difference = ic1 + ic2 - 2 * lcs_ic
925
926 if ic_difference == 0:
927 return _INF
928
929 return 1 / ic_difference
930
931[docs] def lin_similarity(self, other, ic, verbose=False):
932 """
933 Lin Similarity:
934 Return a score denoting how similar two word senses are, based on the
935 Information Content (IC) of the Least Common Subsumer (most specific
936 ancestor node) and that of the two input Synsets. The relationship is
937 given by the equation 2 * IC(lcs) / (IC(s1) + IC(s2)).
938
939 :type other: Synset
940 :param other: The ``Synset`` that this ``Synset`` is being compared to.
941 :type ic: dict
942 :param ic: an information content object (as returned by ``nltk.corpus.wordnet_ic.ic()``).
943 :return: A float score denoting the similarity of the two ``Synset`` objects,
944 in the range 0 to 1.
945 """
946
947 ic1, ic2, lcs_ic = _lcs_ic(self, other, ic)
948 return (2.0 * lcs_ic) / (ic1 + ic2)
949
950 def _iter_hypernym_lists(self):
951 """
952 :return: An iterator over ``Synset`` objects that are either proper
953 hypernyms or instance of hypernyms of the synset.
954 """
955 todo = [self]
956 seen = set()
957 while todo:
958 for synset in todo:
959 seen.add(synset)
960 yield todo
961 todo = [hypernym
962 for synset in todo
963 for hypernym in (synset.hypernyms() +
964 synset.instance_hypernyms())
965 if hypernym not in seen]
966
967 def __repr__(self):
968 return "%s('%s')" % (type(self).__name__, self._name)
969
970 def _related(self, relation_symbol, sort=True):
971 get_synset = self._wordnet_corpus_reader._synset_from_pos_and_offset
972 pointer_tuples = self._pointers[relation_symbol]
973 r = [get_synset(pos, offset) for pos, offset in pointer_tuples]
974 if sort:
975 r.sort()
976 return r
977
978
979######################################################################
980## WordNet Corpus Reader
981######################################################################
982
983[docs]class WordNetCorpusReader(CorpusReader):
984 """
985 A corpus reader used to access wordnet or its variants.
986 """
987
988 _ENCODING = 'utf8'
989
990 #{ Part-of-speech constants
991 ADJ, ADJ_SAT, ADV, NOUN, VERB = 'a', 's', 'r', 'n', 'v'
992 #}
993
994 #{ Filename constants
995 _FILEMAP = {ADJ: 'adj', ADV: 'adv', NOUN: 'noun', VERB: 'verb'}
996 #}
997
998 #{ Part of speech constants
999 _pos_numbers = {NOUN: 1, VERB: 2, ADJ: 3, ADV: 4, ADJ_SAT: 5}
1000 _pos_names = dict(tup[::-1] for tup in _pos_numbers.items())
1001 #}
1002
1003 #: A list of file identifiers for all the fileids used by this
1004 #: corpus reader.
1005 _FILES = ('cntlist.rev', 'lexnames', 'index.sense',
1006 'index.adj', 'index.adv', 'index.noun', 'index.verb',
1007 'data.adj', 'data.adv', 'data.noun', 'data.verb',
1008 'adj.exc', 'adv.exc', 'noun.exc', 'verb.exc', )
1009
1010 def __init__(self, root, omw_reader):
1011 """
1012 Construct a new wordnet corpus reader, with the given root
1013 directory.
1014 """
1015 super(WordNetCorpusReader, self).__init__(root, self._FILES,
1016 encoding=self._ENCODING)
1017
1018 # A index that provides the file offset
1019 # Map from lemma -> pos -> synset_index -> offset
1020 self._lemma_pos_offset_map = defaultdict(dict)
1021
1022 # A cache so we don't have to reconstuct synsets
1023 # Map from pos -> offset -> synset
1024 self._synset_offset_cache = defaultdict(dict)
1025
1026 # A lookup for the maximum depth of each part of speech. Useful for
1027 # the lch similarity metric.
1028 self._max_depth = defaultdict(dict)
1029
1030 # Corpus reader containing omw data.
1031 self._omw_reader = omw_reader
1032
1033 # A cache to store the wordnet data of multiple languages
1034 self._lang_data = defaultdict(list)
1035
1036 self._data_file_map = {}
1037 self._exception_map = {}
1038 self._lexnames = []
1039 self._key_count_file = None
1040 self._key_synset_file = None
1041
1042 # Load the lexnames
1043 for i, line in enumerate(self.open('lexnames')):
1044 index, lexname, _ = line.split()
1045 assert int(index) == i
1046 self._lexnames.append(lexname)
1047
1048 # Load the indices for lemmas and synset offsets
1049 self._load_lemma_pos_offset_map()
1050
1051 # load the exception file data into memory
1052 self._load_exception_map()
1053
1054# Open Multilingual WordNet functions, contributed by
1055# Nasruddin A’aidil Shari, Sim Wei Ying Geraldine, and Soe Lynn
1056
1057[docs] def of2ss(self, of):
1058 ''' take an id and return the synsets '''
1059 return self._synset_from_pos_and_offset(of[-1], int(of[:8]))
1060
1061[docs] def ss2of(self, ss):
1062 ''' return the ILI of the synset '''
1063 return ( "0"*8 + str(ss.offset()) +"-"+ str(ss.pos()))[-10:]
1064
1065
1066 def _load_lang_data(self, lang):
1067 ''' load the wordnet data of the requested language from the file to the cache, _lang_data '''
1068
1069 if lang not in self.langs():
1070 raise WordNetError("Language is not supported.")
1071
1072 if lang in self._lang_data.keys():
1073 return
1074
1075 f = self._omw_reader.open('{0:}/wn-data-{0:}.tab'.format(lang))
1076
1077 self._lang_data[lang].append(defaultdict(list))
1078 self._lang_data[lang].append(defaultdict(list))
1079
1080 for l in f.readlines():
1081 l = l.replace('\n', '')
1082 l = l.replace(' ', '_')
1083 if l[0] != '#':
1084 word = l.split('\t')
1085 self._lang_data[lang][0][word[0]].append(word[2])
1086 self._lang_data[lang][1][word[2]].append(word[0])
1087 f.close()
1088
1089[docs] def langs(self):
1090 ''' return a list of languages supported by Multilingual Wordnet '''
1091 import os
1092 langs = []
1093 fileids = self._omw_reader.fileids()
1094 for fileid in fileids:
1095 file_name, file_extension = os.path.splitext(fileid)
1096 if file_extension == '.tab':
1097 langs.append(file_name.split('-')[-1])
1098
1099 return langs
1100
1101
1102
1103 def _load_lemma_pos_offset_map(self):
1104 for suffix in self._FILEMAP.values():
1105
1106 # parse each line of the file (ignoring comment lines)
1107 for i, line in enumerate(self.open('index.%s' % suffix)):
1108 if line.startswith(' '):
1109 continue
1110
1111 _iter = iter(line.split())
1112 _next_token = lambda: next(_iter)
1113 try:
1114
1115 # get the lemma and part-of-speech
1116 lemma = _next_token()
1117 pos = _next_token()
1118
1119 # get the number of synsets for this lemma
1120 n_synsets = int(_next_token())
1121 assert n_synsets > 0
1122
1123 # get the pointer symbols for all synsets of this lemma
1124 n_pointers = int(_next_token())
1125 _ = [_next_token() for _ in xrange(n_pointers)]
1126
1127 # same as number of synsets
1128 n_senses = int(_next_token())
1129 assert n_synsets == n_senses
1130
1131 # get number of senses ranked according to frequency
1132 _ = int(_next_token())
1133
1134 # get synset offsets
1135 synset_offsets = [int(_next_token()) for _ in xrange(n_synsets)]
1136
1137 # raise more informative error with file name and line number
1138 except (AssertionError, ValueError) as e:
1139 tup = ('index.%s' % suffix), (i + 1), e
1140 raise WordNetError('file %s, line %i: %s' % tup)
1141
1142 # map lemmas and parts of speech to synsets
1143 self._lemma_pos_offset_map[lemma][pos] = synset_offsets
1144 if pos == ADJ:
1145 self._lemma_pos_offset_map[lemma][ADJ_SAT] = synset_offsets
1146
1147 def _load_exception_map(self):
1148 # load the exception file data into memory
1149 for pos, suffix in self._FILEMAP.items():
1150 self._exception_map[pos] = {}
1151 for line in self.open('%s.exc' % suffix):
1152 terms = line.split()
1153 self._exception_map[pos][terms[0]] = terms[1:]
1154 self._exception_map[ADJ_SAT] = self._exception_map[ADJ]
1155
1156 def _compute_max_depth(self, pos, simulate_root):
1157 """
1158 Compute the max depth for the given part of speech. This is
1159 used by the lch similarity metric.
1160 """
1161 depth = 0
1162 for ii in self.all_synsets(pos):
1163 try:
1164 depth = max(depth, ii.max_depth())
1165 except RuntimeError:
1166 print(ii)
1167 if simulate_root:
1168 depth += 1
1169 self._max_depth[pos] = depth
1170
1171[docs] def get_version(self):
1172 fh = self._data_file(ADJ)
1173 for line in fh:
1174 match = re.search(r'WordNet (\d+\.\d+) Copyright', line)
1175 if match is not None:
1176 version = match.group(1)
1177 fh.seek(0)
1178 return version
1179
1180 #////////////////////////////////////////////////////////////
1181 # Loading Lemmas
1182 #////////////////////////////////////////////////////////////
1183
1184[docs] def lemma(self, name, lang='en'):
1185 '''Return lemma object that matches the name'''
1186 # cannot simply split on first '.', e.g.: '.45_caliber.a.01..45_caliber'
1187 separator = SENSENUM_RE.search(name).start()
1188 synset_name, lemma_name = name[:separator+3], name[separator+4:]
1189 synset = self.synset(synset_name)
1190 for lemma in synset.lemmas(lang):
1191 if lemma._name == lemma_name:
1192 return lemma
1193 raise WordNetError('no lemma %r in %r' % (lemma_name, synset_name))
1194
1195[docs] def lemma_from_key(self, key):
1196 # Keys are case sensitive and always lower-case
1197 key = key.lower()
1198
1199 lemma_name, lex_sense = key.split('%')
1200 pos_number, lexname_index, lex_id, _, _ = lex_sense.split(':')
1201 pos = self._pos_names[int(pos_number)]
1202
1203 # open the key -> synset file if necessary
1204 if self._key_synset_file is None:
1205 self._key_synset_file = self.open('index.sense')
1206
1207 # Find the synset for the lemma.
1208 synset_line = _binary_search_file(self._key_synset_file, key)
1209 if not synset_line:
1210 raise WordNetError("No synset found for key %r" % key)
1211 offset = int(synset_line.split()[1])
1212 synset = self._synset_from_pos_and_offset(pos, offset)
1213
1214 # return the corresponding lemma
1215 for lemma in synset._lemmas:
1216 if lemma._key == key:
1217 return lemma
1218 raise WordNetError("No lemma found for for key %r" % key)
1219
1220 #////////////////////////////////////////////////////////////
1221 # Loading Synsets
1222 #////////////////////////////////////////////////////////////
1223
1224[docs] def synset(self, name):
1225 # split name into lemma, part of speech and synset number
1226 lemma, pos, synset_index_str = name.lower().rsplit('.', 2)
1227 synset_index = int(synset_index_str) - 1
1228
1229 # get the offset for this synset
1230 try:
1231 offset = self._lemma_pos_offset_map[lemma][pos][synset_index]
1232 except KeyError:
1233 message = 'no lemma %r with part of speech %r'
1234 raise WordNetError(message % (lemma, pos))
1235 except IndexError:
1236 n_senses = len(self._lemma_pos_offset_map[lemma][pos])
1237 message = "lemma %r with part of speech %r has only %i %s"
1238 if n_senses == 1:
1239 tup = lemma, pos, n_senses, "sense"
1240 else:
1241 tup = lemma, pos, n_senses, "senses"
1242 raise WordNetError(message % tup)
1243
1244 # load synset information from the appropriate file
1245 synset = self._synset_from_pos_and_offset(pos, offset)
1246
1247 # some basic sanity checks on loaded attributes
1248 if pos == 's' and synset._pos == 'a':
1249 message = ('adjective satellite requested but only plain '
1250 'adjective found for lemma %r')
1251 raise WordNetError(message % lemma)
1252 assert synset._pos == pos or (pos == 'a' and synset._pos == 's')
1253
1254 # Return the synset object.
1255 return synset
1256
1257 def _data_file(self, pos):
1258 """
1259 Return an open file pointer for the data file for the given
1260 part of speech.
1261 """
1262 if pos == ADJ_SAT:
1263 pos = ADJ
1264 if self._data_file_map.get(pos) is None:
1265 fileid = 'data.%s' % self._FILEMAP[pos]
1266 self._data_file_map[pos] = self.open(fileid)
1267 return self._data_file_map[pos]
1268
1269 def _synset_from_pos_and_offset(self, pos, offset):
1270 # Check to see if the synset is in the cache
1271 if offset in self._synset_offset_cache[pos]:
1272 return self._synset_offset_cache[pos][offset]
1273
1274 data_file = self._data_file(pos)
1275 data_file.seek(offset)
1276 data_file_line = data_file.readline()
1277 synset = self._synset_from_pos_and_line(pos, data_file_line)
1278 assert synset._offset == offset
1279 self._synset_offset_cache[pos][offset] = synset
1280 return synset
1281
1282 def _synset_from_pos_and_line(self, pos, data_file_line):
1283 # Construct a new (empty) synset.
1284 synset = Synset(self)
1285
1286 # parse the entry for this synset
1287 try:
1288
1289 # parse out the definitions and examples from the gloss
1290 columns_str, gloss = data_file_line.split('|')
1291 gloss = gloss.strip()
1292 definitions = []
1293 for gloss_part in gloss.split(';'):
1294 gloss_part = gloss_part.strip()
1295 if gloss_part.startswith('"'):
1296 synset._examples.append(gloss_part.strip('"'))
1297 else:
1298 definitions.append(gloss_part)
1299 synset._definition = '; '.join(definitions)
1300
1301 # split the other info into fields
1302 _iter = iter(columns_str.split())
1303 _next_token = lambda: next(_iter)
1304
1305 # get the offset
1306 synset._offset = int(_next_token())
1307
1308 # determine the lexicographer file name
1309 lexname_index = int(_next_token())
1310 synset._lexname = self._lexnames[lexname_index]
1311
1312 # get the part of speech
1313 synset._pos = _next_token()
1314
1315 # create Lemma objects for each lemma
1316 n_lemmas = int(_next_token(), 16)
1317 for _ in xrange(n_lemmas):
1318 # get the lemma name
1319 lemma_name = _next_token()
1320 # get the lex_id (used for sense_keys)
1321 lex_id = int(_next_token(), 16)
1322 # If the lemma has a syntactic marker, extract it.
1323 m = re.match(r'(.*?)(\(.*\))?$', lemma_name)
1324 lemma_name, syn_mark = m.groups()
1325 # create the lemma object
1326 lemma = Lemma(self, synset, lemma_name, lexname_index,
1327 lex_id, syn_mark)
1328 synset._lemmas.append(lemma)
1329 synset._lemma_names.append(lemma._name)
1330
1331 # collect the pointer tuples
1332 n_pointers = int(_next_token())
1333 for _ in xrange(n_pointers):
1334 symbol = _next_token()
1335 offset = int(_next_token())
1336 pos = _next_token()
1337 lemma_ids_str = _next_token()
1338 if lemma_ids_str == '0000':
1339 synset._pointers[symbol].add((pos, offset))
1340 else:
1341 source_index = int(lemma_ids_str[:2], 16) - 1
1342 target_index = int(lemma_ids_str[2:], 16) - 1
1343 source_lemma_name = synset._lemmas[source_index]._name
1344 lemma_pointers = synset._lemma_pointers
1345 tups = lemma_pointers[source_lemma_name, symbol]
1346 tups.add((pos, offset, target_index))
1347
1348 # read the verb frames
1349 try:
1350 frame_count = int(_next_token())
1351 except StopIteration:
1352 pass
1353 else:
1354 for _ in xrange(frame_count):
1355 # read the plus sign
1356 plus = _next_token()
1357 assert plus == '+'
1358 # read the frame and lemma number
1359 frame_number = int(_next_token())
1360 frame_string_fmt = VERB_FRAME_STRINGS[frame_number]
1361 lemma_number = int(_next_token(), 16)
1362 # lemma number of 00 means all words in the synset
1363 if lemma_number == 0:
1364 synset._frame_ids.append(frame_number)
1365 for lemma in synset._lemmas:
1366 lemma._frame_ids.append(frame_number)
1367 lemma._frame_strings.append(frame_string_fmt %
1368 lemma._name)
1369 # only a specific word in the synset
1370 else:
1371 lemma = synset._lemmas[lemma_number - 1]
1372 lemma._frame_ids.append(frame_number)
1373 lemma._frame_strings.append(frame_string_fmt %
1374 lemma._name)
1375
1376 # raise a more informative error with line text
1377 except ValueError as e:
1378 raise WordNetError('line %r: %s' % (data_file_line, e))
1379
1380 # set sense keys for Lemma objects - note that this has to be
1381 # done afterwards so that the relations are available
1382 for lemma in synset._lemmas:
1383 if synset._pos == ADJ_SAT:
1384 head_lemma = synset.similar_tos()[0]._lemmas[0]
1385 head_name = head_lemma._name
1386 head_id = '%02d' % head_lemma._lex_id
1387 else:
1388 head_name = head_id = ''
1389 tup = (lemma._name, WordNetCorpusReader._pos_numbers[synset._pos],
1390 lemma._lexname_index, lemma._lex_id, head_name, head_id)
1391 lemma._key = ('%s%%%d:%02d:%02d:%s:%s' % tup).lower()
1392
1393 # the canonical name is based on the first lemma
1394 lemma_name = synset._lemmas[0]._name.lower()
1395 offsets = self._lemma_pos_offset_map[lemma_name][synset._pos]
1396 sense_index = offsets.index(synset._offset)
1397 tup = lemma_name, synset._pos, sense_index + 1
1398 synset._name = '%s.%s.%02i' % tup
1399
1400 return synset
1401
1402 #////////////////////////////////////////////////////////////
1403 # Retrieve synsets and lemmas.
1404 #////////////////////////////////////////////////////////////
1405
1406[docs] def synsets(self, lemma, pos=None, lang='en'):
1407 """Load all synsets with a given lemma and part of speech tag.
1408 If no pos is specified, all synsets for all parts of speech
1409 will be loaded.
1410 If lang is specified, all the synsets associated with the lemma name
1411 of that language will be returned.
1412 """
1413 lemma = lemma.lower()
1414
1415 if lang == 'en':
1416 get_synset = self._synset_from_pos_and_offset
1417 index = self._lemma_pos_offset_map
1418 if pos is None:
1419 pos = POS_LIST
1420 return [get_synset(p, offset)
1421 for p in pos
1422 for form in self._morphy(lemma, p)
1423 for offset in index[form].get(p, [])]
1424
1425 else:
1426 self._load_lang_data(lang)
1427 synset_list = []
1428 for l in self._lang_data[lang][1][lemma]:
1429 if pos is not None and l[-1] != pos:
1430 continue
1431 synset_list.append(self.of2ss(l))
1432 return synset_list
1433
1434[docs] def lemmas(self, lemma, pos=None, lang='en'):
1435 """Return all Lemma objects with a name matching the specified lemma
1436 name and part of speech tag. Matches any part of speech tag if none is
1437 specified."""
1438
1439 if lang == 'en':
1440 lemma = lemma.lower()
1441 return [lemma_obj
1442 for synset in self.synsets(lemma, pos)
1443 for lemma_obj in synset.lemmas()
1444 if lemma_obj.name().lower() == lemma]
1445
1446 else:
1447 self._load_lang_data(lang)
1448 lemmas = []
1449 syn = self.synsets(lemma, lang=lang)
1450 for s in syn:
1451 if pos is not None and s.pos() != pos:
1452 continue
1453 a = Lemma(self, s, lemma, self._lexnames.index(s.lexname()), 0, None)
1454 a._lang = lang
1455 lemmas.append(a)
1456 return lemmas
1457
1458[docs] def all_lemma_names(self, pos=None, lang='en'):
1459 """Return all lemma names for all synsets for the given
1460 part of speech tag and langauge or languages. If pos is not specified, all synsets
1461 for all parts of speech will be used."""
1462
1463 if lang == 'en':
1464 if pos is None:
1465 return iter(self._lemma_pos_offset_map)
1466 else:
1467 return (lemma
1468 for lemma in self._lemma_pos_offset_map
1469 if pos in self._lemma_pos_offset_map[lemma])
1470 else:
1471 self._load_lang_data(lang)
1472 lemma = []
1473 for i in self._lang_data[lang][0]:
1474 if pos is not None and i[-1] != pos:
1475 continue
1476 lemma.extend(self._lang_data[lang][0][i])
1477
1478 lemma = list(set(lemma))
1479 return lemma
1480
1481
1482[docs] def all_synsets(self, pos=None):
1483 """Iterate over all synsets with a given part of speech tag.
1484 If no pos is specified, all synsets for all parts of speech
1485 will be loaded.
1486 """
1487 if pos is None:
1488 pos_tags = self._FILEMAP.keys()
1489 else:
1490 pos_tags = [pos]
1491
1492 cache = self._synset_offset_cache
1493 from_pos_and_line = self._synset_from_pos_and_line
1494
1495 # generate all synsets for each part of speech
1496 for pos_tag in pos_tags:
1497 # Open the file for reading. Note that we can not re-use
1498 # the file poitners from self._data_file_map here, because
1499 # we're defining an iterator, and those file pointers might
1500 # be moved while we're not looking.
1501 if pos_tag == ADJ_SAT:
1502 pos_tag = ADJ
1503 fileid = 'data.%s' % self._FILEMAP[pos_tag]
1504 data_file = self.open(fileid)
1505
1506 try:
1507 # generate synsets for each line in the POS file
1508 offset = data_file.tell()
1509 line = data_file.readline()
1510 while line:
1511 if not line[0].isspace():
1512 if offset in cache[pos_tag]:
1513 # See if the synset is cached
1514 synset = cache[pos_tag][offset]
1515 else:
1516 # Otherwise, parse the line
1517 synset = from_pos_and_line(pos_tag, line)
1518 cache[pos_tag][offset] = synset
1519
1520 # adjective satellites are in the same file as
1521 # adjectives so only yield the synset if it's actually
1522 # a satellite
1523 if pos_tag == ADJ_SAT:
1524 if synset._pos == pos_tag:
1525 yield synset
1526
1527 # for all other POS tags, yield all synsets (this means
1528 # that adjectives also include adjective satellites)
1529 else:
1530 yield synset
1531 offset = data_file.tell()
1532 line = data_file.readline()
1533
1534 # close the extra file handle we opened
1535 except:
1536 data_file.close()
1537 raise
1538 else:
1539 data_file.close()
1540
1541 #////////////////////////////////////////////////////////////
1542 # Misc
1543 #////////////////////////////////////////////////////////////
1544
1545[docs] def lemma_count(self, lemma):
1546 """Return the frequency count for this Lemma"""
1547 # open the count file if we haven't already
1548 if self._key_count_file is None:
1549 self._key_count_file = self.open('cntlist.rev')
1550 # find the key in the counts file and return the count
1551 line = _binary_search_file(self._key_count_file, lemma._key)
1552 if line:
1553 return int(line.rsplit(' ', 1)[-1])
1554 else:
1555 return 0
1556
1557[docs] def path_similarity(self, synset1, synset2, verbose=False, simulate_root=True):
1558 return synset1.path_similarity(synset2, verbose, simulate_root)
1559
1560 path_similarity.__doc__ = Synset.path_similarity.__doc__
1561
1562[docs] def lch_similarity(self, synset1, synset2, verbose=False, simulate_root=True):
1563 return synset1.lch_similarity(synset2, verbose, simulate_root)
1564
1565 lch_similarity.__doc__ = Synset.lch_similarity.__doc__
1566
1567[docs] def wup_similarity(self, synset1, synset2, verbose=False, simulate_root=True):
1568 return synset1.wup_similarity(synset2, verbose, simulate_root)
1569
1570 wup_similarity.__doc__ = Synset.wup_similarity.__doc__
1571
1572[docs] def res_similarity(self, synset1, synset2, ic, verbose=False):
1573 return synset1.res_similarity(synset2, ic, verbose)
1574
1575 res_similarity.__doc__ = Synset.res_similarity.__doc__
1576
1577[docs] def jcn_similarity(self, synset1, synset2, ic, verbose=False):
1578 return synset1.jcn_similarity(synset2, ic, verbose)
1579
1580 jcn_similarity.__doc__ = Synset.jcn_similarity.__doc__
1581
1582[docs] def lin_similarity(self, synset1, synset2, ic, verbose=False):
1583 return synset1.lin_similarity(synset2, ic, verbose)
1584
1585 lin_similarity.__doc__ = Synset.lin_similarity.__doc__
1586
1587 #////////////////////////////////////////////////////////////
1588 # Morphy
1589 #////////////////////////////////////////////////////////////
1590 # Morphy, adapted from Oliver Steele's pywordnet
1591[docs] def morphy(self, form, pos=None):
1592 """
1593 Find a possible base form for the given form, with the given
1594 part of speech, by checking WordNet's list of exceptional
1595 forms, and by recursively stripping affixes for this part of
1596 speech until a form in WordNet is found.
1597
1598 >>> from nltk.corpus import wordnet as wn
1599 >>> print(wn.morphy('dogs'))
1600 dog
1601 >>> print(wn.morphy('churches'))
1602 church
1603 >>> print(wn.morphy('aardwolves'))
1604 aardwolf
1605 >>> print(wn.morphy('abaci'))
1606 abacus
1607 >>> wn.morphy('hardrock', wn.ADV)
1608 >>> print(wn.morphy('book', wn.NOUN))
1609 book
1610 >>> wn.morphy('book', wn.ADJ)
1611 """
1612
1613 if pos is None:
1614 morphy = self._morphy
1615 analyses = chain(a for p in POS_LIST for a in morphy(form, p))
1616 else:
1617 analyses = self._morphy(form, pos)
1618
1619 # get the first one we find
1620 first = list(islice(analyses, 1))
1621 if len(first) == 1:
1622 return first[0]
1623 else:
1624 return None
1625
1626 MORPHOLOGICAL_SUBSTITUTIONS = {
1627 NOUN: [('s', ''), ('ses', 's'), ('ves', 'f'), ('xes', 'x'),
1628 ('zes', 'z'), ('ches', 'ch'), ('shes', 'sh'),
1629 ('men', 'man'), ('ies', 'y')],
1630 VERB: [('s', ''), ('ies', 'y'), ('es', 'e'), ('es', ''),
1631 ('ed', 'e'), ('ed', ''), ('ing', 'e'), ('ing', '')],
1632 ADJ: [('er', ''), ('est', ''), ('er', 'e'), ('est', 'e')],
1633 ADV: []}
1634
1635 MORPHOLOGICAL_SUBSTITUTIONS[ADJ_SAT] = MORPHOLOGICAL_SUBSTITUTIONS[ADJ]
1636
1637 def _morphy(self, form, pos):
1638 # from jordanbg:
1639 # Given an original string x
1640 # 1. Apply rules once to the input to get y1, y2, y3, etc.
1641 # 2. Return all that are in the database
1642 # 3. If there are no matches, keep applying rules until you either
1643 # find a match or you can't go any further
1644
1645 exceptions = self._exception_map[pos]
1646 substitutions = self.MORPHOLOGICAL_SUBSTITUTIONS[pos]
1647
1648 def apply_rules(forms):
1649 return [form[:-len(old)] + new
1650 for form in forms
1651 for old, new in substitutions
1652 if form.endswith(old)]
1653
1654 def filter_forms(forms):
1655 result = []
1656 seen = set()
1657 for form in forms:
1658 if form in self._lemma_pos_offset_map:
1659 if pos in self._lemma_pos_offset_map[form]:
1660 if form not in seen:
1661 result.append(form)
1662 seen.add(form)
1663 return result
1664
1665 # 0. Check the exception lists
1666 if form in exceptions:
1667 return filter_forms([form] + exceptions[form])
1668
1669 # 1. Apply rules once to the input to get y1, y2, y3, etc.
1670 forms = apply_rules([form])
1671
1672 # 2. Return all that are in the database (and check the original too)
1673 results = filter_forms([form] + forms)
1674 if results:
1675 return results
1676
1677 # 3. If there are no matches, keep applying rules until we find a match
1678 while forms:
1679 forms = apply_rules(forms)
1680 results = filter_forms(forms)
1681 if results:
1682 return results
1683
1684 # Return an empty list if we can't find anything
1685 return []
1686
1687 #////////////////////////////////////////////////////////////
1688 # Create information content from corpus
1689 #////////////////////////////////////////////////////////////
1690[docs] def ic(self, corpus, weight_senses_equally = False, smoothing = 1.0):
1691 """
1692 Creates an information content lookup dictionary from a corpus.
1693
1694 :type corpus: CorpusReader
1695 :param corpus: The corpus from which we create an information
1696 content dictionary.
1697 :type weight_senses_equally: bool
1698 :param weight_senses_equally: If this is True, gives all
1699 possible senses equal weight rather than dividing by the
1700 number of possible senses. (If a word has 3 synses, each
1701 sense gets 0.3333 per appearance when this is False, 1.0 when
1702 it is true.)
1703 :param smoothing: How much do we smooth synset counts (default is 1.0)
1704 :type smoothing: float
1705 :return: An information content dictionary
1706 """
1707 counts = FreqDist()
1708 for ww in corpus.words():
1709 counts[ww] += 1
1710
1711 ic = {}
1712 for pp in POS_LIST:
1713 ic[pp] = defaultdict(float)
1714
1715 # Initialize the counts with the smoothing value
1716 if smoothing > 0.0:
1717 for ss in self.all_synsets():
1718 pos = ss._pos
1719 if pos == ADJ_SAT:
1720 pos = ADJ
1721 ic[pos][ss._offset] = smoothing
1722
1723 for ww in counts:
1724 possible_synsets = self.synsets(ww)
1725 if len(possible_synsets) == 0:
1726 continue
1727
1728 # Distribute weight among possible synsets
1729 weight = float(counts[ww])
1730 if not weight_senses_equally:
1731 weight /= float(len(possible_synsets))
1732
1733 for ss in possible_synsets:
1734 pos = ss._pos
1735 if pos == ADJ_SAT:
1736 pos = ADJ
1737 for level in ss._iter_hypernym_lists():
1738 for hh in level:
1739 ic[pos][hh._offset] += weight
1740 # Add the weight to the root
1741 ic[pos][0] += weight
1742 return ic
1743
1744
1745######################################################################
1746## WordNet Information Content Corpus Reader
1747######################################################################
1748
1749[docs]class WordNetICCorpusReader(CorpusReader):
1750 """
1751 A corpus reader for the WordNet information content corpus.
1752 """
1753
1754 def __init__(self, root, fileids):
1755 CorpusReader.__init__(self, root, fileids, encoding='utf8')
1756
1757 # this load function would be more efficient if the data was pickled
1758 # Note that we can't use NLTK's frequency distributions because
1759 # synsets are overlapping (each instance of a synset also counts
1760 # as an instance of its hypernyms)
1761[docs] def ic(self, icfile):
1762 """
1763 Load an information content file from the wordnet_ic corpus
1764 and return a dictionary. This dictionary has just two keys,
1765 NOUN and VERB, whose values are dictionaries that map from
1766 synsets to information content values.
1767
1768 :type icfile: str
1769 :param icfile: The name of the wordnet_ic file (e.g. "ic-brown.dat")
1770 :return: An information content dictionary
1771 """
1772 ic = {}
1773 ic[NOUN] = defaultdict(float)
1774 ic[VERB] = defaultdict(float)
1775 for num, line in enumerate(self.open(icfile)):
1776 if num == 0: # skip the header
1777 continue
1778 fields = line.split()
1779 offset = int(fields[0][:-1])
1780 value = float(fields[1])
1781 pos = _get_pos(fields[0])
1782 if len(fields) == 3 and fields[2] == "ROOT":
1783 # Store root count.
1784 ic[pos][0] += value
1785 if value != 0:
1786 ic[pos][offset] = value
1787 return ic
1788
1789
1790######################################################################
1791# Similarity metrics
1792######################################################################
1793
1794# TODO: Add in the option to manually add a new root node; this will be
1795# useful for verb similarity as there exist multiple verb taxonomies.
1796
1797# More information about the metrics is available at
1798# http://marimba.d.umn.edu/similarity/measures.html
1799
1800[docs]def path_similarity(synset1, synset2, verbose=False, simulate_root=True):
1801 return synset1.path_similarity(synset2, verbose, simulate_root)
1802
1803path_similarity.__doc__ = Synset.path_similarity.__doc__
1804
1805
1806[docs]def lch_similarity(synset1, synset2, verbose=False, simulate_root=True):
1807 return synset1.lch_similarity(synset2, verbose, simulate_root)
1808
1809lch_similarity.__doc__ = Synset.lch_similarity.__doc__
1810
1811
1812[docs]def wup_similarity(synset1, synset2, verbose=False, simulate_root=True):
1813 return synset1.wup_similarity(synset2, verbose, simulate_root)
1814
1815wup_similarity.__doc__ = Synset.wup_similarity.__doc__
1816
1817
1818[docs]def res_similarity(synset1, synset2, ic, verbose=False):
1819 return synset1.res_similarity(synset2, verbose)
1820
1821res_similarity.__doc__ = Synset.res_similarity.__doc__
1822
1823
1824[docs]def jcn_similarity(synset1, synset2, ic, verbose=False):
1825 return synset1.jcn_similarity(synset2, verbose)
1826
1827jcn_similarity.__doc__ = Synset.jcn_similarity.__doc__
1828
1829
1830[docs]def lin_similarity(synset1, synset2, ic, verbose=False):
1831 return synset1.lin_similarity(synset2, verbose)
1832
1833lin_similarity.__doc__ = Synset.lin_similarity.__doc__
1834
1835
1836def _lcs_ic(synset1, synset2, ic, verbose=False):
1837 """
1838 Get the information content of the least common subsumer that has
1839 the highest information content value. If two nodes have no
1840 explicit common subsumer, assume that they share an artificial
1841 root node that is the hypernym of all explicit roots.
1842
1843 :type synset1: Synset
1844 :param synset1: First input synset.
1845 :type synset2: Synset
1846 :param synset2: Second input synset. Must be the same part of
1847 speech as the first synset.
1848 :type ic: dict
1849 :param ic: an information content object (as returned by ``load_ic()``).
1850 :return: The information content of the two synsets and their most
1851 informative subsumer
1852 """
1853 if synset1._pos != synset2._pos:
1854 raise WordNetError('Computing the least common subsumer requires ' + \
1855 '%s and %s to have the same part of speech.' % \
1856 (synset1, synset2))
1857
1858 ic1 = information_content(synset1, ic)
1859 ic2 = information_content(synset2, ic)
1860 subsumers = synset1.common_hypernyms(synset2)
1861 if len(subsumers) == 0:
1862 subsumer_ic = 0
1863 else:
1864 subsumer_ic = max(information_content(s, ic) for s in subsumers)
1865
1866 if verbose:
1867 print("> LCS Subsumer by content:", subsumer_ic)
1868
1869 return ic1, ic2, subsumer_ic
1870
1871
1872# Utility functions
1873
1874[docs]def information_content(synset, ic):
1875 try:
1876 icpos = ic[synset._pos]
1877 except KeyError:
1878 msg = 'Information content file has no entries for part-of-speech: %s'
1879 raise WordNetError(msg % synset._pos)
1880
1881 counts = icpos[synset._offset]
1882 if counts == 0:
1883 return _INF
1884 else:
1885 return -math.log(counts / icpos[0])
1886
1887
1888# get the part of speech (NOUN or VERB) from the information content record
1889# (each identifier has a 'n' or 'v' suffix)
1890
1891def _get_pos(field):
1892 if field[-1] == 'n':
1893 return NOUN
1894 elif field[-1] == 'v':
1895 return VERB
1896 else:
1897 msg = "Unidentified part of speech in WordNet Information Content file for field %s" % field
1898 raise ValueError(msg)
1899
1900
1901# unload corpus after tests
1902[docs]def teardown_module(module=None):
1903 from nltk.corpus import wordnet
1904 wordnet._unload()
1905
1906
1907######################################################################
1908# Demo
1909######################################################################
1910
1911[docs]def demo():
1912 import nltk
1913 print('loading wordnet')
1914 wn = WordNetCorpusReader(nltk.data.find('corpora/wordnet'), None)
1915 print('done loading')
1916 S = wn.synset
1917 L = wn.lemma
1918
1919 print('getting a synset for go')
1920 move_synset = S('go.v.21')
1921 print(move_synset.name(), move_synset.pos(), move_synset.lexname())
1922 print(move_synset.lemma_names())
1923 print(move_synset.definition())
1924 print(move_synset.examples())
1925
1926 zap_n = ['zap.n.01']
1927 zap_v = ['zap.v.01', 'zap.v.02', 'nuke.v.01', 'microwave.v.01']
1928
1929 def _get_synsets(synset_strings):
1930 return [S(synset) for synset in synset_strings]
1931
1932 zap_n_synsets = _get_synsets(zap_n)
1933 zap_v_synsets = _get_synsets(zap_v)
1934
1935 print(zap_n_synsets)
1936 print(zap_v_synsets)
1937
1938 print("Navigations:")
1939 print(S('travel.v.01').hypernyms())
1940 print(S('travel.v.02').hypernyms())
1941 print(S('travel.v.03').hypernyms())
1942
1943 print(L('zap.v.03.nuke').derivationally_related_forms())
1944 print(L('zap.v.03.atomize').derivationally_related_forms())
1945 print(L('zap.v.03.atomise').derivationally_related_forms())
1946 print(L('zap.v.03.zap').derivationally_related_forms())
1947
1948 print(S('dog.n.01').member_holonyms())
1949 print(S('dog.n.01').part_meronyms())
1950
1951 print(S('breakfast.n.1').hypernyms())
1952 print(S('meal.n.1').hyponyms())
1953 print(S('Austen.n.1').instance_hypernyms())
1954 print(S('composer.n.1').instance_hyponyms())
1955
1956 print(S('faculty.n.2').member_meronyms())
1957 print(S('copilot.n.1').member_holonyms())
1958
1959 print(S('table.n.2').part_meronyms())
1960 print(S('course.n.7').part_holonyms())
1961
1962 print(S('water.n.1').substance_meronyms())
1963 print(S('gin.n.1').substance_holonyms())
1964
1965 print(L('leader.n.1.leader').antonyms())
1966 print(L('increase.v.1.increase').antonyms())
1967
1968 print(S('snore.v.1').entailments())
1969 print(S('heavy.a.1').similar_tos())
1970 print(S('light.a.1').attributes())
1971 print(S('heavy.a.1').attributes())
1972
1973 print(L('English.a.1.English').pertainyms())
1974
1975 print(S('person.n.01').root_hypernyms())
1976 print(S('sail.v.01').root_hypernyms())
1977 print(S('fall.v.12').root_hypernyms())
1978
1979 print(S('person.n.01').lowest_common_hypernyms(S('dog.n.01')))
1980 print(S('woman.n.01').lowest_common_hypernyms(S('girlfriend.n.02')))
1981
1982 print(S('dog.n.01').path_similarity(S('cat.n.01')))
1983 print(S('dog.n.01').lch_similarity(S('cat.n.01')))
1984 print(S('dog.n.01').wup_similarity(S('cat.n.01')))
1985
1986 wnic = WordNetICCorpusReader(nltk.data.find('corpora/wordnet_ic'),
1987 '.*\.dat')
1988 ic = wnic.ic('ic-brown.dat')
1989 print(S('dog.n.01').jcn_similarity(S('cat.n.01'), ic))
1990
1991 ic = wnic.ic('ic-semcor.dat')
1992 print(S('dog.n.01').lin_similarity(S('cat.n.01'), ic))
1993
1994 print(S('code.n.03').topic_domains())
1995 print(S('pukka.a.01').region_domains())
1996 print(S('freaky.a.01').usage_domains())
1997
1998
1999if __name__ == '__main__':
2000 demo()