1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
|
#!/usr/bin/env python3
from bs4 import BeautifulSoup, Comment
import requests
import sys
import re
def lookup(word):
r = requests.get("https://www.oxforddictionaries.com/definition/english/" +
word)
soup = BeautifulSoup(r.text, "html.parser")
entry = "<meta charset=\"utf-8\">"
for res in soup.find_all("section", class_="senseGroup"):
entry += res.find_all("span", class_="partOfSpeech")[0].get_text()
# remove comments
for el in res(text=lambda text: isinstance(text, Comment)):
el.extract()
# remove annyoing link
for el in res.find_all("a", class_="moreInformationSynonyms"):
el.extract()
for sense in res.find_all("div", class_="msDict"):
# prepare example sentences for "more" tag
# The class typo is real.
for el in sense.find_all("a", class_="moreInformationExemples"):
#el.name = "button" # a
el["href"] = "#"
el["onclick"] = \
"var els = this.parentNode.getElementsByTagName('ul')[0].getElementsByTagName('li');" + \
"for(var c = 0; c < els.length; c++){" + \
"if (els[c].getAttribute('data-toggled') == 'false') {" + \
"els[c].style = 'display: none;';" + \
"els[c].setAttribute('data-toggled', 'true');" + \
"} else {" + \
"els[c].style = '';" + \
"els[c].setAttribute('data-toggled', 'false');" + \
"}" + \
"}"
el.string = "Examples…"
el.insert_before(soup.new_tag("br"))
# transform div structure into 'examples'-like list
for el in sense.find_all(class_="entrySynList"):
el.name = "ul"
moretag = soup.new_tag("a")
moretag.string = "Synonyms…"
moretag["href"] = "#"
moretag["protect"] = "" # mark
# important: use the 'ul' at index 1 for synonyms
moretag["onclick"] = \
"var els = this.parentNode.getElementsByTagName('ul')[1].getElementsByTagName('li');" + \
"for(var c = 0; c < els.length; c++){" + \
"if (els[c].getAttribute('data-toggled') == 'false') {" + \
"els[c].style = 'display: none;';" + \
"els[c].setAttribute('data-toggled', 'true');" + \
"} else {" + \
"els[c].style = '';" + \
"els[c].setAttribute('data-toggled', 'false');" + \
"}" + \
"}"
el.parent.a.insert_after(moretag)
moretag.insert_before(soup.new_tag("br"))
for syno in el.find_all("div"):
syno.name = "li"
#el.insert_before(soup.new_tag("br"))
# simplify structure
for el in sense.find_all("span") + \
sense.find_all("div") + \
sense.find_all("a"):
# skip "more" tags
if "onclick" not in el.attrs:
el.unwrap()
# remove unused classes
for el in sense.find_all():
del el["class"]
del sense["class"]
# hide examples
for el in sense.find_all("li"):
el["style"] = "display: none;"
entry += str(sense) #.prettify(formatter="html")
#entry += "<br/>"
entry += "<br/>"
# ugly hack. Why is this necessary:
#- for el in sense.find_all("a", class_="moreInformationExemples"):
#- AttributeError: 'NoneType' object has no attribute 'next_element'
soup = BeautifulSoup(entry, "html.parser")
for el in soup.find_all("li"):
if "Get more examples" == el.get_text() or \
"View synonyms" in el.get_text() or \
el.get_text() == "":
el.extract()
text = str(soup)
# another hack, I'm out of time
text = re.sub(r"(\d)([A-Za-z])", r"\1 \2", text)
return text
#lookup("clerk")
print(lookup("smudge"))
|