summaryrefslogtreecommitdiff
path: root/ankivoc.py
blob: da8adb2c75ff8335e0f1339fe360ead1d66a1dab (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
#!/usr/bin/env python3

from bs4 import BeautifulSoup, Comment
import requests
import sys
import re

def lookup(word):
    r = requests.get("https://www.oxforddictionaries.com/definition/english/" +
                     word)
    soup = BeautifulSoup(r.text, "html.parser")

    entry = "<meta charset=\"utf-8\">"

    for res in soup.find_all("section", class_="senseGroup"):
        entry += res.find_all("span", class_="partOfSpeech")[0].get_text()

        # remove comments
        for el in res(text=lambda text: isinstance(text, Comment)):
            el.extract()

        # remove annyoing link
        for el in res.find_all("a", class_="moreInformationSynonyms"):
            el.extract()

        for sense in res.find_all("div", class_="msDict"):
            # prepare example sentences for "more" tag
            # The class typo is real.
            for el in sense.find_all("a", class_="moreInformationExemples"):
                #el.name = "button"  # a
                el["href"] = "#"
                el["onclick"] = \
                    "var els = this.parentNode.getElementsByTagName('ul')[0].getElementsByTagName('li');" + \
                    "for(var c = 0; c < els.length; c++){" + \
                        "if (els[c].getAttribute('data-toggled') == 'false') {" + \
                            "els[c].style = 'display: none;';" + \
                            "els[c].setAttribute('data-toggled', 'true');" + \
                        "} else {" + \
                            "els[c].style = '';" + \
                            "els[c].setAttribute('data-toggled', 'false');" + \
                        "}" + \
                    "}"
                el.string = "Examples…"
                el.insert_before(soup.new_tag("br"))

            # transform div structure into 'examples'-like list
            for el in sense.find_all(class_="entrySynList"):
                el.name = "ul"
                moretag = soup.new_tag("a")
                moretag.string = "Synonyms…"
                moretag["href"] = "#"
                moretag["protect"] = ""  # mark
                # important: use the 'ul' at index 1 for synonyms
                moretag["onclick"] = \
                    "var els = this.parentNode.getElementsByTagName('ul')[1].getElementsByTagName('li');" + \
                    "for(var c = 0; c < els.length; c++){" + \
                        "if (els[c].getAttribute('data-toggled') == 'false') {" + \
                            "els[c].style = 'display: none;';" + \
                            "els[c].setAttribute('data-toggled', 'true');" + \
                        "} else {" + \
                            "els[c].style = '';" + \
                            "els[c].setAttribute('data-toggled', 'false');" + \
                        "}" + \
                    "}"

                el.parent.a.insert_after(moretag)
                moretag.insert_before(soup.new_tag("br"))

                for syno in el.find_all("div"):
                    syno.name = "li"

                #el.insert_before(soup.new_tag("br"))

            # simplify structure
            for el in sense.find_all("span") + \
                    sense.find_all("div") + \
                    sense.find_all("a"):
                # skip "more" tags
                if "onclick" not in el.attrs:
                    el.unwrap()

            # remove unused classes
            for el in sense.find_all():
                del el["class"]
            del sense["class"]

            # hide examples
            for el in sense.find_all("li"):
                el["style"] = "display: none;"

            entry += str(sense) #.prettify(formatter="html")
            #entry += "<br/>"
        entry += "<br/>"

    # ugly hack. Why is this necessary:
    #- for el in sense.find_all("a", class_="moreInformationExemples"):
    #- AttributeError: 'NoneType' object has no attribute 'next_element'
    soup = BeautifulSoup(entry, "html.parser")
    for el in soup.find_all("li"):
        if "Get more examples" == el.get_text() or \
           "View synonyms" in el.get_text() or \
                el.get_text() == "":
            el.extract()

    text = str(soup)
    # another hack, I'm out of time
    text = re.sub(r"(\d)([A-Za-z])", r"\1 \2", text)
    return text

#lookup("clerk")
print(lookup("smudge"))