searxng/searx/engines/btdigg.py

"""
 BTDigg (Videos, Music, Files)

 @website     https://btdigg.org
 @provide-api yes (on demand)

 @using-api   no
 @results     HTML (using search portal)
 @stable      no (HTML can change)
 @parse       url, title, content, seed, leech, magnetlink
"""

from lxml import html
from operator import itemgetter
from searx.engines.xpath import extract_text
from searx.url_utils import quote, urljoin
from searx.utils import get_torrent_size

# engine dependent config
categories = ['videos', 'music', 'files']
paging = True

# search-url
url = 'https://btdigg.org'
search_url = url + '/search?q={search_term}&p={pageno}'


# do search-request
def request(query, params):
    params['url'] = search_url.format(search_term=quote(query),
                                      pageno=params['pageno'] - 1)

    return params


# get response from search-request
def response(resp):
    results = []

    dom = html.fromstring(resp.text)

    search_res = dom.xpath('//div[@id="search_res"]/table/tr')

    # return empty array if nothing is found
    if not search_res:
        return []

    # parse results
    for result in search_res:
        link = result.xpath('.//td[@class="torrent_name"]//a')[0]
        href = urljoin(url, link.attrib.get('href'))
        title = extract_text(link)
        content = extract_text(result.xpath('.//pre[@class="snippet"]')[0])
        content = "<br />".join(content.split("\n"))

        filesize = result.xpath('.//span[@class="attr_val"]/text()')[0].split()[0]
        filesize_multiplier = result.xpath('.//span[@class="attr_val"]/text()')[0].split()[1]
        files = result.xpath('.//span[@class="attr_val"]/text()')[1]
        seed = result.xpath('.//span[@class="attr_val"]/text()')[2]

        # convert seed to int if possible
        if seed.isdigit():
            seed = int(seed)
        else:
            seed = 0

        leech = 0

        # convert filesize to byte if possible
        filesize = get_torrent_size(filesize, filesize_multiplier)

        # convert files to int if possible
        if files.isdigit():
            files = int(files)
        else:
            files = None

        magnetlink = result.xpath('.//td[@class="ttth"]//a')[0].attrib['href']

        # append result
        results.append({'url': href,
                        'title': title,
                        'content': content,
                        'seed': seed,
                        'leech': leech,
                        'filesize': filesize,
                        'files': files,
                        'magnetlink': magnetlink,
                        'template': 'torrent.html'})

    # return results sorted by seeder
    return sorted(results, key=itemgetter('seed'), reverse=True)
update versions.cfg to use the current up-to-date packages 2015-05-02 13:45:17 +00:00			`"""`
			`BTDigg (Videos, Music, Files)`

			`@website https://btdigg.org`
			`@provide-api yes (on demand)`

			`@using-api no`
			`@results HTML (using search portal)`
			`@stable no (HTML can change)`
			`@parse url, title, content, seed, leech, magnetlink`
			`"""`
BTDigg and Mixcloud engines 2015-01-21 17:02:29 +00:00
			`from lxml import html`
			`from operator import itemgetter`
			`from searx.engines.xpath import extract_text`
[enh] py3 compatibility 2016-11-30 17:43:03 +00:00			`from searx.url_utils import quote, urljoin`
add digbt engine Unfortunately, it is quite slow so it is disabled. Furthermore, the display of number of files is wrong on digbt.org, so it is not displayed on searx. 2016-08-13 12:55:47 +00:00			`from searx.utils import get_torrent_size`
BTDigg and Mixcloud engines 2015-01-21 17:02:29 +00:00
			`# engine dependent config`
			`categories = ['videos', 'music', 'files']`
			`paging = True`

			`# search-url`
			`url = 'https://btdigg.org'`
[fix] btdigg 2015-01-25 09:21:44 +00:00			`search_url = url + '/search?q={search_term}&p={pageno}'`
BTDigg and Mixcloud engines 2015-01-21 17:02:29 +00:00

			`# do search-request`
			`def request(query, params):`
			`params['url'] = search_url.format(search_term=quote(query),`
[fix] pep8 compatibilty 2016-01-18 11:47:31 +00:00			`pageno=params['pageno'] - 1)`
BTDigg and Mixcloud engines 2015-01-21 17:02:29 +00:00
			`return params`


			`# get response from search-request`
			`def response(resp):`
			`results = []`

[enh] py3 compatibility 2016-11-30 17:43:03 +00:00			`dom = html.fromstring(resp.text)`
BTDigg and Mixcloud engines 2015-01-21 17:02:29 +00:00
			`search_res = dom.xpath('//div[@id="search_res"]/table/tr')`

			`# return empty array if nothing is found`
			`if not search_res:`
			`return []`

			`# parse results`
			`for result in search_res:`
			`link = result.xpath('.//td[@class="torrent_name"]//a')[0]`
BTDigg's unit test 2015-01-30 18:52:44 +00:00			`href = urljoin(url, link.attrib.get('href'))`
[mod] do not escape html content in engines 2016-12-09 10:44:24 +00:00			`title = extract_text(link)`
			`content = extract_text(result.xpath('.//pre[@class="snippet"]')[0])`
BTDigg and Mixcloud engines 2015-01-21 17:02:29 +00:00			`content = "<br />".join(content.split("\n"))`

			`filesize = result.xpath('.//span[@class="attr_val"]/text()')[0].split()[0]`
			`filesize_multiplier = result.xpath('.//span[@class="attr_val"]/text()')[0].split()[1]`
			`files = result.xpath('.//span[@class="attr_val"]/text()')[1]`
			`seed = result.xpath('.//span[@class="attr_val"]/text()')[2]`

			`# convert seed to int if possible`
			`if seed.isdigit():`
			`seed = int(seed)`
			`else:`
			`seed = 0`

			`leech = 0`

			`# convert filesize to byte if possible`
add digbt engine Unfortunately, it is quite slow so it is disabled. Furthermore, the display of number of files is wrong on digbt.org, so it is not displayed on searx. 2016-08-13 12:55:47 +00:00			`filesize = get_torrent_size(filesize, filesize_multiplier)`
BTDigg and Mixcloud engines 2015-01-21 17:02:29 +00:00
			`# convert files to int if possible`
			`if files.isdigit():`
			`files = int(files)`
			`else:`
			`files = None`

			`magnetlink = result.xpath('.//td[@class="ttth"]//a')[0].attrib['href']`

			`# append result`
			`results.append({'url': href,`
			`'title': title,`
			`'content': content,`
			`'seed': seed,`
			`'leech': leech,`
			`'filesize': filesize,`
			`'files': files,`
			`'magnetlink': magnetlink,`
			`'template': 'torrent.html'})`

			`# return results sorted by seeder`
			`return sorted(results, key=itemgetter('seed'), reverse=True)`