mirror of
https://github.com/searxng/searxng.git
synced 2024-11-14 00:30:15 +01:00
542f7d0d7b
In the past, some files were tested with the standard profile, others with a profile in which most of the messages were switched off ... some files were not checked at all. - ``PYLINT_SEARXNG_DISABLE_OPTION`` has been abolished - the distinction ``# lint: pylint`` is no longer necessary - the pylint tasks have been reduced from three to two 1. ./searx/engines -> lint engines with additional builtins 2. ./searx ./searxng_extra ./tests -> lint all other python files Signed-off-by: Markus Heiser <markus.heiser@darmarit.de>
82 lines
2.4 KiB
Python
82 lines
2.4 KiB
Python
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
"""Bandcamp (Music)
|
|
|
|
@website https://bandcamp.com/
|
|
@provide-api no
|
|
@results HTML
|
|
@parse url, title, content, publishedDate, iframe_src, thumbnail
|
|
|
|
"""
|
|
|
|
from urllib.parse import urlencode, urlparse, parse_qs
|
|
from dateutil.parser import parse as dateparse
|
|
from lxml import html
|
|
|
|
from searx.utils import (
|
|
eval_xpath_getindex,
|
|
eval_xpath_list,
|
|
extract_text,
|
|
)
|
|
|
|
# about
|
|
about = {
|
|
"website": 'https://bandcamp.com/',
|
|
"wikidata_id": 'Q545966',
|
|
"official_api_documentation": 'https://bandcamp.com/developer',
|
|
"use_official_api": False,
|
|
"require_api_key": False,
|
|
"results": 'HTML',
|
|
}
|
|
|
|
categories = ['music']
|
|
paging = True
|
|
|
|
base_url = "https://bandcamp.com/"
|
|
search_string = 'search?{query}&page={page}'
|
|
iframe_src = "https://bandcamp.com/EmbeddedPlayer/{type}={result_id}/size=large/bgcol=000/linkcol=fff/artwork=small"
|
|
|
|
|
|
def request(query, params):
|
|
|
|
search_path = search_string.format(query=urlencode({'q': query}), page=params['pageno'])
|
|
params['url'] = base_url + search_path
|
|
return params
|
|
|
|
|
|
def response(resp):
|
|
|
|
results = []
|
|
dom = html.fromstring(resp.text)
|
|
|
|
for result in eval_xpath_list(dom, '//li[contains(@class, "searchresult")]'):
|
|
|
|
link = eval_xpath_getindex(result, './/div[@class="itemurl"]/a', 0, default=None)
|
|
if link is None:
|
|
continue
|
|
|
|
title = result.xpath('.//div[@class="heading"]/a/text()')
|
|
content = result.xpath('.//div[@class="subhead"]/text()')
|
|
new_result = {
|
|
"url": extract_text(link),
|
|
"title": extract_text(title),
|
|
"content": extract_text(content),
|
|
}
|
|
|
|
date = eval_xpath_getindex(result, '//div[@class="released"]/text()', 0, default=None)
|
|
if date:
|
|
new_result["publishedDate"] = dateparse(date.replace("released ", ""))
|
|
|
|
thumbnail = result.xpath('.//div[@class="art"]/img/@src')
|
|
if thumbnail:
|
|
new_result['img_src'] = thumbnail[0]
|
|
|
|
result_id = parse_qs(urlparse(link.get('href')).query)["search_item_id"][0]
|
|
itemtype = extract_text(result.xpath('.//div[@class="itemtype"]')).lower()
|
|
if "album" == itemtype:
|
|
new_result["iframe_src"] = iframe_src.format(type='album', result_id=result_id)
|
|
elif "track" == itemtype:
|
|
new_result["iframe_src"] = iframe_src.format(type='track', result_id=result_id)
|
|
|
|
results.append(new_result)
|
|
return results
|