From 0bdf090ed70ee0c3bf9e42dab0679fe67a1a50f6 Mon Sep 17 00:00:00 2001 From: Venca24 Date: Thu, 22 Nov 2018 13:00:34 +0100 Subject: [PATCH 1/5] [fix] google videos engine --- searx/engines/google_videos.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/searx/engines/google_videos.py b/searx/engines/google_videos.py index 310b3149..53acf517 100644 --- a/searx/engines/google_videos.py +++ b/searx/engines/google_videos.py @@ -25,7 +25,7 @@ time_range_support = True number_of_results = 10 search_url = 'https://www.google.com/search'\ - '?{query}'\ + '?q={query}'\ '&tbm=vid'\ '&{search_options}' time_range_attr = "qdr:{range}" @@ -69,8 +69,8 @@ def response(resp): # parse results for result in dom.xpath('//div[@class="g"]'): - title = extract_text(result.xpath('.//h3/a')) - url = result.xpath('.//h3/a/@href')[0] + title = extract_text(result.xpath('.//h3')) + url = result.xpath('.//div[@class="r"]/a/@href')[0] content = extract_text(result.xpath('.//span[@class="st"]')) # append result From cee15f03755c8e360883918b38e6080c0dce800e Mon Sep 17 00:00:00 2001 From: Venca24 Date: Thu, 22 Nov 2018 13:18:18 +0100 Subject: [PATCH 2/5] [fix] google videos test --- tests/unit/engines/test_google_videos.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/unit/engines/test_google_videos.py b/tests/unit/engines/test_google_videos.py index a48e9a75..dde4e21b 100644 --- a/tests/unit/engines/test_google_videos.py +++ b/tests/unit/engines/test_google_videos.py @@ -30,16 +30,16 @@ class TestGoogleVideosEngine(SearxTestCase):
-
-
-

Title 2

+
Content 2 From cf26aba93b96bb1171feb60fefb232a9113b85b0 Mon Sep 17 00:00:00 2001 From: Venca24 Date: Fri, 4 Jan 2019 15:48:22 +0100 Subject: [PATCH 3/5] [FIX] google videos thumbnails --- searx/engines/google_videos.py | 20 +++++++++++++++++--- 1 file changed, 17 insertions(+), 3 deletions(-) diff --git a/searx/engines/google_videos.py b/searx/engines/google_videos.py index 53acf517..aa850798 100644 --- a/searx/engines/google_videos.py +++ b/searx/engines/google_videos.py @@ -7,15 +7,16 @@ @using-api no @results HTML @stable no - @parse url, title, content + @parse url, title, content, thumbnail """ from datetime import date, timedelta from json import loads from lxml import html +from searx.engines import logger from searx.engines.xpath import extract_text from searx.url_utils import urlencode - +import re # engine dependent config categories = ['videos'] @@ -73,11 +74,24 @@ def response(resp): url = result.xpath('.//div[@class="r"]/a/@href')[0] content = extract_text(result.xpath('.//span[@class="st"]')) + # get thumbnails + script = str(dom.xpath('//script[contains(., "_setImagesSrc")]')[0].text) + id = result.xpath('.//div[@class="s"]//img/@id')[0] + thumbnails_data = re.findall('s=\'(.*?)(?:\\\\[a-z,1-9,\\\\]+\'|\')\;var ii=\[(?:|[\'vidthumb\d+\',]+)\'' + id, + script) + logger.debug('google video engine: ' + id + ' matched ' + str(len(thumbnails_data)) + ' times (thumbnail)') + tmp = [] + if len(thumbnails_data) != 0: + tmp = re.findall('(data:image/jpeg;base64,[a-z,A-Z,0-9,/,\+]+)', thumbnails_data[0]) + thumbnail = '' + if len(tmp) != 0: + thumbnail = tmp[-1] + # append result results.append({'url': url, 'title': title, 'content': content, - 'thumbnail': '', + 'thumbnail': thumbnail, 'template': 'videos.html'}) return results From 0e493db2fb06fd78ac05f82abd8b1cc86089684f Mon Sep 17 00:00:00 2001 From: Venca24 Date: Fri, 4 Jan 2019 16:04:05 +0100 Subject: [PATCH 4/5] [fix] google videos test --- tests/unit/engines/test_google_videos.py | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/tests/unit/engines/test_google_videos.py b/tests/unit/engines/test_google_videos.py index dde4e21b..3b7edf37 100644 --- a/tests/unit/engines/test_google_videos.py +++ b/tests/unit/engines/test_google_videos.py @@ -33,6 +33,15 @@ class TestGoogleVideosEngine(SearxTestCase): +
+ +
Content 1
@@ -41,12 +50,22 @@ class TestGoogleVideosEngine(SearxTestCase): +
+ +
Content 2
+ """ response = mock.Mock(text=html) results = google_videos.response(response) From 2456b8f57199b0479b063fa3dfb16a585c6a40ed Mon Sep 17 00:00:00 2001 From: Venca24 Date: Sat, 5 Jan 2019 12:12:09 +0100 Subject: [PATCH 5/5] [mod] google videos --- searx/engines/google_videos.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/searx/engines/google_videos.py b/searx/engines/google_videos.py index aa850798..9a41b2df 100644 --- a/searx/engines/google_videos.py +++ b/searx/engines/google_videos.py @@ -13,7 +13,6 @@ from datetime import date, timedelta from json import loads from lxml import html -from searx.engines import logger from searx.engines.xpath import extract_text from searx.url_utils import urlencode import re @@ -79,7 +78,6 @@ def response(resp): id = result.xpath('.//div[@class="s"]//img/@id')[0] thumbnails_data = re.findall('s=\'(.*?)(?:\\\\[a-z,1-9,\\\\]+\'|\')\;var ii=\[(?:|[\'vidthumb\d+\',]+)\'' + id, script) - logger.debug('google video engine: ' + id + ' matched ' + str(len(thumbnails_data)) + ' times (thumbnail)') tmp = [] if len(thumbnails_data) != 0: tmp = re.findall('(data:image/jpeg;base64,[a-z,A-Z,0-9,/,\+]+)', thumbnails_data[0])