From a247cc5261ff838383f76ffcd44a420caea75f70 Mon Sep 17 00:00:00 2001 From: Jeremy Benoist Date: Wed, 1 Feb 2017 23:15:06 +0100 Subject: Some fixes (#250) * Fix LF / CRLF * Fix some typos --- archive.pressthink.org.txt | 22 +++++++++++----------- computerbase.de.txt | 3 ++- edge-online.com.txt | 2 +- heise.de.txt | 3 ++- huffingtonpost.com.txt | 4 ++-- menshealth.com.txt | 4 ++-- oxfordamerican.com.txt | 18 +++++++++--------- suntimes.com.txt | 2 +- 8 files changed, 30 insertions(+), 28 deletions(-) diff --git a/archive.pressthink.org.txt b/archive.pressthink.org.txt index 828a242..ab973a3 100644 --- a/archive.pressthink.org.txt +++ b/archive.pressthink.org.txt @@ -1,11 +1,11 @@ -# Generated by FiveFilters.org's web-based selection tool -# Place this file inside your site_config/custom/ folder - -title: //h3[contains(concat(' ',normalize-space(@class),' '),' title ')] -body: //div[contains(concat(' ',normalize-space(@class),' '),' blogbody ')] -date: //div[contains(concat(' ',normalize-space(@class),' '),' date ')] - -strip: //h3[contains(concat(' ',normalize-space(@class),' '),' title ')] -strip: //span[contains(concat(' ',normalize-space(@class),' '),' posted ')] - -test_url: http://archive.pressthink.org/2003/09/08/basics_master.html +# Generated by FiveFilters.org's web-based selection tool +# Place this file inside your site_config/custom/ folder + +title: //h3[contains(concat(' ',normalize-space(@class),' '),' title ')] +body: //div[contains(concat(' ',normalize-space(@class),' '),' blogbody ')] +date: //div[contains(concat(' ',normalize-space(@class),' '),' date ')] + +strip: //h3[contains(concat(' ',normalize-space(@class),' '),' title ')] +strip: //span[contains(concat(' ',normalize-space(@class),' '),' posted ')] + +test_url: http://archive.pressthink.org/2003/09/08/basics_master.html diff --git a/computerbase.de.txt b/computerbase.de.txt index c6957c3..55ec48f 100644 --- a/computerbase.de.txt +++ b/computerbase.de.txt @@ -4,7 +4,8 @@ author://span[@class="article-authornames"]/a body: //div[@class='article-view__content'] -replace_string("padding-bottom:): " +# this line breaks the parser +#replace_string("padding-bottom:): " strip://div[@class='adbox-wrapper__label'] diff --git a/edge-online.com.txt b/edge-online.com.txt index cf58581..47b80e8 100644 --- a/edge-online.com.txt +++ b/edge-online.com.txt @@ -1,4 +1,4 @@ -title: //meta[@property="og:title"]/@content +title: //meta[@property="og:title"]/@content body: //h2[@class='strapline'] | //article[contains(@class, 'node-article')] date: //time[@pubdate]/@datetime author: //span[@class='author-name'] diff --git a/heise.de.txt b/heise.de.txt index 5bca8ac..869cd9b 100755 --- a/heise.de.txt +++ b/heise.de.txt @@ -42,7 +42,8 @@ strip_id_or_class: ad_ # Some optimizations replace_string(
):

replace_string(

): -replace_string():
single_page_link: //a[contains(@href, '?view=print')] diff --git a/huffingtonpost.com.txt b/huffingtonpost.com.txt index d4618c1..69b6b10 100644 --- a/huffingtonpost.com.txt +++ b/huffingtonpost.com.txt @@ -1,4 +1,4 @@ -title: //meta[@property="og:title"]/@content +title: //meta[@property="og:title"]/@content body: //div[img[starts-with(@id, 'img_caption')]] | //div[@class="big_photo"] | //div[contains(@class, 'entry_body_text')] date: //meta[@name="publish_date"]/@content author: //a[@rel="author"] @@ -15,7 +15,7 @@ strip_id_or_class: contribute-story strip_id_or_class: promo_holder # end early -replace_string(