From 4e067ceabd705201a16b4c92cf4b23f3b990326c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Nicolas=20L=C5=93uillet?= Date: Sun, 13 Jul 2014 10:15:40 +0200 Subject: updated specific configuration for parsing --- .../site_config/standard/smithsonianmag.com.txt | 36 +++++++++++----------- 1 file changed, 18 insertions(+), 18 deletions(-) mode change 100644 => 100755 inc/3rdparty/site_config/standard/smithsonianmag.com.txt (limited to 'inc/3rdparty/site_config/standard/smithsonianmag.com.txt') diff --git a/inc/3rdparty/site_config/standard/smithsonianmag.com.txt b/inc/3rdparty/site_config/standard/smithsonianmag.com.txt old mode 100644 new mode 100755 index 10a3f717..3e8fee95 --- a/inc/3rdparty/site_config/standard/smithsonianmag.com.txt +++ b/inc/3rdparty/site_config/standard/smithsonianmag.com.txt @@ -1,20 +1,20 @@ -# meta data -title://h1[@id = 'articleTitle'] -author:substring-after(//ul[@id = 'byLine']/li[1],'By ') -date:substring-before(substring-after(//ul[@id = 'byLine']/li[last()],','),',') -body://div[@id = 'article-body'] - -# full content -single_page_link://td/li[@class = 'article-singlepage']/a - -# caption clean up -wrap_in(i)://span[@class='articleImageCaptionwide'] -move_into (//span[@class='articleImageCaptionwide'])://div[@id = 'articleImage']/p - - -# clean up -strip://p[@id = 'articlePaginationWrapper'] -strip://ul[contains(@class, 'cat-breadcrumb')] -strip://div [@class= 'viewMorePhotos'] +# meta data +title://h1[@id = 'articleTitle'] +author:substring-after(//ul[@id = 'byLine']/li[1],'By ') +date:substring-before(substring-after(//ul[@id = 'byLine']/li[last()],','),',') +body://div[@id = 'article-body'] + +# full content +single_page_link://td/li[@class = 'article-singlepage']/a + +# caption clean up +wrap_in(i)://span[@class='articleImageCaptionwide'] +move_into (//span[@class='articleImageCaptionwide'])://div[@id = 'articleImage']/p + + +# clean up +strip://p[@id = 'articlePaginationWrapper'] +strip://ul[contains(@class, 'cat-breadcrumb')] +strip://div [@class= 'viewMorePhotos'] test_url: http://www.smithsonianmag.com/history-archaeology/The-Goddess-Goes-Home.html \ No newline at end of file -- cgit v1.2.3