diff options
Diffstat (limited to 'inc/3rdparty/site_config/standard/cbsnews.com.txt')
-rw-r--r-- | inc/3rdparty/site_config/standard/cbsnews.com.txt | 14 |
1 files changed, 14 insertions, 0 deletions
diff --git a/inc/3rdparty/site_config/standard/cbsnews.com.txt b/inc/3rdparty/site_config/standard/cbsnews.com.txt new file mode 100644 index 00000000..4ba3da19 --- /dev/null +++ b/inc/3rdparty/site_config/standard/cbsnews.com.txt | |||
@@ -0,0 +1,14 @@ | |||
1 | date: //meta[@name="published"]/@content | ||
2 | date: //div[@class="timeLine"] | ||
3 | title: //div[@id='contentBody']//h1 | ||
4 | author: //dl[@class="storyBlogByline"]/dd/a | ||
5 | body: //div[@id='storyMediaBox'] | //div[contains(@class, 'storyText')] | ||
6 | |||
7 | # Content Pruning | ||
8 | strip: //div[@class="scrollingArrows"] | ||
9 | strip: //div[@class="timeLine"] | ||
10 | strip: //dl[@class="storyBlogByline"] | ||
11 | |||
12 | prune: no | ||
13 | |||
14 | test_url: http://www.cbsnews.com/8301-201_162-57366361/rescued-americans-dad-proud-of-the-u.s/ \ No newline at end of file | ||