diff --git a/resources/recipes/soldiers.recipe b/resources/recipes/soldiers.recipe index 173ab49925..fb96e5a2ed 100644 --- a/resources/recipes/soldiers.recipe +++ b/resources/recipes/soldiers.recipe @@ -1,4 +1,3 @@ -#!/usr/bin/env python __license__ = 'GPL v3' __copyright__ = '2009, Darko Miletic ' @@ -16,26 +15,23 @@ class Soldiers(BasicNewsRecipe): max_articles_per_feed = 100 no_stylesheets = True use_embedded_content = False - remove_javascript = True simultaneous_downloads = 1 delay = 4 max_connections = 1 encoding = 'utf-8' publisher = 'U.S. Army' category = 'news, politics, war, weapons' - language = 'en' - + language = 'en' INDEX = 'http://www.army.mil/soldiers/' - html2lrf_options = [ - '--comment', description - , '--category', category - , '--publisher', publisher - ] - - html2epub_options = 'publisher="' + publisher + '"\ncomments="' + description + '"\ntags="' + category + '"' + conversion_options = { + 'comment' : description + , 'tags' : category + , 'publisher' : publisher + , 'language' : language + } - keep_only_tags = [dict(name='div', attrs={'id':'rightCol'})] + keep_only_tags = [dict(name='div', attrs={'id':['storyHeader','textArea']})] remove_tags = [ dict(name='div', attrs={'id':['addThis','comment','articleFooter']}) @@ -44,10 +40,6 @@ class Soldiers(BasicNewsRecipe): feeds = [(u'Frontpage', u'http://www.army.mil/rss/feeds/soldiersfrontpage.xml' )] - def preprocess_html(self, soup): - for item in soup.findAll(style=True): - del item['style'] - return soup def get_cover_url(self): cover_url = None @@ -56,3 +48,4 @@ class Soldiers(BasicNewsRecipe): if cover_item: cover_url = cover_item['src'] return cover_url +