Hi there:
Updates for these three newspapers in spanish from Spain
El Periódico de Aragón
El Correo
Heraldo de Aragón
Regards
desUBIKado
Updates for these three newspapers in spanish from Spain
El Periódico de Aragón
Spoiler:
Code:
#!/usr/bin/env python
# -*- coding: utf-8 -*-
__license__ = 'GPL v3'
__copyright__ = '04 December 2010, desUBIKado'
__author__ = 'desUBIKado'
__description__ = 'Daily newspaper from Aragon'
__version__ = 'v0.10'
__date__ = '09, September 2017'
'''
elperiodicodearagon.com
'''
import re
from calibre.web.feeds.news import BasicNewsRecipe
class elperiodicodearagon(BasicNewsRecipe):
title = u'El Periodico de Aragon'
__author__ = u'desUBIKado'
description = u'Noticias desde Aragon'
publisher = u'elperiodicodearagon.com'
category = u'news, politics, Spain, Aragon'
oldest_article = 1
delay = 1
max_articles_per_feed = 100
no_stylesheets = True
use_embedded_content = False
language = 'es'
masthead_url = 'http://pdf.elperiodicodearagon.com/img/logotipo.gif'
encoding = 'iso-8859-1'
remove_empty_feeds = True
remove_javascript = True
feeds = [
(u'Portada', u'http://zetaestaticos.com/aragon/rss/portada_es.xml'),
(u'Arag\xf3n', u'http://zetaestaticos.com/aragon/rss/2_es.xml'),
(u'Internacional', u'http://zetaestaticos.com/aragon/rss/4_es.xml'),
(u'Espa\xf1a', u'http://zetaestaticos.com/aragon/rss/3_es.xml'),
(u'Econom\xeda', u'http://zetaestaticos.com/aragon/rss/5_es.xml'),
(u'Deportes', u'http://zetaestaticos.com/aragon/rss/7_es.xml'),
(u'Real Zaragoza', u'http://zetaestaticos.com/aragon/rss/10_es.xml'),
(u'Tecnyconta Zaragoza', u'http://zetaestaticos.com/aragon/rss/91_es.xml'),
(u'Monta\xf1ismo', u'http://zetaestaticos.com/aragon/rss/354_es.xml'),
(u'Opini\xf3n', u'http://zetaestaticos.com/aragon/rss/103_es.xml'),
(u'Tema del d\xeda', u'http://zetaestaticos.com/aragon/rss/102_es.xml'),
(u'Escenarios', u'http://zetaestaticos.com/aragon/rss/105_es.xml'),
(u'Sociedad', u'http://zetaestaticos.com/aragon/rss/104_es.xml'),
(u'Gente', u'http://zetaestaticos.com/aragon/rss/330_es.xml'),
(u'Espacio 3', u'http://zetaestaticos.com/aragon/rss/328_es.xml'),
(u'Fiestas del Pilar', u'http://zetaestaticos.com/aragon/rss/107_es.xml'),
(u'Semana Santa', u'http://zetaestaticos.com/aragon/rss/385_es.xml')
,(u'La crónica de Valdejal\xf3n', u'http://zetaestaticos.com/aragon/rss/206_es.xml')
,(u'La crónica de Campo de Borja', u'http://zetaestaticos.com/aragon/rss/208_es.xml')
,(u'La crónica de Ejea y sus pueblos', u'http://zetaestaticos.com/aragon/rss/212_es.xml')
,(u'La crónica del Bajo Gállego', u'http://zetaestaticos.com/aragon/rss/205_es.xml')
,(u'La crónica del Campo de Cariñena', u'http://zetaestaticos.com/aragon/rss/207_es.xml')
,(u'La crónica de la Ribera Alta del Ebro', u'http://zetaestaticos.com/aragon/rss/211_es.xml')
,(u'La crónica del Campo de Belchite', u'http://zetaestaticos.com/aragon/rss/331_es.xml')
]
remove_tags_before = dict(name='div' , attrs={'class':'Pagina'})
remove_tags_after = dict(name='div' , attrs={'class':'ComentariosNew'})
keep_only_tags = [dict(name='div', attrs={'class':'Pagina'})]
remove_tags = [
dict(name='nav', attrs={'class':['Compartir','HerramientasConversacion Herramientas']}),
dict(name='h5', attrs={'class':['CintilloBox']}),
dict(name='div', attrs={'class':['BoxMenu BoxMenuConFoto','BxGalerias','ConStick','HerramientasComentarioNew Herramientas','NumeroComentarioNew']}),
dict(name='div', attrs={'class':['BoxPestanas','Box','ColumnaDerecha','NoticiasRelacionadasDeNoticia','CintilloNoticiasRelacionadasDeNoticia']}),
dict(name='a', attrs={'class':['IrA BotonLink']})
]
# Recuperamos la portada de papel (la imagen format=1 tiene mayor resolucion)
def get_cover_url(self):
index = 'http://pdf.elperiodicodearagon.com/edicion.php'
soup = self.index_to_soup(index)
for image in soup.findAll('img',src=True):
if image['src'].startswith('/funciones/img-public.php?key='):
return 'http://pdf.elperiodicodearagon.com' + image['src']
return None
extra_css = '''
h1 {font-family:Arial,Helvetica,sans-serif; font-weight:bold;font-size:28px;}
h2 {font-family:Arial,Helvetica,sans-serif; font-style:italic;font-size:14px;color:#4D4D4D;}
h3 {font-family:Arial,Helvetica,sans-serif; font-weight:bold;font-size:18px;}
'''El Correo
Spoiler:
Code:
#!/usr/bin/env python
__license__ = 'GPL v3'
__copyright__ = '08 Januery 2011, desUBIKado'
__author__ = 'desUBIKado'
__description__ = 'Daily newspaper from Biscay'
__version__ = 'v0.14'
__date__ = '10, September 2017'
'''
http://www.elcorreo.com/
'''
import time
import re
from calibre.web.feeds.recipes import BasicNewsRecipe
class elcorreo(BasicNewsRecipe):
author = 'desUBIKado'
description = 'Daily newspaper from Biscay'
title = u'El Correo'
publisher = 'Vocento'
category = 'News, politics, culture, economy, general interest'
oldest_article = 1
delay = 1
max_articles_per_feed = 100
no_stylesheets = True
use_embedded_content = False
masthead_url = 'http://www.elcorreo.com/vizcaya/noticias/201002/02/Media/logo-elcorreo-nuevo.png'
language = 'es'
timefmt = '[%a, %d %b, %Y]'
encoding = 'utf-8'
remove_empty_feeds = True
remove_javascript = True
feeds = [
(u'Portada', u'http://www.elcorreo.com/rss/atom/portada'),
(u'Mundo', u'http://www.elcorreo.com/rss/atom/?section=internacional'),
(u'Bizkaia', u'http://www.elcorreo.com/rss/atom/?section=bizkaia'),
(u'Guipuzkoa', u'http://www.elcorreo.com/rss/atom/?section=gipuzkoa'),
(u'Araba', u'http://www.elcorreo.com/rss/atom/?section=araba'),
(u'La Rioja', u'http://www.elcorreo.com/rss/atom/?section=larioja'),
(u'Miranda', u'http://www.elcorreo.com/rss/atom/?section=miranda'),
(u'Economía', u'http://www.elcorreo.com/rss/atom/?section=economia'),
(u'Culturas', u'http://www.elcorreo.com/rss/atom/?section=culturas'),
(u'Politica', u'http://www.elcorreo.com/rss/atom/?section=politica'),
(u'Tecnología', u'http://www.elcorreo.com/rss/atom/?section=tecnologia'),
(u'Gente - Estilo', u'http://www.elcorreo.com/rss/atom/?section=gente-estilo'),
(u'Planes', u'http://www.elcorreo.com/rss/atom/?section=planes'),
(u'Athletic', u'http://www.elcorreo.com/rss/atom/?section=athletic'),
(u'Alavés', u'http://www.elcorreo.com/rss/atom/?section=alaves'),
(u'Bilbao Basket', u'http://www.elcorreo.com/rss/atom/?section=bilbaobasket'),
(u'Baskonia', u'http://www.elcorreo.com/rss/atom/?section=baskonia'),
(u'Deportes', u'http://www.elcorreo.com/rss/atom/?section=deportes'),
(u'Jaiak', u'http://www.elcorreo.com/rss/atom/?section=jaiak'),
(u'La Blanca', u'http://www.elcorreo.com/rss/atom/?section=la-blanca-vitoria'),
(u'Aste Nagusia', u'http://www.elcorreo.com/rss/atom/?section=aste-nagusia-bilbao'),
(u'Semana Santa', u'http://www.elcorreo.com/rss/atom/?section=semana-santa'),
(u'Festivales', u'http://www.elcorreo.com/rss/atom/?section=festivales')
]
keep_only_tags = [
dict(name='div', attrs={'class':['col-xs-12 col-sm-12 col-md-8 col-lg-8']})
]
remove_tags = [
dict(name='div', attrs={'class':['voc-topics voc-detail-grid ','voc-newsletter ','voc-author-social']}),
dict(name='section', attrs={'class':['voc-ficha-detail voc-file-sports']})
]
remove_tags_before = dict(name='div' , attrs={'class':'col-xs-12 col-sm-12 col-md-8 col-lg-8'})
remove_tags_after = dict(name='div' , attrs={'class':'col-xs-12 col-sm-12 col-md-8 col-lg-8'})
_processed_links = []
def get_article_url(self, article):
link = article.get('link', None)
if link is None:
return article
# modificamos la url de las noticias de los equipos deportivos para que funcionen, por ejemplo:
# http://athletic.elcorreo.com/noticias/201407/27/muniain-estrella-athletic-para-20140727093046.html
# http://m.elcorreo.com/noticias/201407/27/muniain-estrella-athletic-para-20140727093046.html?external=deportes/athletic
parte = link.split('/')
if parte[2] == 'athletic.elcorreo.com':
link = 'http://www.elcorreo.com/' + parte[3] + '/' + parte[4] + '/' + parte[5] + '/' + parte[6] + '?external=deportes/athletic'
else:
if parte[2] == 'baskonia.elcorreo.com':
link = 'http://www.elcorreo.com/' + parte[3] + '/' + parte[4] + '/' + parte[5] + '/' + parte[6] + '?external=deportes/baskonia'
else:
if parte[2] == 'bilbaobasket.elcorreo.com':
link = 'http://www.elcorreo.com/' + parte[3] + '/' + parte[4] + '/' + parte[5] + '/' + parte[6] + '?external=deportes/bilbaobasket'
else:
if parte[2] == 'alaves.elcorreo.com':
link = 'http://www.elcorreo.com/' + parte[3] + '/' + parte[4] + '/' + parte[5] + '/' + parte[6] + '?external=deportes/alaves'
# A veces el mismo articulo aparece en la versión de Alava y en la de Bizkaia. Por ejemplo:
# http://www.elcorreo.com/alava/deportes/motor/formula-1/201407/27/ecclestone-quiere-briatore-ayude-20140727140820-rc.html
# http://www.elcorreo.com/bizkaia/deportes/motor/formula-1/201407/27/ecclestone-quiere-briatore-ayude-20140727140820-rc.html
# para controlar los duplicados, unificamos las url para que sean siempre de bizkaia (excepto para la sección "araba")
if ((parte[3] == 'alava') and (parte[4] != 'araba')):
link = link.replace('elcorreo.com/alava', 'elcorreo.com/bizkaia')
# Controlamos si el artículo ha sido incluido en otro feed para eliminarlo
if not (link in self._processed_links):
self._processed_links.append(link)
else:
link = None
return link
# Recuperamos la portada de papel (la imagen format=1 tiene mayor resolucion)
def get_cover_url(self):
cover = None
st = time.localtime()
year = str(st.tm_year)
month = "%.2d" % st.tm_mon
day = "%.2d" % st.tm_mday
#http://info.elcorreo.com/pdf/07082013-viz.pdf
cover='http://info.elcorreo.com/pdf/'+ day + month + year +'-viz.pdf'
br = BasicNewsRecipe.get_browser(self)
try:
br.open(cover)
except:
self.log("\nPortada no disponible")
cover ='http://www.elcorreo.com/vizcaya/noticias/201002/02/Media/logo-elcorreo-nuevo.png'
return cover
# Para cambiar el estilo del texto
extra_css = '''
h1 {font-family:Arial,Helvetica,sans-serif; font-weight:bold;font-size:28px;}
h2 {font-family:georgia,serif; font-style:italic; font-weight:normal;font-size:16px;color:#4D4D4D;}
h3 {font-family:georgia,serif; font-weight:bold;font-size:18px;}
'''
preprocess_regexps = [
# Para presentar la imagen de los video incrustados
(re.compile(r'stillURLVideo: \'', re.DOTALL|re.IGNORECASE), lambda match: '</script><img src="'),
(re.compile(r'.jpg\',', re.DOTALL|re.IGNORECASE), lambda match: '.jpg"><SCRIPT TYPE="text/JavaScript"'),
# Para quitar el punto de la lista
(re.compile(r'<li class="destacada">', re.DOTALL|re.IGNORECASE), lambda match: '<div class="destacada"></div>')
]Heraldo de Aragón
Spoiler:
Code:
#!/usr/bin/env python
__license__ = 'GPL v3'
__copyright__ = '04 December 2010, desUBIKado'
__author__ = 'desUBIKado'
__description__ = 'Daily newspaper from Aragon'
__version__ = 'v0.08'
__date__ = '10, September 2017'
'''
http://www.heraldo.es/
'''
import time
import re
from calibre.web.feeds.news import BasicNewsRecipe
class heraldo(BasicNewsRecipe):
author = 'desUBIKado'
description = 'Daily newspaper from Aragon'
title = u'Heraldo de Aragon'
publisher = 'Grupo Heraldo'
category = 'News, politics, culture, economy, general interest'
language = 'es'
timefmt = '[%a, %d %b, %Y]'
oldest_article = 2
delay = 1
max_articles_per_feed = 100
use_embedded_content = False
masthead_url = 'http://aureel.com/es/wp-content/uploads/sites/4/2016/07/Heraldo_de_Arago%CC%81n.png'
remove_empty_feeds = True
remove_javascript = True
no_stylesheets = True
feeds = [
(u'Noticias', u'http://www.heraldo.es/index.php/mod.portadas/mem.rss')
]
keep_only_tags = [dict(name='div', attrs={'class':['row-f2 brd-row-f4 bck-row-f1-f1 padd-t padd-btt con n-marg-btt']}),
dict(name='div', attrs={'id':['dts','com']}),
dict(name='img', attrs={'class':['lazy']})]
remove_tags = [dict(name='a', attrs={'class':['com flo-r','enl-if','enl-df','next_com']}),
dict(name='div', attrs={'class':['brb-b-s con marg-btt','cnt-rel con','col5-f1','tit txt-wh f-s con','con cont-top ','col5-f1 flo-l','cnt-rel brr','caj_part con','caj_topic con']}),
dict(name='div', attrs={'id':['cont-Top-8760','caj-pub','8760-cpt1','caj_topic con','slider-oferplan','cont-Top-']}),
dict(name='form', attrs={'class':'form'}),
dict(name='ul', attrs={'class':['tabs-nav','men_nav con hg_2n','lst-not-f2 con ']}),
dict(name='span', attrs={'class':['flo-r']}),
dict(name='ul', attrs={'id':['cont-tags','pag-1','pag-cnt-I-']})]
remove_tags_before = dict(name='div' , attrs={'id':'dts'})
remove_tags_after = dict(name='div' , attrs={'id':'com'})
def get_cover_url(self):
cover = None
st = time.localtime()
year = str(st.tm_year)
month = "%.2d" % st.tm_mon
day = "%.2d" % st.tm_mday
#http://img.kiosko.net/2017/09/10/es/heraldo_aragon.750.jpg
cover='http://img.kiosko.net/'+ year +'/'+ month + '/' + day +'/es/heraldo_aragon.750.jpg'
br = BasicNewsRecipe.get_browser(self)
try:
br.open(cover)
except:
self.log("\nPortada no disponible")
cover ='http://aureel.com/es/wp-content/uploads/sites/4/2016/07/Heraldo_de_Arago%CC%81n.png'
return cover
extra_css = '''
h1 {font-family:Arial,Helvetica,sans-serif; font-weight:bold;font-size:28px;}
h2 {font-family:georgia,serif; font-style:italic; font-weight:normal;font-size:22px;color:#4D4D4D;}
.ladillo {font-family:georgia,serif; font-weight:bold;font-size:18px;}
.firm, .sp, .fech, ".com flo-r" {font-family:Arial,Helvetica,sans-serif; font-weight:normal;font-size:12px;}
img{margin-bottom: 0.4em}
'''
preprocess_regexps = [
# Para separar los comentarios con una linea en blanco
(re.compile(r'<div class="tit-f2">', re.DOTALL|re.IGNORECASE), lambda match: '<br /><br /><div class="tit-f2">'),
(re.compile(r'<div id="com"', re.DOTALL|re.IGNORECASE), lambda match: '<br><div id="com"'),
# Para ver las imágenes de las noticias
(re.compile(r'<img class="lazy" data-original="', re.DOTALL|re.IGNORECASE), lambda match: '<img src="http://www.heraldo.es')
]Regards
desUBIKado






