import re
import htmlmin
import datetime
import html
from functions import sanitizeUrl
from newspaper import Article, Config
class News():
r = re.compile(r"hours? ago|yesterday|today|last week|month",
flags=re.IGNORECASE)
r_noise = re.compile(r"posted|Product|Google", flags=re.IGNORECASE)
URL = ""
LOWERURL = ""
AUTHORS = []
SCRAP_DATE = None
PUBLISH_DATE = None
SUMMARY = ""
TEXT = ""
HTML = ""
ERROR = False
SOURCE = ""
SOURCENAME = ""
WEIGHT_SUM = 0
MATCHES = False
MATCHING_KW = []
TITLE = ""
CATEGORY = 0
def __init__(self, url, sourceid, sourcename, sourcecategory, kw, lang="en", html=None):
url = sanitizeUrl(url)
self.URL = url
self.LOWERURL = url.lower()
self.SOURCENAME = sourcename
self.SOURCE = sourceid
self.CATEGORY = sourcecategory
self.MATCHING_KW = []
config = Config()
config.language = lang
config.memoize_articles = False
config.keep_article_html = True
config.request_timeout = 15
config.fetch_images = False
config.browser_user_agent = "Mozilla/5.0 (X11; Linux x86_64; rv:47.0) Gecko/20100101 Firefox/47.0"
a = None
try:
a = Article(url=url, config=config)
if html is not None:
a.download(html=html)
else:
a.download()
a.parse()
a.nlp()
except Exception as e:
print("ERROR parsing/downloading/nlp article")
print(e)
print(url)
print(sourcename)
if html is not None:
print(len(html))
self.ERROR = True
self.MATCHES = False
return
a.authors = self.dedup([self.fix_author(
value) for value in a.authors if not self.isComment(value) and not self.isNoise(value)])
self.SCRAP_DATE = datetime.datetime.now()
self.AUTHORS = a.authors
self.PUBLISH_DATE = a.publish_date
self.SUMMARY = a.summary
self.TEXT = a.text
self.TITLE = self.fix_title(a.title)
self.HTML = htmlmin.minify(a.article_html, remove_empty_space=True)
# if html is None:
# self.HTML=htmlmin.minify(a.article_html,remove_empty_space=True)
# else:
# self.HTML=htmlmin.minify(html,remove_empty_space=True)
self.ERROR = False
lowertext = self.TEXT.lower()
ret = match_keywords(kw, lowertext)
self.MATCHES = ret["matches"]
self.WEIGHT_SUM = ret["weight"]
self.MATCHING_KW = ret["kw"]
def get(self):
return {
"url": self.URL,
"lowerurl": self.LOWERURL,
"source": self.SOURCE,
"source_name": self.SOURCENAME,
"authors": self.AUTHORS,
"scrap_date": self.SCRAP_DATE,
"publish_date": self.PUBLISH_DATE,
"summary": self.SUMMARY,
"text": self.TEXT,
"matches": self.MATCHES,
"weight": self.WEIGHT_SUM,
"kw": self.MATCHING_KW,
"html": self.HTML,
"error": self.ERROR,
"deleted": False,
"category": self.CATEGORY,
"title": self.TITLE
}
def fix_title(self, n):
while(html.unescape(n) != n):
n = html.unescape(n)
n = n.replace("&Dquot", '"')
return n
def dedup(self, val):
return list(set(val))
def isNoise(self, v):
if self.r_noise.match(v):
return True
return False
def isComment(self, v):
if self.r.match(v):
return True
return False
def fix_author(self, a):
return a.replace("_", " ").lower()
def match_keywords(keywords, lowertext):
w_sum = 0
matching = []
matches = False
for k in keywords:
num_matches = len([1 for kwp in k["pattern"] if kwp.search(lowertext)])
if num_matches == len(k["pattern"]):
w_sum += k["weight"]
matching.append(k["kw"])
matches = True
ret = ({"matches": matches, "weight": w_sum, "kw": matching})
return ret