1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120
| from nntplib import NNTP, decode_header from urllib.request import urlopen import textwrap import re
class NewsAgent: def __init__(self): self.sources = [] self.destinations = []
def addSource(self, source): self.sources.append(source)
def addDestination(self, dest): self.destinations.append(dest)
def distribute(self): items = [] for source in self.sources: items.extend(source.get_items()) for dest in self.destinations: dest.receive_items(items)
class NewsItem: def __init__(self, title, body): self.title = title self.body = body
class NNTPSource: def __init__(self, servername, group, howmany): self.servername = servername self.group = group self.howmany = howmany
def get_items(self): server = NNTP(self.servername) resp, count, first, last, name = server.group(self.group) start = last - self.howmany + 1 resp, overviews = server.over((start, last)) for id, over in overviews: title = decode_header(over['subject']) resp, info = server.body(id) body = '\n'.join(line.decode('latin') for line in info.lines) + '\n\n' yield NewsItem(title, body) server.quit()
class SimpleWebSource: def __init__(self, url, title_pattern, body_pattern, encoding='utf8'): self.url = url self.title_pattern = re.compile(title_pattern) self.body_pattern = re.compile(body_pattern) self.encoding = encoding
def get_items(self): text = urlopen(self.url).read().decode(self.encoding) titles = self.title_pattern.findall(text) bodies = self.body_pattern.findall(text) for title, body in zip(titles, bodies): yield NewsItem(title, textwrap.fill(body) + '\n')
class PlainDestination: def receive_items(self, items): for item in items: print(item.title) print('-' * len(item.title)) print(item.body)
class HTMLDestination: def __init__(self, filename): self.filename = filename
def receive_items(self, items): out = open(self.filename, 'w') print(""" <html> <head> <title>Today's News</title> <body> <h1> Today's News</h1> """, file=out)
print('<ul>', file=out) id = 0 for item in items: id += 1 print('<li><a href="#{}">{}</a>'.format(id, item.title), file=out) print('</ul>', file=out)
id = 0 for item in items: id += 1 print('<h2><a href="{}"></a></h2>'.format(id, item.title), file=out) print('<pre>{}</pre>'.format(item.body), file=out)
print(""" </body> </html? """, file=out)
def runDefaultSetup(): agent = NewsAgent() reuters_url = '' reuters_title = r'<h4><a href="[^"]*"\s*>(.*?)</a>' reuters_body = r'<div class="item-content">(.*?)</div>' reuters = SimpleWebSource(reuters_url, reuters_title, reuters_body)
agent.addSource(reuters)
clpa_server = 'read80.eternal-september.org' clpa_group = 'eternal-september.software' clpa_howmany = 10 clpa = NNTPSource(clpa_server, clpa_group, clpa_howmany) agent.addSource(clpa)
agent.addDestination(PlainDestination()) agent.addDestination(HTMLDestination('news.html')) agent.distribute()
if __name__ == '__main__': runDefaultSetup()
|
评论· · · · · ·