29 lines
1.0 KiB
Python
29 lines
1.0 KiB
Python
import urllib.request
|
|
from html.parser import HTMLParser
|
|
|
|
class MyParser(HTMLParser):
|
|
def __init__(self):
|
|
super().__init__()
|
|
self.links = []
|
|
|
|
def handle_starttag(self, tag, attrs):
|
|
if tag == 'a':
|
|
for attr in attrs:
|
|
if attr[0] == 'href':
|
|
self.links.append({'url': attr[1], 'text': ''})
|
|
|
|
def handle_data(self, data):
|
|
if self.links:
|
|
self.links[-1]['text'] += data.strip() + ' '
|
|
|
|
req = urllib.request.Request('https://lernplattform-babz-bund.de/goto.php/cat/155364', headers={'Cookie': 'PHPSESSID=130bad4218594384b892580b74197e19'})
|
|
html = urllib.request.urlopen(req).read().decode('utf-8')
|
|
|
|
parser = MyParser()
|
|
parser.feed(html)
|
|
|
|
with open('titles.txt', 'w', encoding='utf-8') as f:
|
|
for l in parser.links:
|
|
if ('cat/' in l['url'] or 'sahs' in l['url'] or 'lm' in l['url'] or 'file' in l['url']) and l['text'].strip():
|
|
f.write(l['text'].strip() + " -> " + l['url'] + '\n')
|