Files
L363-Fortbildung-ABC-CBRN-E…/scratch_titles.py
T

29 lines
1.0 KiB
Python

import urllib.request
from html.parser import HTMLParser
class MyParser(HTMLParser):
def __init__(self):
super().__init__()
self.links = []
def handle_starttag(self, tag, attrs):
if tag == 'a':
for attr in attrs:
if attr[0] == 'href':
self.links.append({'url': attr[1], 'text': ''})
def handle_data(self, data):
if self.links:
self.links[-1]['text'] += data.strip() + ' '
req = urllib.request.Request('https://lernplattform-babz-bund.de/goto.php/cat/155364', headers={'Cookie': 'PHPSESSID=130bad4218594384b892580b74197e19'})
html = urllib.request.urlopen(req).read().decode('utf-8')
parser = MyParser()
parser.feed(html)
with open('titles.txt', 'w', encoding='utf-8') as f:
for l in parser.links:
if ('cat/' in l['url'] or 'sahs' in l['url'] or 'lm' in l['url'] or 'file' in l['url']) and l['text'].strip():
f.write(l['text'].strip() + " -> " + l['url'] + '\n')