Files
L363-Fortbildung-ABC-CBRN-E…/scratch_scrape.py
T

19 lines
701 B
Python

import urllib.request, re
url = 'https://lernplattform-babz-bund.de/goto.php/cat/155364'
req = urllib.request.Request(url, headers={'Cookie': 'PHPSESSID=130bad4218594384b892580b74197e19'})
try:
html = urllib.request.urlopen(req).read().decode('utf-8')
links = re.findall(r'href=[\'\"]([^\'\"]+)[\'\"]', html)
# Filter for interesting links
modules = set()
for link in links:
if 'lm_id' in link or 'ref_id' in link or 'goto.php' in link or 'sahs' in link:
modules.add(link)
with open('links.txt', 'w', encoding='utf-8') as f:
f.write('\n'.join(modules))
print("Found links:", len(modules))
except Exception as e:
print("Error:", e)