Files
L363-Fortbildung-ABC-CBRN-E…/scratch_crawl_babz.py
T

93 lines
2.7 KiB
Python

import urllib.request
import re
import sys
from html.parser import HTMLParser
COOKIE = 'PHPSESSID=130bad4218594384b892580b74197e19'
BASE_URL = 'https://lernplattform-babz-bund.de'
START_CAT = '155364'
visited = set()
categories = set([f'/goto.php/cat/{START_CAT}'])
queue = [f'/goto.php/cat/{START_CAT}']
modules = []
pdfs = []
class IliasParser(HTMLParser):
def __init__(self):
super().__init__()
self.links = []
def handle_starttag(self, tag, attrs):
if tag == 'a':
href = ''
for attr in attrs:
if attr[0] == 'href':
href = attr[1]
if href:
self.links.append({'url': href, 'text': ''})
def handle_data(self, data):
if self.links:
self.links[-1]['text'] += data.strip() + ' '
def fetch(url):
full_url = url if url.startswith('http') else BASE_URL + url
req = urllib.request.Request(full_url, headers={'Cookie': COOKIE})
try:
resp = urllib.request.urlopen(req)
return resp.read().decode('utf-8', errors='ignore')
except Exception as e:
print(f"Failed to fetch {full_url}: {e}")
return ""
def log(msg):
with open('crawl_log.txt', 'a', encoding='utf-8') as f:
f.write(msg + '\n')
log("Starting crawl...")
max_cats = 5000
cats_visited = 0
while queue and cats_visited < max_cats:
url = queue.pop(0)
if url in visited:
continue
visited.add(url)
cats_visited += 1
log(f"Crawling: {url}")
html = fetch(url)
if not html:
continue
parser = IliasParser()
parser.feed(html)
for l in parser.links:
l_url = l['url']
l_text = l['text'].strip()
if not l_text:
continue
if 'goto.php/cat/' in l_url or 'goto.php/crs/' in l_url or 'goto.php/fold/' in l_url:
# subcategory, course, or folder
if l_url not in visited and l_url not in queue:
queue.append(l_url)
elif 'goto.php/sahs/' in l_url or 'goto.php/lm/' in l_url or 'baseClass=ilObjSCORMLearningModuleGUI' in l_url:
modules.append({'url': l_url, 'title': l_text})
log(f" [WBT] {l_text}")
elif 'goto.php/file/' in l_url and 'download' in l_url:
pdfs.append({'url': l_url, 'title': l_text})
log(f" [PDF] {l_text}")
log(f"\nCrawl complete. Visited {cats_visited} categories.")
log(f"Found {len(modules)} Web-Trainings and {len(pdfs)} PDFs.")
with open('crawl_results.txt', 'w', encoding='utf-8') as f:
f.write(f"Modules ({len(modules)}):\n")
for m in modules:
f.write(f"{m['title']} | {m['url']}\n")
f.write(f"\nPDFs ({len(pdfs)}):\n")
for p in pdfs:
f.write(f"{p['title']} | {p['url']}\n")