93 lines
2.7 KiB
Python
93 lines
2.7 KiB
Python
import urllib.request
|
|
import re
|
|
import sys
|
|
from html.parser import HTMLParser
|
|
|
|
COOKIE = 'PHPSESSID=130bad4218594384b892580b74197e19'
|
|
BASE_URL = 'https://lernplattform-babz-bund.de'
|
|
START_CAT = '155364'
|
|
|
|
visited = set()
|
|
categories = set([f'/goto.php/cat/{START_CAT}'])
|
|
queue = [f'/goto.php/cat/{START_CAT}']
|
|
|
|
modules = []
|
|
pdfs = []
|
|
|
|
class IliasParser(HTMLParser):
|
|
def __init__(self):
|
|
super().__init__()
|
|
self.links = []
|
|
def handle_starttag(self, tag, attrs):
|
|
if tag == 'a':
|
|
href = ''
|
|
for attr in attrs:
|
|
if attr[0] == 'href':
|
|
href = attr[1]
|
|
if href:
|
|
self.links.append({'url': href, 'text': ''})
|
|
def handle_data(self, data):
|
|
if self.links:
|
|
self.links[-1]['text'] += data.strip() + ' '
|
|
|
|
def fetch(url):
|
|
full_url = url if url.startswith('http') else BASE_URL + url
|
|
req = urllib.request.Request(full_url, headers={'Cookie': COOKIE})
|
|
try:
|
|
resp = urllib.request.urlopen(req)
|
|
return resp.read().decode('utf-8', errors='ignore')
|
|
except Exception as e:
|
|
print(f"Failed to fetch {full_url}: {e}")
|
|
return ""
|
|
|
|
def log(msg):
|
|
with open('crawl_log.txt', 'a', encoding='utf-8') as f:
|
|
f.write(msg + '\n')
|
|
|
|
log("Starting crawl...")
|
|
max_cats = 5000
|
|
cats_visited = 0
|
|
|
|
while queue and cats_visited < max_cats:
|
|
url = queue.pop(0)
|
|
if url in visited:
|
|
continue
|
|
visited.add(url)
|
|
cats_visited += 1
|
|
|
|
log(f"Crawling: {url}")
|
|
html = fetch(url)
|
|
if not html:
|
|
continue
|
|
|
|
parser = IliasParser()
|
|
parser.feed(html)
|
|
|
|
for l in parser.links:
|
|
l_url = l['url']
|
|
l_text = l['text'].strip()
|
|
if not l_text:
|
|
continue
|
|
|
|
if 'goto.php/cat/' in l_url or 'goto.php/crs/' in l_url or 'goto.php/fold/' in l_url:
|
|
# subcategory, course, or folder
|
|
if l_url not in visited and l_url not in queue:
|
|
queue.append(l_url)
|
|
elif 'goto.php/sahs/' in l_url or 'goto.php/lm/' in l_url or 'baseClass=ilObjSCORMLearningModuleGUI' in l_url:
|
|
modules.append({'url': l_url, 'title': l_text})
|
|
log(f" [WBT] {l_text}")
|
|
elif 'goto.php/file/' in l_url and 'download' in l_url:
|
|
pdfs.append({'url': l_url, 'title': l_text})
|
|
log(f" [PDF] {l_text}")
|
|
|
|
log(f"\nCrawl complete. Visited {cats_visited} categories.")
|
|
log(f"Found {len(modules)} Web-Trainings and {len(pdfs)} PDFs.")
|
|
|
|
with open('crawl_results.txt', 'w', encoding='utf-8') as f:
|
|
f.write(f"Modules ({len(modules)}):\n")
|
|
for m in modules:
|
|
f.write(f"{m['title']} | {m['url']}\n")
|
|
f.write(f"\nPDFs ({len(pdfs)}):\n")
|
|
for p in pdfs:
|
|
f.write(f"{p['title']} | {p['url']}\n")
|