Separate crawlers
This commit is contained in:
@@ -0,0 +1,20 @@
|
|||||||
|
import requests
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
URL = "https://www.bestattung-wels.at/current-deaths/"
|
||||||
|
page = requests.get(URL)
|
||||||
|
|
||||||
|
soup = BeautifulSoup(page.content, "html.parser")
|
||||||
|
death_table = soup.find("table", attrs={"class": "death-table"})
|
||||||
|
|
||||||
|
death_table_data = death_table.find_all("tr")
|
||||||
|
for row in death_table_data:
|
||||||
|
cols = row.find_all('td')
|
||||||
|
cols = [ele.text.strip() for ele in cols]
|
||||||
|
print(cols) # This will print each row as a list
|
||||||
|
|
||||||
|
#for death in death_table_data:
|
||||||
|
# print(death.full-name.text)
|
||||||
|
|
||||||
|
#results = soup.findAll({"id" : lambda L: L and L.startswith('death-row-')})
|
||||||
|
results = soup.find("death-table-body")
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
import requests
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
URL = "https://www.bestattung-aichinger.at/traueranzeigen--5683197-de.html"
|
||||||
|
page = requests.get(URL)
|
||||||
|
|
||||||
|
soup = BeautifulSoup(page.content, "html.parser")
|
||||||
|
death_table = soup.find("ul", attrs={"class": "homepage_unterseiten_layout_todesanzeigen"})
|
||||||
|
|
||||||
|
death_table_data = death_table.find_all("li")
|
||||||
|
for row in death_table_data:
|
||||||
|
date_span = row.find('span', class_='homepage_unterseiten_layout_datum')
|
||||||
|
date_death = date_span.text.replace('†', '').strip() if date_span else None
|
||||||
|
|
||||||
|
title_span = row.find('span', class_='homepage_unterseiten_layout_titel')
|
||||||
|
|
||||||
|
if title_span:
|
||||||
|
title_text = title_span.get_text(separator='\n')
|
||||||
|
title_parts = title_text.split('\n')
|
||||||
|
title_left = title_parts[0].strip() if len(title_parts) > 0 else None
|
||||||
|
title_right = title_parts[1].strip() if len(title_parts) > 1 else None
|
||||||
|
else:
|
||||||
|
title_left = title_right = None
|
||||||
|
|
||||||
|
print(date_death, title_left, title_right)
|
||||||
|
|
||||||
|
# cols = row.find_all('homepage_unterseiten_layout_datum')
|
||||||
|
# cols = [ele.text.strip() for ele in cols]
|
||||||
|
# print(cols) # This will print each row as a list
|
||||||
|
|
||||||
|
#for death in death_table_data:
|
||||||
|
# print(death.full-name.text)
|
||||||
|
|
||||||
|
#results = soup.findAll({"id" : lambda L: L and L.startswith('death-row-')})
|
||||||
|
results = soup.find("death-table-body")
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
import requests
|
|
||||||
from bs4 import BeautifulSoup
|
|
||||||
|
|
||||||
URL = "https://www.bestattung-wels.at/current-deaths/"
|
|
||||||
page = requests.get(URL)
|
|
||||||
|
|
||||||
soup = BeautifulSoup(page.content, "html.parser")
|
|
||||||
death_table = soup.find("table", attrs={"class": "death-table"})
|
|
||||||
|
|
||||||
death_table_data = death_table.find_all("tr")
|
|
||||||
for row in death_table_data:
|
|
||||||
cols = row.find_all('td')
|
|
||||||
cols = [ele.text.strip() for ele in cols]
|
|
||||||
print(cols) # This will print each row as a list
|
|
||||||
|
|
||||||
#for death in death_table_data:
|
|
||||||
# print(death.full-name.text)
|
|
||||||
|
|
||||||
#results = soup.findAll({"id" : lambda L: L and L.startswith('death-row-')})
|
|
||||||
results = soup.find("death-table-body")
|
|
||||||
Reference in New Issue
Block a user