Files
sisvietnamvn_01/sisvietnamvn_main/parse_html.py
T

20 lines
674 B
Python

from bs4 import BeautifulSoup
with open('/home/x79/sisvietnamvn_01/BV_DHYD_HCM/Đội ngũ bác sĩ UMC.html', 'r', encoding='utf-8') as f:
soup = BeautifulSoup(f, 'html.parser')
# Remove script and style tags
for s in soup(['script', 'style', 'noscript', 'meta', 'link']):
s.extract()
main_content = soup.find('main')
if main_content:
print(f"Main content found! Size: {len(str(main_content))} bytes")
# print first 500 chars
print(str(main_content)[:500])
else:
print("No main tag found. Trying to find #__next or body...")
next_div = soup.find(id='__next')
if next_div:
print(f"Found #__next! Size: {len(str(next_div))} bytes")