20 lines
674 B
Python
20 lines
674 B
Python
from bs4 import BeautifulSoup
|
|
|
|
with open('/home/x79/sisvietnamvn_01/BV_DHYD_HCM/Đội ngũ bác sĩ UMC.html', 'r', encoding='utf-8') as f:
|
|
soup = BeautifulSoup(f, 'html.parser')
|
|
|
|
# Remove script and style tags
|
|
for s in soup(['script', 'style', 'noscript', 'meta', 'link']):
|
|
s.extract()
|
|
|
|
main_content = soup.find('main')
|
|
if main_content:
|
|
print(f"Main content found! Size: {len(str(main_content))} bytes")
|
|
# print first 500 chars
|
|
print(str(main_content)[:500])
|
|
else:
|
|
print("No main tag found. Trying to find #__next or body...")
|
|
next_div = soup.find(id='__next')
|
|
if next_div:
|
|
print(f"Found #__next! Size: {len(str(next_div))} bytes")
|