INNER CODE UNIT · Python

fetch_article_content

zillionare/zillionare · scripts/enhanced_news_crawler.py:173

    def fetch_article_content(self, article: NewsArticle) -> bool:
        """获取文章正文内容"""
        try:
            logger.info(f"Fetching content for: {article.title}")
            response = self.session.get(article.url)
            response.raise_for_status()

            soup = BeautifulSoup(response.content, "html.parser")

            # 移除脚本和样式标签
            for script in soup(["script", "style", "nav", "footer", "header"]):
                script.decompose()

            # 尝试找到主要内容区域
            content_selectors = [
                "article",
                ".article-content",
                ".content",

View source record →

📰 Research Paper
Loading…
⏳ Fetching content…