import fetch from 'node-fetch';
import * as cheerio from 'cheerio';

class WebLoader {
  constructor() {
    this.chunkSize = 1000;
    this.chunkOverlap = 200;
  }

  async loadWebPage(url) {
    try {
      // 发送HTTP请求获取网页内容
      const response = await fetch(url);
      
      const html = await response.text();
      
      // 使用Cheerio解析HTML
      const $ = cheerio.load(html);
      
      // 提取主要内容(移除脚本、样式和导航元素)
      $('script, style, nav, footer, aside').remove();
      
      // 获取文本内容
      let text = $('body').text();
      
      // 清理文本:去除多余空格和换行
      text = text.replace(/\s+/g, ' ').trim();
      
      return text;
    } catch (error) {
      console.error(`加载网页失败: ${url}`, error);
      throw error;
    }
  }

  splitText(text) {
    const chunks = [];
    let start = 0;

    // 将文本分割成指定大小的块,带有重叠
    while (start < text.length) {
      let end = start + this.chunkSize;
      if (end > text.length) {
        end = text.length;
      }
      chunks.push(text.slice(start, end));
      start += this.chunkSize - this.chunkOverlap;
    }

    return chunks;
  }

  async processWebPage(url) {
    // 加载网页
    const text = await this.loadWebPage(url);
    
    // 分割文本
    const chunks = this.splitText(text);
    
    // 生成文档对象
    const documents = chunks.map((chunk, index) => {
      const id = `${new URL(url).hostname}_chunk_${index}`;
      return {
        id,
        text: chunk,
        metadata: {
          source: url,
          chunkIndex: index,
          totalChunks: chunks.length
        }
      };
    });

    return documents;
  }

  async processWebPages(urls) {
    let allDocuments = [];
    
    for (const url of urls) {
      const documents = await this.processWebPage(url);
      allDocuments = allDocuments.concat(documents);
    }
    
    return allDocuments;
  }
}

export default WebLoader;