firecrawl/apps/api/src/scraper/WebScraper/single_url.ts

import * as cheerio from "cheerio";
import { extractMetadata } from "./utils/metadata";
import dotenv from "dotenv";
import {
  Document,
  PageOptions,
  FireEngineResponse,
  ExtractorOptions,
} from "../../lib/entities";
import { parseMarkdown } from "../../lib/html-to-markdown";
import { urlSpecificParams } from "./utils/custom/website_params";
import { fetchAndProcessPdf } from "./utils/pdfProcessor";
import { handleCustomScraping } from "./custom/handleCustomScraping";
import { removeUnwantedElements } from "./utils/removeUnwantedElements";
import { scrapWithFetch } from "./scrapers/fetch";
import { scrapWithFireEngine } from "./scrapers/fireEngine";
import { scrapWithPlaywright } from "./scrapers/playwright";
import { scrapWithScrapingBee } from "./scrapers/scrapingBee";

dotenv.config();

const baseScrapers = [
  "fire-engine",
  "scrapingBee",
  "playwright",
  "scrapingBeeLoad",
  "fetch",
] as const;

export async function generateRequestParams(
  url: string,
  wait_browser: string = "domcontentloaded",
  timeout: number = 15000
): Promise<any> {
  const defaultParams = {
    url: url,
    params: { timeout: timeout, wait_browser: wait_browser },
    headers: { "ScrapingService-Request": "TRUE" },
  };

  try {
    const urlKey = new URL(url).hostname.replace(/^www\./, "");
    if (urlSpecificParams.hasOwnProperty(urlKey)) {
      return { ...defaultParams, ...urlSpecificParams[urlKey] };
    } else {
      return defaultParams;
    }
  } catch (error) {
    console.error(`Error generating URL key: ${error}`);
    return defaultParams;
  }
}

/**
 * Get the order of scrapers to be used for scraping a URL
 * If the user doesn't have envs set for a specific scraper, it will be removed from the order.
 * @param defaultScraper The default scraper to use if the URL does not have a specific scraper order defined
 * @returns The order of scrapers to be used for scraping a URL
 */
function getScrapingFallbackOrder(
  defaultScraper?: string,
  isWaitPresent: boolean = false,
  isScreenshotPresent: boolean = false,
  isHeadersPresent: boolean = false
) {
  const availableScrapers = baseScrapers.filter((scraper) => {
    switch (scraper) {
      case "scrapingBee":
      case "scrapingBeeLoad":
        return !!process.env.SCRAPING_BEE_API_KEY;
      case "fire-engine":
        return !!process.env.FIRE_ENGINE_BETA_URL;
      case "playwright":
        return !!process.env.PLAYWRIGHT_MICROSERVICE_URL;
      default:
        return true;
    }
  });

  let defaultOrder = [
    "scrapingBee",
    "fire-engine",
    "playwright",
    "scrapingBeeLoad",
    "fetch",
  ];

  if (isWaitPresent || isScreenshotPresent || isHeadersPresent) {
    defaultOrder = [
      "fire-engine",
      "playwright",
      ...defaultOrder.filter(
        (scraper) => scraper !== "fire-engine" && scraper !== "playwright"
      ),
    ];
  }

  const filteredDefaultOrder = defaultOrder.filter(
    (scraper: (typeof baseScrapers)[number]) =>
      availableScrapers.includes(scraper)
  );
  const uniqueScrapers = new Set(
    defaultScraper
      ? [defaultScraper, ...filteredDefaultOrder, ...availableScrapers]
      : [...filteredDefaultOrder, ...availableScrapers]
  );

  const scrapersInOrder = Array.from(uniqueScrapers);
  return scrapersInOrder as (typeof baseScrapers)[number][];
}

function extractLinks(html: string, baseUrl: string): string[] {
  const $ = cheerio.load(html);
  const links: string[] = [];

  // Parse the base URL to get the origin
  const urlObject = new URL(baseUrl);
  const origin = urlObject.origin;

  $('a').each((_, element) => {
    const href = $(element).attr('href');
    if (href) {
      if (href.startsWith('http://') || href.startsWith('https://')) {
        // Absolute URL, add as is
        links.push(href);
      } else if (href.startsWith('/')) {
        // Relative URL starting with '/', append to origin
        links.push(`${origin}${href}`);
      } else if (!href.startsWith('#') && !href.startsWith('mailto:')) {
        // Relative URL not starting with '/', append to base URL
        links.push(`${baseUrl}/${href}`);
      } else if (href.startsWith('mailto:')) {
        // mailto: links, add as is
        links.push(href);
      }
      // Fragment-only links (#) are ignored
    }
  });

  // Remove duplicates and return
  return [...new Set(links)];
}

export async function scrapSingleUrl(
  urlToScrap: string,
  pageOptions: PageOptions = {
    onlyMainContent: true,
    includeHtml: false,
    includeRawHtml: false,
    waitFor: 0,
    screenshot: false,
    headers: undefined,
  },
  extractorOptions: ExtractorOptions = {
    mode: "llm-extraction-from-markdown",
  },
  existingHtml: string = ""
): Promise<Document> {
  urlToScrap = urlToScrap.trim();

  const attemptScraping = async (
    url: string,
    method: (typeof baseScrapers)[number]
  ) => {
    let scraperResponse: {
      text: string;
      screenshot: string;
      metadata: { pageStatusCode?: number; pageError?: string | null };
    } = { text: "", screenshot: "", metadata: {} };
    let screenshot = "";
    switch (method) {
      case "fire-engine":
        if (process.env.FIRE_ENGINE_BETA_URL) {
          console.log(`Scraping ${url} with Fire Engine`);
          const response = await scrapWithFireEngine({
            url,
            waitFor: pageOptions.waitFor,
            screenshot: pageOptions.screenshot,
            pageOptions: pageOptions,
            headers: pageOptions.headers,
          });
          scraperResponse.text = response.html;
          scraperResponse.screenshot = response.screenshot;
          scraperResponse.metadata.pageStatusCode = response.pageStatusCode;
          scraperResponse.metadata.pageError = response.pageError;
        }
        break;
      case "scrapingBee":
        if (process.env.SCRAPING_BEE_API_KEY) {
          const response = await scrapWithScrapingBee(
            url,
            "domcontentloaded",
            pageOptions.fallback === false ? 7000 : 15000
          );
          scraperResponse.text = response.content;
          scraperResponse.metadata.pageStatusCode = response.pageStatusCode;
          scraperResponse.metadata.pageError = response.pageError;
        }
        break;
      case "playwright":
        if (process.env.PLAYWRIGHT_MICROSERVICE_URL) {
          const response = await scrapWithPlaywright(
            url,
            pageOptions.waitFor,
            pageOptions.headers
          );
          scraperResponse.text = response.content;
          scraperResponse.metadata.pageStatusCode = response.pageStatusCode;
          scraperResponse.metadata.pageError = response.pageError;
        }
        break;
      case "scrapingBeeLoad":
        if (process.env.SCRAPING_BEE_API_KEY) {
          const response = await scrapWithScrapingBee(url, "networkidle2");
          scraperResponse.text = response.content;
          scraperResponse.metadata.pageStatusCode = response.pageStatusCode;
          scraperResponse.metadata.pageError = response.pageError;
        }
        break;
      case "fetch":
        const response = await scrapWithFetch(url);
        scraperResponse.text = response.content;
        scraperResponse.metadata.pageStatusCode = response.pageStatusCode;
        scraperResponse.metadata.pageError = response.pageError;
        break;
    }

    let customScrapedContent: FireEngineResponse | null = null;

    // Check for custom scraping conditions
    const customScraperResult = await handleCustomScraping(
      scraperResponse.text,
      url
    );

    if (customScraperResult) {
      switch (customScraperResult.scraper) {
        case "fire-engine":
          customScrapedContent = await scrapWithFireEngine({
            url: customScraperResult.url,
            waitFor: customScraperResult.waitAfterLoad,
            screenshot: false,
            pageOptions: customScraperResult.pageOptions,
          });
          if (screenshot) {
            customScrapedContent.screenshot = screenshot;
          }
          break;
        case "pdf":
          const { content, pageStatusCode, pageError } =
            await fetchAndProcessPdf(
              customScraperResult.url,
              pageOptions?.parsePDF
            );
          customScrapedContent = {
            html: content,
            screenshot,
            pageStatusCode,
            pageError,
          };
          break;
      }
    }

    if (customScrapedContent) {
      scraperResponse.text = customScrapedContent.html;
      screenshot = customScrapedContent.screenshot;
    }
    //* TODO: add an optional to return markdown or structured/extracted content
    let cleanedHtml = removeUnwantedElements(scraperResponse.text, pageOptions);
    return {
      text: await parseMarkdown(cleanedHtml),
      html: cleanedHtml,
      rawHtml: scraperResponse.text,
      screenshot: scraperResponse.screenshot,
      pageStatusCode: scraperResponse.metadata.pageStatusCode,
      pageError: scraperResponse.metadata.pageError || undefined,
    };
  };

  let { text, html, rawHtml, screenshot, pageStatusCode, pageError } = {
    text: "",
    html: "",
    rawHtml: "",
    screenshot: "",
    pageStatusCode: 200,
    pageError: undefined,
  };
  try {
    let urlKey = urlToScrap;
    try {
      urlKey = new URL(urlToScrap).hostname.replace(/^www\./, "");
    } catch (error) {
      console.error(`Invalid URL key, trying: ${urlToScrap}`);
    }
    const defaultScraper = urlSpecificParams[urlKey]?.defaultScraper ?? "";
    const scrapersInOrder = getScrapingFallbackOrder(
      defaultScraper,
      pageOptions && pageOptions.waitFor && pageOptions.waitFor > 0,
      pageOptions && pageOptions.screenshot && pageOptions.screenshot === true,
      pageOptions && pageOptions.headers && pageOptions.headers !== undefined
    );

    for (const scraper of scrapersInOrder) {
      // If exists text coming from crawler, use it
      if (existingHtml && existingHtml.trim().length >= 100) {
        let cleanedHtml = removeUnwantedElements(existingHtml, pageOptions);
        text = await parseMarkdown(cleanedHtml);
        html = cleanedHtml;
        break;
      }

      const attempt = await attemptScraping(urlToScrap, scraper);
      text = attempt.text ?? "";
      html = attempt.html ?? "";
      rawHtml = attempt.rawHtml ?? "";
      screenshot = attempt.screenshot ?? "";

      if (attempt.pageStatusCode) {
        pageStatusCode = attempt.pageStatusCode;
      }
      if (attempt.pageError && attempt.pageStatusCode >= 400) {
        pageError = attempt.pageError;
      } else if (attempt && attempt.pageStatusCode && attempt.pageStatusCode < 400) {
        pageError = undefined;
      }

      if (text && text.trim().length >= 100) break;
      if (pageStatusCode && pageStatusCode == 404) break;
      const nextScraperIndex = scrapersInOrder.indexOf(scraper) + 1;
      if (nextScraperIndex < scrapersInOrder.length) {
        console.info(`Falling back to ${scrapersInOrder[nextScraperIndex]}`);
      }
    }

    if (!text) {
      throw new Error(`All scraping methods failed for URL: ${urlToScrap}`);
    }

    const soup = cheerio.load(rawHtml);
    const metadata = extractMetadata(soup, urlToScrap);

    let linksOnPage: string[] | undefined;

    linksOnPage = extractLinks(rawHtml, urlToScrap);

    let document: Document;
    if (screenshot && screenshot.length > 0) {
      document = {
        content: text,
        markdown: text,
        html: pageOptions.includeHtml ? html : undefined,
        rawHtml:
          pageOptions.includeRawHtml ||
            extractorOptions.mode === "llm-extraction-from-raw-html"
            ? rawHtml
            : undefined,
        linksOnPage,
        metadata: {
          ...metadata,
          screenshot: screenshot,
          sourceURL: urlToScrap,
          pageStatusCode: pageStatusCode,
          pageError: pageError,
        },
      };
    } else {
      document = {
        content: text,
        markdown: text,
        html: pageOptions.includeHtml ? html : undefined,
        rawHtml:
          pageOptions.includeRawHtml ||
            extractorOptions.mode === "llm-extraction-from-raw-html"
            ? rawHtml
            : undefined,
        metadata: {
          ...metadata,
          sourceURL: urlToScrap,
          pageStatusCode: pageStatusCode,
          pageError: pageError,
        },
        linksOnPage,
      };
    }

    return document;
  } catch (error) {
    console.error(`Error: ${error} - Failed to fetch URL: ${urlToScrap}`);
    return {
      content: "",
      markdown: "",
      html: "",
      linksOnPage: [],
      metadata: {
        sourceURL: urlToScrap,
        pageStatusCode: pageStatusCode,
        pageError: pageError,
      },
    } as Document;
  }
}
Initial commit 2024-04-15 17:01:47 -04:00			`import * as cheerio from "cheerio";`
			`import { extractMetadata } from "./utils/metadata";`
			`import dotenv from "dotenv";`
Nick: refactor 2024-07-03 18:01:17 -03:00			`import {`
			`Document,`
			`PageOptions,`
			`FireEngineResponse,`
			`ExtractorOptions,`
			`} from "../../lib/entities";`
Initial commit 2024-04-15 17:01:47 -04:00			`import { parseMarkdown } from "../../lib/html-to-markdown";`
Nick: 2024-04-28 11:34:25 -07:00			`import { urlSpecificParams } from "./utils/custom/website_params";`
Added check during scraping to deal with pdfs Checks if the URL is a PDF during the scraping process (single_url.ts). TODO: Run integration tests - Does this strat affect the running time? ps. Some comments need to be removed if we decide to proceed with this strategy. 2024-05-13 09:13:42 -03:00			`import { fetchAndProcessPdf } from "./utils/pdfProcessor";`
Nick: 2024-06-04 12:15:39 -07:00			`import { handleCustomScraping } from "./custom/handleCustomScraping";`
Moved to utils/removeUnwantedElements, added unit tests 2024-06-18 09:46:42 -03:00			`import { removeUnwantedElements } from "./utils/removeUnwantedElements";`
Nick: refactor 2024-07-03 18:01:17 -03:00			`import { scrapWithFetch } from "./scrapers/fetch";`
			`import { scrapWithFireEngine } from "./scrapers/fireEngine";`
			`import { scrapWithPlaywright } from "./scrapers/playwright";`
			`import { scrapWithScrapingBee } from "./scrapers/scrapingBee";`
Initial commit 2024-04-15 17:01:47 -04:00
			`dotenv.config();`

Nick: improvements 2024-05-21 18:34:23 -07:00			`const baseScrapers = [`
			`"fire-engine",`
			`"scrapingBee",`
			`"playwright",`
			`"scrapingBeeLoad",`
			`"fetch",`
			`] as const;`

Nick: 2024-04-28 11:34:25 -07:00			`export async function generateRequestParams(`
			`url: string,`
			`wait_browser: string = "domcontentloaded",`
			`timeout: number = 15000`
			`): Promise<any> {`
			`const defaultParams = {`
			`url: url,`
			`params: { timeout: timeout, wait_browser: wait_browser },`
			`headers: { "ScrapingService-Request": "TRUE" },`
			`};`

Update single_url.ts 2024-04-28 12:44:00 -07:00			`try {`
Nick: a lot better 2024-05-09 17:45:16 -07:00			`const urlKey = new URL(url).hostname.replace(/^www\./, "");`
Update single_url.ts 2024-04-28 12:44:00 -07:00			`if (urlSpecificParams.hasOwnProperty(urlKey)) {`
			`return { ...defaultParams, ...urlSpecificParams[urlKey] };`
			`} else {`
			`return defaultParams;`
			`}`
			`} catch (error) {`
			console.error(`Error generating URL key: ${error}`);
Nick: 2024-04-28 11:34:25 -07:00			`return defaultParams;`
			`}`
			`}`
Update single_url.ts 2024-05-21 18:50:42 -07:00
			`/**`
			`* Get the order of scrapers to be used for scraping a URL`
			`* If the user doesn't have envs set for a specific scraper, it will be removed from the order.`
			`* @param defaultScraper The default scraper to use if the URL does not have a specific scraper order defined`
			`* @returns The order of scrapers to be used for scraping a URL`
			`*/`
Nick: 2024-05-31 15:39:54 -07:00			`function getScrapingFallbackOrder(`
			`defaultScraper?: string,`
			`isWaitPresent: boolean = false,`
			`isScreenshotPresent: boolean = false,`
			`isHeadersPresent: boolean = false`
			`) {`
			`const availableScrapers = baseScrapers.filter((scraper) => {`
Update single_url.ts 2024-05-21 18:50:42 -07:00			`switch (scraper) {`
			`case "scrapingBee":`
			`case "scrapingBeeLoad":`
			`return !!process.env.SCRAPING_BEE_API_KEY;`
			`case "fire-engine":`
			`return !!process.env.FIRE_ENGINE_BETA_URL;`
			`case "playwright":`
			`return !!process.env.PLAYWRIGHT_MICROSERVICE_URL;`
			`default:`
			`return true;`
			`}`
			`});`

Nick: 2024-05-31 15:39:54 -07:00			`let defaultOrder = [`
			`"scrapingBee",`
			`"fire-engine",`
			`"playwright",`
			`"scrapingBeeLoad",`
			`"fetch",`
			`];`

			`if (isWaitPresent \|\| isScreenshotPresent \|\| isHeadersPresent) {`
			`defaultOrder = [`
			`"fire-engine",`
			`"playwright",`
			`...defaultOrder.filter(`
			`(scraper) => scraper !== "fire-engine" && scraper !== "playwright"`
			`),`
			`];`
Nick: 2024-05-28 12:56:24 -07:00			`}`

Nick: 2024-05-31 15:39:54 -07:00			`const filteredDefaultOrder = defaultOrder.filter(`
			`(scraper: (typeof baseScrapers)[number]) =>`
			`availableScrapers.includes(scraper)`
			`);`
			`const uniqueScrapers = new Set(`
			`defaultScraper`
			`? [defaultScraper, ...filteredDefaultOrder, ...availableScrapers]`
			`: [...filteredDefaultOrder, ...availableScrapers]`
			`);`
Merge remote-tracking branch 'origin/main' into test/load-testing 2024-06-14 15:14:01 -03:00
Nick: improvements 2024-05-21 18:34:23 -07:00			`const scrapersInOrder = Array.from(uniqueScrapers);`
Nick: 2024-05-31 15:39:54 -07:00			`return scrapersInOrder as (typeof baseScrapers)[number][];`
Nick: improvements 2024-05-21 18:34:23 -07:00			`}`

Caleb: now extracting and returning a list of all links on the page for a customer 2024-07-16 18:38:03 -07:00			`function extractLinks(html: string, baseUrl: string): string[] {`
			`const $ = cheerio.load(html);`
			`const links: string[] = [];`

			`// Parse the base URL to get the origin`
			`const urlObject = new URL(baseUrl);`
			`const origin = urlObject.origin;`

			`$('a').each((_, element) => {`
			`const href = $(element).attr('href');`
			`if (href) {`
			`if (href.startsWith('http://') \|\| href.startsWith('https://')) {`
			`// Absolute URL, add as is`
			`links.push(href);`
			`} else if (href.startsWith('/')) {`
			`// Relative URL starting with '/', append to origin`
			links.push(`${origin}${href}`);
			`} else if (!href.startsWith('#') && !href.startsWith('mailto:')) {`
			`// Relative URL not starting with '/', append to base URL`
			links.push(`${baseUrl}/${href}`);
			`} else if (href.startsWith('mailto:')) {`
			`// mailto: links, add as is`
			`links.push(href);`
			`}`
			`// Fragment-only links (#) are ignored`
			`}`
			`});`

			`// Remove duplicates and return`
			`return [...new Set(links)];`
			`}`

Initial commit 2024-04-15 17:01:47 -04:00			`export async function scrapSingleUrl(`
			`urlToScrap: string,`
Nick: 2024-05-31 15:39:54 -07:00			`pageOptions: PageOptions = {`
			`onlyMainContent: true,`
			`includeHtml: false,`
update to includeRawHtml 2024-06-28 17:07:47 -04:00			`includeRawHtml: false,`
Nick: 2024-05-31 15:39:54 -07:00			`waitFor: 0,`
			`screenshot: false,`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`headers: undefined,`
Nick: 2024-05-31 15:39:54 -07:00			`},`
init 2024-06-28 16:39:09 -04:00			`extractorOptions: ExtractorOptions = {`
Nick: refactor 2024-07-03 18:01:17 -03:00			`mode: "llm-extraction-from-markdown",`
init 2024-06-28 16:39:09 -04:00			`},`
Nick: fixes 2024-05-15 11:28:20 -07:00			`existingHtml: string = ""`
Initial commit 2024-04-15 17:01:47 -04:00			`): Promise<Document> {`
			`urlToScrap = urlToScrap.trim();`

Nick: 2024-04-16 12:06:46 -04:00			`const attemptScraping = async (`
			`url: string,`
Nick: 2024-05-31 15:39:54 -07:00			`method: (typeof baseScrapers)[number]`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`) => {`
			`let scraperResponse: {`
			`text: string;`
			`screenshot: string;`
			`metadata: { pageStatusCode?: number; pageError?: string \| null };`
			`} = { text: "", screenshot: "", metadata: {} };`
init commit 2024-05-29 18:56:57 -04:00			`let screenshot = "";`
Initial commit 2024-04-15 17:01:47 -04:00			`switch (method) {`
Nick: improvements 2024-05-21 18:34:23 -07:00			`case "fire-engine":`
Update single_url.ts 2024-05-21 18:50:42 -07:00			`if (process.env.FIRE_ENGINE_BETA_URL) {`
Nick: 2024-05-28 12:56:24 -07:00			console.log(`Scraping ${url} with Fire Engine`);
Update single_url.ts 2024-06-28 15:45:16 -03:00			`const response = await scrapWithFireEngine({`
Nick: 2024-05-31 15:39:54 -07:00			`url,`
Update single_url.ts 2024-06-28 15:45:16 -03:00			`waitFor: pageOptions.waitFor,`
			`screenshot: pageOptions.screenshot,`
			`pageOptions: pageOptions,`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`headers: pageOptions.headers,`
			`});`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`scraperResponse.text = response.html;`
			`scraperResponse.screenshot = response.screenshot;`
			`scraperResponse.metadata.pageStatusCode = response.pageStatusCode;`
			`scraperResponse.metadata.pageError = response.pageError;`
Update single_url.ts 2024-05-21 18:50:42 -07:00			`}`
Nick: 2024-04-16 12:06:46 -04:00			`break;`
			`case "scrapingBee":`
Initial commit 2024-04-15 17:01:47 -04:00			`if (process.env.SCRAPING_BEE_API_KEY) {`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`const response = await scrapWithScrapingBee(`
Nick: 2024-04-28 11:34:25 -07:00			`url,`
			`"domcontentloaded",`
			`pageOptions.fallback === false ? 7000 : 15000`
			`);`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`scraperResponse.text = response.content;`
			`scraperResponse.metadata.pageStatusCode = response.pageStatusCode;`
			`scraperResponse.metadata.pageError = response.pageError;`
Initial commit 2024-04-15 17:01:47 -04:00			`}`
			`break;`
Nick: 2024-04-16 12:06:46 -04:00			`case "playwright":`
Initial commit 2024-04-15 17:01:47 -04:00			`if (process.env.PLAYWRIGHT_MICROSERVICE_URL) {`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`const response = await scrapWithPlaywright(`
			`url,`
			`pageOptions.waitFor,`
			`pageOptions.headers`
			`);`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`scraperResponse.text = response.content;`
			`scraperResponse.metadata.pageStatusCode = response.pageStatusCode;`
			`scraperResponse.metadata.pageError = response.pageError;`
Initial commit 2024-04-15 17:01:47 -04:00			`}`
			`break;`
Nick: 2024-04-16 12:06:46 -04:00			`case "scrapingBeeLoad":`
Initial commit 2024-04-15 17:01:47 -04:00			`if (process.env.SCRAPING_BEE_API_KEY) {`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`const response = await scrapWithScrapingBee(url, "networkidle2");`
			`scraperResponse.text = response.content;`
			`scraperResponse.metadata.pageStatusCode = response.pageStatusCode;`
			`scraperResponse.metadata.pageError = response.pageError;`
Initial commit 2024-04-15 17:01:47 -04:00			`}`
			`break;`
Nick: 2024-04-16 12:06:46 -04:00			`case "fetch":`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`const response = await scrapWithFetch(url);`
			`scraperResponse.text = response.content;`
			`scraperResponse.metadata.pageStatusCode = response.pageStatusCode;`
			`scraperResponse.metadata.pageError = response.pageError;`
Initial commit 2024-04-15 17:01:47 -04:00			`break;`
			`}`
Caleb: initially pulled inspiration code from https://github.com/mishushakov/llm-scraper 2024-04-28 13:59:35 -07:00
Update single_url.ts 2024-06-28 15:51:18 -03:00			`let customScrapedContent: FireEngineResponse \| null = null;`
Nick: 2024-06-04 12:15:39 -07:00
Added custom scraping conditions for readme docs 2024-05-29 13:39:43 -03:00			`// Check for custom scraping conditions`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`const customScraperResult = await handleCustomScraping(`
			`scraperResponse.text,`
			`url`
			`);`
Nick: 2024-06-04 12:15:39 -07:00
Update single_url.ts 2024-06-28 15:51:18 -03:00			`if (customScraperResult) {`
[feat] improved the scrape for gdrive pdfs 2024-06-04 17:47:28 -03:00			`switch (customScraperResult.scraper) {`
			`case "fire-engine":`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`customScrapedContent = await scrapWithFireEngine({`
			`url: customScraperResult.url,`
			`waitFor: customScraperResult.waitAfterLoad,`
			`screenshot: false,`
			`pageOptions: customScraperResult.pageOptions,`
			`});`
bugfix screenshot for readme pages 2024-06-05 15:34:42 -03:00			`if (screenshot) {`
			`customScrapedContent.screenshot = screenshot;`
			`}`
missing breaks 2024-06-05 15:02:28 -03:00			`break;`
[feat] improved the scrape for gdrive pdfs 2024-06-04 17:47:28 -03:00			`case "pdf":`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`const { content, pageStatusCode, pageError } =`
			`await fetchAndProcessPdf(`
			`customScraperResult.url,`
			`pageOptions?.parsePDF`
			`);`
			`customScrapedContent = {`
			`html: content,`
			`screenshot,`
			`pageStatusCode,`
			`pageError,`
			`};`
missing breaks 2024-06-05 15:02:28 -03:00			`break;`
[feat] improved the scrape for gdrive pdfs 2024-06-04 17:47:28 -03:00			`}`
Nick: 2024-06-04 12:15:39 -07:00			`}`

Added custom scraping conditions for readme docs 2024-05-29 13:39:43 -03:00			`if (customScrapedContent) {`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`scraperResponse.text = customScrapedContent.html;`
Update single_url.ts 2024-06-03 15:24:40 -03:00			`screenshot = customScrapedContent.screenshot;`
Added custom scraping conditions for readme docs 2024-05-29 13:39:43 -03:00			`}`
Nick: a lot better 2024-05-09 17:45:16 -07:00			`//* TODO: add an optional to return markdown or structured/extracted content`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`let cleanedHtml = removeUnwantedElements(scraperResponse.text, pageOptions);`
			`return {`
			`text: await parseMarkdown(cleanedHtml),`
Fixed includeHTML to use cleanedHtml as response 2024-06-18 16:26:54 -03:00			`html: cleanedHtml,`
Nick: metadata fixes and lock duration for bull decreased to 2 hrs 2024-06-25 15:21:14 -03:00			`rawHtml: scraperResponse.text,`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`screenshot: scraperResponse.screenshot,`
			`pageStatusCode: scraperResponse.metadata.pageStatusCode,`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`pageError: scraperResponse.metadata.pageError \|\| undefined,`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`};`
Initial commit 2024-04-15 17:01:47 -04:00			`};`
Nick: metadata fixes and lock duration for bull decreased to 2 hrs 2024-06-25 15:21:14 -03:00
Update single_url.ts 2024-06-28 15:51:18 -03:00			`let { text, html, rawHtml, screenshot, pageStatusCode, pageError } = {`
			`text: "",`
			`html: "",`
			`rawHtml: "",`
			`screenshot: "",`
			`pageStatusCode: 200,`
			`pageError: undefined,`
			`};`
Initial commit 2024-04-15 17:01:47 -04:00			`try {`
Nick: a lot better 2024-05-09 17:45:16 -07:00			`let urlKey = urlToScrap;`
			`try {`
			`urlKey = new URL(urlToScrap).hostname.replace(/^www\./, "");`
			`} catch (error) {`
			console.error(`Invalid URL key, trying: ${urlToScrap}`);
Nick: mvp 2024-04-23 15:28:32 -07:00			`}`
Nick: a lot better 2024-05-09 17:45:16 -07:00			`const defaultScraper = urlSpecificParams[urlKey]?.defaultScraper ?? "";`
Nick: 2024-05-31 15:39:54 -07:00			`const scrapersInOrder = getScrapingFallbackOrder(`
			`defaultScraper,`
			`pageOptions && pageOptions.waitFor && pageOptions.waitFor > 0,`
			`pageOptions && pageOptions.screenshot && pageOptions.screenshot === true,`
			`pageOptions && pageOptions.headers && pageOptions.headers !== undefined`
			`);`
Nick: a lot better 2024-05-09 17:45:16 -07:00
			`for (const scraper of scrapersInOrder) {`
Nick: 4x speed 2024-05-13 20:45:11 -07:00			`// If exists text coming from crawler, use it`
Nick: fixes 2024-05-15 11:28:20 -07:00			`if (existingHtml && existingHtml.trim().length >= 100) {`
			`let cleanedHtml = removeUnwantedElements(existingHtml, pageOptions);`
			`text = await parseMarkdown(cleanedHtml);`
Fixed includeHTML to use cleanedHtml as response 2024-06-18 16:26:54 -03:00			`html = cleanedHtml;`
Nick: 4x speed 2024-05-13 20:45:11 -07:00			`break;`
			`}`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00
			`const attempt = await attemptScraping(urlToScrap, scraper);`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`text = attempt.text ?? "";`
			`html = attempt.html ?? "";`
			`rawHtml = attempt.rawHtml ?? "";`
			`screenshot = attempt.screenshot ?? "";`
Nick: refactor 2024-07-03 18:01:17 -03:00
fixed edge cases 2024-06-14 09:46:55 -03:00			`if (attempt.pageStatusCode) {`
			`pageStatusCode = attempt.pageStatusCode;`
			`}`
bugfixed pageStatusCode 2024-07-02 10:51:35 -03:00			`if (attempt.pageError && attempt.pageStatusCode >= 400) {`
fixed edge cases 2024-06-14 09:46:55 -03:00			`pageError = attempt.pageError;`
Update single_url.ts 2024-07-03 18:38:17 -03:00			`} else if (attempt && attempt.pageStatusCode && attempt.pageStatusCode < 400) {`
Update single_url.ts 2024-07-01 18:21:15 -03:00			`pageError = undefined;`
fixed edge cases 2024-06-14 09:46:55 -03:00			`}`
Update single_url.ts 2024-06-28 15:51:18 -03:00
Nick: 4x speed 2024-05-13 20:45:11 -07:00			`if (text && text.trim().length >= 100) break;`
fixed edge cases 2024-06-14 09:46:55 -03:00			`if (pageStatusCode && pageStatusCode == 404) break;`
Nick: improvements 2024-05-21 18:34:23 -07:00			`const nextScraperIndex = scrapersInOrder.indexOf(scraper) + 1;`
			`if (nextScraperIndex < scrapersInOrder.length) {`
			console.info(`Falling back to ${scrapersInOrder[nextScraperIndex]}`);
			`}`
Initial commit 2024-04-15 17:01:47 -04:00			`}`

Update single_url.ts 2024-05-09 17:52:46 -07:00			`if (!text) {`
Nick: a lot better 2024-05-09 17:45:16 -07:00			throw new Error(`All scraping methods failed for URL: ${urlToScrap}`);
Initial commit 2024-04-15 17:01:47 -04:00			`}`

Nick: metadata fixes and lock duration for bull decreased to 2 hrs 2024-06-25 15:21:14 -03:00			`const soup = cheerio.load(rawHtml);`
Initial commit 2024-04-15 17:01:47 -04:00			`const metadata = extractMetadata(soup, urlToScrap);`
init commit 2024-05-29 18:56:57 -04:00
Caleb: now extracting and returning a list of all links on the page for a customer 2024-07-16 18:38:03 -07:00			`let linksOnPage: string[] \| undefined;`

			`linksOnPage = extractLinks(rawHtml, urlToScrap);`

init commit 2024-05-29 18:56:57 -04:00			`let document: Document;`
Nick: 2024-05-31 15:39:54 -07:00			`if (screenshot && screenshot.length > 0) {`
init commit 2024-05-29 18:56:57 -04:00			`document = {`
			`content: text,`
			`markdown: text,`
			`html: pageOptions.includeHtml ? html : undefined,`
Nick: refactor 2024-07-03 18:01:17 -03:00			`rawHtml:`
			`pageOptions.includeRawHtml \|\|`
Caleb: now extracting and returning a list of all links on the page for a customer 2024-07-16 18:38:03 -07:00			`extractorOptions.mode === "llm-extraction-from-raw-html"`
Nick: refactor 2024-07-03 18:01:17 -03:00			`? rawHtml`
			`: undefined,`
Caleb: now extracting and returning a list of all links on the page for a customer 2024-07-16 18:38:03 -07:00			`linksOnPage,`
Nick: 2024-05-31 15:39:54 -07:00			`metadata: {`
			`...metadata,`
			`screenshot: screenshot,`
			`sourceURL: urlToScrap,`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`pageStatusCode: pageStatusCode,`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`pageError: pageError,`
Nick: 2024-05-31 15:39:54 -07:00			`},`
			`};`
			`} else {`
init commit 2024-05-29 18:56:57 -04:00			`document = {`
			`content: text,`
			`markdown: text,`
			`html: pageOptions.includeHtml ? html : undefined,`
Nick: refactor 2024-07-03 18:01:17 -03:00			`rawHtml:`
			`pageOptions.includeRawHtml \|\|`
Caleb: now extracting and returning a list of all links on the page for a customer 2024-07-16 18:38:03 -07:00			`extractorOptions.mode === "llm-extraction-from-raw-html"`
Nick: refactor 2024-07-03 18:01:17 -03:00			`? rawHtml`
			`: undefined,`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`metadata: {`
			`...metadata,`
			`sourceURL: urlToScrap,`
			`pageStatusCode: pageStatusCode,`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`pageError: pageError,`
Added metadata.pageStatusCode and metadata.pageError properties to the responses 2024-06-13 17:08:40 -03:00			`},`
Caleb: now extracting and returning a list of all links on the page for a customer 2024-07-16 18:38:03 -07:00			`linksOnPage,`
Nick: 2024-05-31 15:39:54 -07:00			`};`
init commit 2024-05-29 18:56:57 -04:00			`}`
Nick: a lot better 2024-05-09 17:45:16 -07:00
			`return document;`
Initial commit 2024-04-15 17:01:47 -04:00			`} catch (error) {`
			console.error(`Error: ${error} - Failed to fetch URL: ${urlToScrap}`);
			`return {`
			`content: "",`
changed to `includeHtml` 2024-05-06 19:45:56 -03:00			`markdown: "",`
			`html: "",`
Caleb: now extracting and returning a list of all links on the page for a customer 2024-07-16 18:38:03 -07:00			`linksOnPage: [],`
fixed edge cases 2024-06-14 09:46:55 -03:00			`metadata: {`
			`sourceURL: urlToScrap,`
			`pageStatusCode: pageStatusCode,`
Update single_url.ts 2024-06-28 15:51:18 -03:00			`pageError: pageError,`
fixed edge cases 2024-06-14 09:46:55 -03:00			`},`
Initial commit 2024-04-15 17:01:47 -04:00			`} as Document;`
			`}`
			`}`