
    Sj                    d   U d Z ddlZddlZddlZddlmZmZmZmZm	Z	m
Z
mZmZmZ ddlZddlmZ ddlZddlZddlZddlZddlZddlZddlZddlmZ d Z e       Z ej4                  d      Zej8                  ed<    ed	      Z G d
 dej>                        Z  G d dej>                        Z! G d dej>                        Z" G d dej>                        Z# G d dej>                  ee         Z$ G d dej>                        Z% G d dej>                        Z& G d dej>                        Z' G d dej>                        Z( G d dej>                        Z) G d dej>                        Z* G d  d!ej>                        Z+ G d" d#ej>                        Z, G d$ d%ej>                        Z- G d& d'ej>                        Z. G d( d)ej>                        Z/ G d* d+ej>                        Z0 G d, d-ej>                        Z1 G d. d/ej>                        Z2 G d0 d1ej>                        Z3 G d2 d3e(      Z4 G d4 d5e$e   ee         Z5 G d6 d7ej>                        Z6 G d8 d9ej>                        Z7 G d: d;ej>                        Z8 G d< d=ej>                        Z9 G d> d?ej>                        Z: G d@ dAej>                        Z; G dB dCej>                        Z< G dD dEej>                        Z= G dF dGej>                        Z> G dH dIej>                  ee         Z? G dJ dKej>                        Z@ G dL dMej>                        ZA G dN dOej>                        ZB G dP dQej>                        ZC G dR dSej>                        ZD G dT dUej>                        ZE G dV dWej>                        ZF G dX dYej>                        ZG G dZ d[ej>                        ZH G d\ dMej>                        ZA G d] dGej>                        Z> G d^ d_      ZI G d` da      ZJ G db dceI      ZK G dd deeJ      ZLy)fa  
FirecrawlApp Module

This module provides a class `FirecrawlApp` for interacting with the Firecrawl API.
It includes methods to scrape URLs, perform searches, initiate and monitor crawl jobs,
and check the status of these jobs. The module uses requests for HTTP communication
and handles retries for certain HTTP status codes.

Classes:
    - FirecrawlApp: Main class for interacting with the Firecrawl API.
    N)	AnyDictOptionalListUnionCallableLiteralTypeVarGeneric)datetime)Fieldc                     	 ddl m}  t        j                  j	                  t
              } | t        j                  j                  |d            j                         }t        j                  d|t        j                        }|r|j                  d      j                         S y # t        $ r t        d       Y y w xY w)Nr   )Pathz__init__.pyz"^__version__ = ['\"]([^'\"]*)['\"]   z&Failed to get version from __init__.py)pathlibr   ospathdirname__file__join	read_textresearchMgroupstrip	Exceptionprint)r   package_pathversion_fileversion_matchs       M/root/.hermes/venv/lib/python3.12/site-packages/firecrawl/firecrawl.backup.pyget_versionr#      s    	WW__X.l"'',,|]CDNNPlii E|UWUYUYZm	$$Q'--/
/ 
	 45s   B"B& &B=<B=	firecrawlloggerTc                   :    e Zd ZU dZdZed   ed<   dZee	   ed<   y)AgentOptionszConfiguration for the agent.FIRE-1modelNprompt)
__name__
__module____qualname____doc__r*   r	   __annotations__r+   r   str     r"   r(   r(   Q   s"    &'E78' FHSM r3   r(   c                   &    e Zd ZU dZdZed   ed<   y)AgentOptionsExtract2Configuration for the agent in extract operations.r)   r*   Nr,   r-   r.   r/   r*   r	   r0   r2   r3   r"   r5   r5   V       <'E78'r3   r5   c                   2    e Zd ZU dZee   ed<   ee   ed<   y)ActionsResultz,Result of actions performed during scraping.screenshotspdfsN)r,   r-   r.   r/   r   r1   r0   r2   r3   r"   r:   r:   Z   s    6c
s)Or3   r:   c                       e Zd ZU dZdZee   ed<   eed<   eed<   dZee	ee
f      ed<    ej                  dd      Zee
   ed	<   y)
ChangeTrackingDataz.
    Data for the change tracking format.
    NpreviousScrapeAtchangeStatus
visibilitydiffjsonalias
json_field)r,   r-   r.   r/   r?   r   r1   r0   rB   r   r   pydanticr   rF   r2   r3   r"   r>   r>   _   sU     '+hsm*O%)D(4S>
") .t6 BJBr3   r>   c                   @   e Zd ZU dZdZee   ed<   dZee   ed<   dZ	ee   ed<   dZ
ee   ed<   dZeee      ed<   dZee   ed<    ej                   dd	
      Zee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   y)FirecrawlDocumentz-Document retrieved or processed by Firecrawl.NurlmarkdownhtmlrawHtmllinksextractrC   rD   rF   
screenshotmetadataactionstitledescriptionchangeTracking)r,   r-   r.   r/   rJ   r   r1   r0   rK   rL   rM   rN   r   rO   r&   rG   r   rF   rP   rQ   r   rR   r:   rS   rT   rU   r>   r2   r3   r"   rI   rI   i   s    7C#"Hhsm"D(3-!GXc]!!%E8DI%GXa[,hnnT@J@ $J$"Hhsm"'+GXm$+E8C=!%K#%37NH/07r3   rI   c                   @    e Zd ZU dZdZee   ed<   dZee	e      ed<   y)LocationConfigz$Location configuration for scraping.Ncountry	languages)
r,   r-   r.   r/   rX   r   r1   r0   rY   r   r2   r3   r"   rW   rW   y   s&    .!GXc]!%)IxS	")r3   rW   c                   x    e Zd ZU dZeed<   dZeeeef      ed<   dZ	eeeef      ed<   dZ
eeed         ed<   y)WebhookConfigzConfiguration for webhooks.rJ   NheadersrQ   )	completedfailedpagestartedevents)r,   r-   r.   r/   r1   r0   r\   r   r   rQ   ra   r   r	   r2   r3   r"   r[   r[   ~   sR    %	H(,GXd38n%,)-HhtCH~&-PTFHT'"JKLMTr3   r[   c                       e Zd ZU dZdZeeed         ed<    e	j                  dd      Zee   ed<   dZee   ed<   dZee   ed	<   y)
ChangeTrackingOptionsz"Configuration for change tracking.N)zgit-diffrC   modesschemarD   schema_fieldr+   tag)r,   r-   r.   r/   rd   r   r   r	   r0   rG   r   rf   r   r+   r1   rg   r2   r3   r"   rc   rc      sV    ,9=E8D!3456="0(..X"FL(3-F FHSM C#r3   rc   c                      e Zd ZU dZdZeeed         ed<   dZ	ee
eef      ed<   dZeee      ed<   dZeee      ed<   dZee   ed<   dZee   ed	<   d
Zee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZeed      ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   y)ScrapeOptions#Parameters for scraping operations.N
rK   rL   rM   contentrN   rP   screenshot@fullPagerO   rC   rU   formatsr\   includeTagsexcludeTagsonlyMainContentwaitFor0u  timeoutlocationmobileskipTlsVerificationremoveBase64ImagesblockAdsbasicstealthautoproxychangeTrackingOptionsmaxAgestoreInCacheparsePDF)r,   r-   r.   r/   rn   r   r   r	   r0   r\   r   r1   ro   rp   rq   boolrr   intrt   ru   rW   rv   rw   rx   ry   r~   r   rc   r   r   r   r2   r3   r"   ri   ri      s2   - eiGXd7  $_  `  a  b  i(,GXd38n%,'+K$s)$+'+K$s)$+&*OXd^*!GXc]!"GXc]")-Hh~&-!FHTN!*.$.)--#Hhtn#;?E8G678?=A8$9:A FHSM #'L(4.'#Hhtn#r3   ri   c                   J    e Zd ZU dZed   ed<   dZee   ed<   dZ	ee
   ed<   y)
WaitActionz'Wait action to perform during scraping.waittypeNmillisecondsselector)r,   r-   r.   r/   r	   r0   r   r   r   r   r1   r2   r3   r"   r   r      s+    1
&/"&L(3-&"Hhsm"r3   r   c                   J    e Zd ZU dZed   ed<   dZee   ed<   dZ	ee
   ed<   y)ScreenshotActionz-Screenshot action to perform during scraping.rP   r   NfullPagequality)r,   r-   r.   r/   r	   r0   r   r   r   r   r   r2   r3   r"   r   r      s,    7
,
#Hhtn#!GXc]!r3   r   c                   ,    e Zd ZU dZed   ed<   eed<   y)ClickActionz(Click action to perform during scraping.clickr   r   Nr,   r-   r.   r/   r	   r0   r1   r2   r3   r"   r   r      s    2
'
Mr3   r   c                   ,    e Zd ZU dZed   ed<   eed<   y)WriteActionz(Write action to perform during scraping.writer   textNr   r2   r3   r"   r   r      s    2
'

Ir3   r   c                   ,    e Zd ZU dZed   ed<   eed<   y)PressActionz(Press action to perform during scraping.pressr   keyNr   r2   r3   r"   r   r      s    2
'
	Hr3   r   c                   F    e Zd ZU dZed   ed<   ed   ed<   dZee   ed<   y)ScrollActionz)Scroll action to perform during scraping.scrollr   )updown	directionNr   )	r,   r-   r.   r/   r	   r0   r   r   r1   r2   r3   r"   r   r      s)    3
(
|$$"Hhsm"r3   r   c                   "    e Zd ZU dZed   ed<   y)ScrapeActionz)Scrape action to perform during scraping.scraper   N)r,   r-   r.   r/   r	   r0   r2   r3   r"   r   r      s    3
(
r3   r   c                   ,    e Zd ZU dZed   ed<   eed<   y)ExecuteJavascriptActionz5Execute javascript action to perform during scraping.executeJavascriptr   scriptNr   r2   r3   r"   r   r      s    ?
%
&&Kr3   r   c                   d    e Zd ZU dZed   ed<   dZeed      ed<   dZee	   ed<   dZ
ee   ed<   y)		PDFActionz&PDF action to perform during scraping.pdfr   N)A0A1A2A3A4A5A6LetterLegalTabloidLedgerformat	landscapescale)r,   r-   r.   r/   r	   r0   r   r   r   r   r   floatr2   r3   r"   r   r      s@    0
%.rvFHWmnov $Ix~$!E8E?!r3   r   c                   &    e Zd ZU dZdZed   ed<   y)ExtractAgentr6   r)   r*   Nr7   r2   r3   r"   r   r      r8   r3   r   c                       e Zd ZU dZdZee   ed<    ej                  dd      Z
ee   ed<   dZee   ed<   dZee   ed<   y)	
JsonConfigzConfiguration for extraction.Nr+   re   rD   rf   systemPromptagent)r,   r-   r.   r/   r+   r   r1   r0   rG   r   rf   r   r   r   r   r2   r3   r"   r   r      sK    ' FHSM "0(..X"FL(3-F"&L(3-&$(E8L!(r3   r   c                       e Zd ZU dZdZee   ed<   dZee   ed<   dZ	ee
eeeeeeeeeef	         ed<   dZee   ed<   dZee   ed<   y)ScrapeParamsrj   NrO   jsonOptionsrR   r   webhook)r,   r-   r.   r/   rO   r   r   r0   r   rR   r   r   r   r   r   r   r   r   r   r   r   r   r(   r   r[   r2   r3   r"   r   r      s    -$(GXj!((,K*%, koGXd5-={KYdfr  uA  CZ  \e  "e  f  g  h  o$(E8L!('+GXm$+r3   r   c                   H    e Zd ZU dZdZeed<   dZee	   ed<   dZ
ee	   ed<   y)ScrapeResponsez"Response from scraping operations.TsuccessNwarningerror)r,   r-   r.   r/   r   r   r0   r   r   r1   r   r2   r3   r"   r   r      s+    ,GT!GXc]!E8C=r3   r   c                   v    e Zd ZU dZdZee   ed<   dZee   ed<   dZ	e
ed<   dZee   ed<   dZeee      ed<   y)	BatchScrapeResponsez&Response from batch scrape operations.NidrJ   Tr   r   invalidURLs)r,   r-   r.   r/   r   r   r1   r0   rJ   r   r   r   r   r   r2   r3   r"   r   r      sL    0BC#GTE8C='+K$s)$+r3   r   c                   |    e Zd ZU dZdZeed<   ed   ed<   eed<   eed<   eed<   e	ed	<   d
Z
ee   ed<   ee   ed<   y
)BatchScrapeStatusResponsez)Response from batch scrape status checks.Tr   scrapingr]   r^   	cancelledstatusr]   totalcreditsUsed	expiresAtNnextdatar,   r-   r.   r/   r   r   r0   r	   r   r   r   r   r1   r   rI   r2   r3   r"   r   r      sK    3GTBCCNJD(3-
 
!!r3   r   c                   |   e Zd ZU dZdZeee      ed<   dZ	eee      ed<   dZ
ee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed	<   dZee   ed
<   dZee   ed<   dZee   ed<   dZeeeef      ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   y)CrawlParamsz#Parameters for crawling operations.NincludePathsexcludePathsmaxDepthmaxDiscoveryDepthlimitallowBackwardLinkscrawlEntireDomainallowExternalLinksignoreSitemapscrapeOptionsr   deduplicateSimilarURLsignoreQueryParametersregexOnFullURLdelaymaxConcurrencyallowSubdomains)r,   r-   r.   r/   r   r   r   r1   r0   r   r   r   r   r   r   r   r   r   r   r   ri   r   r   r[   r   r   r   r   r   r   r2   r3   r"   r   r      s   -(,L(49%,(,L(49%,"Hhsm"'+x}+E8C=)--(,x~,)--$(M8D>(-1M8M*137GXeC./07-1HTN1,08D>0%)NHTN)E8C=$(NHSM(&*OXd^*r3   r   c                   \    e Zd ZU dZdZee   ed<   dZee   ed<   dZ	e
ed<   dZee   ed<   y)CrawlResponsez"Response from crawling operations.Nr   rJ   Tr   r   )r,   r-   r.   r/   r   r   r1   r0   rJ   r   r   r   r2   r3   r"   r   r     s9    ,BC#GTE8C=r3   r   c                   |    e Zd ZU dZdZeed<   ed   ed<   eed<   eed<   eed<   e	ed	<   d
Z
ee   ed<   ee   ed<   y
)CrawlStatusResponsez"Response from crawl status checks.Tr   r   r   r]   r   r   r   Nr   r   r   r2   r3   r"   r   r     sK    ,GTBCCNJD(3-
 
!!r3   r   c                   <    e Zd ZU dZeeeef      ed<   ee   ed<   y)CrawlErrorsResponsez2Response from crawl/batch scrape error monitoring.errorsrobotsBlockedN)r,   r-   r.   r/   r   r   r1   r0   r2   r3   r"   r   r   #  s"    <c3h  9r3   r   c                       e Zd ZU dZdZee   ed<   dZee	   ed<   dZ
ee	   ed<   dZee	   ed<   dZee   ed<   dZee   ed	<   dZee	   ed
<   y)	MapParamsz"Parameters for mapping operations.Nr   r   includeSubdomainssitemapOnlyr   rs   rt   useIndex)r,   r-   r.   r/   r   r   r1   r0   r   r   r   r   r   r   rt   r   r2   r3   r"   r   r   (  sh    , FHSM $(M8D>((,x~,"&K$&E8C="GXc]"#Hhtn#r3   r   c                   N    e Zd ZU dZdZeed<   dZee	e
      ed<   dZee
   ed<   y)MapResponsez!Response from mapping operations.Tr   NrN   r   )r,   r-   r.   r/   r   r   r0   rN   r   r   r1   r   r2   r3   r"   r   r   2  s0    +GT!%E8DI%E8C=r3   r   c                       e Zd ZU dZdZee   ed<    ej                  dd      Z
ee   ed<   dZee   ed<   dZee   ed<   dZee   ed	<   dZee   ed
<   dZee   ed<   dZee   ed<   dZee   ed<   y)ExtractParamsz0Parameters for extracting information from URLs.Nr+   re   rD   rf   r   r   enableWebSearchr   originshowSourcesr   )r,   r-   r.   r/   r+   r   r1   r0   rG   r   rf   r   r   r   r   r   r   r   r   r   ri   r2   r3   r"   r   r   8  s    : FHSM "0(..X"FL(3-F"&L(3-&)--&*OXd^*(,x~, FHSM "&K$&-1M8M*1r3   r   c                       e Zd ZU dZdZee   ed<   dZee	d      ed<   dZ
ee   ed<   dZeed<   dZee   ed	<   dZee   ed
<   dZee   ed<   dZeeeef      ed<   y)ExtractResponsez!Response from extract operations.Nr   
processingr]   r^   r   r   Tr   r   r   r   sources)r,   r-   r.   r/   r   r   r1   r0   r   r	   r   r   r   r   r   r&   r   r   r  r   r   r2   r3   r"   r   r   D  s    +BEIFHW@ABI$(Ix!(GTD(1+E8C=!GXc]!(,GXd38n%,r3   r   c                       e Zd ZU eed<   dZee   ed<   dZee   ed<   dZ	ee   ed<   dZ
ee   ed<   d	Zee   ed
<   dZee   ed<   dZee   ed<   dZee   ed<   dZee   ed<   y)SearchParamsquery   r   NtbsfilterenlangusrX   ru   apir   i`  rt   r   )r,   r-   r.   r1   r0   r   r   r   r  r	  r  rX   ru   r   rt   r   ri   r2   r3   r"   r  r  O  s    JE8C=C# FHSM D(3-!GXc]!"Hhsm"!FHSM!"GXc]"-1M8M*1r3   r  c                   X    e Zd ZU dZdZeed<   ee   ed<   dZ	e
e   ed<   dZe
e   ed<   y)SearchResponsez Response from search operations.Tr   r   Nr   r   )r,   r-   r.   r/   r   r   r0   r   rI   r   r   r1   r   r2   r3   r"   r  r  [  s7    *GT
 
!!!GXc]!E8C=r3   r  c                   b    e Zd ZU dZdZee   ed<   dZee	   ed<   dZ
ee	   ed<   dZee	   ed	<   y)
GenerateLLMsTextParamsz;
    Parameters for the LLMs.txt generation operation.
    
   maxUrlsFshowFullTextTcacheN,_GenerateLLMsTextParams__experimental_stream)r,   r-   r.   r/   r  r   r   r0   r  r   r  r  r2   r3   r"   r  r  b  sB      GXc]#(L(4.( E8D> ,08D>0r3   r  c                       e Zd ZU dZdZee   ed<   dZee   ed<   dZ	ee   ed<   dZ
ee   ed	<   dZee   ed
<   dZee   ed<   y)DeepResearchParamsz5
    Parameters for the deep research operation.
       r   i  	timeLimit   r  NanalysisPromptr   -_DeepResearchParams__experimental_streamSteps)r,   r-   r.   r/   r   r   r   r0   r  r  r  r1   r   r  r   r2   r3   r"   r  r  k  s^      Hhsm"Ix}"GXc]$(NHSM("&L(3-&155r3   r  c                   :    e Zd ZU dZeed<   eed<   dZee   ed<   y)DeepResearchResponsez4
    Response from the deep research operation.
    r   r   Nr   )	r,   r-   r.   r/   r   r0   r1   r   r   r2   r3   r"   r  r  v  s!     MGE8C=r3   r  c                       e Zd ZU dZeed<   dZeee	e
f      ed<   e	ed<   dZee	   ed<   e	ed<   eed<   eed	<   eee	e
f      ed
<   eee	e
f      ed<   ee	   ed<   y)DeepResearchStatusResponsez;
    Status response from the deep research operation.
    r   Nr   r   r   r   currentDepthr   
activitiesr  	summaries)r,   r-   r.   r/   r   r0   r   r   r   r1   r   r   r   r   r2   r3   r"   r!  r!  ~  sx     M%)D(4S>
")KE8C=NMT#s(^$$$sCx.!!Cyr3   r!  c                   >    e Zd ZU dZdZeed<   eed<   dZe	e   ed<   y)GenerateLLMsTextResponsez-Response from LLMs.txt generation operations.Tr   r   Nr   )
r,   r-   r.   r/   r   r   r0   r1   r   r   r2   r3   r"   r&  r&    s"    7GTGE8C=r3   r&  c                   ,    e Zd ZU eed<   dZee   ed<   y)"GenerateLLMsTextStatusResponseDatallmstxtNllmsfulltxt)r,   r-   r.   r1   r0   r*  r   r2   r3   r"   r(  r(    s    L!%K#%r3   r(  c                   b    e Zd ZU dZdZeed<   dZee	   ed<   e
d   ed<   dZee   ed<   eed	<   y)
GenerateLLMsTextStatusResponsez4Status response from LLMs.txt generation operations.Tr   Nr   r  r   r   r   )r,   r-   r.   r/   r   r   r0   r   r   r(  r	   r   r1   r2   r3   r"   r,  r,    s>    >GT9=D(5
6=788E8C=Nr3   r,  c                   ^    e Zd ZU dZeed<   eeee	f      ed<   dZ
ee   ed<   dZee   ed<   y)r  z-
    Response from the search operation.
    r   r   Nr   r   )r,   r-   r.   r/   r   r0   r   r   r1   r   r   r   r   r2   r3   r"   r  r    s=     M
tCH~
!GXc]!E8C=r3   c                       e Zd ZU dZdZee   ed<    ej                  dd      Z
ee   ed<   dZee   ed<   dZee   ed	<   dZee   ed
<   dZee   ed<   dZee   ed<   dZeeeef      ed<   y)r   z/
    Parameters for the extract operation.
    Nr+   re   rD   rf   system_promptFallow_external_linksenable_web_searchr   show_sourcesr   )r,   r-   r.   r/   r+   r   r1   r0   rG   r   rf   r   r/  r0  r   r1  r   r2  r   r   r2   r3   r"   r   r     s     !FHSM "0(..X"FL(3-F#'M8C='+0(4.0(-x~-&+OXd^+#(L(4.(&*E8DcN#*r3   c            2       N   e Zd Zddee   dee   ddfdZdddddddddddddddddddddddded	eeed
         deeeef      deee      deee      dee	   dee
   dee
   dee   dee	   dee	   dee	   dee	   deed      dee	   dee   dee   deeeeeeeeeeeef	         dee   dee
   dee	   dee	   d ee   dee   f0d!Zddddddddd"d#ed$ee
   d%ee   d&ee   d'ee   d(ee   dee   dee
   d)ee   defd*Zddddddddddddddddddd+dd,ded-eee      d.eee      d/ee
   d0ee
   d$ee
   d1ee	   d2ee	   d3ee	   d4ee	   d)ee   d5eeeef      d6ee	   d7ee	   d8ee	   d9ee
   d:ee	   d;ee
   dee	   d<ee
   d=ee   de f,d>Z!dddddddddddddddddddd?ded-eee      d.eee      d/ee
   d0ee
   d$ee
   d1ee	   d2ee	   d3ee	   d4ee	   d)ee   d5eeeef      d6ee	   d7ee	   d8ee	   d9ee
   d:ee	   d;ee
   dee	   d=ee   de"f*d@Z#dAede fdBZ$dAede%fdCZ&dAedeeef   fdDZ'dddddddddddddddddddd?ded-eee      d.eee      d/ee
   d0ee
   d$ee
   d1ee	   d2ee	   d3ee	   d4ee	   d)ee   d5eeeef      d6ee	   d7ee	   d8ee	   d9ee
   d:ee	   d;ee
   dee	   d=ee   ddEf*dFZ(ddddddddGdedHee   d4ee	   dIee	   dJee	   d$ee
   dee
   dKee	   de)fdLZ*dddddddddddddddddd+ddddMdNee   d	eeedO         deeeef      deee      deee      dee	   dee
   dee
   dee   dee	   dee	   dee	   dee	   deed      dee   dee   deeeeeeeeeeeef	         d ee   d<ee
   d;ee
   dee	   d=ee   de+f.dPZ,dddddddddddddddddddddQdNee   d	eeedO         deeeef      deee      deee      dee	   dee
   dee
   dee   dee	   dee	   dee	   dee	   deed      dee   dee   deeeeeeeeeeeef	         d ee   d;ee
   d=ee   dee	   de-f,dRZ.dddddddddddddddddddddSdNee   d	eeedO         deeeef      deee      deee      dee	   dee
   dee
   dee   dee	   dee	   dee	   dee	   deed      dee   dee   deeeeeeeeeeeef	         d ee   d;ee
   dee	   d=ee   ddEf,dTZ/dAede+fdUZ0dAede%fdVZ1	 dddddWdWdWddXdNeee      dYee   dZee   d[ee   d3ee	   d\ee	   d]ee	   d eeeef      de2e   fd^Z3d_ede2e   fd`Z4	 dddddWdWdWddXdNeee      dYee   dZee   d[ee   d3ee	   d\ee	   d]ee	   d eeeef      de2e   fdaZ5dddddbdedcee
   ddee	   deee	   dfee	   de6fdgZ7dddddbdedcee
   ddee	   deee	   dfee	   de8fdhZ9dAede6fdiZ:	 dd=ee   deeef   fdjZ;	 	 ddedkeeef   deeef   dle
dme<de=j|                  fdnZ?	 	 ddedeeef   dle
dme<de=j|                  f
doZ@	 	 ddedeeef   dle
dme<de=j|                  f
dpZAdAedeeef   d<e
de fdqZBdre=j|                  dseddfdtZCdue
dsedvedwedef
dxZDdddddddddyd#ed/ee
   dzee
   dcee
   d{ee   d[ee   d|ee	   d}eeEeeef   gdf      d~eeEeeef   gdf      deFfdZGdddddddd#ed/ee
   dzee
   dcee
   d{ee   d[ee   d|ee	   deeef   fdZHdAedeFfdZIdeeef   deddfdZJd ZKy)FirecrawlAppNapi_keyapi_urlreturnc                 6   |xs t        j                  d      | _        |xs t        j                  dd      | _        d| j                  v r,| j                   t        j                  d       t        d      t        j                  d| j                          y)	z
        Initialize the FirecrawlApp instance with API key, API URL.

        Args:
            api_key (Optional[str]): API key for authenticating with the Firecrawl API.
            api_url (Optional[str]): Base URL for the Firecrawl API.
        FIRECRAWL_API_KEYFIRECRAWL_API_URLzhttps://api.firecrawl.devzapi.firecrawl.devNz%No API key provided for cloud servicezNo API key providedz'Initialized FirecrawlApp with API URL: )r   getenvr5  r6  r%   r   
ValueErrordebug)selfr5  r6  s      r"   __init__zFirecrawlApp.__init__  s{     @")),?"@]")),?A\"] $,,.4<<3GNNBC233>t||nMNr3   rs   )rn   r\   include_tagsexclude_tagsonly_main_contentwait_forrt   ru   rv   skip_tls_verificationremove_base64_images	block_adsr~   	parse_pdfrO   json_optionsrR   change_tracking_optionsmax_agestore_in_cachezero_data_retentionr   rJ   rn   rk   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rz   rG  rO   rH  rR   rI  rJ  rK  rL  r   c                \   | j                  |d       | j                         }|dt         d}|r||d<   |r||d<   |r||d<   |r||d<   |||d	<   |r||d
<   |r||d<   |	r|	j                  dd      |d<   |
|
|d<   |||d<   |||d<   |||d<   |r||d<   |||d<   |d| j	                  |      }t        |t              rd|v r| j	                  |d         |d<   t        |t              r|n|j                  dd      |d<   |d| j	                  |      }t        |t              rd|v r| j	                  |d         |d<   t        |t              r|n|j                  dd      |d<   |r6|D cg c]'  }t        |t              r|n|j                  dd      ) c}|d<   |r(t        |t              r|n|j                  dd      |d<   |||d<   |||d<   |||d<   ||j                  dd      |d<   |j                  |       d|v r)|d   r$d|d   v r| j	                  |d   d         |d   d<   d|v r)|d   r$d|d   v r| j	                  |d   d         |d   d<   t        j                  | j                   d||||dz  d z   nd!      }|j                  d"k(  rW	 |j                         }|j                  d#      rd$|v rt        d)i |d$   S d%|v rt        d&|d%          t        d&|       | j!                  |d(       yc c}w # t        $ r t        d'      w xY w)*a  
        Scrape and extract content from a URL.

        Args:
          url (str): Target URL to scrape
          formats (Optional[List[Literal["markdown", "html", "rawHtml", "content", "links", "screenshot", "screenshot@fullPage", "extract", "json"]]]): Content types to retrieve (markdown/html/etc)
          headers (Optional[Dict[str, str]]): Custom HTTP headers
          include_tags (Optional[List[str]]): HTML tags to include
          exclude_tags (Optional[List[str]]): HTML tags to exclude
          only_main_content (Optional[bool]): Extract main content only
          wait_for (Optional[int]): Wait for a specific element to appear
          timeout (Optional[int]): Request timeout (ms)
          location (Optional[LocationConfig]): Location configuration
          mobile (Optional[bool]): Use mobile user agent
          skip_tls_verification (Optional[bool]): Skip TLS verification
          remove_base64_images (Optional[bool]): Remove base64 images
          block_ads (Optional[bool]): Block ads
          proxy (Optional[Literal["basic", "stealth", "auto"]]): Proxy type (basic/stealth)
          extract (Optional[JsonConfig]): Content extraction settings
          json_options (Optional[JsonConfig]): JSON extraction settings
          actions (Optional[List[Union[WaitAction, ScreenshotAction, ClickAction, WriteAction, PressAction, ScrollAction, ScrapeAction, ExecuteJavascriptAction, PDFAction]]]): Actions to perform
          change_tracking_options (Optional[ChangeTrackingOptions]): Change tracking settings
          zero_data_retention (Optional[bool]): Whether to delete data after scrape is done
          agent (Optional[AgentOptions]): Agent configuration for FIRE-1 model


        Returns:
          ScrapeResponse with:
          * Requested content formats
          * Page metadata
          * Extraction results
          * Success/error status

        Raises:
          Exception: If scraping fails
        
scrape_urlpython-sdk@rJ   r   rn   r\   ro   rp   Nrq   rr   rt   Tby_aliasexclude_noneru   rv   rw   rx   ry   r~   r   re   rO   r   rR   r   r   r   zeroDataRetentionr   
/v1/scrape     @@r  r\   rC   rt      r   r   r   Failed to scrape URL. Error: +Failed to parse Firecrawl response as JSON.z
scrape URLr2   )_validate_kwargs_prepare_headersversiondict_ensure_schema_dict
isinstanceupdaterequestspostr6  status_coderC   getr   r   r<  _handle_error)r>  rJ   rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rG  rO   rH  rR   rI  rJ  rK  rL  r   kwargs_headersscrape_paramsactionresponseresponse_jsons                                 r"   rN  zFirecrawlApp.scrape_url  sS   @ 	fl3((* #G9-
 '.M)$'.M)$+7M-(+7M-((/@M+,'/M)$'.M)$(0tRV(WM*%&,M(# ,3HM/0+2FM./ (1M*%%*M'" (1M*%..w7G'4(X-@$($<$<WX=N$O!2<Wd2KwQXQ]Q]gkz~Q]QM)$#33LAL,-(l2J)-)A)A,xBX)YX&;ElTX;Y<_k_p_pz~  NR_p  `SM-( MT  (U  CI*VT2JPVP[P[eix|P[P}(}  (UM)$"PZ[rtxPy5L  @W  @\  @\  fj  y}  @\  @~M12&-M(#%,:M.)*1DM-.%*ZZDZ%QM'"V$%-	*BxS`ajSkGk151I1I-XaJbckJl1mM)$X.M)mM.Jx[hiv[wOw595M5Mm\iNjksNt5uM-(2 ==||nJ'-4-@Wv%)d	
 3&	O ( $$Y/Fm4K)BM&,ABB-#&CMRYDZC[$\]]#&CM?$STT x6O (UH  O MNNOs   4,L(2L #L L+)r   r  r	  r  rX   ru   rt   scrape_optionsr  r   r  r	  r  rX   rm  c                   | j                  |
d       i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d	<   |	|	j                  d
d
      |d<   |j                  |
       |j                  d      }t	        dd|i|}|j                  d
d
      }dt
         |d<   |r||d<   t        j                  | j                   ddd| j                   i|      }|j                  dk(  rT	 |j                         }|j                  d      rd|v rt        di |S d|v rt        d|d          t        d|       | j                  |d       y# t        $ r t        d      w xY w)aS  
        Search for content using Firecrawl.

        Args:
            query (str): Search query string
            limit (Optional[int]): Max results (default: 5)
            tbs (Optional[str]): Time filter (e.g. "qdr:d")
            filter (Optional[str]): Custom result filter
            lang (Optional[str]): Language code (default: "en")
            country (Optional[str]): Country code (default: "us") 
            location (Optional[str]): Geo-targeting
            timeout (Optional[int]): Request timeout in milliseconds
            scrape_options (Optional[ScrapeOptions]): Result scraping configuration
            **kwargs: Additional keyword arguments for future compatibility

        Returns:
            SearchResponse: Response containing:
                * success (bool): Whether request succeeded
                * data (List[FirecrawlDocument]): Search results
                * warning (Optional[str]): Warning message if any
                * error (Optional[str]): Error message if any

        Raises:
            Exception: If search fails or response cannot be parsed
        r   Nr   r  r	  r  rX   ru   rt   TrQ  r   integrationr  rO  r   
/v1/searchAuthorizationBearer r\   rC   rX  r   r   r   zSearch failed. Error: rZ  r2   )r[  r^  ra  re  r  r]  rb  rc  r6  r5  rd  rC   r  r   r<  rf  )r>  r  r   r  r	  r  rX   ru   rt   rm  rg  search_params_integrationfinal_paramsparams_dictrk  rl  s                    r"   r   zFirecrawlApp.searchd  s   N 	fh/  %*M'"?#&M% &,M(#$(M&!'.M)$(0M*%'.M)$%-;-@-@$]a-@-bM/* 	V$$((7 $A%A=A"''D'I"-gY 7H)5K& ==||nJ'$~&>?
 3&	O ( $$Y/Fm4K):M::-#&<]7=S<T$UVV#&<]O$LMM x2  O MNNOs   7/E '#E E2   )include_pathsexclude_paths	max_depthmax_discovery_depthr   allow_backward_linkscrawl_entire_domainr0  ignore_sitemaprm  r   deduplicate_similar_urlsignore_query_parametersregex_on_full_urlr   allow_subdomainsmax_concurrencyrL  poll_intervalidempotency_keyry  rz  r{  r|  r}  r~  r0  r  r   r  r  r  r   r  r  r  r  c                &   | j                  |d       i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   n|||d	<   |	|	|d
<   |
|
|d<   ||j                  dd      |d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |j                  |       |j                  d      }t	        d i |}|j                  dd      }||d<   dt
         |d<   |r||d<   | j                  |      }| j                  | j                   d||      }|j                  dk(  r3	 |j                         j                  d      }| j                  |||      S | j                  |d       y#  t        d      xY w)!a  
        Crawl a website starting from a URL.

        Args:
            url (str): Target URL to start crawling from
            include_paths (Optional[List[str]]): Patterns of URLs to include
            exclude_paths (Optional[List[str]]): Patterns of URLs to exclude
            max_depth (Optional[int]): Maximum crawl depth
            max_discovery_depth (Optional[int]): Maximum depth for finding new URLs
            limit (Optional[int]): Maximum pages to crawl
            allow_backward_links (Optional[bool]): DEPRECATED: Use crawl_entire_domain instead
            crawl_entire_domain (Optional[bool]): Follow parent directory links
            allow_external_links (Optional[bool]): Follow external domain links
            ignore_sitemap (Optional[bool]): Skip sitemap.xml processing
            scrape_options (Optional[ScrapeOptions]): Page scraping configuration
            webhook (Optional[Union[str, WebhookConfig]]): Notification webhook settings
            deduplicate_similar_urls (Optional[bool]): Remove similar URLs
            ignore_query_parameters (Optional[bool]): Ignore URL parameters
            regex_on_full_url (Optional[bool]): Apply regex to full URLs
            delay (Optional[int]): Delay in seconds between scrapes
            allow_subdomains (Optional[bool]): Follow subdomains
            max_concurrency (Optional[int]): Maximum number of concurrent scrapes
            zero_data_retention (Optional[bool]): Whether to delete data after 24 hours
            poll_interval (Optional[int]): Seconds between status checks (default: 2)
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests
            **kwargs: Additional parameters to pass to the API

        Returns:
            CrawlStatusResponse with:
            * Crawling status and progress
            * Crawled page contents
            * Success/error information

        Raises:
            Exception: If crawl fails
        	crawl_urlNr   r   r   r   r   r   r   r   r   TrQ  r   r   r   r   r   r   r   r   rT  ro  rJ   rO  r   	/v1/crawlrX  r   rZ  start crawl jobr2   )r[  r^  ra  re  r   r]  r\  _post_requestr6  rd  rC   r   _monitor_job_statusrf  )r>  rJ   ry  rz  r{  r|  r   r}  r~  r0  r  rm  r   r  r  r  r   r  r  rL  r  r  rg  crawl_paramsru  rv  rw  r\   rk  r   s                                 r"   r  zFirecrawlApp.crawl_url  sX   ~ 	fk2 $+8L($+8L( '0L$*0CL,-$)L!*0CL,-!-1EL-.+1EL-.%,:L)%,:,?,?\`,?,aL)&-L##/5ML12".4KL01(->L)*$)L!'.>L*+&-<L)**0CL,-F##''6 #2\2"''D'I E"-gY 7H)5K& ''8%%i&@+wW3&P]]_((. ++BGGx):;	P"MOOs   >F F)ry  rz  r{  r|  r   r}  r~  r0  r  rm  r   r  r  r  r   r  r  rL  r  c                   | j                  |d       i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   n|||d	<   |	|	|d
<   |
|
|d<   ||j                  dd      |d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |j                  |       t        di |}|j                  dd      }||d<   dt         |d<   | j                  |      }| j                  | j                   d||      }|j                  dk(  r	 t        di |j                         S | j                  |d       y#  t        d      xY w)a  
        Start an asynchronous crawl job.

        Args:
            url (str): Target URL to start crawling from
            include_paths (Optional[List[str]]): Patterns of URLs to include
            exclude_paths (Optional[List[str]]): Patterns of URLs to exclude
            max_depth (Optional[int]): Maximum crawl depth
            max_discovery_depth (Optional[int]): Maximum depth for finding new URLs
            limit (Optional[int]): Maximum pages to crawl
            allow_backward_links (Optional[bool]): DEPRECATED: Use crawl_entire_domain instead
            crawl_entire_domain (Optional[bool]): Follow parent directory links
            allow_external_links (Optional[bool]): Follow external domain links
            ignore_sitemap (Optional[bool]): Skip sitemap.xml processing
            scrape_options (Optional[ScrapeOptions]): Page scraping configuration
            webhook (Optional[Union[str, WebhookConfig]]): Notification webhook settings
            deduplicate_similar_urls (Optional[bool]): Remove similar URLs
            ignore_query_parameters (Optional[bool]): Ignore URL parameters
            regex_on_full_url (Optional[bool]): Apply regex to full URLs
            delay (Optional[int]): Delay in seconds between scrapes
            allow_subdomains (Optional[bool]): Follow subdomains
            max_concurrency (Optional[int]): Maximum number of concurrent scrapes
            zero_data_retention (Optional[bool]): Whether to delete data after 24 hours
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests
            **kwargs: Additional parameters to pass to the API

        Returns:
            CrawlResponse with:
            * success - Whether crawl started successfully
            * id - Unique identifier for the crawl job
            * url - Status check URL for the crawl
            * error - Error message if start failed

        Raises:
            Exception: If crawl initiation fails
        async_crawl_urlNr   r   r   r   r   r   r   r   r   TrQ  r   r   r   r   r   r   r   r   rT  rJ   rO  r   r  rX  rZ  r  r2   )r[  r^  ra  r   r]  r\  r  r6  rd  r   rC   r   rf  )r>  rJ   ry  rz  r{  r|  r   r}  r~  r0  r  rm  r   r  r  r  r   r  r  rL  r  rg  r  rv  rw  r\   rk  s                              r"   r  zFirecrawlApp.async_crawl_urlE  s%   | 	f&78 $+8L($+8L( '0L$*0CL,-$)L!*0CL,-!-1EL-.+1EL-.%,:L)%,:,?,?\`,?,aL)&-L##/5ML12".4KL01(->L)*$)L!'.>L*+&-<L)**0CL,-F# #2\2"''D'I E"-gY 7H ''8%%i&@+wW3&P$7x}}77 x):;P"MOOs   &E Er   c                 2   d| }| j                         }| j                  | j                   | |      }|j                  dk(  rr	 |j	                         }|d   dk(  rd|v r|d   }d|v rt        |d         dk(  rn|j                  d      }|st        j                  d	       n~	 | j                  ||      }|j                  dk7  r#t        j                  d
|j                          n9	 |j	                         }	|j                  |	j                  dg              |	}d|v r||d<   |j                  d      |j                  d      |j                  d      |j                  d      |j                  d      |j                  d      d}d|v r|d   |d<   d|v r|d   |d<   t        ddd|v rdndi|S | j                  |d       y#  t        d      xY w#  t        d      xY w# t
        $ r"}
t        j                  d|
        Y d}
~
d}
~
ww xY w)a"  
        Check the status and results of a crawl job.

        Args:
            id: Unique identifier for the crawl job

        Returns:
            CrawlStatusResponse containing:

            Status Information:
            * status - Current state (scraping/completed/failed/cancelled)
            * completed - Number of pages crawled
            * total - Total pages to crawl
            * creditsUsed - API credits consumed
            * expiresAt - Data expiration timestamp
            
            Results:
            * data - List of crawled documents
            * next - URL for next page of results (if paginated)
            * success - Whether status check succeeded
            * error - Error message if failed

        Raises:
            Exception: If status check fails
        
/v1/crawl/rX  rZ  r   r]   r   r   r   Expected 'next' URL is missing.Failed to fetch next page: !Error during pagination request: Nr   r   r   r   r   r]   r   r   r   r   r   FTcheck crawl statusr2   )r\  _get_requestr6  rd  rC   r   lenre  r%   r   r   extendr   rf  r>  r   endpointr\   rk  status_datar   next_urlstatus_response	next_dataes              r"   check_crawl_statuszFirecrawlApp.check_crawl_status  sP   4  t$'')$$~hZ%@'J3&P&mmo 8$3[(&v.D K/{623q8!#.??6#:'"NN+LM!".2.?.?'.RO.::cA &/J?KfKfJg-h i %`,;,@,@,B	 !KK	fb(AB*3K# !K/* +/K' &//(3$1(__[9*}=(__[9#/H +%$/$8!$#.v#6 & !(K!7T 
 x)=>aP"MOO$`&/2]&_ _  ) ""LL+LQC)PQ!"s=   G (AG+ -G =#G+ GG((G+ +	H4HHc                     | j                         }| j                  | j                   d| d|      }|j                  dk(  r	 t	        di |j                         S | j                  |d       y#  t        d      xY w)aJ  
        Returns information about crawl errors.

        Args:
            id (str): The ID of the crawl job

        Returns:
            CrawlErrorsResponse containing:
            * errors (List[Dict[str, str]]): List of errors with fields:
                - id (str): Error ID
                - timestamp (str): When the error occurred
                - url (str): URL that caused the error
                - error (str): Error message
            * robotsBlocked (List[str]): List of URLs blocked by robots.txt

        Raises:
            Exception: If error check fails
        r  /errorsrX  rZ  zcheck crawl errorsNr2   r\  r  r6  rd  r   rC   r   rf  r>  r   r\   rk  s       r"   check_crawl_errorszFirecrawlApp.check_crawl_errors  s    & '')$$~Zt7%KWU3&P*=X]]_== x)=>P"MOO   A/ /A<c                     | j                         }| j                  | j                   d| |      }|j                  dk(  r	 |j	                         S | j                  |d       y#  t        d      xY w)}  
        Cancel an asynchronous crawl job.

        Args:
            id (str): The ID of the crawl job to cancel

        Returns:
            Dict[str, Any] containing:
            * success (bool): Whether cancellation was successful
            * error (str, optional): Error message if cancellation failed

        Raises:
            Exception: If cancellation fails
        r  rX  rZ  zcancel crawl jobN)r\  _delete_requestr6  rd  rC   r   rf  r  s       r"   cancel_crawlzFirecrawlApp.cancel_crawl1  sz     '')''4<<.
2$(GQ3&P}}& x);<P"MOOs   A% %A2CrawlWatcherc                    | j                   |fi d|d|d|d|d|d|d|d|	d	|
d
|d|d|d|d|d|d|d|d|d||}|j                  r"|j                  rt        |j                  |       S t	        d      )aG  
        Initiate a crawl job and return a CrawlWatcher to monitor the job via WebSocket.

        Args:
            url (str): Target URL to start crawling from
            include_paths (Optional[List[str]]): Patterns of URLs to include
            exclude_paths (Optional[List[str]]): Patterns of URLs to exclude
            max_depth (Optional[int]): Maximum crawl depth
            max_discovery_depth (Optional[int]): Maximum depth for finding new URLs
            limit (Optional[int]): Maximum pages to crawl
            allow_backward_links (Optional[bool]): DEPRECATED: Use crawl_entire_domain instead
            crawl_entire_domain (Optional[bool]): Follow parent directory links
            allow_external_links (Optional[bool]): Follow external domain links
            ignore_sitemap (Optional[bool]): Skip sitemap.xml processing
            scrape_options (Optional[ScrapeOptions]): Page scraping configuration
            webhook (Optional[Union[str, WebhookConfig]]): Notification webhook settings
            deduplicate_similar_urls (Optional[bool]): Remove similar URLs
            ignore_query_parameters (Optional[bool]): Ignore URL parameters
            regex_on_full_url (Optional[bool]): Apply regex to full URLs
            delay (Optional[int]): Delay in seconds between scrapes
            allow_subdomains (Optional[bool]): Follow subdomains
            max_concurrency (Optional[int]): Maximum number of concurrent scrapes
            zero_data_retention (Optional[bool]): Whether to delete data after 24 hours
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests
            **kwargs: Additional parameters to pass to the API

        Returns:
            CrawlWatcher: An instance to monitor the crawl job via WebSocket

        Raises:
            Exception: If crawl job fails to start
        ry  rz  r{  r|  r   r}  r~  r0  r  rm  r   r  r  r  r   r  r  rL  r  Crawl job failed to start)r  r   r   r  r   )r>  rJ   ry  rz  r{  r|  r   r}  r~  r0  r  rm  r   r  r  r  r   r  r  rL  r  rg  crawl_responses                          r"   crawl_url_and_watchz FirecrawlApp.crawl_url_and_watchJ  s   r .--
'
 (
  	

 !4
 
 "6
 !4
 "6
 *
 *
 
 &>
 %<
 0
  !
" .#
$ ,%
& !4'
( ,+
. !!n&7&7 1 1488788r3   )r   r  include_subdomainssitemap_onlyr   rt   	use_indexr   r  r  r  c                   | j                  |	d       i }
|||
d<   |||
d<   |||
d<   |||
d<   |||
d<   |||
d<   |||
d	<   |
j                  |	       |
j                  d
      }t        di |
}|j	                  dd      }||d<   dt
         |d<   |r||d
<   t        j                  | j                   ddd| j                   i|      }|j                  dk(  rT	 |j                         }|j                  d      rd|v rt        di |S d|v rt        d|d          t        d|       | j                  |d       y# t        $ r t        d      w xY w)a}  
        Map and discover links from a URL.

        Args:
            url (str): Target URL to map
            search (Optional[str]): Filter pattern for URLs
            ignore_sitemap (Optional[bool]): Skip sitemap.xml processing
            include_subdomains (Optional[bool]): Include subdomain links
            sitemap_only (Optional[bool]): Only use sitemap.xml
            limit (Optional[int]): Maximum URLs to return
            timeout (Optional[int]): Request timeout in milliseconds
            **kwargs: Additional parameters to pass to the API

        Returns:
            MapResponse: Response containing:
                * success (bool): Whether request succeeded
                * links (List[str]): Discovered URLs
                * error (Optional[str]): Error message if any

        Raises:
            Exception: If mapping fails or response cannot be parsed
        map_urlNr   r   r   r   r   rt   r   ro  TrQ  rJ   rO  r   /v1/maprq  rr  rs  rX  r   rN   r   zMap failed. Error: rZ  mapr2   )r[  ra  re  r   r^  r]  rb  rc  r6  r5  rd  rC   r   r   r<  rf  )r>  rJ   r   r  r  r  r   rt   r  rg  
map_paramsru  rv  rw  rk  rl  s                   r"   r  zFirecrawlApp.map_url  s   F 	fi0 
 #)Jx %*8J').@J*+#(4J}%"'Jw$+Jy! %.Jz" 	&!!~~m4 !.:."''D'I E"-gY 7H)5K& ==||nG$$~&>?
 3&	O ( $$Y/G}4L&777-#&9-:P9Q$RSS#&9-$IJJ x/  O MNNOs   "/E #E E)rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rO   rH  rR   r   r  r  rL  r  urls	rK   rL   rM   rl   rN   rP   rm   rO   rC   c                   | j                  |d       i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d	<   |	|	j                  d
d
      |d<   |
|
|d<   |||d<   |||d<   |||d<   |||d<   |d| j                  |      }t        |t              rd|v r| j                  |d         |d<   t        |t              r|n|j                  d
d
      |d<   |d| j                  |      }t        |t              rd|v r| j                  |d         |d<   t        |t              r|n|j                  d
d
      |d<   |r6|D cg c]'  }t        |t              r|n|j                  d
d
      ) c}|d<   ||j                  d
d
      |d<   |||d<   |||d<   |j	                  |       t        d!i |}|j                  d
d
      }||d<   dt         |d<   d|v r)|d   r$d|d   v r| j                  |d   d         |d   d<   d|v r)|d   r$d|d   v r| j                  |d   d         |d   d<   | j                  |      }| j                  | j                   d||      }|j                  dk(  r3	 |j                         j                  d      }| j                  |||      S | j                  |d        yc c}w #  t        d      xY w)"aF  
        Batch scrape multiple URLs and monitor until completion.

        Args:
            urls (List[str]): URLs to scrape
            formats (Optional[List[Literal]]): Content formats to retrieve
            headers (Optional[Dict[str, str]]): Custom HTTP headers
            include_tags (Optional[List[str]]): HTML tags to include
            exclude_tags (Optional[List[str]]): HTML tags to exclude
            only_main_content (Optional[bool]): Extract main content only
            wait_for (Optional[int]): Wait time in milliseconds
            timeout (Optional[int]): Request timeout in milliseconds
            location (Optional[LocationConfig]): Location configuration
            mobile (Optional[bool]): Use mobile user agent
            skip_tls_verification (Optional[bool]): Skip TLS verification
            remove_base64_images (Optional[bool]): Remove base64 encoded images
            block_ads (Optional[bool]): Block advertisements
            proxy (Optional[Literal]): Proxy type to use
            extract (Optional[JsonConfig]): Content extraction config
            json_options (Optional[JsonConfig]): JSON extraction config
            actions (Optional[List[Union]]): Actions to perform
            agent (Optional[AgentOptions]): Agent configuration
            max_concurrency (Optional[int]): Maximum number of concurrent scrapes
            poll_interval (Optional[int]): Seconds between status checks (default: 2)
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests
            **kwargs: Additional parameters to pass to the API

        Returns:
            BatchScrapeStatusResponse with:
            * Scraping status and progress
            * Scraped content for each URL
            * Success/error information

        Raises:
            Exception: If batch scrape fails
        batch_scrape_urlsNrn   r\   ro   rp   rq   rr   rt   TrQ  ru   rv   rw   rx   ry   r~   re   rO   r   rR   r   r   rT  r  rO  r   /v1/batch/scraperX  r   rZ  start batch scrape jobr2   )r[  r^  r_  r`  ra  r   r]  r\  r  r6  rd  rC   re  r   r  rf  )r>  r  rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rO   rH  rR   r   r  r  rL  r  rg  ri  rj  rv  rw  rk  r   s                                 r"   r  zFirecrawlApp.batch_scrape_urls  s   @ 	f&9: '.M)$'.M)$#+7M-(#+7M-((/@M+,'/M)$'.M)$(0tRV(WM*%&,M(# ,3HM/0+2FM./ (1M*%%*M'"..w7G'4(X-@$($<$<WX=N$O!2<Wd2KwQXQ]Q]gkz~Q]QM)$#33LAL,-(l2J)-)A)A,xBX)YX&;ElTX;Y<_k_p_pz~  NR_p  `SM-( MT  (U  CI*VT2JPVP[P[eix|P[P}(}  (UM)$%*ZZDZ%QM'"&.=M*+*1DM-. 	V$ $4m4"''D'I"F"-gY 7H#I(>8{[dOeCe/3/G/GT]H^_gHh/iK	"8,K'K,F8WbcpWqKq373K3KKXeLfgoLp3qK&x0 ''8%%6F&GV]^3&P]]_((. ++BGGx)ABC (U:P"MOOs   ,KK K)rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rO   rH  rR   r   r  r  rL  c                   | j                  |d       i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d	<   |	|	j                  d
d
      |d<   |
|
|d<   |||d<   |||d<   |||d<   |||d<   |d| j                  |      }t        |t              rd|v r| j                  |d         |d<   t        |t              r|n|j                  d
d
      |d<   |d| j                  |      }t        |t              rd|v r| j                  |d         |d<   t        |t              r|n|j                  d
d
      |d<   |r6|D cg c]'  }t        |t              r|n|j                  d
d
      ) c}|d<   ||j                  d
d
      |d<   |||d<   |||d<   |j	                  |       t        d i |}|j                  d
d
      }||d<   dt         |d<   d|v r)|d   r$d|d   v r| j                  |d   d         |d   d<   d|v r)|d   r$d|d   v r| j                  |d   d         |d   d<   | j                  |      }| j                  | j                   d||      }|j                  dk(  r	 t        d i |j                         S | j                  |d       yc c}w #  t        d      xY w)!a|  
        Initiate a batch scrape job asynchronously.

        Args:
            urls (List[str]): URLs to scrape
            formats (Optional[List[Literal]]): Content formats to retrieve
            headers (Optional[Dict[str, str]]): Custom HTTP headers
            include_tags (Optional[List[str]]): HTML tags to include
            exclude_tags (Optional[List[str]]): HTML tags to exclude
            only_main_content (Optional[bool]): Extract main content only
            wait_for (Optional[int]): Wait time in milliseconds
            timeout (Optional[int]): Request timeout in milliseconds
            location (Optional[LocationConfig]): Location configuration
            mobile (Optional[bool]): Use mobile user agent
            skip_tls_verification (Optional[bool]): Skip TLS verification
            remove_base64_images (Optional[bool]): Remove base64 encoded images
            block_ads (Optional[bool]): Block advertisements
            proxy (Optional[Literal]): Proxy type to use
            extract (Optional[JsonConfig]): Content extraction config
            json_options (Optional[JsonConfig]): JSON extraction config
            actions (Optional[List[Union]]): Actions to perform
            agent (Optional[AgentOptions]): Agent configuration
            max_concurrency (Optional[int]): Maximum number of concurrent scrapes
            zero_data_retention (Optional[bool]): Whether to delete data after 24 hours
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests
            **kwargs: Additional parameters to pass to the API

        Returns:
            BatchScrapeResponse with:
            * success - Whether job started successfully
            * id - Unique identifier for the job
            * url - Status check URL
            * error - Error message if start failed

        Raises:
            Exception: If job initiation fails
        async_batch_scrape_urlsNrn   r\   ro   rp   rq   rr   rt   TrQ  ru   rv   rw   rx   ry   r~   re   rO   r   rR   r   r   rT  r  rO  r   r  rX  rZ  r  r2   )r[  r^  r_  r`  ra  r   r]  r\  r  r6  rd  r   rC   r   rf  )r>  r  rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rO   rH  rR   r   r  r  rL  rg  ri  rj  rv  rw  rk  s                               r"   r  z$FirecrawlApp.async_batch_scrape_urls  s   @ 	f&?@ '.M)$'.M)$#+7M-(#+7M-((/@M+,'/M)$'.M)$(0tRV(WM*%&,M(# ,3HM/0+2FM./ (1M*%%*M'"..w7G'4(X-@$($<$<WX=N$O!2<Wd2KwQXQ]Q]gkz~Q]QM)$#33LAL,-(l2J)-)A)A,xBX)YX&;ElTX;Y<_k_p_pz~  NR_p  `SM-( MT  (U  CI*VT2JPVP[P[eix|P[P}(}  (UM)$%*ZZDZ%QM'"&.=M*+*1DM-. 	V$ $4m4"''D'I"F"-gY 7H#I(>8{[dOeCe/3/G/GT]H^_gHh/iK	"8,K'K,F8WbcpWqKq373K3KKXeLfgoLp3qK&x0 ''8%%6F&GV]^3&P*=X]]_== x)ABA (U:P"MOOs   ,J2J7 7K)rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rO   rH  rR   r   r  rL  r  c                    | j                  |d       i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d	<   |	|	j                  d
d
      |d<   |
|
|d<   |||d<   |||d<   |||d<   |||d<   |d| j                  |      }t        |t              rd|v r| j                  |d         |d<   t        |t              r|n|j                  d
d
      |d<   |d| j                  |      }t        |t              rd|v r| j                  |d         |d<   t        |t              r|n|j                  d
d
      |d<   |r6|D cg c]'  }t        |t              r|n|j                  d
d
      ) c}|d<   ||j                  d
d
      |d<   |||d<   |||d<   |j	                  |       t        d!i |}|j                  d
d
      }||d<   dt         |d<   d|v r)|d   r$d|d   v r| j                  |d   d         |d   d<   d|v r)|d   r$d|d   v r| j                  |d   d         |d   d<   | j                  |      }| j                  | j                   d||      }|j                  dk(  rS	 t        d!i |j                         }|j                  r"|j                  rt        |j                  |       S t!        d      | j#                  |d        yc c}w #  t!        d      xY w)"a  
        Initiate a batch scrape job and return a CrawlWatcher to monitor the job via WebSocket.

        Args:
            urls (List[str]): URLs to scrape
            formats (Optional[List[Literal]]): Content formats to retrieve
            headers (Optional[Dict[str, str]]): Custom HTTP headers
            include_tags (Optional[List[str]]): HTML tags to include
            exclude_tags (Optional[List[str]]): HTML tags to exclude
            only_main_content (Optional[bool]): Extract main content only
            wait_for (Optional[int]): Wait time in milliseconds
            timeout (Optional[int]): Request timeout in milliseconds
            location (Optional[LocationConfig]): Location configuration
            mobile (Optional[bool]): Use mobile user agent
            skip_tls_verification (Optional[bool]): Skip TLS verification
            remove_base64_images (Optional[bool]): Remove base64 encoded images
            block_ads (Optional[bool]): Block advertisements
            proxy (Optional[Literal]): Proxy type to use
            extract (Optional[JsonConfig]): Content extraction config
            json_options (Optional[JsonConfig]): JSON extraction config
            actions (Optional[List[Union]]): Actions to perform
            agent (Optional[AgentOptions]): Agent configuration
            max_concurrency (Optional[int]): Maximum number of concurrent scrapes
            zero_data_retention (Optional[bool]): Whether to delete data after 24 hours
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests
            **kwargs: Additional parameters to pass to the API

        Returns:
            CrawlWatcher: An instance to monitor the batch scrape job via WebSocket

        Raises:
            Exception: If batch scrape job fails to start
        batch_scrape_urls_and_watchNrn   r\   ro   rp   rq   rr   rt   TrQ  ru   rv   rw   rx   ry   r~   re   rO   r   rR   r   r   rT  r  rO  r   r  rX   Batch scrape job failed to startrZ  r  r2   )r[  r^  r_  r`  ra  r   r]  r\  r  r6  rd  r   rC   r   r   r  r   rf  )r>  r  rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rO   rH  rR   r   r  rL  r  rg  ri  rj  rv  rw  rk  r  s                                r"   r  z(FirecrawlApp.batch_scrape_urls_and_watch  s   x 	f&CD '.M)$'.M)$#+7M-(#+7M-((/@M+,'/M)$'.M)$(0tRV(WM*%&,M(# ,3HM/0+2FM./ (1M*%%*M'"..w7G'4(X-@$($<$<WX=N$O!2<Wd2KwQXQ]Q]gkz~Q]QM)$#33LAL,-(l2J)-)A)A,xBX)YX&;ElTX;Y<_k_p_pz~  NR_p  `SM-( MT  (U  CI*VT2JPVP[P[eix|P[P}(}  (UM)$%*ZZDZ%QM'"&.=M*+*1DM-. 	V$ $4m4"''D'I"F"-gY 7H#I(>8{[dOeCe/3/G/GT]H^_gHh/iK	"8,K'K,F8WbcpWqKq373K3KKXeLfgoLp3qK&x0 ''8%%6F&GV]^3&P!4!Gx}}!G!))n.?.?'(9(94@@#$FGG x)ABI (UBP"MOOs   ,K+AK0 K0 0K=c                 <   d| }| j                         }| j                  | j                   | |      }|j                  dk(  rw	 |j	                         }|d   dk(  rd|v r|d   }d|v rt        |d         dk(  rn|j                  d      }|st        j                  d	       n~	 | j                  ||      }|j                  dk7  r#t        j                  d
|j                          n9	 |j	                         }	|j                  |	j                  dg              |	}d|v r||d<   t        di d|v rdnd|j                  d      |j                  d      |j                  d      |j                  d      |j                  d      |j                  d      |j                  d      |j                  d      d	S | j                  |d       y#  t        d      xY w#  t        d      xY w# t
        $ r"}
t        j                  d|
        Y d}
~
d}
~
ww xY w)a>  
        Check the status of a batch scrape job using the Firecrawl API.

        Args:
            id (str): The ID of the batch scrape job.

        Returns:
            BatchScrapeStatusResponse: The status of the batch scrape job.

        Raises:
            Exception: If the status check request fails.
        /v1/batch/scrape/rX  rZ  r   r]   r   r   r   r  r  r  Nr   FTr   r   r   )	r   r   r   r]   r   r   r   r   r   zcheck batch scrape statusr2   )r\  r  r6  rd  rC   r   r  re  r%   r   r   r  r   rf  r  s              r"   check_batch_scrape_statusz&FirecrawlApp.check_batch_scrape_status  s,    'rd+'')$$~hZ%@'J3&P&mmo 8$3[(&v.D K/{623q8!#.??6#:'"NN+LM!".2.?.?'.RO.::cA &/J?KfKfJg-h i %`,;,@,@,B	 !KK	fb(AB*3K# !K/* +/K', 
$+{$:5%//(3$1(__[9*}=(__[9#/#/$1
0 
 
 x)DEQP"MOO$`&/2]&_ _  ) ""LL+LQC)PQ!"s=   G (AG0 -G  =#G0 G G--G0 0	H9HHc                     | j                         }| j                  | j                   d| d|      }|j                  dk(  r	 t	        di |j                         S | j                  |d       y#  t        d      xY w)aV  
        Returns information about batch scrape errors.

        Args:
            id (str): The ID of the crawl job.

        Returns:
            CrawlErrorsResponse containing:
            * errors (List[Dict[str, str]]): List of errors with fields:
              * id (str): Error ID
              * timestamp (str): When the error occurred
              * url (str): URL that caused the error
              * error (str): Error message
            * robotsBlocked (List[str]): List of URLs blocked by robots.txt

        Raises:
            Exception: If the error check request fails
        r  r  rX  rZ  zcheck batch scrape errorsNr2   r  r  s       r"   check_batch_scrape_errorsz&FirecrawlApp.check_batch_scrape_errors  s    & '')$$~5Frd'%RT[\3&P*=X]]_== x)DEP"MOOr  Fr+   re   r/  r0  r1  r2  r   r+   re   r/  r1  r2  c                   | j                  |	d       | j                         }
|s|st        d      |s|st        d      |r| j                  |      }|xs g ||||dt	                d}|r||d<   |r||d<   |r||d<   |j                  |	       	 | j                  | j                   d	||
      }|j                  d
k(  r	 |j                         }|d   r|j                  d      }|st        d      	 | j                  | j                   d| |
      }|j                  d
k(  rB	 |j                         }|d   dk(  rt        di |S |d   dv r)t        d|d    d|d          | j                  |d       t        j                   d       t        d|d          | j                  |d       	 t        dd      S #  t        d      xY w#  t        d      xY w# t        $ r}t        t#        |      d      d}~ww xY w)a  
        Extract structured information from URLs.

        Args:
            urls (Optional[List[str]]): URLs to extract from
            prompt (Optional[str]): Custom extraction prompt
            schema (Optional[Any]): JSON schema/Pydantic model
            system_prompt (Optional[str]): System context
            allow_external_links (Optional[bool]): Follow external links
            enable_web_search (Optional[bool]): Enable web search
            show_sources (Optional[bool]): Include source URLs
            agent (Optional[Dict[str, Any]]): Agent configuration
            **kwargs: Additional parameters to pass to the API

        Returns:
            ExtractResponse[Any] with:
            * success (bool): Whether request succeeded
            * data (Optional[Any]): Extracted data matching schema
            * error (Optional[str]): Error message if any

        Raises:
            ValueError: If prompt/schema missing or extraction fails
        rO   #Either prompt or schema is required!Either urls or prompt is requiredrO  r  r   r   r   re   r   r+   r   r   /v1/extractrX  rZ  r   r   )Job ID not returned from extract request./v1/extract/r   r]   r^   r   Extract job 	. Error: r   zextract-statusrx  Failed to extract. Error:   NFzInternal server error.r   r   r2   )r[  r\  r<  r_  r#   ra  r  r6  rd  rC   r   re  r  r   rf  timesleepr1   )r>  r  r+   re   r/  r0  r1  r2  r   rg  r\   request_datark  r   job_idr  r  r  s                     r"   rO   zFirecrawlApp.extract  sl   H 	fi0'')fBCCF@AA--f5F JB"60'#KM?3
 %+L"+8L($)L! 	F#)	*))<<.,H
 ##s*T#==?D 	?!XXd^F!'(STT *.*;*;#||nLA#+ +66#=`.=.B.B.D  +84C'6'E'E E!,X!6:Q!Q&/,{8?T>UU^_jkr_s^t0u&v v ..@PQ

1# & $&@g$PQQ""8Y7 u4LMMCT#&QSS`&/2]&_ _  	*SVS))	*sP   /G" G AG" (G 8G" A)G" GG" GG" "	H+H  Hr  c                 J   | j                         }	 | j                  | j                   d| |      }|j                  dk(  r	 t	        di |j                         S | j                  |d       y#  t        d      xY w# t        $ r}t        t        |      d      d}~ww xY w)a$  
        Retrieve the status of an extract job.

        Args:
            job_id (str): The ID of the extract job.

        Returns:
            ExtractResponse[Any]: The status of the extract job.

        Raises:
            ValueError: If there is an error retrieving the status.
        r  rX  rZ  zget extract statusr  Nr2   )
r\  r  r6  rd  r   rC   r   rf  r<  r1   )r>  r  r\   rk  r  s        r"   get_extract_statuszFirecrawlApp.get_extract_statusl  s     '')
	*((DLL>fX)NPWXH##s*T*=X]]_== ""8-ABT#&QSS  	*SVS))	*s/   0A? A/ A? /A<<A? ?	B"BB"c                   | j                         }	|}|r| j                  |      }|||||dt         d}
|r||
d<   |r||
d<   |r||
d<   	 | j                  | j                   d|
|	      }|j
                  dk(  r	 t        di |j                         S | j                  |d	       y#  t        d      xY w# t        $ r}t        t        |      d
      d}~ww xY w)a  
        Initiate an asynchronous extract job.

        Args:
            urls (List[str]): URLs to extract information from
            prompt (Optional[str]): Custom extraction prompt
            schema (Optional[Any]): JSON schema/Pydantic model
            system_prompt (Optional[str]): System context
            allow_external_links (Optional[bool]): Follow external links
            enable_web_search (Optional[bool]): Enable web search
            show_sources (Optional[bool]): Include source URLs
            agent (Optional[Dict[str, Any]]): Agent configuration
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests

        Returns:
            ExtractResponse[Any] with:
            * success (bool): Whether request succeeded
            * data (Optional[Any]): Extracted data matching schema
            * error (Optional[str]): Error message if any

        Raises:
            ValueError: If job initiation fails
        rO  r  r+   r   r   r  rX  rZ  zasync extractr  Nr2   )r\  r_  r]  r  r6  rd  r   rC   r   rf  r<  r1   )r>  r  r+   re   r/  r0  r1  r2  r   r\   r  rk  r  s                r"   async_extractzFirecrawlApp.async_extract  s   D '')--f5F "60'#G9-
 %+L"+8L($)L!
	*))T\\N+*FV]^H##s*T*=X]]_== ""8_=T#&QSS  	*SVS))	*s0   /B8 <B( B8 (B55B8 8	CCCmax_urlsshow_full_textr  experimental_streamr  r  r  r  c                   t        ||||      }| j                  |||||      }|j                  r|j                  st	        dddd      S |j                  }	 | j                  |      }	|	j                  dk(  r|	S |	j                  dk(  r|	S |	j                  d	k7  rt	        dd
dd      S t        j                  d       g)a  
        Generate LLMs.txt for a given URL and poll until completion.

        Args:
            url (str): Target URL to generate LLMs.txt from
            max_urls (Optional[int]): Maximum URLs to process (default: 10)
            show_full_text (Optional[bool]): Include full text in output (default: False)
            cache (Optional[bool]): Whether to use cached content if available (default: True)
            experimental_stream (Optional[bool]): Enable experimental streaming

        Returns:
            GenerateLLMsTextStatusResponse with:
            * Generated LLMs.txt content
            * Full version if requested
            * Generation status
            * Success/error information

        Raises:
            Exception: If generation fails
        r  r  r  __experimental_streamr  Fz#Failed to start LLMs.txt generationr^    r   r   r   r   r]   r  /LLMs.txt generation job terminated unexpectedlyrx  )	r  async_generate_llms_textr   r   r,  check_generate_llms_text_statusr   r  r  )
r>  rJ   r  r  r  r  paramsrk  r  r   s
             r"   generate_llms_textzFirecrawlApp.generate_llms_text  s    8 ('"5	
 00) 3 1 
 x{{1;	  99&AF}}+(*,.5!K# 	  JJqM r3   c                   t        ||||      }| j                         }d|i|j                  dd      }dt         |d<   	 | j	                  | j
                   d||      }	|	j                         }
t        d|       t        d	|
       |
j                  d
      r	 t        di |
S | j                  |
d       	 t        dd      S #  t        d      xY w# t        $ r}t        t        |            d}~ww xY w)a  
        Initiate an asynchronous LLMs.txt generation operation.

        Args:
            url (str): The target URL to generate LLMs.txt from. Must be a valid HTTP/HTTPS URL.
            max_urls (Optional[int]): Maximum URLs to process (default: 10)
            show_full_text (Optional[bool]): Include full text in output (default: False)
            cache (Optional[bool]): Whether to use cached content if available (default: True)
            experimental_stream (Optional[bool]): Enable experimental streaming

        Returns:
            GenerateLLMsTextResponse: A response containing:
            * success (bool): Whether the generation initiation was successful
            * id (str): The unique identifier for the generation job
            * error (str, optional): Error message if initiation failed

        Raises:
            Exception: If the generation job initiation fails.
        r  rJ   TrQ  rO  r   /v1/llmstxt	json_datark  r   rZ  zstart LLMs.txt generationNFInternal server errorr  r2   )r  r\  r^  r]  r  r6  rC   r   re  r&  r   rf  r<  r1   )r>  rJ   r  r  r  r  r  r\   r  reqrk  r  s               r"   r  z%FirecrawlApp.async_generate_llms_text  s   6 ('"5	
 '')CQ6;;4;#PQ	 +G95	(	%$$~[%A9gVCxxzH+y)*h'||I&S3?h?? ""8-HI ()
 	
S#$QRR  	%SV$$	%s1   AC 
C	 )C 	CC 	C;"C66C;c                    | j                         }	 | j                  | j                   d| |      }|j                  dk(  r	 |j	                         }t        di |S |j                  dk(  rt        d      | j                  |d       	 t        dd	d
d      S # t        $ r}t        dt        |             d}~ww xY w# t        $ r}t        t        |            d}~ww xY w)a;  
        Check the status of a LLMs.txt generation operation.

        Args:
            id (str): The unique identifier of the LLMs.txt generation job to check status for.

        Returns:
            GenerateLLMsTextStatusResponse: A response containing:
            * success (bool): Whether the generation was successful
            * status (str): Status of generation ("processing", "completed", "failed")
            * data (Dict[str, str], optional): Generated text with fields:
              * llmstxt (str): Generated LLMs.txt content
              * llmsfulltxt (str, optional): Full version if requested
            * error (str, optional): Error message if generation failed
            * expiresAt (str): When the generated data expires

        Raises:
            Exception: If the status check fails.
        /v1/llmstxt/rX  zFFailed to parse Firecrawl response as GenerateLLMsTextStatusResponse: N  z!LLMs.txt generation job not foundz check LLMs.txt generation statusFr  r^   r  r  r2   )
r\  r  r6  rd  rC   r,  r   r1   rf  r<  )r>  r   r\   rk  r  r  s         r"   r  z,FirecrawlApp.check_generate_llms_text_statusI  s    ( '')	%((DLL>bT)JGTH##s*w (I9FIFF %%, CDD""8-OP .eCZckwyzz ! w#&lmpqrmslt$uvvw  	%SV$$	%s;   0C B ,C 	B?#B::B??C 	C$CC$c                 P    |rdd| j                    |dS dd| j                    dS )a$  
        Prepare the headers for API requests.

        Args:
            idempotency_key (Optional[str]): A unique key to ensure idempotency of requests.

        Returns:
            Dict[str, str]: The headers including content type, authorization, and optionally idempotency key.
        zapplication/jsonrr  )Content-Typerq  zx-idempotency-key)r  rq  )r5  )r>  r  s     r"   r\  zFirecrawlApp._prepare_headerso  sB      2#*4<<.!9%4  /&t||n5
 	
r3   r   retriesbackoff_factorc                     t        |      D ]]  }t        j                  |||d|v r|d   |d   dz  dz   nd      }|j                  dk(  rt	        j
                  |d|z  z         [|c S  S )a^  
        Make a POST request with retries.

        Args:
            url (str): The URL to send the POST request to.
            data (Dict[str, Any]): The JSON data to include in the POST request.
            headers (Dict[str, str]): The headers to include in the POST request.
            retries (int): Number of retries for the request.
            backoff_factor (float): Backoff factor for retries.

        Returns:
            requests.Response: The response from the POST request.

        Raises:
            requests.RequestException: If the request fails after the specified retries.
        rt   NrV  r  rW    rx  )rangerb  rc  rd  r  r  )r>  rJ   r   r\   r  r  attemptrk  s           r"   r  zFirecrawlApp._post_request  s    . W~ 	 G}}S'qz  C  rC  HL  MV  HW  HcPTU^P_bhPhklPl  im  oH##s*

>Q'\:;	  r3   c                     t        |      D ]G  }t        j                  ||      }|j                  dk(  rt	        j
                  |d|z  z         E|c S  S )a	  
        Make a GET request with retries.

        Args:
            url (str): The URL to send the GET request to.
            headers (Dict[str, str]): The headers to include in the GET request.
            retries (int): Number of retries for the request.
            backoff_factor (float): Backoff factor for retries.

        Returns:
            requests.Response: The response from the GET request.

        Raises:
            requests.RequestException: If the request fails after the specified retries.
        r\   r  rx  )r  rb  re  rd  r  r  r>  rJ   r\   r  r  r  rk  s          r"   r  zFirecrawlApp._get_request  sV    * W~ 	 G||C9H##s*

>Q'\:;	  r3   c                     t        |      D ]G  }t        j                  ||      }|j                  dk(  rt	        j
                  |d|z  z         E|c S  S )a  
        Make a DELETE request with retries.

        Args:
            url (str): The URL to send the DELETE request to.
            headers (Dict[str, str]): The headers to include in the DELETE request.
            retries (int): Number of retries for the request.
            backoff_factor (float): Backoff factor for retries.

        Returns:
            requests.Response: The response from the DELETE request.

        Raises:
            requests.RequestException: If the request fails after the specified retries.
        r  r  rx  )r  rb  deleterd  r  r  r  s          r"   r  zFirecrawlApp._delete_request  sV    * W~ 	 GsG<H##s*

>Q'\:;	  r3   c                    	 | j                    d| }| j                  ||      }|j                  dk(  r	 |j                         }|d   dk(  rd|v rw|d   }d|v r^t        |d         dk(  rnL| j                  |d   |      }	 |j                         }|j                  |j                  dg              d|v r^||d<   t        di |S t	        d	      |d   d
v r"t        |d      }t        j                  |       n#t	        d|d          | j                  |d       #  t	        d      xY w#  t	        d      xY w)a  
        Monitor the status of a crawl job until completion.

        Args:
            id (str): The ID of the crawl job.
            headers (Dict[str, str]): The headers to include in the status check requests.
            poll_interval (int): Seconds between status checks.

        Returns:
            CrawlStatusResponse: The crawl results if the job is completed successfully.

        Raises:
            Exception: If the job fails or an error occurs during status checks.
        r  rX  rZ  r   r]   r   r   r   z,Crawl job completed but no data was returnedactivepausedpendingqueuedwaitingr   rx  z)Crawl job failed or was stopped. Status: r  r2   )r6  r  rd  rC   r   r  r  re  r   maxr  r  rf  )r>  r   r\   r  r6  r  r  r   s           r"   r  z FirecrawlApp._monitor_job_status  s   & j5G"//AO**c1T"1"6"6"8K x(K7,*62$3";v#671< %.2.?.?F@SU\.]O`.=.B.B.D !KK(CD %3 /3F+2A[AA'(VWW *.nn"%mA"6MJJ}-#&OP[\dPeOf$ghh""?4HI? T#&QSS`&/2]&_ _s   D! D1 !D.1D>rk  rj  c                    	 |j                         }|j                  dd      }|j                  dd      }| j                  |j                  |||      }t        j                  j                  ||      #  	 |j                  dd }|j                         rd| }d|j                   }nd	|j                   }d
}n # t
        $ r d|j                   }d
}Y nw xY wY xY w)ah  
        Handle errors from API responses.

        Args:
            response (requests.Response): The response object from the API request.
            action (str): Description of the action that was being performed.

        Raises:
            Exception: An exception with a message containing the status code and error details from the response.
        r   No error message provided.details%No additional error details provided.Nr  z#Server returned non-JSON response: zFull response status: z+Server returned empty response with status zNo additional details availablez0Server returned unreadable response with status )rk  )
rC   re  r   r   rd  r<  _get_error_messagerb  
exceptions	HTTPError)r>  rk  rj  rl  error_messageerror_detailsresponse_textmessages           r"   rf  zFirecrawlApp._handle_error	  s   	B$MMOM)--g7STM)--i9`aM ))(*>*>Wde !!++Gh+GG#	B
B (ds 3 &&(&I-$YM&<X=Q=Q<R$SM&QRZRfRfQg$hM$EM B"RS[SgSgRh i ABs0   4A5 5C 8AB>=C >CC CC rd  r  r  c                     |dk(  rd| d| d| S |dk(  rd| d| d| S |dk(  rd| d	| d| S |d
k(  rd| d| d| S |dk(  rd| d| d| S d| d| d| d| S )a  
        Generate a standardized error message based on HTTP status code.
        
        Args:
            status_code (int): The HTTP status code from the response
            action (str): Description of the action that was being performed
            error_message (str): The error message from the API response
            error_details (str): Additional error details from the API response
            
        Returns:
            str: A formatted error message
        i  zPayment Required: Failed to z. z - i  z!Website Not Supported: Failed to i  zRequest Timeout: Failed to z as the request timed out. i  zConflict: Failed to z due to a conflict. r  z!Internal Server Error: Failed to zUnexpected error during z: Status code r2   r>  rd  rj  r  r  s        r"   r  zFirecrawlApp._get_error_message9	  s     #1&M?#m_]]C6vhbsS`RabbC08STaSbbefsetuuC)&1Em_TWXeWfggC6vhbsS`Rabb-fX^K=PRS`Raaderdsttr3   )r{  
time_limitr  analysis_promptr/  (_FirecrawlApp__experimental_stream_stepson_activity	on_sourcer  r  r  r  r  c                T   i }
|||
d<   |||
d<   |||
d<   |||
d<   |||
d<   |||
d<   t        di |
}
| j                  ||||||      }|j                  d	      rd
|vr|S |d
   }d}d}	 | j                  |      }|r)d|v r%|d   |d }|D ]
  } ||        t	        |d         }|	r)d|v r%|d   |d }|D ]
  } |	|        t	        |d         }|d   dk(  r|S |d   dk(  rt        d|j                  d             |d   dk7  rnt        j                  d       dddS a  
        Initiates a deep research operation on a given query and polls until completion.

        Args:
            query (str): Research query or topic to investigate
            max_depth (Optional[int]): Maximum depth of research exploration
            time_limit (Optional[int]): Time limit in seconds for research
            max_urls (Optional[int]): Maximum number of URLs to process
            analysis_prompt (Optional[str]): Custom prompt for analysis
            system_prompt (Optional[str]): Custom system prompt
            __experimental_stream_steps (Optional[bool]): Enable experimental streaming
            on_activity (Optional[Callable]): Progress callback receiving {type, status, message, timestamp, depth}
            on_source (Optional[Callable]): Source discovery callback receiving {url, title, description}

        Returns:
            DeepResearchStatusResponse containing:
            * success (bool): Whether research completed successfully
            * status (str): Current state (processing/completed/failed)
            * error (Optional[str]): Error message if failed
            * id (str): Unique identifier for the research job
            * data (Any): Research findings and analysis
            * sources (List[Dict]): List of discovered sources
            * activities (List[Dict]): Research progress log
            * summaries (List[str]): Generated research summaries

        Raises:
            Exception: If research fails
        Nr   r  r  r  r   __experimental_streamSteps)r{  r  r  r  r/  r   r   r   r#  r  r   r]   r^   zDeep research failed. Error: r   r  rx  Fz)Deep research job terminated unexpectedlyr  r2   )r  async_deep_researchre  check_deep_research_statusr  r   r  r  )r>  r  r{  r  r  r  r/  r  r  r  research_paramsrk  r  last_activity_countlast_source_countr   new_activitiesactivitynew_sourcessources                       r"   deep_researchzFirecrawlApp.deep_researchS	  s   P  *3OJ'!+5OK()1OI&&0?O,-$.;ON+&2<WO89,??++!+' , 
 ||I&$h*>O$44V<F|v5!'!56I6J!K . *H)*&)&*>&?#Y&0$Y/0A0BC) &Ff%&$'y(9$:!h;.!X-"?

7@S?T UVV!\1JJqM- 0 !+VWWr3   )r{  r  r  r  r/  r  c                `   i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   t        di |}| j                         }	d|i|j                  d	d	
      }
dt         |
d<   d|
v r3|
d   }|r,d|v r(t	        |d   d      r|d   j                         |
d   d<   	 | j                  | j                   d|
|	      }|j                  dk(  r	 |j                         S | j                  |d       	 dddS #  t        d      xY w# t        $ r}t        t        |            d}~ww xY w)  
        Initiates an asynchronous deep research operation.

        Args:
            query (str): Research query or topic to investigate
            max_depth (Optional[int]): Maximum depth of research exploration
            time_limit (Optional[int]): Time limit in seconds for research
            max_urls (Optional[int]): Maximum number of URLs to process
            analysis_prompt (Optional[str]): Custom prompt for analysis
            system_prompt (Optional[str]): Custom system prompt
            __experimental_stream_steps (Optional[bool]): Enable experimental streaming

        Returns:
            Dict[str, Any]: A response containing:
            * success (bool): Whether the research initiation was successful
            * id (str): The unique identifier for the research job
            * error (str, optional): Error message if initiation failed

        Raises:
            Exception: If the research initiation fails.
        Nr   r  r  r  r   r  r  TrQ  rO  r   r   re   /v1/deep-researchrX  rZ  zstart deep researchFr  r  r2   )r  r\  r^  r]  hasattrre   r  r6  rd  rC   r   rf  r<  r1   )r>  r  r{  r  r  r  r/  r  r!  r\   r  	json_optsrk  r  s                 r"   r  z FirecrawlApp.async_deep_research	  s   >  *3OJ'!+5OK()1OI&&0?O,-$.;ON+&2<WO89,??'')e^';';TX\';']^	 +G95	( I%!-0IX2wy?RT\7]5>x5H5O5O5Q	-(2
	%))T\\N:K*LiY`aH##s*S#==?* ""8-BC !+BCCS#$QRR  	%SV$$	%s0   #/D C; #D ;DD 	D-D((D-c                 t   | j                         }	 | j                  | j                   d| |      }|j                  dk(  r	 |j	                         S |j                  dk(  rt        d      | j                  |d       	 dd	d
S #  t        d      xY w# t
        $ r}t        t        |            d}~ww xY w)  
        Check the status of a deep research operation.

        Args:
            id (str): The ID of the deep research operation.

        Returns:
            DeepResearchResponse containing:

            Status:
            * success - Whether research completed successfully
            * status - Current state (processing/completed/failed)
            * error - Error message if failed
            
            Results:
            * id - Unique identifier for the research job
            * data - Research findings and analysis
            * sources - List of discovered sources
            * activities - Research progress log
            * summaries - Generated research summaries

        Raises:
            Exception: If the status check fails.
        /v1/deep-research/rX  rZ  r  zDeep research job not foundzcheck deep research statusNFr  r  )	r\  r  r6  rd  rC   r   rf  r<  r1   )r>  r   r\   rk  r  s        r"   r   z'FirecrawlApp.check_deep_research_status	  s    2 '')	%((DLL>9KB4)PRYZH##s*S#==?* %%, =>>""8-IJ !+BCCS#$QRR
  	%SV$$	%s/   0B B ,B BB 	B7B22B7rg  method_namec           	          |syh dh dh dh dh dh dh dh dd}|j                  |t                     }t        |j                               |z
  }|r!t        d	| d
dj	                  |       d      y)a  
        Validate additional keyword arguments before they are passed to the API.
        This provides early validation before the Pydantic model validation.

        Args:
            kwargs (Dict[str, Any]): Additional keyword arguments to validate
            method_name (str): Name of the method these kwargs are for

        Raises:
            ValueError: If kwargs contain invalid or unsupported parameters
        N>   r   r~   rv   rR   rO   rn   rJ  rt   ru   rC  rF  ro  rA  r@  rH  rB  rE  rD  rI  >	   r  r  r   r	  rX   rt   ru   ro  rm  >   r   r   r{  ro  rz  ry  r  rm  r  r|  r}  r0  r  r  >   r   r   rt   ro  r  r  r  >   r   r+   re   ro  r2  r/  r1  r0  >   r   r~   rv   rR   rO   rn   r\   rt   r   ru   rC  rF  rA  r@  rH  rB  rE  rD  )rN  r   r  r  rO   r  r  r  zUnsupported parameter(s) for z: z, zC. Please refer to the API documentation for the correct parameters.)re  setkeysr<  r   )r>  rg  r1  method_paramsallowed_paramsunknown_paramss         r"   r[  zFirecrawlApp._validate_kwargs$
  s     T | } R"@(F,J%
2 '**;> V[[]+n<<[MDIIVdLeKf  gj  k  l  l r3   c                    ||S t        |t              r8t        |d      r|j                         S t        |d      r|j	                         S t        |t
              r3|j                         D ci c]  \  }}|| j                  |       c}}S t        |t        t        f      r|D cg c]  }| j                  |       c}S |S c c}}w c c}w )zw
        Utility to ensure a schema is a dict, not a Pydantic model class. Recursively checks dicts and lists.
        model_json_schemare   )
r`  r   r,  r9  re   r^  itemsr_  listtuple)r>  re   kvs       r"   r_  z FirecrawlApp._ensure_schema_dictW
  s     >Mfd#v23//11*}}&fd#?E||~Ntq!At//22NNftUm,9?@AD,,Q/@@ O@s   0C*CNNN         ?)Lr,   r-   r.   r   r1   r?  r   r	   r   r   r   rW   r   r   r   r   r   r   r   r   r   r   r   rc   r(   r   r   rN  ri   r  r   r[   r   r  r   r  r  r   r  r  r  r   r  r   r  r   r  r  r  r  r   rO   r  r  r,  r  r&  r  r  r\  r   rb  Responser  r  r  r  rf  r  r   r!  r(  r  r   r[  r_  r2   r3   r"   r4  r4    s   O Ox} OX\ O, mq04040404&*%*15%)4837(,CG(,,015 swGK%)-126,03Y7Y7 d7  ,g  $h  i  j	Y7
 d38n-Y7 #49-Y7 #49-Y7  (~Y7 smY7 c]Y7 ~.Y7 TNY7 $,D>Y7 #+4.Y7  ~Y7  G$>?@!Y7"  ~#Y7$ j)%Y7& #:.'Y7( d55E{T_alnz  }I  Kb  dm  *m  $n  o  p)Y7* &..C%D+Y7, c]-Y7. %TN/Y70 "*$1Y72 L)3Y74 (,5Y7~ $(!%$("&%)&*%*6:]3]3 C=	]3
 #]3 SM]3 3-]3 c]]3 sm]3 c]]3 %]3]3 (]3F .2-1#'-1#/3.2/3)-267;3726,0#+/)-.2'()-/@<@<  S	*	@<
  S	*@< C=@< &c]@< }@< 'tn@< &d^@< 'tn@< !@< !/@< %] 234@< #+4.@<  "*$!@<" $D>#@<$ }%@<& #4.'@<( "#)@<* &d^+@<,  }-@<. "#/@<2 
3@<L .2-1#'-1#/3.2/3)-267;3726,0#+/)-.2)--z<z<  S	*	z<
  S	*z< C=z< &c]z< }z< 'tnz< &d^z< 'tnz< !z< !/z< %] 234z< #+4.z<  "*$!z<" $D>#z<$ }%z<& #4.'z<( "#)z<* &d^+z<, "#-z<0 
1z<xQ?S Q?-@ Q?f?S ?-@ ?:=s =tCH~ =: 2615'+15#'372637-16:;?7;6:04#'/3-126-1-S9S9 $DI.	S9
 $DI.S9  }S9 "*#S9 C=S9 #+4.S9 "*$S9 #+4.S9 %TNS9 %]3S9 eC$678S9 '/tnS9  &.d^!S9"  (~#S9$ C=%S9& 'tn'S9( &c])S9* "*$+S9, &c]-S90 
1S9r %)-115+/#'%*(,X0X0 SM	X0
 %TNX0 !)X0 #4.X0 C=X0 c]X0  ~X0 %X0| W[,0,0,0,0"&!&-1!%04/3$(?C(,-1 os(,'()-.2)-1KC3iKC $w  (Q   R  S  T	KC
 $sCx.)KC tCy)KC tCy)KC $D>KC 3-KC #KC >*KC KC  (~KC 'tnKC D>KC   :;<!KC" *%#KC$ z*%KC& $uZ1A;P[]hjv  yE  G^  `i  &i   j  k  l'KC( %)KC*  }+KC, "#-KC. &d^/KC0 "#1KC4 
#5KCb W[,0,0,0,0"&!&-1!%04/3$(?C(,-1 os(,)-)-.2/JC3iJC $w  (Q   R  S  T	JC
 $sCx.)JC tCy)JC tCy)JC $D>JC 3-JC #JC >*JC JC  (~JC 'tnJC D>JC   :;<!JC" *%#JC$ z*%JC& $uZ1A;P[]hjv  yE  G^  `i  &i   j  k  l'JC( %)JC* "#+JC, "#-JC. &d^/JC2 
3JC` W[,0,0,0,0"&!&-1!%04/3$(?C(,-1 os(,)-.2)-/JC3iJC $w  (Q   R  S  T	JC
 $sCx.)JC tCy)JC tCy)JC $D>JC 3-JC #JC >*JC JC  (~JC 'tnJC D>JC   :;<!JC" *%#JC$ z*%JC& $uZ1A;P[]hjv  yE  G^  `i  &i   j  k  l'JC( %)JC* "#+JC, &d^-JC. "#/JC2 
3JCX<FC <F4M <F|FC F4G F> )-qN %)$(+/3805+0.2qN49%qN SM	qN
 SMqN $C=qN #+4.qN  (~qN #4.qN DcN+qN )-qNf* *1E *8 )-B* %)$(+/3805+0.2B*49%B* SM	B*
 SMB* $C=B* #+4.B*  (~B* #4.B* DcN+B* 8Gs7KB*P '+-1$(26CC sm	C
 %TNC D>C "*$C <ZCR '+-1$(268
8
 sm	8

 %TN8
 D>8
 "*$8
 <T8
t${# ${:X ${P .2
%c]
6:38n
: $' sCx. #s(^	
  " -5,=,=F $' #s(^ 	
 "
 -5,=,=B $' #s(^ 	
 "
 -5,=,=:2J2J #s(^2J 	2J $7	2Jh#H''#H #H !#HJuc u3 us ucf ukn u< (,(,&*-1+/:>FJDH^X^X  }	^X
 !^X sm^X &c]^X $C=^X *2$^X "(DcN+;T+A"BC^X  $sCx.)94)? @A^X Nh^XH (,(,&*-1+/:>EDED  }	ED
 !ED smED &c]ED $C=ED *2$ED DHS>EDN(DS (D5O (DT.ltCH~ .lC .lD .lfr3   r4  c                       e Zd ZdZdedefdZddZddZd	ed
e	e
eef   gdf   ddfdZd	ede
eef   ddfdZde
eef   ddfdZy)r  a  
    A class to watch and handle crawl job events via WebSocket connection.

    Attributes:
        id (str): The ID of the crawl job to watch
        app (FirecrawlApp): The FirecrawlApp instance
        data (List[Dict[str, Any]]): List of crawled documents/data
        status (str): Current status of the crawl job
        ws_url (str): WebSocket URL for the crawl job
        event_handlers (dict): Dictionary of event type to list of handler functions
    r   appc                     || _         || _        g | _        d| _        |j                  j                  dd       d| | _        g g g d| _        y )Nr   httpwsr  )doner   document)r   rF  r   r   r6  replacews_urlevent_handlers)r>  r   rF  s      r"   r?  zCrawlWatcher.__init__u
  sU    *,	 ,,VT:;:bTJ
r3   r7  Nc                   K   t        j                  | j                  ddd| j                  j                   fg      4 d{   }| j                  |       d{    ddd      d{    y7 .7 7 	# 1 d{  7  sw Y   yxY ww)zU
        Establishes WebSocket connection and starts listening for messages.
        Nrq  rr  )max_sizeadditional_headers
websocketsconnectrM  rF  r5  _listenr>  	websockets     r"   rT  zCrawlWatcher.connect
  s      %%KK!0GDHH<L<L;M2N OP
 	* 	* ,,y)))	* 	* 	*
 *	* 	* 	* 	*sZ   ABA2BA8A4A8!B,A6-B4A86B8B
>B?B
Bc                    K   |2 3 d{   }t        j                  |      }| j                  |       d{    87 37 6 yw)z
        Listens for incoming WebSocket messages and handles them.

        Args:
            websocket: The WebSocket connection object
        NrC   loads_handle_messager>  rW  r  msgs       r"   rU  zCrawlWatcher._listen
  H      ' 	, 	,'**W%C&&s+++	,+ '1   AA=A*A?AAAA
event_typehandlerc                 ^    || j                   v r| j                   |   j                  |       yy)z
        Adds an event handler function for a specific event type.

        Args:
            event_type (str): Type of event to listen for ('done', 'error', or 'document')
            handler (Callable): Function to handle the event
        N)rN  append)r>  r`  ra  s      r"   add_event_listenerzCrawlWatcher.add_event_listener
  s0     ,,,
+227; -r3   detailc                 Z    || j                   v r| j                   |   D ]
  } ||        yy)z
        Dispatches an event to all registered handlers for that event type.

        Args:
            event_type (str): Type of event to dispatch
            detail (Dict[str, Any]): Event details/data to pass to handlers
        N)rN  )r>  r`  re  ra  s       r"   dispatch_eventzCrawlWatcher.dispatch_event
  s8     ,,,..z:    -r3   r]  c                   K   |d   dk(  r<d| _         | j                  d| j                   | j                  | j                  d       y|d   dk(  r@d| _         | j                  d| j                   | j                  |d   | j                  d       y|d   dk(  rl|d	   d
   | _         | j                  j	                  |d	   j                  d	g              | j                  D ]!  }| j                  d|| j                  d       # y|d   dk(  rA| j                  j                  |d	          | j                  d|d	   | j                  d       yyw)z
        Handles incoming WebSocket messages based on their type.

        Args:
            msg (Dict[str, Any]): The message to handle
        r   rJ  r]   r   r   r   r   r^   r   r   r   r   catchupr   r   rK  r   r   Nr   rg  r   r   r  re  rc  r>  r]  docs      r"   r[  zCrawlWatcher._handle_message
  2     v;& %DK4;;		Y]Y`Y`(ab[G#"DKDKK]`ah]iquqxqx)yz[I%f+h/DKIIS[__VR89yy N##JDGG0LMN[J&IIS[)
S[,PQ '   EEr7  N)r,   r-   r.   r/   r1   r4  r?  rT  rU  r   r   r   rd  rg  r[  r2   r3   r"   r  r  i
  s    


3 

\ 

	*	,	<S 	<8T#s(^DTVZDZ;[ 	<`d 	<
  
 d38n 
  
 Rc3h RD Rr3   r  c            0       z   e Zd ZdZ	 	 	 d|dededeeef   deeeef      dede	d	eeef   fd
Z
	 d}dedeeef   deeef   dede	d	eeef   fdZ	 d}dedeeef   dede	d	eeef   f
dZdej                  ded	dfdZdedededed	ef
dZ	 	 d~dedee   dee   d	dfdZ	 	 d~dee   dee   dee   d	dfdZddddddddddddddddddddedeeed         deeeef      deee      deee      d ee   d!ee   d"ee   d#ee   d$ee   d%ee   d&ee   d'ee   d(eed)      d*ee   d+ee   d,ee   d-eeeeeeeee e!e"e#f	         d.ee$   d	e%e   f(d/Z&dddddddddddddddddd0dd1dee   deeed2         deeeef      deee      deee      d ee   d!ee   d"ee   d#ee   d$ee   d%ee   d&ee   d'ee   d(eed)      d+ee   d,ee   d-eeeeeeeee e!e"e#f	         d.ee$   d3ee   dee   d	e'f*d4Z(dddddddddddddddddddd5dee   deeed2         deeeef      deee      deee      d ee   d!ee   d"ee   d#ee   d$ee   d%ee   d&ee   d'ee   d(eed)      d+ee   d,ee   d-eeeeeeeee e!e"e#f	         d.ee$   d6ee   dee   d	e)f*d7Z*ddddddddddddddddd0dd8ded9eee      d:eee      d;ee   d<ee   d=ee   d>ee   d?ee   d@ee   dAee   dBee+   dCeeee,f      dDee   dEee   dFee   dGee   dHee   d3ee   dee   d	e-f(dIZ.ddddddddddddddddd0dd8ded9eee      d:eee      d;ee   d<ee   d=ee   d>ee   d?ee   d@ee   dAee   dBee+   dCeeee,f      dDee   dEee   dFee   dGee   dHee   d3ee   dee   d	e/f(dJZ0dKed	e-fdLZ1ddKedeeef   d3ed	e-fdMZ2ddddddddNdedOee   dAee   dPee   dQee   d=ee   d"ee   dee3   d	e4fdRZ5	 dddddSdSdSddTdeee      dUee   dVee   dWee   d@ee   dXee   dYee   d.eeeef      d	e6e   fdZZ7dKed	e'fd[Z8dKed	e9fd\Z:dKed	e9fd]Z;dKed	eeef   fd^Z<d_ed	e6e   fd`Z=	 dddddSdSdSddTdeee      dUee   dVee   dWee   d@ee   dXee   dYee   d.eeeef      d	e6e   fdaZ>ddddbdedcee   ddee   deee   d	e?f
dfZ@dddddgdedcee   ddee   dhee   deee   d	eAfdiZBdKed	e?fdjZCdddddddddkdled;ee   dmee   dcee   dnee   dWee   doee   dpeeDeeef   gdf      dqeeDeeef   gdf      d	eEfdrZFdddddddsdled;ee   dmee   dcee   dnee   dWee   doee   d	eeef   fdtZGdKed	eEfduZHddddddddddv	dled=ee   dwee   dxee   dyee   dzee   d#ee   d"ee   dBee+   deeeeef   eIf      d	eJfd{ZKy)AsyncFirecrawlAppz
    Asynchronous version of FirecrawlApp that implements async methods using aiohttp.
    Provides non-blocking alternatives to all FirecrawlApp operations.
    NmethodrJ   r\   r   r  r  r7  c           	      &  K   t        j                         4 d{   }t        |      D ]  }	 |j                  ||||      4 d{   }	|	j                  dk(  r5t        j                  |d|z  z         d{    	 ddd      d{    f|	j                  dk\  r| j                  |	d| d       d{    |	j                          d{   cddd      d{    c cddd      d{    S  t        d	      7 7 7 7 7 T7 >7 07 # 1 d{  7  sw Y   nxY w# t         j                  $ r9}
||dz
  k(  r|
t        j                  |d|z  z         d{  7   Y d}
~
Id}
~
ww xY w# 1 d{  7  sw Y   yxY ww)
a  
        Generic async request method with exponential backoff retry logic.

        Args:
            method (str): The HTTP method to use (e.g., "GET" or "POST").
            url (str): The URL to send the request to.
            headers (Dict[str, str]): Headers to include in the request.
            data (Optional[Dict[str, Any]]): The JSON data to include in the request body (only for POST requests).
            retries (int): Maximum number of retry attempts (default: 3).
            backoff_factor (float): Factor to calculate delay between retries (default: 0.5).
                Delay will be backoff_factor * (2 ** retry_count).

        Returns:
            Dict[str, Any]: The parsed JSON response from the server.

        Raises:
            aiohttp.ClientError: If the request fails after all retries.
            Exception: If max retries are exceeded or other errors occur.
        N)ru  rJ   r\   rC   r  rx  i,  zmake z requestr   zMax retries exceeded)aiohttpClientSessionr  requestr   asyncior  rf  rC   ClientErrorr   )r>  ru  rJ   r\   r   r  r  sessionr  rk  r  s              r"   _async_requestz AsyncFirecrawlApp._async_request
  s    6 ((* 	4 	4g > II&%3d  /   5 5!#??c1")--!w,0O"PPP$5 5 5 $??c1"&"4"4Xvhh?W"XXX%-]]_45 5 5	4 	4 	4I 233!	45 Q	5 Y45	45 5 5 5 ** I'A+-!--!w,(GHHHI	4 	4 	4s-  FDFE<D-D
	D-.D:D

;D D-DD-E<(D9D
:DD
DD-!D
"D-&E<(F4D5F:E<FD-
DD-DDD-FD(DD($D-+E<-E9 (E4(E+
)E4.E<4E99E<<FFF
Fc                 J   K   | j                  d|||||       d{   S 7 w)a  
        Make an async POST request with exponential backoff retry logic.

        Args:
            url (str): The URL to send the POST request to.
            data (Dict[str, Any]): The JSON data to include in the request body.
            headers (Dict[str, str]): Headers to include in the request.
            retries (int): Maximum number of retry attempts (default: 3).
            backoff_factor (float): Factor to calculate delay between retries (default: 0.5).
                Delay will be backoff_factor * (2 ** retry_count).

        Returns:
            Dict[str, Any]: The parsed JSON response from the server.

        Raises:
            aiohttp.ClientError: If the request fails after all retries.
            Exception: If max retries are exceeded or other errors occur.
        POSTNr}  )r>  rJ   r   r\   r  r  s         r"   _async_post_requestz%AsyncFirecrawlApp._async_post_request
  s)     * ((gtWn]]]]   #!#c                 J   K   | j                  d||d||       d{   S 7 w)a  
        Make an async GET request with exponential backoff retry logic.

        Args:
            url (str): The URL to send the GET request to.
            headers (Dict[str, str]): Headers to include in the request.
            retries (int): Maximum number of retry attempts (default: 3).
            backoff_factor (float): Factor to calculate delay between retries (default: 0.5).
                Delay will be backoff_factor * (2 ** retry_count).

        Returns:
            Dict[str, Any]: The parsed JSON response from the server.

        Raises:
            aiohttp.ClientError: If the request fails after all retries.
            Exception: If max retries are exceeded or other errors occur.
        GETNr  )r>  rJ   r\   r  r  s        r"   _async_get_requestz$AsyncFirecrawlApp._async_get_request  s)     ( ((WdG^\\\\r  rk  rj  c                 P  K   	 |j                          d{   }|j                  dd      }|j                  dd      }| j                  |j                  |||       d{   }t        j                  |      7 c#  t        j                  d|j                         xY w7 Bw)aR  
        Handle errors from async API responses with detailed error messages.

        Args:
            response (aiohttp.ClientResponse): The response object from the failed request
            action (str): Description of the action that was being attempted

        Raises:
            aiohttp.ClientError: With a detailed error message based on the response status:
                - 402: Payment Required
                - 408: Request Timeout
                - 409: Conflict
                - 500: Internal Server Error
                - Other: Unexpected error with status code
        Nr   r  r  r  ?Failed to parse Firecrawl error response as JSON. Status code: )rC   re  rw  r{  r   _get_async_error_messager>  rk  rj  
error_datar  r  r  s          r"   rf  zAsyncFirecrawlApp._handle_error$  s      	{'}}.J&NN74PQM&NN96]^M 55hoov}^kll!!'** /	{%%(ghphwhwgx&yzzls8   B&A= A;(A=  !B&!B$"B&;A= =$B!!B&rd  r  r  c                 2   K   | j                  ||||      S wa  
        Generate a standardized error message based on HTTP status code for async operations.
        
        Args:
            status_code (int): The HTTP status code from the response
            action (str): Description of the action that was being performed
            error_message (str): The error message from the API response
            error_details (str): Additional error details from the API response
            
        Returns:
            str: A formatted error message
        r  r  s        r"   r  z*AsyncFirecrawlApp._get_async_error_message?        &&{FM=YY   r  r  AsyncCrawlWatcherc                    K   | j                  |||       d{   }|j                  d      rd|v rt        |d   |       S t        d      7 3w)a  
        Initiate an async crawl job and return an AsyncCrawlWatcher to monitor progress via WebSocket.

        Args:
          url (str): Target URL to start crawling from
          params (Optional[CrawlParams]): See CrawlParams model for configuration:
            URL Discovery:
            * includePaths - Patterns of URLs to include
            * excludePaths - Patterns of URLs to exclude
            * maxDepth - Maximum crawl depth
            * maxDiscoveryDepth - Maximum depth for finding new URLs
            * limit - Maximum pages to crawl

            Link Following:
            * allowBackwardLinks - DEPRECATED: Use crawlEntireDomain instead
            * crawlEntireDomain - Follow parent directory links
            * allowExternalLinks - Follow external domain links  
            * ignoreSitemap - Skip sitemap.xml processing

            Advanced:
            * scrapeOptions - Page scraping configuration
            * webhook - Notification webhook settings
            * deduplicateSimilarURLs - Remove similar URLs
            * ignoreQueryParameters - Ignore URL parameters
            * regexOnFullURL - Apply regex to full URLs
          idempotency_key (Optional[str]): Unique key to prevent duplicate requests

        Returns:
          AsyncCrawlWatcher: An instance to monitor the crawl job via WebSocket

        Raises:
          Exception: If crawl job fails to start
        Nr   r   r  )r  re  r  r   )r>  rJ   r  r  r  s        r"   r  z%AsyncFirecrawlApp.crawl_url_and_watchN  sY     L  $33CQQi(T^-C$^D%94@@788	 R   AA4Ar  c                    K   | j                  |||       d{   }|j                  d      rd|v rt        |d   |       S t        d      7 3w)a  
        Initiate an async batch scrape job and return an AsyncCrawlWatcher to monitor progress.

        Args:
            urls (List[str]): List of URLs to scrape
            params (Optional[ScrapeParams]): See ScrapeParams model for configuration:

              Content Options:
              * formats - Content formats to retrieve
              * includeTags - HTML tags to include
              * excludeTags - HTML tags to exclude
              * onlyMainContent - Extract main content only
              
              Request Options:
              * headers - Custom HTTP headers
              * timeout - Request timeout (ms)
              * mobile - Use mobile user agent
              * proxy - Proxy type
              
              Extraction Options:
              * extract - Content extraction config
              * jsonOptions - JSON extraction config
              * actions - Actions to perform
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests

        Returns:
            AsyncCrawlWatcher: An instance to monitor the batch scrape job via WebSocket

        Raises:
            Exception: If batch scrape job fails to start
        Nr   r   r  )r  re  r  r   )r>  r  r  r  batch_responses        r"   r  z-AsyncFirecrawlApp.batch_scrape_urls_and_watchz  sY     H  $;;D&/ZZi(T^-C$^D%94@@>??	 [r  rs   )rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rG  rO   rH  rR   r   rn   rk   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rz   rG  rO   rH  rR   r   c                ^  K   | j                  |d       | j                         }|dt         d}|r||d<   |r||d<   |r||d<   |r||d<   |||d	<   |r||d
<   |r||d<   |	r|	j                  dd      |d<   |
|
|d<   |||d<   |||d<   |||d<   |r||d<   |||d<   |d| j	                  |      }t        |t              rd|v r| j	                  |d         |d<   t        |t              r|n|j                  dd      |d<   |d| j	                  |      }t        |t              rd|v r| j	                  |d         |d<   t        |t              r|n|j                  dd      |d<   |r6|D cg c]'  }t        |t              r|n|j                  dd      ) c}|d<   ||j                  dd      |d<   d|v r)|d   r$d|d   v r| j	                  |d   d         |d   d<   d|v r)|d   r$d|d   v r| j	                  |d   d         |d   d<   d}| j                  | j                   | ||       d{   }|j                  d      rd|v rt        di |d   S d|v rt        d|d          |j                  dt        |            }t        d|       c c}w 7 jw) av  
        Scrape a single URL asynchronously.

        Args:
          url (str): Target URL to scrape
          formats (Optional[List[Literal["markdown", "html", "rawHtml", "content", "links", "screenshot", "screenshot@fullPage", "extract", "json"]]]): Content types to retrieve (markdown/html/etc)
          headers (Optional[Dict[str, str]]): Custom HTTP headers
          include_tags (Optional[List[str]]): HTML tags to include
          exclude_tags (Optional[List[str]]): HTML tags to exclude
          only_main_content (Optional[bool]): Extract main content only
          wait_for (Optional[int]): Wait for a specific element to appear
          timeout (Optional[int]): Request timeout (ms)
          location (Optional[LocationConfig]): Location configuration
          mobile (Optional[bool]): Use mobile user agent
          skip_tls_verification (Optional[bool]): Skip TLS verification
          remove_base64_images (Optional[bool]): Remove base64 images
          block_ads (Optional[bool]): Block ads
          proxy (Optional[Literal["basic", "stealth", "auto"]]): Proxy type (basic/stealth)
          extract (Optional[JsonConfig]): Content extraction settings
          json_options (Optional[JsonConfig]): JSON extraction settings
          actions (Optional[List[Union[WaitAction, ScreenshotAction, ClickAction, WriteAction, PressAction, ScrollAction, ScrapeAction, ExecuteJavascriptAction, PDFAction]]]): Actions to perform
          agent (Optional[AgentOptions]): Agent configuration for FIRE-1 model
          **kwargs: Additional parameters to pass to the API

        Returns:
            ScrapeResponse with:
            * success - Whether scrape was successful
            * markdown - Markdown content if requested
            * html - HTML content if requested
            * rawHtml - Raw HTML content if requested
            * links - Extracted links if requested
            * screenshot - Screenshot if requested
            * extract - Extracted data if requested
            * json - JSON data if requested
            * error - Error message if scrape failed

        Raises:
            Exception: If scraping fails
        rN  rO  rP  rn   r\   ro   rp   Nrq   rr   rt   TrQ  ru   rv   rw   rx   ry   r~   r   re   rO   r   rR   r   rU  r   r   r   rY  r2   )r[  r\  r]  r^  r_  r`  r  r6  re  r   r   r1   )r>  rJ   rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rG  rO   rH  rR   r   rg  rh  ri  rj  r  rk  error_contents                              r"   rN  zAsyncFirecrawlApp.scrape_url  s    ~ 	fl3((* #G9-
 '.M)$'.M)$+7M-(+7M-((/@M+,'/M)$'.M)$(0tRV(WM*%&,M(# ,3HM/0+2FM./ (1M*%%*M'" (1M*%..w7G'4(X-@$($<$<WX=N$O!2<Wd2KwQXQ]Q]gkz~Q]QM)$#33LAL,-(l2J)-)A)A,xBX)YX&;ElTX;Y<_k_p_pz~  NR_p  `SM-( MT  (U  CI*VT2JPVP[P[eix|P[P}(}  (UM)$%*ZZDZ%QM'"%-	*BxS`ajSkGk151I1I-XaJbckJl1mM)$X.M)mM.Jx[hiv[wOw595M5Mm\iNjksNt5uM-(2  11||nXJ'
 
 <<	"v'9!5HV$455 ;HW<M;NOPP %LL#h-@M;M?KLL/ (U
s    E4J-6,J&"BJ- J+A+J-rx  )rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rO   rH  rR   r   r  r  r  r  c                  K   | j                  |d       i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d	<   |	|	j                  d
d
      |d<   |
|
|d<   |||d<   |||d<   |||d<   |||d<   |d| j                  |      }t        |t              rd|v r| j                  |d         |d<   t        |t              r|n|j                  d
d
      |d<   |d| j                  |      }t        |t              rd|v r| j                  |d         |d<   t        |t              r|n|j                  d
d
      |d<   |$|D cg c]  }|j                  d
d
       c}|d<   ||j                  d
d
      |d<   |j	                  |       t        di |}|j                  d
d
      }||d<   dt         |d<   d|v r)|d   r$d|d   v r| j                  |d   d         |d   d<   d|v r)|d   r$d|d   v r| j                  |d   d         |d   d<   | j                  |      }| j                  | j                   d||       d{   }|j                  d      r-	 |j                  d      }| j                  |||       d{   S | j                  |d       yc c}w 7 Z#  t        d      xY w7 .w) a  
        Asynchronously scrape multiple URLs and monitor until completion.

        Args:
            urls (List[str]): URLs to scrape
            formats (Optional[List[Literal]]): Content formats to retrieve
            headers (Optional[Dict[str, str]]): Custom HTTP headers
            include_tags (Optional[List[str]]): HTML tags to include
            exclude_tags (Optional[List[str]]): HTML tags to exclude
            only_main_content (Optional[bool]): Extract main content only
            wait_for (Optional[int]): Wait time in milliseconds
            timeout (Optional[int]): Request timeout in milliseconds
            location (Optional[LocationConfig]): Location configuration
            mobile (Optional[bool]): Use mobile user agent
            skip_tls_verification (Optional[bool]): Skip TLS verification
            remove_base64_images (Optional[bool]): Remove base64 encoded images
            block_ads (Optional[bool]): Block advertisements
            proxy (Optional[Literal]): Proxy type to use
            extract (Optional[JsonConfig]): Content extraction config
            json_options (Optional[JsonConfig]): JSON extraction config
            actions (Optional[List[Union]]): Actions to perform
            agent (Optional[AgentOptions]): Agent configuration
            poll_interval (Optional[int]): Seconds between status checks (default: 2)
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests
            **kwargs: Additional parameters to pass to the API

        Returns:
            BatchScrapeStatusResponse with:
            * Scraping status and progress
            * Scraped content for each URL
            * Success/error information

        Raises:
            Exception: If batch scrape fails
        r  Nrn   r\   ro   rp   rq   rr   rt   TrQ  ru   rv   rw   rx   ry   r~   re   rO   r   rR   r   r  rO  r   r  r   r   rZ  r  r2   )r[  r^  r_  r`  ra  r   r]  r\  r  r6  re  r   _async_monitor_job_statusrf  )r>  r  rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rO   rH  rR   r   r  r  rg  ri  rj  rv  rw  rk  r   s                               r"   r  z#AsyncFirecrawlApp.batch_scrape_urls.  s    z 	f&9: '.M)$'.M)$#+7M-(#+7M-((/@M+,'/M)$'.M)$(0tRV(WM*%&,M(# ,3HM/0+2FM./ (1M*%%*M'"..w7G'4(X-@$($<$<WX=N$O!2<Wd2KwQXQ]Q]gkz~Q]QM)$#33LAL,-(l2J)-)A)A,xBX)YX&;ElTX;Y<_k_p_pz~  NR_p  `SM-(dk'lZ`TPT(U'lM)$%*ZZDZ%QM'" 	V$ $4m4"''D'I"F"-gY 7H#I(>8{[dOeCe/3/G/GT]H^_gHh/iK	"8,K'K,F8WbcpWqKq373K3KKXeLfgoLp3qK&x0 ''811||n,-
 
 <<	"P\\$' 77G]SSSx)ABC (m(
P"MOOSsC   EK
J1/C,K
J6K
2J8 K
KK
8KK
)rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rO   rH  rR   r   rL  r  rL  c                  K   | j                  |d       i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d	<   |	|	j                  d
d
      |d<   |
|
|d<   |||d<   |||d<   |||d<   |||d<   |d| j                  |      }t        |t              rd|v r| j                  |d         |d<   t        |t              r|n|j                  d
d
      |d<   |d| j                  |      }t        |t              rd|v r| j                  |d         |d<   t        |t              r|n|j                  d
d
      |d<   |r6|D cg c]'  }t        |t              r|n|j                  d
d
      ) c}|d<   ||j                  d
d
      |d<   |||d<   |j	                  |       t        d i |}|j                  d
d
      }||d<   dt         |d<   d|v r)|d   r$d|d   v r| j                  |d   d         |d   d<   d|v r)|d   r$d|d   v r| j                  |d   d         |d   d<   | j                  |      }| j                  | j                   d||       d{   }|j                  d      dk(  r	 t        d i |j                         S | j                  |d       d{    yc c}w 7 R#  t        d      xY w7 w)!a*  
        Initiate a batch scrape job asynchronously.

        Args:
            urls (List[str]): URLs to scrape
            formats (Optional[List[Literal]]): Content formats to retrieve
            headers (Optional[Dict[str, str]]): Custom HTTP headers
            include_tags (Optional[List[str]]): HTML tags to include
            exclude_tags (Optional[List[str]]): HTML tags to exclude
            only_main_content (Optional[bool]): Extract main content only
            wait_for (Optional[int]): Wait time in milliseconds
            timeout (Optional[int]): Request timeout in milliseconds
            location (Optional[LocationConfig]): Location configuration
            mobile (Optional[bool]): Use mobile user agent
            skip_tls_verification (Optional[bool]): Skip TLS verification
            remove_base64_images (Optional[bool]): Remove base64 encoded images
            block_ads (Optional[bool]): Block advertisements
            proxy (Optional[Literal]): Proxy type to use
            extract (Optional[JsonConfig]): Content extraction config
            json_options (Optional[JsonConfig]): JSON extraction config
            actions (Optional[List[Union]]): Actions to perform
            agent (Optional[AgentOptions]): Agent configuration
            zero_data_retention (Optional[bool]): Whether to delete data after 24 hours
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests
            **kwargs: Additional parameters to pass to the API

        Returns:
            BatchScrapeResponse with:
            * success - Whether job started successfully
            * id - Unique identifier for the job
            * url - Status check URL
            * error - Error message if start failed

        Raises:
            Exception: If job initiation fails
        r  Nrn   r\   ro   rp   rq   rr   rt   TrQ  ru   rv   rw   rx   ry   r~   re   rO   r   rR   r   rT  r  rO  r   r  rd  rX  rZ  r  r2   )r[  r^  r_  r`  ra  r   r]  r\  r  r6  re  r   rC   r   rf  )r>  r  rn   r\   r@  rA  rB  rC  rt   ru   rv   rD  rE  rF  r~   rO   rH  rR   r   rL  r  rg  ri  rj  rv  rw  rk  s                              r"   r  z)AsyncFirecrawlApp.async_batch_scrape_urls  s    | 	f&?@ '.M)$'.M)$#+7M-(#+7M-((/@M+,'/M)$'.M)$(0tRV(WM*%&,M(# ,3HM/0+2FM./ (1M*%%*M'"..w7G'4(X-@$($<$<WX=N$O!2<Wd2KwQXQ]Q]gkz~Q]QM)$#33LAL,-(l2J)-)A)A,xBX)YX&;ElTX;Y<_k_p_pz~  NR_p  `SM-( MT  (U  CI*VT2JPVP[P[eix|P[P}(}  (UM)$%*ZZDZ%QM'"*1DM-. 	V$ $4m4"''D'I"F"-gY 7H#I(>8{[dOeCe/3/G/GT]H^_gHh/iK	"8,K'K,F8WbcpWqKq373K3KKXeLfgoLp3qK&x0 ''811||n,-
 
 <<&#-P*=X]]_== $$X/GHHHE (U,
P"MOOHsC   EK,KC3K4K5KK	 &K<K=K	KK)ry  rz  r{  r|  r   r}  r~  r0  r  rm  r   r  r  r  r   r  r  r  ry  rz  r{  r|  r   r}  r~  r0  r  rm  r   r  r  r  r   r  c                  K   | j                  |d       i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   n|||d	<   |	|	|d
<   |
|
|d<   ||j                  dd      |d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |j                  |       t        di |}|j                  dd      }||d<   dt         |d<   | j                  |      }| j                  | j                   d||       d{   }|j                  d      r-	 |j                  d      }| j                  |||       d{   S | j                  |d       d{    y7 ]#  t        d      xY w7 17 w)a
  
        Crawl a website starting from a URL.

        Args:
            url (str): Target URL to start crawling from
            include_paths (Optional[List[str]]): Patterns of URLs to include
            exclude_paths (Optional[List[str]]): Patterns of URLs to exclude
            max_depth (Optional[int]): Maximum crawl depth
            max_discovery_depth (Optional[int]): Maximum depth for finding new URLs
            limit (Optional[int]): Maximum pages to crawl
            allow_backward_links (Optional[bool]): DEPRECATED: Use crawl_entire_domain instead
            crawl_entire_domain (Optional[bool]): Follow parent directory links
            allow_external_links (Optional[bool]): Follow external domain links
            ignore_sitemap (Optional[bool]): Skip sitemap.xml processing
            scrape_options (Optional[ScrapeOptions]): Page scraping configuration
            webhook (Optional[Union[str, WebhookConfig]]): Notification webhook settings
            deduplicate_similar_urls (Optional[bool]): Remove similar URLs
            ignore_query_parameters (Optional[bool]): Ignore URL parameters
            regex_on_full_url (Optional[bool]): Apply regex to full URLs
            delay (Optional[int]): Delay in seconds between scrapes
            allow_subdomains (Optional[bool]): Follow subdomains
            poll_interval (Optional[int]): Seconds between status checks (default: 2)
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests
            **kwargs: Additional parameters to pass to the API

        Returns:
            CrawlStatusResponse with:
            * Crawling status and progress
            * Crawled page contents
            * Success/error information

        Raises:
            Exception: If crawl fails
        r  Nr   r   r   r   r   r   r   r   r   TrQ  r   r   r   r   r   r   r   rJ   rO  r   r  r   r   rZ  r  r2   )r[  r^  ra  r   r]  r\  r  r6  re  r   r  rf  )r>  rJ   ry  rz  r{  r|  r   r}  r~  r0  r  rm  r   r  r  r  r   r  r  r  rg  r  rv  rw  r\   rk  r   s                              r"   r  zAsyncFirecrawlApp.crawl_urlE  s2    v 	fk2 $+8L($+8L( '0L$*0CL,-$)L!*0CL,-!-1EL-.+1EL-.%,:L)%,:,?,?\`,?,aL)&-L##/5ML12".4KL01(->L)*$)L!'.>L*+ 	F# #2\2"''D'I E"-gY 7H''811\\N)
$k7< < <<	"P\\$' 77G]SSS$$X/@AAA<P"MOOSAsH   DFE+F$E- 5FE=F%E?&F-E::F?Fc                  K   i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   n|||d<   |	|	|d	<   |
|
|d
<   ||j                  dd      |d<   |||d<   |||d<   |||d<   |||d<   |||d<   |||d<   |j                  |       t        di |}|j                  dd      }||d<   dt         |d<   | j	                  |      }| j                  | j                   d||       d{   }|j                  d      r	 t        di |S | j                  |d       d{    y7 <#  t        d      xY w7 w)a  
        Start an asynchronous crawl job.

        Args:
            url (str): Target URL to start crawling from
            include_paths (Optional[List[str]]): Patterns of URLs to include
            exclude_paths (Optional[List[str]]): Patterns of URLs to exclude
            max_depth (Optional[int]): Maximum crawl depth
            max_discovery_depth (Optional[int]): Maximum depth for finding new URLs
            limit (Optional[int]): Maximum pages to crawl
            allow_backward_links (Optional[bool]): DEPRECATED: Use crawl_entire_domain instead
            crawl_entire_domain (Optional[bool]): Follow parent directory links
            allow_external_links (Optional[bool]): Follow external domain links
            ignore_sitemap (Optional[bool]): Skip sitemap.xml processing
            scrape_options (Optional[ScrapeOptions]): Page scraping configuration
            webhook (Optional[Union[str, WebhookConfig]]): Notification webhook settings
            deduplicate_similar_urls (Optional[bool]): Remove similar URLs
            ignore_query_parameters (Optional[bool]): Ignore URL parameters
            regex_on_full_url (Optional[bool]): Apply regex to full URLs
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests
            **kwargs: Additional parameters to pass to the API

        Returns:
            CrawlResponse with:
            * success - Whether crawl started successfully
            * id - Unique identifier for the crawl job
            * url - Status check URL for the crawl
            * error - Error message if start failed

        Raises:
            Exception: If crawl initiation fails
        Nr   r   r   r   r   r   r   r   r   TrQ  r   r   r   r   r   r   r   rJ   rO  r   r  r   rZ  r  r2   )r^  ra  r   r]  r\  r  r6  re  r   r   rf  )r>  rJ   ry  rz  r{  r|  r   r}  r~  r0  r  rm  r   r  r  r  r   r  r  r  rg  r  rv  rw  r\   rk  s                             r"   r  z!AsyncFirecrawlApp.async_crawl_url  s
    p  $+8L($+8L( '0L$*0CL,-$)L!*0CL,-!-1EL-.+1EL-.%,:L)%,:,?,?\`,?,aL)&-L##/5ML12".4KL01(->L)*$)L!'.>L*+ 	F# #2\2"''D'I E"-gY 7H ''811\\N)
$


 
 <<	"P$0x00 $$X/@AAA
P"MOOAs6   C9E;D8<E
D: E2E
3E:EEr   c           
      $  K   | j                         }d| }| j                  | j                   | |       d{   }|j                  d      dk(  rd|v r|d   }d|v r}t	        |d         dk(  rnk|j                  d      }|st
        j                  d       nB| j                  ||       d{   }|j                  |j                  dg              |}d|v r}||d<   t        |j                  d      |j                  d	      |j                  d      |j                  d
      |j                  d      |j                  d      d|v rdnd      }d|v r|j                  d      |_	        d|v r|j                  d      |_
        |S 7 P7 ܭw)a5  
        Check the status and results of an asynchronous crawl job.

        Args:
            id (str): Unique identifier for the crawl job

        Returns:
            CrawlStatusResponse containing:
            Status Information:
            * status - Current state (scraping/completed/failed/cancelled)
            * completed - Number of pages crawled
            * total - Total pages to crawl
            * creditsUsed - API credits consumed
            * expiresAt - Data expiration timestamp
            
            Results:
            * data - List of crawled documents
            * next - URL for next page of results (if paginated)
            * success - Whether status check succeeded
            * error - Error message if failed

        Raises:
            Exception: If status check fails
        r  Nr   r]   r   r   r   r  r   r   r   r   FT)r   r   r]   r   r   r   r   )r\  r  r6  re  r  r%   r   r  r   r   r   	r>  r   r\   r  r  r   r  r  rk  s	            r"   r  z$AsyncFirecrawlApp.check_crawl_status2  s    2 '')t$ 33||nXJ'
 

 ??8$3$"6*+;v./14*v6H#'HI&*&=&=h&P PIKK	fb 9:"+K + '+F#&??8,//'*!ook2#6!ook2($3E
 k!(__W5HN[ 'OOF3HMI
 !Qs)   9FFA5F1F2+FB.FFc                 z  K   	 | j                  | j                   d| |       d{   }|j                  d      dk(  rd|v r|d   }d|v r}t        |d         dk(  rnk|j                  d      }|st        j                  d       nB| j                  ||       d{   }|j                  |j                  dg              |}d|v r}||d<   t        di |S t        d	      |j                  d      d
v r(t        j                  t        |d             d{    nt        d|d          07 7 7 w)a  
        Monitor the status of an asynchronous job until completion.

        Args:
            id (str): The ID of the job to monitor
            headers (Dict[str, str]): Headers to include in status check requests
            poll_interval (int): Seconds between status checks (default: 2)

        Returns:
            CrawlStatusResponse: The job results if completed successfully

        Raises:
            Exception: If the job fails or an error occurs during status checks
        r  Nr   r]   r   r   r   r  z&Job completed but no data was returnedr  rx  z#Job failed or was stopped. Status: r2   )r  r6  re  r  r%   r   r  r   r   rz  r  r	  )r>  r   r\   r  r  r   r  r  s           r"   r  z+AsyncFirecrawlApp._async_monitor_job_statust  sU      $ 7 7<<.
2$/! K
 x(K7[(&v.D K/{623q8!#.??6#:'"NN+LM!*.*A*A(G*T$T	IMM&"$=>&/ !K/ +/K'.===#$LMM*.nnmmCq$9:::"EkRZF[E\ ]^^5  %U ;s;   &D;D4A5D;D7+D;AD;D9D;7D;9D;)r   r  r  r  r   rt   r  r   r  r  c                  K   i }	|r"|	j                  |j                  dd             |||	d<   |||	d<   |||	d<   |||	d<   |||	d<   |||	d	<   t        di |	}
|
j                  dd      }||d
<   dt         |d<   d}| j	                  | j
                   | |dd| j                   i       d{   }|j                  d      rd|v rt        di |S d|v rt        d|d          t        d|       7 Gw)a  
        Asynchronously map and discover links from a URL.

        Args:
          url (str): Target URL to map
          params (Optional[MapParams]): See MapParams model:
            Discovery Options:
            * search - Filter pattern for URLs
            * ignoreSitemap - Skip sitemap.xml
            * includeSubdomains - Include subdomain links
            * sitemapOnly - Only use sitemap.xml
            
            Limits:
            * limit - Max URLs to return
            * timeout - Request timeout (ms)

        Returns:
          MapResponse with:
          * Discovered URLs
          * Success/error status

        Raises:
          Exception: If mapping fails
        TrQ  Nr   r   r   r   r   rt   rJ   rO  r   r  rq  rr  r  r   rN   r   zFailed to map URL. Error: r2   )
ra  r^  r   r]  r  r6  r5  re  r   r   )r>  rJ   r   r  r  r  r   rt   r  r  rv  rw  r  rk  s                 r"   r  zAsyncFirecrawlApp.map_url  sq    F 
fkk4dkKL #)Jx %*8J').@J*+#(4J}%"'Jw$+Jy! !.:."''D'I E"-gY 7H 11||nXJ'$~&>? 2 
 
 <<	"w(':*** 8'9J8KLMM8
CDD
s   B6D8D 9ADFr  r+   re   r/  r1  r2  c                  K   | j                         }	|s|st        d      |s|st        d      |r| j                  |      }|xs g ||||dt                d}
|r||
d<   |r||
d<   |r||
d<   | j	                  | j
                   d|
|	       d	{   }|j                  d
      r|j                  d      }|st        d      	 | j                  | j
                   d| |	       d	{   }|d   dk(  rt        di |S |d   dv rt        d|d    d|d          t        j                  d       d	{    xt        d|j                  d             7 7 r7 &w)aW  
        Asynchronously extract structured information from URLs.

        Args:
            urls (Optional[List[str]]): URLs to extract from
            prompt (Optional[str]): Custom extraction prompt
            schema (Optional[Any]): JSON schema/Pydantic model
            system_prompt (Optional[str]): System context
            allow_external_links (Optional[bool]): Follow external links
            enable_web_search (Optional[bool]): Enable web search
            show_sources (Optional[bool]): Include source URLs
            agent (Optional[Dict[str, Any]]): Agent configuration

        Returns:
          ExtractResponse with:
          * Structured data matching schema
          * Source information if requested
          * Success/error status

        Raises:
          ValueError: If prompt/schema missing or extraction fails
        r  r  rO  r  r+   r   r   r  Nr   r   r  r  r   r]   r  r  r  r   rx  r  r2   )r\  r<  r_  r#   r  r6  re  r   r  r   rz  r  )r>  r  r+   re   r/  r0  r1  r2  r   r\   r  rk  r  r  s                 r"   rO   zAsyncFirecrawlApp.extract  s    D '')fBCCF@AA--f5F JB"60'#KM?3
 %+L"+8L($)L!11||nK(
 
 <<	"\\$'F KLL$($;$;||nL9% 
 x(K7*9[99 *.EE#l;x3H2IS^_fSgRh$ijjmmA&&&  8g9N8OPQQ1
 's8   BE$EAE$-E .AE$;E"<#E$ E$"E$c           
        K   | j                         }d| }| j                  | j                   | |       d{   }|d   dk(  rd|v r|d   }d|v r}t        |d         dk(  rnk|j	                  d      }|st
        j                  d       nB| j                  ||       d{   }|j                  |j	                  dg              |}d|v r}||d<   t        |j	                  d      |j	                  d	      |j	                  d      |j	                  d
      |j	                  d      |j	                  d            }d|v r|d   |d<   d|v r|d   |d<   dd|v rdi|S di|S 7 .7 ƭw)a0  
        Check the status of an asynchronous batch scrape job.

        Args:
            id (str): The ID of the batch scrape job

        Returns:
            BatchScrapeStatusResponse containing:
            Status Information:
            * status - Current state (scraping/completed/failed/cancelled)
            * completed - Number of URLs scraped
            * total - Total URLs to scrape
            * creditsUsed - API credits consumed
            * expiresAt - Data expiration timestamp
            
            Results:
            * data - List of scraped documents
            * next - URL for next page of results (if paginated)
            * success - Whether status check succeeded
            * error - Error message if failed

        Raises:
            Exception: If status check fails
        r  Nr   r]   r   r   r   r  r   r   r   r  r   r   FT)	r\  r  r6  r  re  r%   r   r  r   r  s	            r"   r  z+AsyncFirecrawlApp.check_batch_scrape_statusB  s    2 '')&rd+ 33||nXJ'
 

 x K/$"6*+;v./14*v6H#'HI&*&=&=h&P PIKK	fb 9:"+K + '+F#,??8,//'*!ook2#6!ook2(
 k! +G 4HW[ *62HV ; 6u

 	
<@

 	
G
 !Qs)   9E.E)A)E.%E,&+E.BE.,E.c                    K   | j                         }| j                  | j                   d| d|       d{   S 7 w)a[  
        Get information about errors from an asynchronous batch scrape job.

        Args:
          id (str): The ID of the batch scrape job

        Returns:
          CrawlErrorsResponse containing:
            errors (List[Dict[str, str]]): List of errors with fields:
              * id (str): Error ID
              * timestamp (str): When the error occurred
              * url (str): URL that caused the error
              * error (str): Error message
          * robotsBlocked (List[str]): List of URLs blocked by robots.txt

        Raises:
          Exception: If error check fails
        r  r  Nr\  r  r6  r>  r   r\   s      r"   r  z+AsyncFirecrawlApp.check_batch_scrape_errors  sK     & ''),,||n-bT9
 
 	
 
   6?=?c                    K   | j                         }| j                  | j                   d| d|       d{   S 7 w)a_  
        Get information about errors from an asynchronous crawl job.

        Args:
            id (str): The ID of the crawl job

        Returns:
            CrawlErrorsResponse containing:
            * errors (List[Dict[str, str]]): List of errors with fields:
                - id (str): Error ID
                - timestamp (str): When the error occurred
                - url (str): URL that caused the error
                - error (str): Error message
            * robotsBlocked (List[str]): List of URLs blocked by robots.txt

        Raises:
            Exception: If error check fails
        r  r  Nr  r  s      r"   r  z$AsyncFirecrawlApp.check_crawl_errors  sJ     & ''),,||nJrd'2
 
 	
 
r  c                   K   | j                         }t        j                         4 d{   }|j                  | j                   d| |      4 d{   }|j                          d{   cddd      d{    cddd      d{    S 7 i7 @7 *7 7 # 1 d{  7  sw Y   nxY wddd      d{  7   y# 1 d{  7  sw Y   yxY ww)r  Nr  r  )r\  rw  rx  r  r6  rC   )r>  r   r\   r|  rk  s        r"   r  zAsyncFirecrawlApp.cancel_crawl  s      '')((* 	- 	-g~~j&Ew~W - -[c%]]_,- - -	- 	- 	--,-	-- - -	- 	- 	- 	- 	-s   )CBC'CBCB.B/B2C>B?CCBCCBCCB1	%B(&B1	-C4C?C CCCCCr  c                    K   | j                         }	 | j                  | j                   d| |       d{   S 7 # t        $ r}t	        t        |            d}~ww xY ww)a<  
        Check the status of an asynchronous extraction job.

        Args:
            job_id (str): The ID of the extraction job

        Returns:
            ExtractResponse[Any] with:
            * success (bool): Whether request succeeded
            * data (Optional[Any]): Extracted data matching schema
            * error (Optional[str]): Error message if any
            * warning (Optional[str]): Warning message if any
            * sources (Optional[List[str]]): Source URLs if requested

        Raises:
            ValueError: If status check fails
        r  Nr\  r  r6  r   r<  r1   )r>  r  r\   r  s       r"   r  z$AsyncFirecrawlApp.get_extract_status  sm     $ '')	%00<<.VH5     	%SV$$	%1   A$$? =? A$? 	A!AA!!A$c          	        K   | j                         }	|s|st        d      |s|st        d      |r| j                  |      }t        |xs g ||||dt               }
|r||
d<   |r||
d<   |r||
d<   	 | j                  | j                   d|
|	       d	{   S 7 # t        $ r}t        t        |            d	}~ww xY ww)
a  
        Initiate an asynchronous extraction job without waiting for completion.

        Args:
            urls (Optional[List[str]]): URLs to extract from
            prompt (Optional[str]): Custom extraction prompt
            schema (Optional[Any]): JSON schema/Pydantic model
            system_prompt (Optional[str]): System context
            allow_external_links (Optional[bool]): Follow external links
            enable_web_search (Optional[bool]): Enable web search
            show_sources (Optional[bool]): Include source URLs
            agent (Optional[Dict[str, Any]]): Agent configuration
            idempotency_key (Optional[str]): Unique key to prevent duplicate requests

        Returns:
            ExtractResponse[Any] with:
            * success (bool): Whether request succeeded
            * data (Optional[Any]): Extracted data matching schema
            * error (Optional[str]): Error message if any

        Raises:
            ValueError: If job initiation fails
        r  r  rO  r  r+   r   r   r  N)	r\  r<  r_  r   r]  r  r6  r   r1   )r>  r  r+   re   r/  r0  r1  r2  r   r\   r  r  s               r"   r  zAsyncFirecrawlApp.async_extract  s     D '')fBCCF@AA--f5F&3-$ 	*
 %+L"+8L($)L!	%11<<.,   
  	%SV$$	%s<   A3C6#B  BB  CB   	C)B==CCr  r  r  r  r  r  c                  K   i }|||d<   |||d<   |||d<   | j                  ||||       d{   }|j                  d      rd|vr|S |d   }	 | j                  |       d{   }|d   d	k(  r|S |d   d
k(  rt        d|j                  d             |d   dk7  rnt	        j
                  d       d{    ot        dd      S 7 7 i7 w)a  
        Generate LLMs.txt for a given URL and monitor until completion.

        Args:
            url (str): Target URL to generate LLMs.txt from
            max_urls (Optional[int]): Maximum URLs to process (default: 10)
            show_full_text (Optional[bool]): Include full text in output (default: False)
            experimental_stream (Optional[bool]): Enable experimental streaming

        Returns:
            GenerateLLMsTextStatusResponse containing:
            * success (bool): Whether generation completed successfully
            * status (str): Status of generation (processing/completed/failed)
            * data (Dict[str, str], optional): Generated text with fields:
                - llmstxt (str): Generated LLMs.txt content
                - llmsfulltxt (str, optional): Full version if requested
            * error (str, optional): Error message if generation failed
            * expiresAt (str): When the generated data expires

        Raises:
            Exception: If generation fails
        Nr  r  r  r  r   r   r   r]   r^   z#LLMs.txt generation failed. Error: r   r  rx  Fr  r  )r  re  r  r   rz  r  r,  )	r>  rJ   r  r  r  r  rk  r  r   s	            r"   r  z$AsyncFirecrawlApp.generate_llms_text-  s#    :  (F9%%3F>"*.AF*+66) 3	 7 
 
 ||I&$h*>O$??GGFh;.!X-"EfjjQXFYEZ [\\!\1--"""  .eCtuu-
 H #s4   0CC5C(C)AC=C>CCCr  r  c                \  K   i }|||d<   |||d<   |||d<   t        ||||      }| j                         }d|i|j                  dd      }d	t         |d
<   	 | j	                  | j
                   d||       d{   S 7 # t        $ r}	t        t        |	            d}	~	ww xY ww)a>  
        Initiate an asynchronous LLMs.txt generation job without waiting for completion.

        Args:
            url (str): Target URL to generate LLMs.txt from
            max_urls (Optional[int]): Maximum URLs to process (default: 10)
            show_full_text (Optional[bool]): Include full text in output (default: False)
            cache (Optional[bool]): Whether to use cached content if available (default: True)
            experimental_stream (Optional[bool]): Enable experimental streaming

        Returns:
            GenerateLLMsTextResponse containing:
            * success (bool): Whether job started successfully
            * id (str): Unique identifier for the job
            * error (str, optional): Error message if start failed

        Raises:
            ValueError: If job initiation fails
        Nr  r  r  r  rJ   TrQ  rO  r   r  )	r  r\  r^  r]  r  r6  r   r<  r1   )
r>  rJ   r  r  r  r  r  r\   r  r  s
             r"   r  z*AsyncFirecrawlApp.async_generate_llms_textj  s     6  (F9%%3F>"*.AF*+''"5	
 '')CQ6;;4;#PQ	 +G95	(	%11<<.,   
  	%SV$$	%s<   AB,#B  BB B,B 	B)B$$B))B,c                    K   | j                         }	 | j                  | j                   d| |       d{   S 7 # t        $ r}t	        t        |            d}~ww xY ww)a  
        Check the status of an asynchronous LLMs.txt generation job.

        Args:
            id (str): The ID of the generation job

        Returns:
            GenerateLLMsTextStatusResponse containing:
            * success (bool): Whether generation completed successfully
            * status (str): Status of generation (processing/completed/failed)
            * data (Dict[str, str], optional): Generated text with fields:
                - llmstxt (str): Generated LLMs.txt content
                - llmsfulltxt (str, optional): Full version if requested
            * error (str, optional): Error message if generation failed
            * expiresAt (str): When the generated data expires

        Raises:
            ValueError: If status check fails
        r  Nr  r>  r   r\   r  s       r"   r  z1AsyncFirecrawlApp.check_generate_llms_text_status  sm     ( '')	%00<<.RD1     	%SV$$	%r  )r{  r  r  r  r/  -_AsyncFirecrawlApp__experimental_stream_stepsr  r  r  r  r  r  r  r  c                  K   i }
|||
d<   |||
d<   |||
d<   |||
d<   |||
d<   |||
d<   t        di |
}
| j                  ||||||       d{   }|j                  d	      rd
|vr|S |d
   }d}d}	 | j                  |       d{   }|r)d|v r%|d   |d }|D ]
  } ||        t	        |d         }|	r)d|v r%|d   |d }|D ]
  } |	|        t	        |d         }|d   dk(  r|S |d   dk(  rt        d|j                  d             |d   dk7  rnt        j                  d       d{    t        dd      S 7 7 7 wr  )	r  r  re  r   r  r   rz  r  r!  )r>  r  r{  r  r  r  r/  r  r  r  r!  rk  r  r"  r#  r   r$  r%  r&  r'  s                       r"   r(  zAsyncFirecrawlApp.deep_research  s    P  *3OJ'!+5OK()1OI&&0?O,-$.;ON+&2<WO89,??11!+' 2 
 
 ||I&$h*>O$::6BBF|v5!'!56I6J!K . *H)*&)&*>&?#Y&0$Y/0A0BC) &Ff%&$'y(9$:!h;.!X-"?

7@S?T UVV!\1--"""- 0 *%?jkkO
  C* #s7   AEE9EEB*E9E:EEE)r{  r  r  r  r/  r  c                ~  K   i }|||d<   |||d<   |||d<   |||d<   |||d<   |||d<   t        di |}| j                         }	d|i|j                  d	d	
      }
dt         |
d<   	 | j	                  | j
                   d|
|	       d{   S 7 # t        $ r}t        t        |            d}~ww xY ww)r*  Nr   r  r  r  r   r  r  TrQ  rO  r   r+  r2   )	r  r\  r^  r]  r  r6  r   r<  r1   )r>  r  r{  r  r  r  r/  r  r!  r\   r  r  s               r"   r  z%AsyncFirecrawlApp.async_deep_research  s    >  *3OJ'!+5OK()1OI&&0?O,-$.;ON+&2<WO89,??'')e^';';TX\';']^	 +G95	(	%11<<. 12   
  	%SV$$	%s<   A+B=.#B BB B=B 	B:!B55B::B=c                    K   | j                         }	 | j                  | j                   d| |       d{   S 7 # t        $ r}t	        t        |            d}~ww xY ww)r/  r0  Nr  r  s       r"   r   z,AsyncFirecrawlApp.check_deep_research_statusZ  sn     2 '')	%00<<. 22$7     	%SV$$	%r  )	r   r  r	  r  rX   ru   rt   rm  r  r  r	  r  rX   c       	           K   i }|
rDt        |
t              r|j                  |
       n"|j                  |
j                  dd             |||d<   |||d<   |||d<   |||d<   |||d<   |||d	<   |||d
<   |	|	j                  dd      |d<   |j                  |       t        dd|i|}|j                  dd      }dt         |d<   | j                  | j                   d|dd| j                   i       d{   S 7 w)a  
        Asynchronously search for content using Firecrawl.

        Args:
            query (str): Search query string
            limit (Optional[int]): Max results (default: 5)
            tbs (Optional[str]): Time filter (e.g. "qdr:d")
            filter (Optional[str]): Custom result filter
            lang (Optional[str]): Language code (default: "en")
            country (Optional[str]): Country code (default: "us") 
            location (Optional[str]): Geo-targeting
            timeout (Optional[int]): Request timeout in milliseconds
            scrape_options (Optional[ScrapeOptions]): Result scraping configuration
            params (Optional[Union[Dict[str, Any], SearchParams]]): Additional search parameters
            **kwargs: Additional keyword arguments for future compatibility

        Returns:
            SearchResponse: Response containing:
                * success (bool): Whether request succeeded
                * data (List[FirecrawlDocument]): Search results
                * warning (Optional[str]): Warning message if any
                * error (Optional[str]): Error message if any

        Raises:
            Exception: If search fails or response cannot be parsed
        TrQ  Nr   r  r	  r  rX   ru   rt   r   r  rO  r   rp  rq  rr  r2   )r`  r^  ra  r  r]  r  r6  r5  )r>  r  r   r  r	  r  rX   ru   rt   rm  r  rg  rt  rv  rw  s                  r"   r   zAsyncFirecrawlApp.search|  sa    R &$'$$V,$$V[[$T[%RS %*M'"?#&M% &,M(#$(M&!'.M)$(0M*%'.M)$%-;-@-@$]a-@-bM/* 	V$ $A%A=A"''D'I"-gY 7H--||nJ'~67
 
 	
 
s   DD
DD
)NrB  rC  rA  r?  )rx  r@  )Lr,   r-   r.   r/   r1   r   r   r   r   r   r}  r  r  rw  ClientResponserf  r  r   r  r   r   r  r	   r   rW   r   r   r   r   r   r   r   r   r   r   r   r(   r   rN  r   r  r   r  ri   r[   r   r  r   r  r  r  r   r   r  r   rO   r  r   r  r  r  r  r  r,  r  r&  r  r  r   r!  r(  r  r   r  r  r   r2   r3   r"   rt  rt  
  s    .2$'+4+4 +4 #s(^	+4
 4S>*+4 +4 "+4 -1cN+4^ 7:^^"&sCx.^;?S>^^.3^>B38n^2 7:]]%)#s(^]].3]>B38n],+G,B,B +C +TX +6Z# Zs Z[^ Zor Zwz Z$ -1-1	*9*9 [)*9 &c]	*9 7J	*9^ .2-1	(@s)(@ \*(@ &c]	(@ 7J	(@\ mq04040404&*%*15%)4837(,CG(,,015 sw,0+HMHM d7  ,g  $h  i  j	HM
 d38n-HM #49-HM #49-HM  (~HM smHM c]HM ~.HM TNHM $,D>HM #+4.HM  ~HM  G$>?@!HM"  ~#HM$ j)%HM& #:.'HM( d55E{T_alnz  }I  Kb  dm  *m  $n  o  p)HM* L)+HM, (,-HM\ W[,0,0,0,0"&!&-1!%04/3$(?C(,-1 os(,'()--HC3iHC $w  (Q   R  S  T	HC
 $sCx.)HC tCy)HC tCy)HC $D>HC 3-HC #HC >*HC HC  (~HC 'tnHC D>HC   :;<!HC" *%#HC$ z*%HC& $uZ1A;P[]hjv  yE  G^  `i  &i   j  k  l'HC( %)HC*  }+HC, "#-HC0 
#1HC^ W[,0,0,0,0"&!&-1!%04/3$(?C(,-1 os(,.2)--JI3iJI $w  (Q   R  S  T	JI
 $sCx.)JI tCy)JI tCy)JI $D>JI 3-JI #JI >*JI JI  (~JI 'tnJI D>JI   :;<!JI" *%#JI$ z*%JI& $uZ1A;P[]hjv  yE  G^  `i  &i   j  k  l'JI( %)JI* &d^+JI, "#-JI0 
1JI` .2-1#'-1#/3.2/3)-267;3726,0#+/'()-+uBuB  S	*	uB
  S	*uB C=uB &c]uB }uB 'tnuB &d^uB 'tnuB !uB !/uB %] 234uB #+4.uB  "*$!uB" $D>#uB$ }%uB& #4.'uB(  })uB* "#+uB. 
/uBx .2-1#'-1#/3.2/3)-267;3726,0#+/'()-+sBsB  S	*	sB
  S	*sB C=sB &c]sB }sB 'tnsB &d^sB 'tnsB !sB !/sB %] 234sB #+4.sB  "*$!sB" $D>#sB$ }%sB& #4.'sB(  })sB* "#+sB. 
/sBj@3 @3F @D)_# )_S#X )__b )_k~ )_^ !%)--1'+#!&&*HEHE 	HE
 !HE %TNHE tnHE }HE #HE #HE 0;HEX )-WR %)$(+/3805+0.2WR49%WR SM	WR
 SMWR $C=WR #+4.WR  (~WR #4.WR DcN+WR 8Gs7KWRrB
# B
:S B
H
# 
:M 
2
3 
3F 
2-S -T#s(^ -(%s %s7K %: )-D% %)$(+/3805+0.2D%49%D% SM	D%
 SMD% $C=D% #+4.D%  (~D% #4.D% DcN+D% 8Gs7KD%T '+-126;v;v sm	;v
 %TN;v "*$;v <Z;vB '+-1$(265%5% sm	5%
 %TN5% D>5% "*$5% <T5%n% %@^ %B (,(,&*-1+/:>FJDH^l^l  }	^l
 !^l sm^l &c]^l $C=^l *2$^l "(DcN+;T+A"BC^l  $sCx.)94)? @A^l Nh^lH (,(,&*-1+/:>:%:%  }	:%
 !:% sm:% &c]:% $C=:% *2$:% DHS>:%x %3  %;U  %L $(!%$("&%)&*%*6:DHN
N
 C=	N

 #N
 SMN
 3-N
 c]N
 smN
 c]N
 %]3N
 U4S><#?@AN
 (N
r3   rt  c            
            e Zd ZdZdedef fdZddZddZd	e	ee
f   ddfd
Zdej                  deddfdZdededededef
dZ xZS )r  zO
    Async version of CrawlWatcher that properly handles async operations.
    r   rF  c                 &    t         |   ||       y r@  )superr?  )r>  r   rF  	__class__s      r"   r?  zAsyncCrawlWatcher.__init__  s    S!r3   r7  Nc                   K   t        j                  | j                  dd| j                  j                   fg      4 d{   }| j                  |       d{    ddd      d{    y7 .7 7 	# 1 d{  7  sw Y   yxY ww)z[
        Establishes async WebSocket connection and starts listening for messages.
        rq  rr  )rQ  NrR  rV  s     r"   rT  zAsyncCrawlWatcher.connect  s      %%KK!0GDHH<L<L;M2N OP
 	* 	* ,,y)))		* 	* 	* *		* 	* 	* 	*sZ   A BA1BA7A3A7 B+A5,B3A75B7B	=B >B	Bc                    K   |2 3 d{   }t        j                  |      }| j                  |       d{    87 37 6 yw)z
        Listens for incoming WebSocket messages and handles them asynchronously.

        Args:
            websocket: The WebSocket connection object
        NrY  r\  s       r"   rU  zAsyncCrawlWatcher._listen  r^  r_  r]  c                   K   |d   dk(  r<d| _         | j                  d| j                   | j                  | j                  d       y|d   dk(  r@d| _         | j                  d| j                   | j                  |d   | j                  d       y|d   dk(  rl|d	   d
   | _         | j                  j	                  |d	   j                  d	g              | j                  D ]!  }| j                  d|| j                  d       # y|d   dk(  rA| j                  j                  |d	          | j                  d|d	   | j                  d       yyw)z
        Handles incoming WebSocket messages based on their type asynchronously.

        Args:
            msg (Dict[str, Any]): The message to handle
        r   rJ  r]   ri  r   r^   rj  rk  r   r   rK  rl  Nrm  rn  s      r"   r[  z!AsyncCrawlWatcher._handle_message  rp  rq  rk  rj  c                 d  K   	 |j                          d{   }|j                  dd      }|j                  dd      }| j
                  j                  |j                  |||       d{   }t        j                  |      7 m#  t        j                  d|j                         xY w7 Bw)z9
        Handle errors from async API responses.
        Nr   r  r  r  r  )rC   re  rw  r{  r   rF  r  r  s          r"   rf  zAsyncCrawlWatcher._handle_error  s     	{'}}.J&NN74PQM&NN96]^M
 99(//6S`bopp!!'** /	{%%(ghphwhwgx&yzz qs8   B0B B(B  +B0+B.,B0B $B++B0rd  r  r  c                 2   K   | j                  ||||      S wr  r  r  s        r"   r  z*AsyncCrawlWatcher._get_async_error_message  r  r  rr  )r,   r-   r.   r/   r1   rt  r?  rT  rU  r   r   r[  rw  r  rf  r   r  __classcell__)r  s   @r"   r  r    s    "3 "%6 "*	,Rc3h RD R,+G,B,B +C +TX + Z# Zs Z[^ Zor Zwz Zr3   r  )Mr/   loggingr   r  typingr   r   r   r   r   r   r	   r
   r   rC   r   r   warningsrb  rG   rS  rw  rz  r   r#   r]  	getLoggerr%   Loggerr0   r&   	BaseModelr(   r5   r:   r>   rI   rW   r[   rc   ri   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r  r  r  r  r  r!  r&  r(  r,  r4  r  rt  r  r2   r3   r"   <module>r     s  
  	  X X X   	       
 -+'++K8 8CLJ!8%% !
((,, (H&& 
C++ C8**GAJ 8 *X'' *
UH&& UH.. $H&& $(### #"x)) "($$ 
($$ 
($$ 
#8%% #8%% h00 
""" "(8%% ()## ),= , &q)71:  ,(,, ,	" 2 2 	"+($$ +( H&&  	"(,, 	"(,, 
$"" $ ($$  
2H&& 
2	-h(('!* 	-
28%% 
2 X''  1X// 1	6++ 	6 8--  !3!3  x11  &);); &X%7%7  X''  +H&& +q" q"fEYR YRvF
 F
P8OZ OZr3   