From ec440c3386c80d871aa5a47ef2767d7cb07a9a82 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Wed, 22 Jul 2026 04:31:40 +0000 Subject: [PATCH 1/3] chore: plan dependency and quality updates --- dist/google-lens-python-2023.3.18.tar.gz | Bin 0 -> 4760 bytes google_lens_python.egg-info/PKG-INFO | 13 +++++++++++++ google_lens_python.egg-info/SOURCES.txt | 9 +++++++++ google_lens_python.egg-info/dependency_links.txt | 1 + google_lens_python.egg-info/requires.txt | 2 ++ google_lens_python.egg-info/top_level.txt | 1 + 6 files changed, 26 insertions(+) create mode 100644 dist/google-lens-python-2023.3.18.tar.gz create mode 100644 google_lens_python.egg-info/PKG-INFO create mode 100644 google_lens_python.egg-info/SOURCES.txt create mode 100644 google_lens_python.egg-info/dependency_links.txt create mode 100644 google_lens_python.egg-info/requires.txt create mode 100644 google_lens_python.egg-info/top_level.txt diff --git a/dist/google-lens-python-2023.3.18.tar.gz b/dist/google-lens-python-2023.3.18.tar.gz new file mode 100644 index 0000000000000000000000000000000000000000..c8955b8f2b66881b5c3703d0c1b6e14447820176 GIT binary patch literal 4760 zcmV;J5@+oniwFpvM__6K|7UM+XKZCHY-Mh9EpT~sXm4&UGB7eTE;BAMI4*Qyascf; zX>;2+viZzkfmJ?8s*!2k&T2~UZ8q0`{w>%j`{Tug4-f|@~d23dF}e!?GA?dXT0C*_Iv#&?E1+g zTuBu35T<|fZC?FnY#lpmG3^~Z8;<`x96uYGBl!PdI53{*>koe|eShhSJy&?qUa*N* zzPH!!_6O#`>^b1c-~D8 z@V55{`Ju>VE^OsaF_q@bqAx0{~r$qVE^66wEu6^ z|GyJ4w|UI>{w~7E@x2M_n?2(_UyBLb)(s=?l`4sGlC0M}+)UVWc1)dEz^x0u6fE{x zC@x`yV9pwEM1qIbia8#8L!jAP0Bt-^U{na>z14>O7&_6)HQv3}o&@!`v06afe`Fag= zv$xz^CddTnuJ{C>2BrQ-0D;HY`(zWvVvRkHPGCw~1URV|8^%|1xx=oUc*WwCV0`Yl zPQ1Z^s_T3Me1(nyhc14I-;>ZabpZI|U%!j|uRq?-{{}k$dzAdIU&;SQ2WD3rzRv$J z|NHRj`O9~&%(eZi=6|E^aG2+R-5%zDqh43%e~)oJEhY&nwJ?nBd(4R#-%34JJZ9W= z{i}#={A6b;;+IkpmPDB1NLcsHOFn*hYZ}JWr;rqZ;9Qquo@N;Tg4rz&SjxLKM5$Q= zCzu$<=g*&`l>wi#0ZS+TZWYHtG}+&W&#PpPDdoOOGW-4Rcz@eeOz4TQ2%(P=P;3~d zKFlO)5a<9;0zZt2*3YsXn?ExPQ$V%|{WZ0v+~uCltWfY6KT1R~-SijCi6spxDW{PV zJ47y>C!fMF=>D{DtJO{sf67jX+sGZoaH1lY0neq-A>xPv**=?ZX4v|ftpy-s6V?IJ z1NkTR%LT|$gQ7TuiAv}zK%s~dHwGjTSF;Sg)j$aQvA<7xW&RW_8*L*1P;sj?Tv2tp z0(HG|K(3*PlhA{SZHH2NSgA0}_hRmNNHAik2$cXLvxX<2bXzQed$r7f#Caz#V9eKY z0l>Bzse+1Yz7AYL8u~a4VTZQhphtV`5ND=UQ!K7yW{X%59M*CcEbw7111qL>xnsz^ z$VEL~fHo(gsPsknF93zmnJ2jVF^v414|)p@0nw_l=^*rfg6SCtm{-CDeuzIY5R{dU z7aayza1c|J%-0U|WI;k)_;(_rI0dcXG$VHQ#)*GO=FkD~b7DW-oVV`n7uqHxqTV_d z+yrxacX*nAd5Ze=-jCtC5{xUG7!Ly15fS)Hcax%q>aP?Cdgf>`PU~V=h=MxSFP&Kn z0W{T=NZ^6I+ObFdMgRhcv=PUu!#yg$Pl63Oi*P1*(okxOO7k55VXE zpaq};3tm_u+a2&{f}EAL;^7kHfHQa-hKpnv;LlTy;~=eIgFT7D$X~=)C{HvVupQrW zKyz$n`&NS5L=IokAn7|%BN}a*1QgXx4K8h>Amb-7>NF?@DJ=M7&vIcA4>+XFt^+zE zXTag1vs}fbe_-exM6<)zK5z#A3Zg1V<}R?k1N@3Im5cBuF{y)O>;pR^0a^zTNW6;@ za?}SW36P($;t)-HwNf!5BLq0K0Q-e01&y)6EBT#*6iH`%jLAqC7=jdPGy#?Dne3DT zt$!()M;a*SM#cxZG*wfy!pI0pA{hk}0$X{3ju0uuumlR{G7LIouNC@0Z~AiLA{AhA zmkz>gfJbzqMkf;j^gtaVl|YTTa6C{aaUFp0Xi|u8U?z4^fWUXeS_)8d22oI;6w^Uu zpp2kGGSTSZW|D_67XTC-0|)%A#N{DKH3U18%<+6s(*^o#@@8n-0~68@Xkd`+fPqLp z;e686kY%oeJ4X^S&;TLcS%Z>r^wva#Pqv8N6f##ULAQ{{@J~8bv>)_k@D-ICqIx>? z$|3Cnn@9RF1?Vv39*pn_2#oWTYtB-vOaRaYzhnH$a6!WXSW(q@Oub6d`#-<#^S{CN z{!dTu|2)e6&q3|;zrk>%H-WzS{U6yl!^W9f@rvbg55l3pe^2|r*ZUv+{Q29`Acot%;|Jz{-C=)F$^VA>{f|ewvR8bJNW@8C1{>opS-){d)wr== zn$=jFnd^8L(X9jUsjU-=+dIa701c3n6?$UF{n^a%oOm|N>7?8FM`+OTU&ntP|MjKg z|9zeJSH^$%T8aPT@xegH|Hrtrsq6K>j{iFT>-ewZ|GnaWaV_@1^?$clef~4*>*qg@ zbIqfnp6W4g2N#KmFeDK(GHE<)Yd#=1s+8w2kQXpN{`J{_E>oj{gsNL-ywQ-|zPNmHL1E zepkQ$^*EOb_n|P4bDjT)Ld`EK`Q=lIBh$#bo(RA#3D_LVdr@gM z&t{QCQvTYF2SMluAr_EgD++M3*3X(>;E(Jw0%O1UQ#rF1F~r$SLYGZhR)>wuL0Sg> z0I||JMP-_DuGP*uWFMZgkAaQ&rM*);0jIUIb}XtE2bA&9X7kX$ibRNId%``w#NPwf z&1fnZWY!v?E2YwHXeS@!pljo zJ21O#_C{D2{yvy1{$#huUOS;!_}BZ?+WeAD(l!bLUy67Zgjh!m5Sm3+=mar|zrzAR zVS<(Lc4)$@Ybtvb$x2+{dM>#avkaK9+28mjKOv9|f5MtqQ2gQa-CKfR(3WE!^0kP8 zE%|iVg|!F;s!x@TqoPar%g{o#z5=WdR70--2y5v{5JF-t(&B$%govITy_YBgsKlZY zczLX)mtWjak*77{6-zifbK@4oegNTzn6FFkXHs-kSmE< zjU|#SKMX|_V9_}&O6CaH&dP}?bTh)>!7X?a;@k# zJ4{XeGyeV3?2rLZ8lB3JPdMmPW5PZ)aNH-r@e^S!@5#~fE&_nNxk(ooA33&ovEq

?;;7Qpd18W3nNlN!m9dcb&j(p0%@L~_0EgYgPhaO_)*2PK8z5nq8*+VjZS%N zND5+WC36UfyI2L#iN!j)Ere8g*}_<$;>ni#-*)<$%fBtSJinn}>l!{6@V|7T1PpwW z5kgHO+7(wSIhQIyn@^Wxf;3Fo&t<aiFUMhJQ%fdK+viGkhNCH zjVQxJC6^kV>Tw3fSyaRgs9_aQ)WkCO62MFWsEG*3KFPsn#5&+(&KWnVfZNi#yUo$v zJ;xm+Q1G8fb&;EgF%}L%^Mzozlx053JR`@Dy1LwT3veTD&QQ5Hl7?5Ir>YGnd?CQa zh&t7+H&tR>+HEG6QUqLD0uI_v^sv0CEPUQd`*tX0Y!gMW0?Y2V+5fO^ZQ@r0gy3N2 zU+UZ@S(E-4H`5ENO2~P3(kYt@^c|q+e-RQ7?o8srPB_qnpnCTJ1?b<8VKUd1jF@jg z=hpru)qD0fa;*T$iM63|wFK+kmN1tlIFkf2NYBF9we+HPEmJW+Oy-arp8ek^+dQX? zwIf^G$YLd%+!x$gs@d#21)KFK%Xaj;lkMm}vK`%1w$B6n@h7>Ne8F3c)vEq-mH=dK z2yw27*J;0VlN0D|PHxBXQY@F$4DKAQi|Gj#LyBPOlEdkh;S~C{{PLwV*%9EPJz_`|j4*FJ!n zigeo)ne($5ZdxuB79Qn;92j2!M#Z#>wy3g2TFPHozDJ^dt*{0-4%SKEO01h8THhA- zjHqZ%SSzEq!+bF5;`e4aZ?-END@__WQ@bI!GubSx7z)bp*Tivcbx?}pa-B9%W3;_m zS)pk*gx4U2es2Giz+z+N!+&mCp$$(mC=XiFpd=~{LKK$5cYRCxyd0epfZFBG3A#_t zj*m`Gi)T?tmCDY2D5$oS3XoIU=nVkxva8ItslQ{5eF?6*Wr0U0OfyM#Drq`$Z7Tni zo9LXM8LXY15OVcs`kG^O09xK5GeA!lbOWfcctBg6NXDSs&#*UYjr+w)SQi*QrTR%; zT=9rSiG`ZENZew=pDAcWMf+x`b?i@5oM=K$E)Yt&X>PQe6z6I(;l-^*gA(n6EhBp- zX+=N+{&U@wUmEQia9K~!AL zB{bhE4Q#512Ze_MUvd$qfTTf2n0?9y{^(B!J2!>q|)IxQ9iG=Wbjl>U*gFraMG|WtYt@v1s zjXNS@76{+GI;}aANgfbWOm}ZG$J^rIe1kN9b-X(blN}N4VFv#@wD$*^yMC)izyEiy z_y6?ozka*-|7w5#bTe6Vt%9=^m?Ilq--A4@(4)#j*)0{}v?S1$dIU$W1i0=jE$KKt zYok}*dSu@Kpa1}*A#SYz literal 0 HcmV?d00001 diff --git a/google_lens_python.egg-info/PKG-INFO b/google_lens_python.egg-info/PKG-INFO new file mode 100644 index 0000000..50c37ce --- /dev/null +++ b/google_lens_python.egg-info/PKG-INFO @@ -0,0 +1,13 @@ +Metadata-Version: 2.1 +Name: google-lens-python +Version: 2023.3.18 +Summary: A Python package to reverse image search in Google Lens +Author: Anhy Krishna Fitiavana +Author-email: fitiavana.krishna@gmail.com +Keywords: python,google,scraping +Classifier: Development Status :: 5 - Production/Stable +Classifier: Intended Audience :: Developers +Classifier: Programming Language :: Python :: 3 +Classifier: Operating System :: OS Independent + +A Python package to reverse image search in Google Lens, with the ability to search by file path or by url. diff --git a/google_lens_python.egg-info/SOURCES.txt b/google_lens_python.egg-info/SOURCES.txt new file mode 100644 index 0000000..320fb3d --- /dev/null +++ b/google_lens_python.egg-info/SOURCES.txt @@ -0,0 +1,9 @@ +README.md +setup.py +google_lens_python.egg-info/PKG-INFO +google_lens_python.egg-info/SOURCES.txt +google_lens_python.egg-info/dependency_links.txt +google_lens_python.egg-info/requires.txt +google_lens_python.egg-info/top_level.txt +googlelens/__init__.py +googlelens/googlelens.py \ No newline at end of file diff --git a/google_lens_python.egg-info/dependency_links.txt b/google_lens_python.egg-info/dependency_links.txt new file mode 100644 index 0000000..8b13789 --- /dev/null +++ b/google_lens_python.egg-info/dependency_links.txt @@ -0,0 +1 @@ + diff --git a/google_lens_python.egg-info/requires.txt b/google_lens_python.egg-info/requires.txt new file mode 100644 index 0000000..9d981c3 --- /dev/null +++ b/google_lens_python.egg-info/requires.txt @@ -0,0 +1,2 @@ +bs4 +requests diff --git a/google_lens_python.egg-info/top_level.txt b/google_lens_python.egg-info/top_level.txt new file mode 100644 index 0000000..87f42b5 --- /dev/null +++ b/google_lens_python.egg-info/top_level.txt @@ -0,0 +1 @@ +googlelens From cb13bb1e1da7a4a81a037a9536cc8c2b8973f4ad Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Wed, 22 Jul 2026 04:32:53 +0000 Subject: [PATCH 2/3] chore: bump dependencies and harden parser logic --- googlelens/googlelens.py | 139 +++++++++++++++++++++++---------------- requirements.txt | 4 +- setup.py | 2 +- 3 files changed, 84 insertions(+), 61 deletions(-) diff --git a/googlelens/googlelens.py b/googlelens/googlelens.py index f8f2e0c..74b31ac 100644 --- a/googlelens/googlelens.py +++ b/googlelens/googlelens.py @@ -1,5 +1,6 @@ import re import json +from typing import Any, Optional from requests import Session from bs4 import BeautifulSoup @@ -32,22 +33,30 @@ def __get_prerender_script(self, page: str): soup = BeautifulSoup(page, 'html.parser') # Find the script containing 'AF_initDataCallback' with specific key and hash values - prerender_script = list(filter( - lambda s: ( - 'AF_initDataCallback(' in s.text and - re.search(r"key: 'ds:(\d+)'", s.text).group(1) == "0"), - soup.find_all('script') - ))[0].text + prerender_script: Optional[str] = None + for script in soup.find_all('script'): + if script.text is None or 'AF_initDataCallback(' not in script.text: + continue + key_match = re.search(r"key: 'ds:(\d+)'", script.text) + if key_match and key_match.group(1) == "0": + prerender_script = script.text + break + + if prerender_script is None: + raise ValueError("Unable to find Google Lens prerender data in response.") # Clean up the script content to prepare it for JSON parsing prerender_script = prerender_script.replace( "AF_initDataCallback(", "").replace(");", "") # Extract hash value and replace the corresponding fields in the script for JSON formatting - hash = re.search(r"hash: '(\d+)'", prerender_script).group(1) + hash_match = re.search(r"hash: '(\d+)'", prerender_script) + if hash_match is None: + raise ValueError("Unable to parse Google Lens prerender hash.") + script_hash = hash_match.group(1) prerender_script = prerender_script.replace( - f"key: 'ds:0', hash: '{hash}', data:", - f"\"key\": \"ds:0\", \"hash\": \"{hash}\", \"data\":" + f"key: 'ds:0', hash: '{script_hash}', data:", + f"\"key\": \"ds:0\", \"hash\": \"{script_hash}\", \"data\":" ).replace("sideChannel:", "\"sideChannel\":") # Parse the cleaned prerender script into a JSON object @@ -56,6 +65,17 @@ def __get_prerender_script(self, page: str): # Return the relevant data section for further processing return prerender_script['data'][1] + @staticmethod + def __safe_get(value: Any, path: list[int]) -> Any: + current = value + for index in path: + if not isinstance(current, list) or not isinstance(index, int): + return None + if index < 0 or index >= len(current): + return None + current = current[index] + return current + def __parse_prerender_script(self, prerender_script): """ Parses the prerendered script to extract match and similar items. @@ -73,58 +93,58 @@ def __parse_prerender_script(self, prerender_script): } # Extract the best match information if available - try: + match_title = self.__safe_get(prerender_script, [0, 1, 8, 12, 0, 0, 0]) + match_thumbnail = self.__safe_get(prerender_script, [0, 1, 8, 12, 0, 2, 0, 0]) + match_page_url = self.__safe_get(prerender_script, [0, 1, 8, 12, 0, 2, 0, 4]) + if isinstance(match_title, str) and isinstance(match_thumbnail, str) and isinstance(match_page_url, str): data["match"] = { - "title": prerender_script[0][1][8][12][0][0][0], # Extract item title - "thumbnail": prerender_script[0][1][8][12][0][2][0][0], # Extract thumbnail URL - "pageURL": prerender_script[0][1][8][12][0][2][0][4] # Extract page URL + "title": match_title, # Extract item title + "thumbnail": match_thumbnail, # Extract thumbnail URL + "pageURL": match_page_url # Extract page URL } - except IndexError: - # If data is unavailable, continue without a match - pass # Determine which section to use for extracting visual matches if data["match"] is not None: - visual_matches = prerender_script[1][1][8][8][0][12] + visual_matches = self.__safe_get(prerender_script, [1, 1, 8, 8, 0, 12]) else: - try: - visual_matches = prerender_script[0][1][8][8][0][12] - except IndexError: - return data + visual_matches = self.__safe_get(prerender_script, [0, 1, 8, 8, 0, 12]) + + if not isinstance(visual_matches, list): + return data # Iterate through the visual matches and extract relevant details for match in visual_matches: # Safely extract thumbnail URL if available - thumbnail_url = match[0][0] if ( - isinstance(match[0], list) and len(match[0]) > 0 and - isinstance(match[0][0], str) - ) else None + thumbnail_url = self.__safe_get(match, [0, 0]) + if not isinstance(thumbnail_url, str): + thumbnail_url = None # Safely extract price if available - price = match[0][7][1] if ( - isinstance(match[0], list) and len(match[0]) > 7 and - isinstance(match[0][7], list) and len(match[0][7]) > 1 and - isinstance(match[0][7][1], str) - ) else None + price = self.__safe_get(match, [0, 7, 1]) + if not isinstance(price, str): + price = None # Clean price by removing any special characters (e.g., currency signs) price = re.sub(r"[^\d.]", "", price) if price is not None else None # Safely extract currency if available - currency = match[0][7][5] if ( - isinstance(match[0], list) and len(match[0]) > 7 and - isinstance(match[0][7], list) and len(match[0][7]) > 5 and - isinstance(match[0][7][5], str) - ) else None + currency = self.__safe_get(match, [0, 7, 5]) + if not isinstance(currency, str): + currency = None + + title = self.__safe_get(match, [3]) + similarity_score = self.__safe_get(match, [1]) + page_url = self.__safe_get(match, [5]) + source_website = self.__safe_get(match, [14]) # Append the extracted information to the "similar" matches list data["similar"].append( { - "title": match[3], # Extract item title - "similarity score": match[1], # Extract similarity (?) score + "title": title, # Extract item title + "similarity score": similarity_score, # Extract similarity (?) score "thumbnail": thumbnail_url, # Thumbnail URL - "pageURL": match[5], # Extract page URL - "sourceWebsite": match[14], # Extract source website name + "pageURL": page_url, # Extract page URL + "sourceWebsite": source_website, # Extract source website name "price": price, # Price (cleaned) "currency": currency # Currency symbol } @@ -143,24 +163,26 @@ def search_by_file(self, file_path: str): Returns: The parsed search results after extracting and processing the response. """ - multipart = { - 'encoded_image': (file_path, open(file_path, 'rb')), - 'image_content': '' - } + with open(file_path, 'rb') as image_file: + multipart = { + 'encoded_image': (file_path, image_file), + 'image_content': '' + } - # Build the parameter dictionary - params = { - "hl": "en", # Adjust host language here - "gl": "us", # Adjust the geolocation parameter here - } - - # Send a POST request to upload the file - response = self.session.post( - self.url + "/upload", - files=multipart, - params=params, - allow_redirects=False # Must be false to capture the 302 response - ) + # Build the parameter dictionary + params = { + "hl": "en", # Adjust host language here + "gl": "us", # Adjust the geolocation parameter here + } + + # Send a POST request to upload the file + response = self.session.post( + self.url + "/upload", + files=multipart, + params=params, + allow_redirects=False, # Must be false to capture the 302 response + timeout=30 + ) # Check if the request was successful if response.status_code != 302: # Expecting a 302 for redirect @@ -177,7 +199,7 @@ def search_by_file(self, file_path: str): return None # Or handle the error appropriately # Proceed with the redirect - response = self.session.get(search_url) + response = self.session.get(search_url, timeout=30) # Extract the prerendered JavaScript content for further parsing. prerender_script = self.__get_prerender_script(response.text) @@ -205,7 +227,8 @@ def search_by_url(self, url: str): response = self.session.get( self.url + "/uploadbyurl", params=params, - allow_redirects=True + allow_redirects=True, + timeout=30 ) # Extract the prerendered JavaScript content for further parsing diff --git a/requirements.txt b/requirements.txt index a98ae43..a7fa2d1 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,2 +1,2 @@ -requests -beautifulsoup4 \ No newline at end of file +requests>=2.34.2 +beautifulsoup4>=4.15.0 \ No newline at end of file diff --git a/setup.py b/setup.py index 99c89b1..63cc93e 100644 --- a/setup.py +++ b/setup.py @@ -12,7 +12,7 @@ description=DESCRIPTION, long_description=LONG_DESCRIPTION, packages=find_packages(), - install_requires=['requests', 'bs4'], + install_requires=['requests>=2.34.2', 'beautifulsoup4>=4.15.0'], keywords=['python', 'google', 'scraping'], classifiers=[ "Development Status :: 5 - Production/Stable", From 94a65a7ea1d082bec01f657389848347c839c286 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Wed, 22 Jul 2026 04:33:14 +0000 Subject: [PATCH 3/3] chore: refresh package metadata dependencies --- google_lens_python.egg-info/requires.txt | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/google_lens_python.egg-info/requires.txt b/google_lens_python.egg-info/requires.txt index 9d981c3..166da77 100644 --- a/google_lens_python.egg-info/requires.txt +++ b/google_lens_python.egg-info/requires.txt @@ -1,2 +1,2 @@ -bs4 -requests +beautifulsoup4>=4.15.0 +requests>=2.34.2