Spaces:
Sleeping
Sleeping
Download app.py from mr2along/FaceRecognition: direct link, hf CLI and curl.
- Browser
- Download file 9.87 kB
-
https://huggingface.co/spaces/mr2along/FaceRecognition/resolve/main/app.py
- Command line
-
hf download hf://spaces/mr2along/FaceRecognition/app.py
-
curl -L -o app.py https://huggingface.co/spaces/mr2along/FaceRecognition/resolve/main/app.py
9.87 kB
| import subprocess | |
| import sys | |
| import json | |
| import gradio as gr | |
| from bs4 import BeautifulSoup | |
| from urllib.parse import urljoin | |
| # ========================================================= | |
| # CONFIG | |
| # ========================================================= | |
| TIMEOUT = 45000 | |
| BASE_URL = ( | |
| "https://en.xchina.co/models/type-10/" | |
| "sort-debut/1.html" | |
| ) | |
| #TOTAL_PAGE = 148 | |
| TOTAL_PAGE = 4 | |
| # ========================================================= | |
| # INSTALL PLAYWRIGHT | |
| # ========================================================= | |
| def setup_playwright(): | |
| print("="*50) | |
| print("Checking Playwright") | |
| try: | |
| import playwright | |
| print("Playwright OK") | |
| except Exception: | |
| return False | |
| try: | |
| result = subprocess.run( | |
| [ | |
| sys.executable, | |
| "-m", | |
| "playwright", | |
| "install", | |
| "chromium" | |
| ], | |
| stdout=subprocess.PIPE, | |
| stderr=subprocess.STDOUT, | |
| text=True, | |
| timeout=300 | |
| ) | |
| print(result.stdout) | |
| return result.returncode == 0 | |
| except Exception as e: | |
| print(e) | |
| return False | |
| PLAYWRIGHT_READY = setup_playwright() | |
| # ========================================================= | |
| # NORMALIZE | |
| # ========================================================= | |
| def normalize_url(url): | |
| url = (url or "").strip() | |
| if not url: | |
| return "" | |
| if not url.startswith( | |
| ( | |
| "http://", | |
| "https://" | |
| ) | |
| ): | |
| url = "https://" + url | |
| return url | |
| # ========================================================= | |
| # CLEAN HTML | |
| # ========================================================= | |
| def clean_html( | |
| html, | |
| base_url | |
| ): | |
| soup = BeautifulSoup( | |
| html, | |
| "html.parser" | |
| ) | |
| for tag in soup.find_all( | |
| [ | |
| "script", | |
| "iframe", | |
| "noscript" | |
| ] | |
| ): | |
| tag.decompose() | |
| for a in soup.find_all("a"): | |
| href = a.get("href") | |
| if href: | |
| a["href"] = urljoin( | |
| base_url, | |
| href | |
| ) | |
| for img in soup.find_all("img"): | |
| src = ( | |
| img.get("src") | |
| or | |
| img.get("data-src") | |
| ) | |
| if src: | |
| img["src"] = urljoin( | |
| base_url, | |
| src | |
| ) | |
| return soup.prettify() | |
| # ========================================================= | |
| # PLAYWRIGHT RENDER | |
| # ========================================================= | |
| def render_html(url): | |
| if not PLAYWRIGHT_READY: | |
| return "" | |
| from playwright.sync_api import sync_playwright | |
| try: | |
| with sync_playwright() as p: | |
| browser = p.chromium.launch( | |
| headless=True, | |
| args=[ | |
| "--no-sandbox", | |
| "--disable-dev-shm-usage", | |
| "--disable-blink-features=AutomationControlled" | |
| ] | |
| ) | |
| context = browser.new_context( | |
| viewport={ | |
| "width":1440, | |
| "height":1000 | |
| }, | |
| user_agent=( | |
| "Mozilla/5.0 " | |
| "(Windows NT 10.0; Win64; x64) " | |
| "Chrome/130 Safari/537.36" | |
| ), | |
| locale="en-US" | |
| ) | |
| page = context.new_page() | |
| page.goto( | |
| url, | |
| wait_until="domcontentloaded", | |
| timeout=TIMEOUT | |
| ) | |
| try: | |
| page.wait_for_load_state( | |
| "networkidle", | |
| timeout=15000 | |
| ) | |
| except: | |
| pass | |
| page.wait_for_timeout( | |
| 3000 | |
| ) | |
| # auto scroll | |
| try: | |
| page.evaluate( | |
| """ | |
| async()=>{ | |
| let sleep = | |
| ms=>new Promise( | |
| r=>setTimeout(r,ms) | |
| ); | |
| let last=0; | |
| for(let i=0;i<20;i++){ | |
| let h=document.body.scrollHeight; | |
| window.scrollTo(0,h); | |
| await sleep(500); | |
| if(h===last) | |
| break; | |
| last=h; | |
| } | |
| window.scrollTo(0,0); | |
| } | |
| """ | |
| ) | |
| except: | |
| pass | |
| html = page.content() | |
| browser.close() | |
| return clean_html( | |
| html, | |
| url | |
| ) | |
| except Exception as e: | |
| print(e) | |
| return "" | |
| # ========================================================= | |
| # EXTRACT MODELS | |
| # ========================================================= | |
| def extract_models( | |
| html, | |
| base_url | |
| ): | |
| soup = BeautifulSoup( | |
| html, | |
| "html.parser" | |
| ) | |
| models = [] | |
| cards = soup.select( | |
| ".item.actor" | |
| ) | |
| for card in cards: | |
| name = "" | |
| image = "" | |
| href = "" | |
| work = "" | |
| # ------------------------- | |
| # NAME | |
| # ------------------------- | |
| text = card.select_one( | |
| ".text" | |
| ) | |
| if text: | |
| name = text.get_text( | |
| " ", | |
| strip=True | |
| ) | |
| if "(" in name: | |
| name = name.split( | |
| "(" | |
| )[0].strip() | |
| # ------------------------- | |
| # HREF | |
| # ------------------------- | |
| a = card.find( | |
| "a" | |
| ) | |
| if a and a.get( | |
| "href" | |
| ): | |
| href = urljoin( | |
| base_url, | |
| a["href"] | |
| ) | |
| # ------------------------- | |
| # IMAGE | |
| # ------------------------- | |
| img = card.find( | |
| "img" | |
| ) | |
| if img: | |
| image = ( | |
| img.get("src") | |
| or | |
| img.get("data-src") | |
| or | |
| "" | |
| ) | |
| image = urljoin( | |
| base_url, | |
| image | |
| ) | |
| # ------------------------- | |
| # WORKS | |
| # ------------------------- | |
| tags = card.select_one( | |
| ".tags" | |
| ) | |
| if tags: | |
| value = tags.get_text( | |
| " ", | |
| strip=True | |
| ) | |
| if "Works:" in value: | |
| work = value.replace( | |
| "Works:", | |
| "" | |
| ).strip() | |
| models.append( | |
| { | |
| "name": name, | |
| "image": image, | |
| "href": href, | |
| "work": work | |
| } | |
| ) | |
| return models | |
| # ========================================================= | |
| # REMOVE DUPLICATE | |
| # ========================================================= | |
| def remove_duplicate( | |
| data | |
| ): | |
| result = [] | |
| seen = set() | |
| for item in data: | |
| href = item.get( | |
| "href" | |
| ) | |
| if href not in seen: | |
| seen.add( | |
| href | |
| ) | |
| result.append( | |
| item | |
| ) | |
| return result | |
| # ========================================================= | |
| # CRAWL ALL 148 PAGE | |
| # ========================================================= | |
| def crawl_all(): | |
| all_models = [] | |
| for page in range( | |
| 1, | |
| TOTAL_PAGE + 1 | |
| ): | |
| url = BASE_URL.replace( | |
| "/1.html", | |
| f"/{page}.html" | |
| ) | |
| print( | |
| "SCAN PAGE:", | |
| page | |
| ) | |
| html = render_html( | |
| url | |
| ) | |
| if not html: | |
| continue | |
| models = extract_models( | |
| html, | |
| url | |
| ) | |
| print( | |
| "FOUND:", | |
| len(models) | |
| ) | |
| all_models.extend( | |
| models | |
| ) | |
| all_models = remove_duplicate( | |
| all_models | |
| ) | |
| return all_models | |
| # ========================================================= | |
| # SAVE JSON | |
| # ========================================================= | |
| def start_crawl(): | |
| models = crawl_all() | |
| with open( | |
| "models.json", | |
| "w", | |
| encoding="utf-8" | |
| ) as f: | |
| json.dump( | |
| models, | |
| f, | |
| indent=4, | |
| ensure_ascii=False | |
| ) | |
| return json.dumps( | |
| models, | |
| indent=4, | |
| ensure_ascii=False | |
| ) | |
| # ========================================================= | |
| # GRADIO UI | |
| # ========================================================= | |
| with gr.Blocks( | |
| title="Xchina Model Crawler" | |
| ) as demo: | |
| gr.Markdown( | |
| """ | |
| # 🌐 Model Crawler | |
| Tự động quét: | |
| - 148 trang model | |
| - Name | |
| - Image | |
| - Href | |
| - Works | |
| Xuất: | |
| models.json | |
| """ | |
| ) | |
| btn = gr.Button( | |
| "🚀 Start Crawl", | |
| variant="primary" | |
| ) | |
| output = gr.Code( | |
| label="JSON Result", | |
| language="json", | |
| lines=40 | |
| ) | |
| btn.click( | |
| fn=start_crawl, | |
| inputs=[], | |
| outputs=output | |
| ) | |
| # ========================================================= | |
| # RUN | |
| # ========================================================= | |
| if __name__ == "__main__": | |
| print( | |
| "🚀 Starting Gradio" | |
| ) | |
| print( | |
| "Playwright:", | |
| PLAYWRIGHT_READY | |
| ) | |
| demo.launch( | |
| server_name="0.0.0.0", | |
| server_port=7860 | |
| ) |