summaryrefslogtreecommitdiff
path: root/scripts/scan.py
blob: c4b426a385ab074105da4a5608c9ff5118e3e086 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
import json
import requests

from pathlib import Path
from os import makedirs, listdir

ROOT_PATH = Path(__file__).parents[1]

DATA_DIR = ROOT_PATH / "data"

TLDS_URL = "https://data.iana.org/TLD/tlds-alpha-by-domain.txt"

NAMES_ENDPOINT = "https://nameberry.com/nameberry/api/v1/search"


class NameScanner:
    def fetch_tlds(self):
        with open(DATA_DIR / "tlds.txt", "w") as f:
            print("Fetching TLDs...")
            res = requests.get(TLDS_URL)
            tlds = [
                tld.lower() for tld in res.text.strip().split("\n")[1:] if len(tld) < 4
            ]
            f.write("\n".join(tlds).strip())
            TLDS = set(tlds)

    def fetch_homepage(self, domain):
        print(f"Fetching {domain}...", end="")
        res = requests.get(f"http://{domain}", timeout=5)
        if res.ok:
            print(f"Found {domain}!")
            with open(DATA_DIR / "homepages" / f"{domain}.html", "w") as f:
                f.write(res.text)
        else:
            print("x")

    def fetch_names(self, suffix, count=5000):
        res = requests.post(
            NAMES_ENDPOINT,
            json={
                "starts_with": "",
                "ends_with": suffix,
                "contains": "",
                "syllables": "",
                "origin_id": "",
                "derivation": "",
                "page": 1,
                "per_page": count,
            },
        )
        if res.ok:
            j = res.json()
            print(f"found {j['advanced_name_count']}")
            makedirs(DATA_DIR / "names", exist_ok=True)
            with open(DATA_DIR / "names" / f"{suffix}.json", "w") as f:
                f.write(res.text)

    def fetch_all(self):
        with open(DATA_DIR / "tlds.txt", "r") as f:
            for line in f:
                d = line.strip()
                print(f"Fetching {d}...", end="")
                self.fetch_names(d)


ns = NameScanner()
ns.fetch_names()