30 lines
1.2 KiB
Python
30 lines
1.2 KiB
Python
import cloudscraper, re, os, sys
|
|
|
|
OUT = "/Users/drjones/astraea-books/.case-verify"
|
|
os.makedirs(OUT, exist_ok=True)
|
|
|
|
def scrape(url, fname):
|
|
s = cloudscraper.create_scraper(browser={'browser':'chrome','platform':'darwin','mobile':False})
|
|
try:
|
|
r = s.get(url, timeout=90)
|
|
print(fname, "status:", r.status_code, "len:", len(r.text))
|
|
if r.status_code == 200 and len(r.text) > 20000:
|
|
open(os.path.join(OUT, fname + ".html"), "w").write(r.text)
|
|
html = r.text
|
|
print(" has 'Wash.2d':", "Wash.2d" in html, "| has 'opinion':", "opinion" in html.lower())
|
|
return html
|
|
else:
|
|
print(" short/failed, head:", r.text[:200].replace("\n"," "))
|
|
return None
|
|
except Exception as e:
|
|
print(fname, "ERR:", e)
|
|
return None
|
|
|
|
url = sys.argv[1] if len(sys.argv) > 1 else "https://law.justia.com/cases/washington/supreme-court/1997/64471-3-1.html"
|
|
fname = sys.argv[2] if len(sys.argv) > 2 else "justia_little"
|
|
h = scrape(url, fname)
|
|
if h:
|
|
for marker in ['id="opinion"', 'class="page-content', 'justia-citation', '<article', 'case-name']:
|
|
i = h.find(marker)
|
|
print(marker, "->", i)
|