Sitelet https://github.com/python26/python-scraping/commit/4260ca46d30b53f07590356ebc3aad1c138ce271
Skip to content

Commit 4260ca4

Browse files
Ryan MitchellRyan Mitchell
authored andcommitted
Initial commit of code from book
1 parent e939535 commit 4260ca4

98 files changed

Lines changed: 1629 additions & 0 deletions

File tree

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.

‎chapter1/.DS_Store‎

6 KB
Binary file not shown.

‎chapter1/1-basicExample.py‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,3 @@
1+
from urllib.request import urlopen
2+
html = urlopen("http://www.pythonscraping.com/exercises/exercise1.html")
3+
print(html.read())

‎chapter1/2-beautifulSoup.py‎

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
1+
from urllib.request import urlopen
2+
from bs4 import BeautifulSoup
3+
4+
html = urlopen("http://www.pythonscraping.com/exercises/exercise1.html")
5+
bsObj = BeautifulSoup(html.read());
6+
print(bsObj.h1)

‎chapter1/3-exceptionHandling.py‎

Lines changed: 26 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,26 @@
1+
from urllib.request import urlopen
2+
from urllib.error import HTTPError
3+
from bs4 import BeautifulSoup
4+
import sys
5+
6+
7+
def getTitle(url):
8+
try:
9+
html = urlopen(url)
10+
except HTTPError as e:
11+
print(e)
12+
return None
13+
try:
14+
bsObj = BeautifulSoup(html.read())
15+
title = bsObj.body.h1
16+
except AttributeError as e:
17+
return None
18+
return title
19+
20+
title = getTitle("http://www.pythonscraping.com/exercises/exercise1.html")
21+
if title == None:
22+
print("Title could not be found")
23+
else:
24+
print(title)
25+
26+

‎chapter10/1-seleniumBasic.py‎

Lines changed: 8 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,8 @@
1+
from selenium import webdriver
2+
import time
3+
4+
driver = webdriver.PhantomJS(executable_path='')
5+
driver.get("http://pythonscraping.com/pages/javascript/ajaxDemo.html")
6+
time.sleep(3)
7+
print(driver.find_element_by_id("content").text)
8+
driver.close()

‎chapter10/2-waitForLoad.py‎

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,11 @@
1+
from selenium.webdriver.common.by import By
2+
from selenium.webdriver.support.ui import WebDriverWait
3+
from selenium.webdriver.support import expected_conditions as EC
4+
5+
driver = webdriver.PhantomJS(executable_path='')
6+
driver.get("http://pythonscraping.com/pages/javascript/ajaxDemo.html")
7+
try:
8+
element = WebDriverWait(driver, 10).until(EC.presence_of_element_located((By.ID, "loadedButton")))
9+
finally:
10+
print(driver.find_element_by_id("content").text)
11+
driver.close()

‎chapter10/3-javascriptRedirect.py‎

Lines changed: 23 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,23 @@
1+
from selenium import webdriver
2+
import time
3+
from selenium.webdriver.remote.webelement import WebElement
4+
from selenium.common.exceptions import StaleElementReferenceException
5+
6+
def waitForLoad(driver):
7+
elem = driver.find_element_by_tag_name("html")
8+
count = 0
9+
while True:
10+
count += 1
11+
if count > 20:
12+
print("Timing out after 10 seconds and returning")
13+
return
14+
time.sleep(.5)
15+
try:
16+
elem == driver.find_element_by_tag_name("html")
17+
except StaleElementReferenceException:
18+
return
19+
20+
driver = webdriver.PhantomJS(executable_path='<Path to Phantom JS>')
21+
driver.get("http://pythonscraping.com/pages/javascript/redirectDemo1.html")
22+
waitForLoad(driver)
23+
print(driver.page_source)

‎chapter11/1-basicImage.py‎

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
1+
from PIL import Image, ImageFilter
2+
3+
kitten = Image.open("kitten.jpg")
4+
blurryKitten = kitten.filter(ImageFilter.GaussianBlur)
5+
blurryKitten.save("kitten_blurred.jpg")
6+
blurryKitten.show()

‎chapter11/2-cleanImage.py‎

Lines changed: 19 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,19 @@
1+
from PIL import Image
2+
import subprocess
3+
4+
def cleanFile(filePath, newFilePath):
5+
image = Image.open(filePath)
6+
7+
#Set a threshold value for the image, and save
8+
image = image.point(lambda x: 0 if x<143 else 255)
9+
image.save(newFilePath)
10+
11+
#call tesseract to do OCR on the newly created image
12+
subprocess.call(["tesseract", newFilePath, "output"])
13+
14+
#Open and read the resulting data file
15+
outputFile = open("output.txt", 'r')
16+
print(outputFile.read())
17+
outputFile.close()
18+
19+
cleanFile("text_2.png", "text_2_clean.png")

‎chapter11/3-readWebImages.py‎

Lines changed: 36 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,36 @@
1+
import time
2+
from urllib.request import urlretrieve
3+
import subprocess
4+
from selenium import webdriver
5+
6+
driver = webdriver.PhantomJS(executable_path='/Users/ryan/Documents/pythonscraping/code/headless/phantomjs-1.9.8-macosx/bin/phantomjs')
7+
#driver = webdriver.Firefox()
8+
driver.get("http://www.amazon.com/War-Peace-Leo-Nikolayevich-Tolstoy/dp/1427030200")
9+
time.sleep(2)
10+
11+
driver.find_element_by_id("sitbLogoImg").click()
12+
#The easiest way to get exactly one of every page
13+
imageList = set()
14+
15+
#Wait for the page to load
16+
time.sleep(10)
17+
print(driver.find_element_by_id("sitbReaderRightPageTurner").get_attribute("style"))
18+
while "pointer" in driver.find_element_by_id("sitbReaderRightPageTurner").get_attribute("style"):
19+
#While we can click on the right arrow, move through the pages
20+
driver.find_element_by_id("sitbReaderRightPageTurner").click()
21+
time.sleep(2)
22+
#Get any new pages that have loaded (multiple pages can load at once)
23+
pages = driver.find_elements_by_xpath("//div[@class='pageImage']/div/img")
24+
for page in pages:
25+
image = page.get_attribute("src")
26+
imageList.add(image)
27+
28+
driver.quit()
29+
30+
#Start processing the images we've collected URLs for with Tesseract
31+
for image in sorted(imageList):
32+
urlretrieve(image, "page.jpg")
33+
p = subprocess.Popen(["tesseract", "page.jpg", "page"], stdout=subprocess.PIPE,stderr=subprocess.PIPE)
34+
p.wait()
35+
f = open("page.txt", "r")
36+
print(f.read())

0 commit comments

Comments
 (0)