-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathprogram.py
More file actions
61 lines (42 loc) · 1.26 KB
/
Copy pathprogram.py
File metadata and controls
61 lines (42 loc) · 1.26 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
""" Scrape the EconTalk website for podcast transcripts."""
import requests
from time import sleep
from bs4 import BeautifulSoup
def main():
tx_urls = get_urls()
save_transcript_pages(tx_urls[:])
print("Finished.")
def save_transcript_pages(tx_urls):
""" Append each transcript to corpus. """
for url in tx_urls:
page = build_page_from_url(url)
sleep(1)
save_to_file(page)
def build_page_from_url(url):
""" Scrape transcripts from EconTalk. """
print("Downloading {}...".format(url), flush=True)
resp = requests.get(url)
html = resp.text
soup = BeautifulSoup(html, 'lxml')
transcript = soup.select("#unique")[0].get_text()
return transcript
def get_urls():
""" Get the urls from text file. """
urls = []
with open('data/urls.txt', 'r') as fin:
for line in fin:
urls.append(line.strip())
return urls
def save_to_file(page):
with open('data/corpus.txt', 'a') as fout:
fout.write(page)
# def clean_line(text):
# text = text.replace('\n', ' ').replace('\t', ' ')
# size = len(text) + 1
# while size > len(text):
# size = len(text)
# text = text.replace(' ', ' ')
#
# return text.strip()
if __name__ == '__main__':
main()