Repository navigation
Expand file tree
/
Copy pathSelenium2Pdf.py
More file actions
98 lines (76 loc) · 3.45 KB
/
Copy pathSelenium2Pdf.py
File metadata and controls
98 lines (76 loc) · 3.45 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
# Selenium2Pdf.py - Basic example of using Selenium to Generate PDF
# which includes page formatting, headers, footers
# Dependencies
# * python3 -m pip install selenium==4.1.3
# * Download and place or create symlink to chromedriver in current directory
# URL: (http://chromedriver.chromium.org/)
# Author: Timothy.c.quinn@gmail.com (https://github.com/JavaScriptDude)
# License: MIT
import sys
import os
import json
import base64
import time
from datetime import datetime
from selenium import webdriver
from selenium.webdriver.support.ui import WebDriverWait
def main(args):
# sw = StopWatch()
url = 'https://en.wikipedia.org/wiki/Headless_browser'
out_file = f'z_test_{datetime.now().strftime("%y%m%d-%H%M%S.%f")}.pdf'
out_path = os.path.split(sys.argv[0])[0]
if out_path == '': out_path = os.getcwd()
out_path_full = f"{out_path}/{out_file}"
wd_dcap = webdriver.DesiredCapabilities.CHROME.copy()
wd_opts = webdriver.chrome.options.Options()
# Note: headless must be enabled for PDF to avoid the
# ambiguous 'Printing is not available' error.
wd_opts.add_argument('--headless')
wd_opts.add_argument('--disable-gpu')
chr_svc = webdriver.chrome.service.Service('./chromedriver')
with webdriver.Chrome(service=chr_svc, options=wd_opts, desired_capabilities=wd_dcap) as driver:
driver.get(url)
# (optional) Wait for document.readyState = complete
# WebDriverWait(driver, timeout=5, poll_frequency=0.5).until(_waitForDocReady)
assert driver.page_source != '<html><head></head><body></body></html>' \
,f"Url could not be loaded: {url}"
# For full options see:
# https://chromedevtools.github.io/devtools-protocol/tot/Page/#method-printToPDF
result = send_cmd(driver, "Page.printToPDF", params={
'landscape': True
,'margin':{'top':'1cm', 'right':'1cm', 'bottom':'1cm', 'left':'1cm'}
,'format': 'A4'
,'displayHeaderFooter': True
,'headerTemplate': '<span style="font-size:8px; margin-left: 5px">Page <span class=pageNumber></span> of <span class=totalPages></span></span>'
,'footerTemplate': f'<span style="font-size:8px; margin-left: 5px">Generated on {datetime.now().strftime("%m/%d/%Y at %H:%M")} - url = <span class=url></span></span>'
,'scale': 1
})
with open(out_path_full, 'wb') as file:
file.write(base64.b64decode(result['data']))
if not os.path.isfile(out_path_full):
raise Exception(f"PDF WAS NOT GENERATED: {out_path_full}")
print(f"PDF Generated - ./{out_path_full}") # . Time: {sw.elapsed(1)}s
def _waitForDocReady(driver):
rs = driver.execute_script('return document.readyState;')
if rs == 'complete': return True
return False
def send_cmd(driver, cmd, params={}):
response = driver.command_executor._request(
'POST'
,f"{driver.command_executor._url}/session/{driver.session_id}/chromium/send_command_and_get_result"
,json.dumps({'cmd': cmd, 'params': params}))
if response.get('status'): raise Exception(response.get('value'))
return response.get('value')
class StopWatch:
def __init__(self):
self.start()
def start(self):
self._startTime = time.time()
def getStartTime(self):
return self._startTime
def elapsed(self, prec=3):
prec = 3 if prec is None or not isinstance(prec, int) else prec
diff= time.time() - self._startTime
return round(diff, prec)
if __name__ == '__main__':
main(sys.argv[1:])