FazBrowse GitHub Viewer
|
Trending
|
URL:
|
Home
Tools:
[Download Repo ZIP]
[View Raw Code]
[Original HTTPS Page]
python-scrapyd-api/scrapyd_api/wrapper.py at master · githubleaflife/python-scrapyd-api · GitHub
githubleaflife
/
python-scrapyd-api
Public
forked from
djm/python-scrapyd-api
Notifications
You must be signed in to change notification settings
Fork
0
Star
0
Code
Pull requests
0
Actions
Projects
Security and quality
0
Insights
Additional navigation options
Code
Pull requests
Actions
Projects
Security and quality
Insights
Expand file tree
Breadcrumbs
python-scrapyd-api
/
scrapyd_api
/
wrapper.py
Copy path
More file actions
More file actions
Latest commit
History
History
History
198 lines (176 loc) · 6.63 KB
Breadcrumbs
python-scrapyd-api
/
scrapyd_api
/
wrapper.py
Copy path
File metadata and controls
198 lines (176 loc) · 6.63 KB
Raw
Copy raw file
Download raw file
Open symbols panel
Edit and raw actions
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
from
__future__
import
unicode_literals
from
copy
import
deepcopy
from
.
import
constants
from
.
client
import
Client
from
.
compat
import
(
iteritems
,
urljoin
)
class
ScrapydAPI
(
object
):
"""
Provides a thin Pythonic wrapper around the Scrapyd API. The public methods
come in two types: first class, those that wrap a Scrapyd API endpoint
directly; and derived, those that use a one or more Scrapyd API endpoint(s)
to provide functionality that is unique to this wrapper.
"""
def
__init__
(
self
,
target
=
'http://localhost:6800'
,
auth
=
None
,
endpoints
=
None
,
client
=
None
,
timeout
=
None
):
"""
Instantiates the ScrapydAPI wrapper for use.
Args:
target (str): the hostname/port to hit with requests.
auth (str, str): a 2-item tuple containing user/pass details. Only
used when `client` is not passed.
endpoints: a dictionary of custom endpoints to apply on top of
the pre-existing defaults.
client: a pre-instantiated requests-like client. By default, we use
our own client. Override for your own needs.
timeout: timeout for client requests in seconds, either as a float
or a (connect timeout, read timeout) tuple
"""
if
endpoints
is
None
:
endpoints
=
{}
if
client
is
None
:
client
=
Client
()
client
.
auth
=
auth
self
.
target
=
target
self
.
client
=
client
self
.
timeout
=
timeout
self
.
endpoints
=
deepcopy
(
constants
.
DEFAULT_ENDPOINTS
)
self
.
endpoints
.
update
(
endpoints
)
def
_build_url
(
self
,
endpoint
):
"""
Builds the absolute URL using the target and desired endpoint.
"""
try
:
path
=
self
.
endpoints
[
endpoint
]
except
KeyError
:
msg
=
'Unknown endpoint `{0}`'
raise
ValueError
(
msg
.
format
(
endpoint
))
absolute_url
=
urljoin
(
self
.
target
,
path
)
return
absolute_url
def
add_version
(
self
,
project
,
version
,
egg
):
"""
Adds a new project egg to the Scrapyd service. First class, maps to
Scrapyd's add version endpoint.
"""
url
=
self
.
_build_url
(
constants
.
ADD_VERSION_ENDPOINT
)
data
=
{
'project'
:
project
,
'version'
:
version
}
files
=
{
'egg'
:
egg
}
json
=
self
.
client
.
post
(
url
,
data
=
data
,
files
=
files
,
timeout
=
self
.
timeout
)
return
json
[
'spiders'
]
def
cancel
(
self
,
project
,
job
,
signal
=
None
):
"""
Cancels a job from a specific project. First class, maps to
Scrapyd's cancel job endpoint.
"""
url
=
self
.
_build_url
(
constants
.
CANCEL_ENDPOINT
)
data
=
{
'project'
:
project
,
'job'
:
job
,
}
if
signal
is
not
None
:
data
[
'signal'
]
=
signal
json
=
self
.
client
.
post
(
url
,
data
=
data
,
timeout
=
self
.
timeout
)
return
json
[
'prevstate'
]
def
delete_project
(
self
,
project
):
"""
Deletes all versions of a project. First class, maps to Scrapyd's
delete project endpoint.
"""
url
=
self
.
_build_url
(
constants
.
DELETE_PROJECT_ENDPOINT
)
data
=
{
'project'
:
project
,
}
self
.
client
.
post
(
url
,
data
=
data
,
timeout
=
self
.
timeout
)
return
True
def
delete_version
(
self
,
project
,
version
):
"""
Deletes a specific version of a project. First class, maps to
Scrapyd's delete version endpoint.
"""
url
=
self
.
_build_url
(
constants
.
DELETE_VERSION_ENDPOINT
)
data
=
{
'project'
:
project
,
'version'
:
version
}
self
.
client
.
post
(
url
,
data
=
data
,
timeout
=
self
.
timeout
)
return
True
def
job_status
(
self
,
project
,
job_id
):
"""
Retrieves the 'status' of a specific job specified by its id. Derived,
utilises Scrapyd's list jobs endpoint to provide the answer.
"""
all_jobs
=
self
.
list_jobs
(
project
)
for
state
in
constants
.
JOB_STATES
:
job_ids
=
[
job
[
'id'
]
for
job
in
all_jobs
[
state
]]
if
job_id
in
job_ids
:
return
state
return
''
# Job not found, state unknown.
def
list_jobs
(
self
,
project
):
"""
Lists all known jobs for a project. First class, maps to Scrapyd's
list jobs endpoint.
"""
url
=
self
.
_build_url
(
constants
.
LIST_JOBS_ENDPOINT
)
params
=
{
'project'
:
project
}
jobs
=
self
.
client
.
get
(
url
,
params
=
params
,
timeout
=
self
.
timeout
)
return
jobs
def
list_projects
(
self
):
"""
Lists all deployed projects. First class, maps to Scrapyd's
list projects endpoint.
"""
url
=
self
.
_build_url
(
constants
.
LIST_PROJECTS_ENDPOINT
)
json
=
self
.
client
.
get
(
url
,
timeout
=
self
.
timeout
)
return
json
[
'projects'
]
def
list_spiders
(
self
,
project
):
"""
Lists all known spiders for a specific project. First class, maps
to Scrapyd's list spiders endpoint.
"""
url
=
self
.
_build_url
(
constants
.
LIST_SPIDERS_ENDPOINT
)
params
=
{
'project'
:
project
}
json
=
self
.
client
.
get
(
url
,
params
=
params
,
timeout
=
self
.
timeout
)
return
json
[
'spiders'
]
def
list_versions
(
self
,
project
):
"""
Lists all deployed versions of a specific project. First class, maps
to Scrapyd's list versions endpoint.
"""
url
=
self
.
_build_url
(
constants
.
LIST_VERSIONS_ENDPOINT
)
params
=
{
'project'
:
project
}
json
=
self
.
client
.
get
(
url
,
params
=
params
,
timeout
=
self
.
timeout
)
return
json
[
'versions'
]
def
schedule
(
self
,
project
,
spider
,
settings
=
None
,
**
kwargs
):
"""
Schedules a spider from a specific project to run. First class, maps
to Scrapyd's scheduling endpoint.
"""
url
=
self
.
_build_url
(
constants
.
SCHEDULE_ENDPOINT
)
data
=
{
'project'
:
project
,
'spider'
:
spider
}
data
.
update
(
kwargs
)
if
settings
:
setting_params
=
[]
for
setting_name
,
value
in
iteritems
(
settings
):
setting_params
.
append
(
'{0}={1}'
.
format
(
setting_name
,
value
))
data
[
'setting'
]
=
setting_params
json
=
self
.
client
.
post
(
url
,
data
=
data
,
timeout
=
self
.
timeout
)
return
json
[
'jobid'
]
def
daemon_status
(
self
):
"""
Displays the load status of a service.
:rtype: dict
"""
url
=
self
.
_build_url
(
constants
.
DAEMON_STATUS_ENDPOINT
)
json
=
self
.
client
.
get
(
url
,
timeout
=
self
.
timeout
)
return
json
Back
|
FazBrowse Home
|
New Git URL