FazBrowse GitHub Viewer
|
Trending
|
URL:
|
Home
Tools:
[Download Repo ZIP]
[View Raw Code]
[Original HTTPS Page]
TheAlgorithms-Python/web_programming/emails_from_url.py at master · windNight/TheAlgorithms-Python · GitHub
windNight
/
TheAlgorithms-Python
Public
forked from
TheAlgorithms/Python
Notifications
You must be signed in to change notification settings
Fork
1
Star
0
Code
Pull requests
0
Actions
Projects
Security and quality
0
Insights
Additional navigation options
Code
Pull requests
Actions
Projects
Security and quality
Insights
Expand file tree
Breadcrumbs
TheAlgorithms-Python
/
web_programming
/
emails_from_url.py
Copy path
More file actions
More file actions
Latest commit
History
History
History
103 lines (84 loc) · 2.83 KB
Breadcrumbs
TheAlgorithms-Python
/
web_programming
/
emails_from_url.py
Copy path
File metadata and controls
103 lines (84 loc) · 2.83 KB
Raw
Copy raw file
Download raw file
Open symbols panel
Edit and raw actions
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
"""Get the site emails from URL."""
__author__
=
"Muhammad Umer Farooq"
__license__
=
"MIT"
__version__
=
"1.0.0"
__maintainer__
=
"Muhammad Umer Farooq"
__email__
=
"contact@muhammadumerfarooq.me"
__status__
=
"Alpha"
import
re
from
html
.
parser
import
HTMLParser
from
urllib
import
parse
import
requests
class
Parser
(
HTMLParser
):
def
__init__
(
self
,
domain
:
str
):
HTMLParser
.
__init__
(
self
)
self
.
data
=
[]
self
.
domain
=
domain
def
handle_starttag
(
self
,
tag
:
str
,
attrs
:
str
)
->
None
:
"""
This function parse html to take takes url from tags
"""
# Only parse the 'anchor' tag.
if
tag
==
"a"
:
# Check the list of defined attributes.
for
name
,
value
in
attrs
:
# If href is defined, and not empty nor # print it.
if
name
==
"href"
and
value
!=
"#"
and
value
!=
""
:
# If not already in data.
if
value
not
in
self
.
data
:
url
=
parse
.
urljoin
(
self
.
domain
,
value
)
self
.
data
.
append
(
url
)
# Get main domain name (example.com)
def
get_domain_name
(
url
:
str
)
->
str
:
"""
This function get the main domain name
>>> get_domain_name("https://a.b.c.d/e/f?g=h,i=j#k")
'c.d'
>>> get_domain_name("Not a URL!")
''
"""
return
"."
.
join
(
get_sub_domain_name
(
url
).
split
(
"."
)[
-
2
:])
# Get sub domain name (sub.example.com)
def
get_sub_domain_name
(
url
:
str
)
->
str
:
"""
>>> get_sub_domain_name("https://a.b.c.d/e/f?g=h,i=j#k")
'a.b.c.d'
>>> get_sub_domain_name("Not a URL!")
''
"""
return
parse
.
urlparse
(
url
).
netloc
def
emails_from_url
(
url
:
str
=
"https://github.com"
)
->
list
:
"""
This function takes url and return all valid urls
"""
# Get the base domain from the url
domain
=
get_domain_name
(
url
)
# Initialize the parser
parser
=
Parser
(
domain
)
try
:
# Open URL
r
=
requests
.
get
(
url
)
# pass the raw HTML to the parser to get links
parser
.
feed
(
r
.
text
)
# Get links and loop through
valid_emails
=
set
()
for
link
in
parser
.
data
:
# open URL.
# read = requests.get(link)
try
:
read
=
requests
.
get
(
link
)
# Get the valid email.
emails
=
re
.
findall
(
"[a-zA-Z0-9]+@"
+
domain
,
read
.
text
)
# If not in list then append it.
for
email
in
emails
:
valid_emails
.
add
(
email
)
except
ValueError
:
pass
except
ValueError
:
exit
(
-
1
)
# Finally return a sorted list of email addresses with no duplicates.
return
sorted
(
valid_emails
)
if
__name__
==
"__main__"
:
emails
=
emails_from_url
(
"https://github.com"
)
print
(
f"
{
len
(
emails
)
}
emails found:"
)
print
(
"
\n
"
.
join
(
sorted
(
emails
)))
Back
|
FazBrowse Home
|
New Git URL