Tôi đang tìm một cách nhanh chóng để lấy mã phản hồi HTTP từ một URL (tức là 200, 404, v.v.). Tôi không chắc nên sử dụng thư viện nào.
Tôi đang tìm một cách nhanh chóng để lấy mã phản hồi HTTP từ một URL (tức là 200, 404, v.v.). Tôi không chắc nên sử dụng thư viện nào.
Câu trả lời:
Cập nhật bằng cách sử dụng thư viện yêu cầu tuyệt vời . Lưu ý rằng chúng tôi đang sử dụng yêu cầu HEAD, yêu cầu này sẽ diễn ra nhanh hơn sau đó là yêu cầu GET hoặc POST đầy đủ.
import requests
try:
r = requests.head("https://stackoverflow.com")
print(r.status_code)
# prints the int of the status code. Find more at httpstatusrappers.com :)
except requests.ConnectionError:
print("failed to connect")
requestscung cấp 403cho liên kết của bạn, mặc dù nó vẫn hoạt động trong trình duyệt.
Đây là một giải pháp sử dụng httplibthay thế.
import httplib
def get_status_code(host, path="/"):
""" This function retreives the status code of a website by requesting
HEAD data from the host. This means that it only requests the headers.
If the host cannot be reached or something else goes wrong, it returns
None instead.
"""
try:
conn = httplib.HTTPConnection(host)
conn.request("HEAD", path)
return conn.getresponse().status
except StandardError:
return None
print get_status_code("stackoverflow.com") # prints 200
print get_status_code("stackoverflow.com", "/nonexistant") # prints 404
exceptkhối đó ít nhất StandardErrorđể bạn không bắt nhầm những thứ như KeyboardInterrupt.
curl -I http://www.amazon.com/.
Bạn nên sử dụng urllib2, như sau:
import urllib2
for url in ["http://entrian.com/", "http://entrian.com/does-not-exist/"]:
try:
connection = urllib2.urlopen(url)
print connection.getcode()
connection.close()
except urllib2.HTTPError, e:
print e.getcode()
# Prints:
# 200 [from the try block]
# 404 [from the except block]
http://entrian.com/thành http://entrian.com/blogtrong ví dụ của mình, kết quả 200 sẽ đúng mặc dù nó liên quan đến chuyển hướng đến http://entrian.com/blog/(lưu ý dấu gạch chéo sau).
Trong tương lai, đối với những người sử dụng python3 trở lên, đây là một mã khác để tìm mã phản hồi.
import urllib.request
def getResponseCode(url):
conn = urllib.request.urlopen(url)
return conn.getcode()
Các urllib2.HTTPErrorngoại lệ không chứa một getcode()phương pháp. Sử dụng codethuộc tính thay thế.
Đây là một httplibgiải pháp hoạt động giống như urllib2. Bạn chỉ có thể cung cấp cho nó một URL và nó chỉ hoạt động. Không cần phải phân chia các URL của bạn thành tên máy chủ và đường dẫn. Chức năng này đã làm điều đó.
import httplib
import socket
def get_link_status(url):
"""
Gets the HTTP status of the url or returns an error associated with it. Always returns a string.
"""
https=False
url=re.sub(r'(.*)#.*$',r'\1',url)
url=url.split('/',3)
if len(url) > 3:
path='/'+url[3]
else:
path='/'
if url[0] == 'http:':
port=80
elif url[0] == 'https:':
port=443
https=True
if ':' in url[2]:
host=url[2].split(':')[0]
port=url[2].split(':')[1]
else:
host=url[2]
try:
headers={'User-Agent':'Mozilla/5.0 (X11; Ubuntu; Linux x86_64; rv:26.0) Gecko/20100101 Firefox/26.0',
'Host':host
}
if https:
conn=httplib.HTTPSConnection(host=host,port=port,timeout=10)
else:
conn=httplib.HTTPConnection(host=host,port=port,timeout=10)
conn.request(method="HEAD",url=path,headers=headers)
response=str(conn.getresponse().status)
conn.close()
except socket.gaierror,e:
response="Socket Error (%d): %s" % (e[0],e[1])
except StandardError,e:
if hasattr(e,'getcode') and len(e.getcode()) > 0:
response=str(e.getcode())
if hasattr(e, 'message') and len(e.message) > 0:
response=str(e.message)
elif hasattr(e, 'msg') and len(e.msg) > 0:
response=str(e.msg)
elif type('') == type(e):
response=e
else:
response="Exception occurred without a good error message. Manually check the URL to see the status. If it is believed this URL is 100% good then file a issue for a potential bug."
return response