我试图下载一个文本文件给定的url和端口。这在python 2上可以这样做:
Python 2.7.12 (default, Sep 26 2016, 09:46:23) [GCC 4.2.1 Compatible
Apple LLVM 7.3.0 (clang-703.0.31)] on darwin Type "help", "copyright",
"credits" or "license" for more information.
>>> import urllib
>>> foo = urllib.urlopen("http://catnet-ip.icc.cat:8080/")
>>> foo.read()
'SOURCETABLE 200 OKrnServer: NTRIP Trimble NTRIP CasterrnContent-Type: text/plainrnContent-Length: 2884rnDate:
02/Nov/2016:12:52:19 UTCrnrnSTR;VRS_RTK_2_3;Virtual RTK ver RTCM
2.3;RTCM 2.3;1(1),3(6),18(1),19(1),23(5),24(5);2;GPS;Catnet;ESP;41.3;2.09;1;1;Trimble
GPSNet;None;B;N;3900;;rnSTR;VRS_RTK_3_0;Virtual RTK ver RTCM
3.0;RTCM 3;1004(1),1005/1007(5),PBS(10);2;GPS;Catnet;ESP;41.3;2.09;1;1;Trimble
GPSNet;None;B;N;1100;;rnSTR;VRS_DGPS;Virtual DGPS ver RTCM 2.3;RTCM
2.3;1(1),3(6),22(6),23/24(5),16(59);0;GPS;Catnet;ESP;41.3;2.09;1;1;Trimble
GPSNet;None;B;N;640;;rn
...
与wget:
类似Python 2.7.12 (default, Sep 26 2016, 09:46:23) [GCC 4.2.1 Compatible
Apple LLVM 7.3.0 (clang-703.0.31)] on darwin Type "help", "copyright",
"credits" or "license" for more information.
>>> import wget
>>> foo = wget.download("http://catnet-ip.icc.cat:8080/", bar=None)
>>> foo
>>> ' (1).'
>>> exit()
$ less (1).
SOURCETABLE 200 OKrnServer: NTRIP Trimble NTRIP CasterrnContent-Type: text/plainrnContent-Length: 2884rnDate:
02/Nov/2016:12:52:19 UTCrnrnSTR;VRS_RTK_2_3;Virtual RTK ver RTCM
2.3;RTCM 2.3;1(1),3(6),18(1),19(1),23(5),24(5);2;GPS;Catnet;ESP;41.3;2.09;1;1;Trimble
GPSNet;None;B;N;3900;;rnSTR;VRS_RTK_3_0;Virtual RTK ver RTCM
3.0;RTCM 3;1004(1),1005/1007(5),PBS(10);2;GPS;Catnet;ESP;41.3;2.09;1;1;Trimble
GPSNet;None;B;N;1100;;rnSTR;VRS_DGPS;Virtual DGPS ver RTCM 2.3;RTCM
2.3;1(1),3(6),22(6),23/24(5),16(59);0;GPS;Catnet;ESP;41.3;2.09;1;1;Trimble
GPSNet;None;B;N;640;;rn
...
但是在python 3上都失败了,错误是"http.client. net "。BadStatusLine: SOURCETABLE 200 OK"
Python 3.5.2 (v3.5.2:4def2a2901a5, Jun 26 2016, 10:47:25) [GCC 4.2.1 (Apple Inc. build 5666) (dot 3)] on darwin Type "help", "copyright", "credits" or "license" for more information.
>>> import urllib.request
>>> foo = urllib.request.urlopen("http://catnet-ip.icc.cat:8080/")
Traceback (most recent call last):
File "<stdin>", line 1, in <module>
File /Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 163, in urlopen
return opener.open(url, data, timeout)
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 466, in open
response = self._open(req, data)
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 484, in _open
'_open', req) File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 444, in _call_chain
result = func(*args)
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 1282, in http_open
return self.do_open(http.client.HTTPConnection, req)
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 1257, in do_open
r = h.getresponse()
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/http/client.py", line 1197, in getresponse
response.begin()
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/http/client.py", line 297, in begin
version, status, reason = self._read_status()
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/http/client.py", line 279, in _read_status
raise BadStatusLine(line)
http.client.BadStatusLine: SOURCETABLE 200 OK
:
Python 3.5.2 (v3.5.2:4def2a2901a5, Jun 26 2016, 10:47:25) [GCC 4.2.1 (Apple Inc. build 5666) (dot 3)] on darwin Type "help", "copyright", "credits" or "license" for more information.
>>> import wget
>>> wget.download("http://catnet-ip.icc.cat:8080/")
Traceback (most recent call last):
File "<stdin>", line 1, in <module>
File "/Users/toni/Downloads/wget-2.0/wget.py", line 308, in download
(tmpfile, headers) = urllib.urlretrieve(url, tmpfile, callback)
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 188, in urlretrieve
with contextlib.closing(urlopen(url, data)) as fp:
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 163, in urlopen
return opener.open(url, data, timeout)
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 466, in open
response = self._open(req, data)
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 484, in _open
'_open', req)
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 444, in _call_chain
result = func(*args)
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 1282, in http_open
return self.do_open(http.client.HTTPConnection, req)
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/urllib/request.py", line 1257, in do_open
r = h.getresponse()
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/http/client.py", line 1197, in getresponse
response.begin()
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/http/client.py", line 297, in begin
version, status, reason = self._read_status()
File "/Library/Frameworks/Python.framework/Versions/3.5/lib/python3.5/http/client.py", line 279, in _read_status
raise BadStatusLine(line)
http.client.BadStatusLine: SOURCETABLE 200 OK
从http协议的python文档,我猜这是由于urllib和wget理解标签"SOURCETABLE"在文件的第一个位置,我想加载一些http代码。这个标签总是存在于我想要下载的文件中(ntrip脚轮),但是我找不到解决这个问题的方法。
我在使用另一台NTRIP服务器时遇到了同样的问题。根据RFC 2616, SOURCETABLE 200 OK
不是有效的HTTP状态码。叹息。我的解决方案:curl
,特别是pycurl
。
import sys
import pycurl
def handle_write(buf):
sys.stdout.write(buf.decode("iso-8859-1"))
host = "http://catnet-ip.icc.cat:8080/"
curl = pycurl.Curl()
curl.setopt(pycurl.URL, host)
curl.setopt(pycurl.TIMEOUT, 20)
curl.setopt(pycurl.CONNECTTIMEOUT, 3)
curl.setopt(pycurl.HEADERFUNCTION, sys.stdout.write)
curl.setopt(pycurl.WRITEFUNCTION, handle_write)
curl.perform()
curl.close()
结果:
SOURCETABLE 200 OK
Server: GNSS Spider 7.4.0.8125/1.0
Date: Wed, 18 Dec 2019 21:29:11 GMT Standard Time
Content-Type: text/plain
Content-Length: 2667
STR;VRS_RTK_3_0;VRS_RTK_3_0;RTCM
...
ENDSOURCETABLE
- 通过urllib3
import urllib3
http = urllib3.PoolManager()
foo = http.request('GET',"http://example.com:2101/")
print(foo.data)
- 通过urllib
- 按下一种方法编辑文件,并尝试如下
- 通过http模块您可以将python http库修改为额外的状态代码,您需要编辑
{python programm folder}/lib/http/client.py
中的文件。
import urllib.request
f = urllib.request.urlopen('http://splcare.in:2101')
print(f.read())
查找代码
的行if not version.startswith("HTTP/") :
self._close_conn()
raise BadStatusLine(line)
修改
if not (version.startswith("HTTP/") or version.startswith("SOURCETABLE")):
self._close_conn()
raise BadStatusLine(line)
以下行
的另一个修改elif version.startswith("HTTP/1."):
self.version = 11 # use HTTP/1.1 code for HTTP/1.x where x>=1
else:
raise UnknownProtocol(version)
修改elif version.startswith("HTTP/1."):
self.version = 11 # use HTTP/1.1 code for HTTP/1.x where x>=1
elif version.startswith("SOURCETABLE"):
self.version = 12 # Use NTRIP
else:
raise UnknownProtocol(version)
现在您可以使用以下代码在python控制台上获得源表,但您需要编辑更多文件以支持NTRIP。
import requests
url = "http://example.com:2101"
username = "xxxx"
password = "xxxx"
response = requests.get(url, auth=(username, password))
print(response.status_code)
print(response.content)