{"id":177,"date":"2024-04-29T15:51:14","date_gmt":"2024-04-29T07:51:14","guid":{"rendered":"http:\/\/www.shifengzhi.asia\/?p=177"},"modified":"2024-04-29T15:51:14","modified_gmt":"2024-04-29T07:51:14","slug":"%e7%88%ac%e8%99%ab%e7%ae%80%e5%8d%95%e5%85%a5%e9%97%a8python-%e5%b8%a6%e7%ae%80%e5%8d%95%e5%ae%9e%e6%88%98","status":"publish","type":"post","link":"https:\/\/boki.shifengzhi.top\/?p=177","title":{"rendered":"\u722c\u866b\u7b80\u5355\u5165\u95e8[python \u5e26\u7b80\u5355\u5b9e\u6218]"},"content":{"rendered":"<p>\u4e0b\u9762\u662f\u722c\u53d6\u4e66\u7684\u4ef7\u683c\u548c\u4e66\u540d\u7684\u7b80\u5355\u793a\u4f8b\uff1a<\/p>\n<pre><code class=\"language-python\">import requests\nfrom bs4 import BeautifulSoup\ncontent = requests.get(&quot;http:\/\/books.toscrape.com\/&quot;).text # \u62ff\u5230\u7f51\u9875\u6e90\u7801\nsoup = BeautifulSoup(content, &quot;html.parser&quot;)\nall_prices = soup.find_all(&quot;p&quot;, attrs={&quot;class&quot;: &quot;price_color&quot;}) # \u83b7\u53d6\u6240\u6709\u7c7b\u540d\u4e3a\u201cprice_color\u201d\u7684 p \u6807\u7b7e\nfor price in all_prices:\n    print(price.string) #\u62ff\u5230\u6bcf\u4e2a\u6807\u7b7e\u91cc\u7684\u6587\u5b57\u5185\u5bb9\nall_titles = soup.find_all(&quot;h3&quot;)\nfor title in all_titles:\n    all_links = title.find_all(&quot;a&quot;)\n    for link in all_links:\n        print(link.string)<\/code><\/pre>\n<p>\u5982\u679c<code>h3<\/code>\u91cc\u53ea\u6709\u4e00\u4e2a<code>a<\/code>\u6807\u7b7e\uff0c\u53ef\u4ee5\u7528<code>find()<\/code>\u4ee3\u66ff<code>find_all()<\/code>\uff0c\u91cc\u9762\u4e5f\u4e0d\u9700\u8981\u5faa\u73af\u4e86\uff0c\u5982\u4e0b\uff1a<\/p>\n<pre><code class=\"language-python\">for title in all_titles:\n    link = title.find(&quot;a&quot;)\n    print(link.string)<\/code><\/pre>\n<p>\u518d\u6765\u4e00\u4e2a\u722c\u866b\u7ecf\u5178\u6848\u4f8b\uff0c\u722c\u8c46\u74e3\u7535\u5f71TOP250\u7684\u5f71\u7247\u540d\uff0c\u5b8c\u6574\u4ee3\u7801\u5982\u4e0b:<\/p>\n<pre><code class=\"language-python\">import requests\nfrom bs4 import BeautifulSoup\n\nheaders = {\n    &quot;User-Agent&quot;: &quot;Mozilla\/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit\/537.36 (KHTML, like Gecko) Chrome\/123.0.0.0 Safari\/537.36&quot;\n}\nfor i in range(0, 250, 25):\n    response = requests.get(f&#039;https:\/\/movie.douban.com\/top250?start={i}&#039;, headers=headers)\n    html = response.text\n    soup = BeautifulSoup(html, &#039;html.parser&#039;)\n    all_titles = soup.find_all(&#039;span&#039;, attrs={&#039;class&#039;: &#039;title&#039;})\n    for title in all_titles:\n        title_string = title.string\n        if &#039;\/&#039; not in title_string:\n            print(title_string)<\/code><\/pre>\n<hr \/>\n<p>\u4e0b\u9762\u6765\u8fdb\u9636\u4e00\u70b9\uff0c\u6211\u4eec\u6765\u722c\u4e00\u4e0b\u56fe\u7247<\/p>\n<p>\u4e3a\u4e86\u5b89\u5168\u5c31\u4e0d\u505a\u8fc7\u591a\u89e3\u91ca\u4e86<\/p>\n<pre><code class=\"language-python\">import requests\nfrom bs4 import BeautifulSoup\nfrom urllib.parse import urljoin\n\nheaders = {\n    &#039;Referer&#039;: &#039;https:\/\/www.pixiv.net\/&#039;,\n    &#039;User-Agent&#039;: &#039;Mozilla\/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit\/537.36 (KHTML, like Gecko) Chrome\/123.0.0.0 Safari\/537.36 Edg\/123.0.0.0&#039;\n}\nurl = &#039;https:\/\/www.toopic.cn\/&#039;\npath = &#039;F:\/\/spiderImages&#039;\nresponse = requests.get(url, headers=headers)\nhtml = response.text\nsoup = BeautifulSoup(html, &#039;html.parser&#039;)\nimage_links = []\nfor img in soup.find_all(&#039;img&#039;, attrs={&#039;class&#039;: &#039;lazy&#039;}):\n    img_url = img[&#039;data-original&#039;]\n    # \u5982\u679c\u56fe\u7247\u94fe\u63a5\u662f\u76f8\u5bf9\u8def\u5f84\uff0c\u5219\u5c06\u5176\u8f6c\u6362\u4e3a\u7edd\u5bf9\u8def\u5f84\n    if not img_url.startswith((&#039;http:\/\/&#039;, &#039;https:\/\/&#039;)):\n        img_url = urljoin(url, img_url)\n    image_links.append(img_url)\nfor i, image_link in enumerate(image_links):\n    response = requests.get(image_link, headers=headers)\n    with open(f&#039;{path}\/img{i + 1}.jpg&#039;, &#039;wb&#039;) as file:\n        file.write(response.content)<\/code><\/pre>\n","protected":false},"excerpt":{"rendered":"<p>\u4e0b\u9762\u662f\u722c\u53d6\u4e66\u7684\u4ef7\u683c\u548c\u4e66\u540d\u7684\u7b80\u5355\u793a\u4f8b\uff1a import requests from bs4 import Beau [&hellip;]<\/p>\n","protected":false},"author":1,"featured_media":0,"comment_status":"open","ping_status":"open","sticky":false,"template":"","format":"standard","meta":{"footnotes":""},"categories":[3],"tags":[22],"class_list":["post-177","post","type-post","status-publish","format-standard","hentry","category-3","tag-22"],"_links":{"self":[{"href":"https:\/\/boki.shifengzhi.top\/index.php?rest_route=\/wp\/v2\/posts\/177","targetHints":{"allow":["GET"]}}],"collection":[{"href":"https:\/\/boki.shifengzhi.top\/index.php?rest_route=\/wp\/v2\/posts"}],"about":[{"href":"https:\/\/boki.shifengzhi.top\/index.php?rest_route=\/wp\/v2\/types\/post"}],"author":[{"embeddable":true,"href":"https:\/\/boki.shifengzhi.top\/index.php?rest_route=\/wp\/v2\/users\/1"}],"replies":[{"embeddable":true,"href":"https:\/\/boki.shifengzhi.top\/index.php?rest_route=%2Fwp%2Fv2%2Fcomments&post=177"}],"version-history":[{"count":1,"href":"https:\/\/boki.shifengzhi.top\/index.php?rest_route=\/wp\/v2\/posts\/177\/revisions"}],"predecessor-version":[{"id":178,"href":"https:\/\/boki.shifengzhi.top\/index.php?rest_route=\/wp\/v2\/posts\/177\/revisions\/178"}],"wp:attachment":[{"href":"https:\/\/boki.shifengzhi.top\/index.php?rest_route=%2Fwp%2Fv2%2Fmedia&parent=177"}],"wp:term":[{"taxonomy":"category","embeddable":true,"href":"https:\/\/boki.shifengzhi.top\/index.php?rest_route=%2Fwp%2Fv2%2Fcategories&post=177"},{"taxonomy":"post_tag","embeddable":true,"href":"https:\/\/boki.shifengzhi.top\/index.php?rest_route=%2Fwp%2Fv2%2Ftags&post=177"}],"curies":[{"name":"wp","href":"https:\/\/api.w.org\/{rel}","templated":true}]}}