Python 基础语法
Python 速览
1. 注释
- 单行注释:以#开头,注释内容直到行尾。
- 多行注释:用三个单引号或三个双引号括起来的注释内容。
# 这是一个单行注释
print("Hello, World!") # 这是一个单行注释
# 多行注释
"""
1. 注释1
2. 注释2
"""
大约 3 分钟
# 这是一个单行注释
print("Hello, World!") # 这是一个单行注释
# 多行注释
"""
1. 注释1
2. 注释2
"""
if 语句:根据条件判断是否执行某些代码。if 语句后可有 0 个或多个 elif 语句,最后可跟 0 个或 1 个 else 语句。
# if、elif、else语句
age = 18
if age >= 18:
print("成年人")
elif age >= 12:
print("青少年")
else:
print("未成年人")
for 语句:用于遍历序列(如列表、元组、字符串等)中的每个元素。
# 常规循环
words = ['cat', 'window', 'defenestrate']
for word in words:
print(word, len(word))
"""
cat 3
window 6
defenestrate 12
"""
# 循环对象
users = {
'Hans': 'active',
'Éléonore': 'inactive',
'景太郎': 'active'
}
for user,status in users.copy().items(): # 操作修改时,提前copy一份进行for
print(user)
if status == 'active':
print(user)
else:
del users[user]
print(users)
while 语句:条件为真时循环执行,条件变为假时结束。
# while语句:输出斐波那契数列中小于10的数
a, b = 0, 1
while a < 10:
print(a, end=' ')
a, b = b, a + b
# 0 1 1 2 3 5 8
# a, b = b, a + b 为序列解包:右侧先整体求值,再依次赋值给左侧
range()函数:用于生成一个整数序列,通常用于 for 循环中。
# range()函数
for i in range(5):
print(i)
print("------")
# 指定数字开始
for i in list(range(3, 8)): # 生成[3,8)列表
print(i) # 3 4 5 6 7
print("------")
# 指定步长
for i in list(range(3, 8, 2)): # 生成范围[3,8),步长为2的列表
print(i) # 3 5 7
break 和 continue 语句
# break
for i in list(range(1, 10)):
if i % 3 == 0: break
print(i) # 1 2
# continue
for i in list(range(1, 10)):
if i % 3 == 0: continue
print(i) # 1 2 4 5 7 8
循环的 else 子句
模块是一个包含 Python 定义和语句的文件,文件名就是模块名(去掉 .py 后缀)。
# 引入指定目录中的模块
from modules.fb import fib as fb
# as 重命名
fb(500)
import json
# 格式化字符串字面量
year = 2026
event = "Referendum"
# 在字符串前加f或F,即可使用变量
print(f'Results of the {year} {event}')
# 冒号后可跟格式化描述符:.3f保留3位小数、>n右对齐补空格、<n左对齐
import math
print(f'pi ≈ {math.pi:.3f}') # pi ≈ 3.142
print(f'[{92.5:>8.1f}]') # [ 92.5],右对齐,总宽度8,保留1位小数
table2 = {'Sjoerd': 4127, 'Jack': 4098, 'Dcab': 7678}
for name, phone in table2.items():
print(f'{name:8} ==> {phone:6d}')
# Sjoerd ==> 4127
# Jack ==> 4098
# Dcab ==> 7678
# str.format()方法
print("今年是{}年".format(year)) # 今年是2026年
print('{1} and {0}'.format('spam', 'eggs')) # eggs and spam
print('This {food} is {adjective}.'.format(
food='spam', adjective='absolutely horrible')) # This spam is absolutely horrible.
table = {'Sjoerd': 4127, 'Jack': 4098, 'Dcab': 8637678}
print('Jack: {0[Jack]:d}; Sjoerd: {0[Sjoerd]:d}; '
'Dcab: {0[Dcab]:d}'.format(table)) # Jack: 4098; Sjoerd: 4127; Dcab: 8637678
for x in range(1, 11):
print('{0:d} {1:d} {2:d}'.format(x, x * x, x * x * x))
语法错误(SyntaxError)是最基本的错误类型,Python 执行代码前会先解析语法,语法有问题时直接报错,程序不会运行。
# 缺少冒号
if True print("hello")
# SyntaxError: expected ':'
类提供了把数据和功能绑定在一起的方法,类的实例化会创建一个新的对象,每个对象都有自己的属性和方法。
对象之间相互独立,多个名称(甚至是多个作用域内的多个名称)可以绑定到同一对象。
# 赋值不会拷贝对象,只是让多个名称指向同一个对象
a = [1, 2, 3]
b = a # b和a绑定的是同一个列表
b.append(4)
print(a) # [1, 2, 3, 4],通过b修改,a也"变"了
print(a is b) # True,is判断两个名称是否指向同一个对象
# 想要独立的副本,需创建新对象(切片是浅拷贝)
c = a[:]
c.append(5)
print(a) # [1, 2, 3, 4],不受c影响
print(c) # [1, 2, 3, 4, 5]
import os
import shutil
import glob
import sys
import argparse # 提供了一种更复杂的机制处理命令行参数
import re # 提供正则表达式工具
import math # 提供数学函数
import random # 随机数
import statistics # 基本的统计
from urllib.request import urlopen # 从url检索数据
from datetime import date # 操作日期和时间
import zlib # 压缩库
# os库:提供了很多操作系统交互的函数
print(os.getcwd()) # 返回当前目录
print(dir(os)) # 返回由模块的所有函数组成的列表
# 调用系统命令
os.system("ls") # 可执行系统命令
# os.system("mkdir files") # 可执行系统命令
# shutil库:提供文件和目录管理接口
shutil.copyfile('file.txt', 'dataCopy.txt') # 拷贝文件,目标文件不存在时会创建
# shutil.move('dataCopy.txt','files') # 移动文件,目标文件夹中存在时会报错
# glob库:查询文件
print(glob.glob('*.py')) # ['类.py', '输入输出.py', 'main.py', '标准库.py']
# sys库:命令行参数存储于sys的argv属性中
print(sys.argv) # 获取执行python时的命令行参数,例如python3 demo.py one two three,会打印['demo.py', 'one', 'two', 'three']
# parser = argparse.ArgumentParser(
# prog='top',
# description='Show top lines from each file')
# parser.add_argument('filenames', nargs='+')
# parser.add_argument('-l', '--lines', type=int, default=10)
# args = parser.parse_args()
# print(args)
# 终止脚本:exit()
# sys.exit()
# re库:提供正则表达式处理
filter_str = re.findall(r'\bf[a-z]*', 'which foot or hand fell fastest')
print(filter_str) # ['foot', 'fell', 'fastest']
# math库:数学工具函数
cos = math.cos(math.pi / 2) # 6.123233995736766e-17
print(cos)
sin = math.sin(math.pi / 2) # 1
print(sin)
log = math.log(1024, 2)
print(log) # 10
# random库:随机数
number1 = random.random()
print("random:", number1) # [0.0, 1.0)中的浮点随机数
ls = random.sample(range(0, 100), 10) # 无重复值的10个随机数
print("sample:", ls) # 随机10个数字
rg = random.randrange(6)
print("rg:", rg)
# statistic库:基本的统计属性
data = [1, 2, 3]
avg_number = statistics.mean(data)
print("均值:", avg_number) # 均值: 2
mid_number = statistics.median(data)
print("中位数:", mid_number) # 中位数: 2
var_number = statistics.variance(data)
print("方差:", var_number) # 方差: 1
# 获取网页
with urlopen('https://www.baidu.com/') as response:
html = response.read()
# 写入文件中
with open('baidu.html', 'wb') as f:
f.write(html)
f.close()
print(html)
# datetime库:操作日期和时间
now = date.today()
print("今天日期:", now.today()) # 今天日期: 2026-01-23
print("年份:", now.year) # 时间: 2026
print("月份:", now.month) # 时间: 1
# zip库:数据压缩,支持zlib、gzip、bz2、lzma、zipfile、tarfile
s = '测试文本内容'.encode('utf-8')
print(f"原文长度:{len(s)}") # 原文长度:41
print(f"原文:{s.decode('utf-8')}")
t = zlib.compress(s)
print(f"压缩后长度:{len(t)}") # 压缩后长度:37
print(f"压缩后:{t}")
rt = zlib.decompress(t)
print(f"解压缩后:{rt.decode('utf-8')}") # 测试文本内容
import requests
import json
def get_zhihu_hot_list():
url = "https://api.zhihu.com/topstory/hot-list" # 移动端获取热搜榜单数据api地址
# 模拟请求头
headers = {
"User-Agent": "ZhihuHybrid com.zhihu.android/Futureve/6.59.0 Mozilla/5.0 (Linux; Android 10; SM-G9650 Build/QP1A.190711.020; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/83.0.4103.106 Mobile Safari/537.36",
}
try:
# 获取api数据
response = requests.get(url, headers=headers)
# 检测https请求状态
response.raise_for_status()
# 最终接口数据
data = response.json()
if "data" in data:
print(f"{'排名':<5}{'热度':<15}{'标题'}")
print("-" * 80)
# enumerate(list):同时获取索引和值
for index, item in enumerate(data["data"], 1):
# 获取字典中的数据
target = item.get("target", {})
title = target.get("title", "No Title")
# 容错
hot_value = item.get("detail_text", "")
if not hot_value: # 找不到数据时
hot_value = "N/A"
print(f"{index:<5}{hot_value:<15}{title}")
else:
print("Failed to retrieve data format expected.")
except requests.exceptions.RequestException as e:
print(f"Error fetching data: {e}")
except json.JSONDecodeError:
print("Error decoding JSON response")
if __name__ == "__main__":
print("---------欢迎启动python爬虫程序✈️,将爬取的网站信息为:知乎热搜榜---------")
get_zhihu_hot_list()
# Playwright动态网页爬取
from playwright.async_api import async_playwright
# 运行下载浏览器
import asyncio
# 引入html解析工具
from bs4 import BeautifulSoup
async def snail_website():
async with async_playwright() as p:
browser = await p.chromium.launch(headless=False) # 启动chrome浏览器
# 模拟浏览器信息,绕过403
context = await browser.new_context(
user_agent='Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
extra_http_headers={
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
'Accept-Encoding': 'gzip, deflate, br',
'Connection': 'keep-alive',
'Upgrade-Insecure-Requests': '1',
'Sec-Fetch-Dest': 'document',
'Sec-Fetch-Mode': 'navigate',
'Sec-Fetch-Site': 'none',
'Sec-Fetch-User': '?1',
'Cache-Control': 'max-age=0'
}
)
page = await browser.new_page() # 打开新页面
await page.goto("https://www.weather.com.cn/weather1d/101210101.shtml")
# 获取页面标题
title = await page.title()
# 获取内容
content = await page.content()
# 格式化html
soup = BeautifulSoup(content, 'html.parser')
# 找到当前温度div
temp_div = soup.find('div', class_='mySkyNull')
# 取出其中的温度数值
temp = temp_div.find('div', class_='tem').text
print(f"当前温度:{temp}")
# 运行爬虫函数
asyncio.run(snail_website())