diff --git a/q2_1_crawler/movie.html b/q2_1_crawler/movie.html new file mode 100644 index 0000000..c8a0a99 --- /dev/null +++ b/q2_1_crawler/movie.html @@ -0,0 +1,102 @@ +[ + { + "id": "1", + "title": "阿甘正传", + "director": "Frank Darabont", + "year": "2023", + "rating": "8.5", + "duration": "112", + "genre": "科幻", + "actors_count": "4" + }, + { + "id": "2", + "title": "忠犬八公的故事", + "director": "陈凯歌", + "year": "2023", + "rating": "6.9", + "duration": "102", + "genre": "动画", + "actors_count": "4" + }, + { + "id": "3", + "title": "放牛班的春天", + "director": "Robert Zemeckis", + "year": "1993", + "rating": "7.6", + "duration": "175", + "genre": "动画", + "actors_count": "5" + }, + { + "id": "4", + "title": "星际穿越", + "director": "James Cameron", + "year": "2024", + "rating": "9.1", + "duration": "92", + "genre": "喜剧", + "actors_count": "5" + }, + { + "id": "5", + "title": "千与千寻", + "director": "宫崎骏", + "year": "2007", + "rating": "8.3", + "duration": "146", + "genre": "悬疑", + "actors_count": "4" + }, + { + "id": "6", + "title": "肖申克的救赎", + "director": "Christopher Nolan", + "year": "1997", + "rating": "8.3", + "duration": "117", + "genre": "爱情", + "actors_count": "5" + }, + { + "id": "7", + "title": "盗梦空间", + "director": "Lasse Hallström", + "year": "2001", + "rating": "7.5", + "duration": "131", + "genre": "悬疑", + "actors_count": "3" + }, + { + "id": "8", + "title": "泰坦尼克号", + "director": "Rajkumar Hirani", + "year": "1994", + "rating": "6.8", + "duration": "163", + "genre": "喜剧", + "actors_count": "4" + }, + { + "id": "9", + "title": "三傻大闹宝莱坞", + "director": "Christophe Barratier", + "year": "1997", + "rating": "6.8", + "duration": "107", + "genre": "冒险", + "actors_count": "5" + }, + { + "id": "10", + "title": "霸王别姬", + "director": "Christopher Nolan", + "year": "2006", + "rating": "9.1", + "duration": "98", + "genre": "科幻", + "actors_count": "2" + } +] \ No newline at end of file diff --git a/q2_1_crawler/movie.json b/q2_1_crawler/movie.json new file mode 100644 index 0000000..c8a0a99 --- /dev/null +++ b/q2_1_crawler/movie.json @@ -0,0 +1,102 @@ +[ + { + "id": "1", + "title": "阿甘正传", + "director": "Frank Darabont", + "year": "2023", + "rating": "8.5", + "duration": "112", + "genre": "科幻", + "actors_count": "4" + }, + { + "id": "2", + "title": "忠犬八公的故事", + "director": "陈凯歌", + "year": "2023", + "rating": "6.9", + "duration": "102", + "genre": "动画", + "actors_count": "4" + }, + { + "id": "3", + "title": "放牛班的春天", + "director": "Robert Zemeckis", + "year": "1993", + "rating": "7.6", + "duration": "175", + "genre": "动画", + "actors_count": "5" + }, + { + "id": "4", + "title": "星际穿越", + "director": "James Cameron", + "year": "2024", + "rating": "9.1", + "duration": "92", + "genre": "喜剧", + "actors_count": "5" + }, + { + "id": "5", + "title": "千与千寻", + "director": "宫崎骏", + "year": "2007", + "rating": "8.3", + "duration": "146", + "genre": "悬疑", + "actors_count": "4" + }, + { + "id": "6", + "title": "肖申克的救赎", + "director": "Christopher Nolan", + "year": "1997", + "rating": "8.3", + "duration": "117", + "genre": "爱情", + "actors_count": "5" + }, + { + "id": "7", + "title": "盗梦空间", + "director": "Lasse Hallström", + "year": "2001", + "rating": "7.5", + "duration": "131", + "genre": "悬疑", + "actors_count": "3" + }, + { + "id": "8", + "title": "泰坦尼克号", + "director": "Rajkumar Hirani", + "year": "1994", + "rating": "6.8", + "duration": "163", + "genre": "喜剧", + "actors_count": "4" + }, + { + "id": "9", + "title": "三傻大闹宝莱坞", + "director": "Christophe Barratier", + "year": "1997", + "rating": "6.8", + "duration": "107", + "genre": "冒险", + "actors_count": "5" + }, + { + "id": "10", + "title": "霸王别姬", + "director": "Christopher Nolan", + "year": "2006", + "rating": "9.1", + "duration": "98", + "genre": "科幻", + "actors_count": "2" + } +] \ No newline at end of file diff --git a/q2_1_crawler/q2_1.py b/q2_1_crawler/q2_1.py new file mode 100644 index 0000000..22dbf75 --- /dev/null +++ b/q2_1_crawler/q2_1.py @@ -0,0 +1,48 @@ +import requests +from bs4 import BeautifulSoup as bs +import json + +url = 'https://exam.detr.top/exam-b/movies' +headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36 Edg/149.0.0.0', + 'Referer':'https://exam.detr.top/exam-b/movies'} +req = requests.get(url, headers=headers) +req.encoding="utf-8" + +data=[] + +soup=bs(req.text,"html.parser") +# print(soup) +#id, title, director, year, rating, duration, genre, actors-count + +items=soup.select('.item-row') +# movie_list=[] +for i in range(len(items)): + id=items[i].find("td",class_="item-id").get_text() + title=items[i].find("td",class_="item-title").get_text().strip() + director=items[i].find("td",class_="item-director").get_text().strip() + year=items[i].find("td",class_="item-year").get_text() + rating=items[i].find("td",class_="item-rating").get_text() + duration=items[i].find("td",class_="item-duration").get_text() + genre=items[i].find("td",class_="item-genre").get_text() + actors_count=items[i].find("td",class_="item-actors-count").get_text() + + data.append({ + "id":id, + "title":title, + "director":director, + "year":year, + "rating":rating, + "duration":duration, + "genre":genre, + "actors_count":actors_count + + }) + +print(data) + + +with open('movie.json', 'w', encoding='utf-8') as f: + json.dump(data, f, ensure_ascii=False, indent=2) + +with open("movie.html","w",encoding='utf-8') as f: + json.dump(data, f, ensure_ascii=False, indent=2) \ No newline at end of file diff --git a/q2_1_crawler/q2_2.py b/q2_1_crawler/q2_2.py new file mode 100644 index 0000000..f30c2f7 --- /dev/null +++ b/q2_1_crawler/q2_2.py @@ -0,0 +1,43 @@ +# ① 找出评分最高和最低的电影,打印电影名 + 评分。 +# ② 统计各类型的电影数量,用字典格式输出。 +# ③ 统计各导演的电影数量,用字典格式输出。 +# ④ 统计 2020 年(含)以后上映的电影数量。 + +import json + + + +with open('movie.json', 'r', encoding='utf-8') as f: + data=json.load(f) + # print(data) + + +sort_movie=sorted(data,key=lambda x:x["rating"]) +min=sort_movie[0] +max=sort_movie[-1] +print("评分最低的电影",min["title"],min["rating"]) +print("评分最高的电影",max["title"],max["rating"]) + +genre_shu={} +for g in data: + ge=g["genre"] + if ge in genre_shu: + genre_shu[ge]+=1 + else: + genre_shu[ge]=1 +print("各类型的电影数量",genre_shu) + +director_shu={} +for d in data: + di=d["director"] + if di in director_shu: + director_shu[di]+=1 + else: + director_shu[di]=1 +print("各导演的电影数量",director_shu) + +a=0 +for y in data: + if int(y["year"]) >= 2020: + a+=1 +print("2020 年(含)以后上映的电影数量",a) \ No newline at end of file diff --git a/q3/q3_1/q3_1_image_label.zip b/q3/q3_1/q3_1_image_label.zip new file mode 100644 index 0000000..5b08d84 Binary files /dev/null and b/q3/q3_1/q3_1_image_label.zip differ diff --git a/q3/q3_2/q3_2_takeout_reviews.json b/q3/q3_2/q3_2_takeout_reviews.json new file mode 100644 index 0000000..94c7925 --- /dev/null +++ b/q3/q3_2/q3_2_takeout_reviews.json @@ -0,0 +1 @@ +[{"id":32,"annotations":[{"id":20,"completed_by":1,"result":[{"value":{"choices":["正面"]},"id":"Onq13IcOtQ","from_name":"sentiment","to_name":"text","type":"choices","origin":"manual"}],"was_cancelled":false,"ground_truth":false,"created_at":"2026-06-25T07:10:52.845728Z","updated_at":"2026-06-25T07:10:52.845818Z","draft_created_at":"2026-06-25T07:10:22.361376Z","lead_time":12.684999999999999,"prediction":{},"result_count":1,"unique_id":"0dd85e72-c68d-4302-ab92-11c8114bd5bb","import_id":null,"last_action":null,"bulk_created":false,"task":32,"project":12,"updated_by":1,"parent_prediction":null,"parent_annotation":null,"last_created_by":null}],"file_upload":"9ce97d9b-reviews.json","drafts":[],"predictions":[],"data":{"id":1,"text":"外卖小哥送得超快,餐盒还是热的,炸鸡酥脆多汁,酸辣粉也很正宗,分量足,五星好评!"},"meta":{},"created_at":"2026-06-25T07:10:04.754168Z","updated_at":"2026-06-25T07:10:52.903174Z","allow_skip":true,"inner_id":1,"total_annotations":1,"cancelled_annotations":0,"total_predictions":0,"comment_count":0,"unresolved_comment_count":0,"last_comment_updated_at":null,"project":12,"updated_by":1,"comment_authors":[]},{"id":33,"annotations":[{"id":21,"completed_by":1,"result":[{"value":{"choices":["负面"]},"id":"cJxGS89PGV","from_name":"sentiment","to_name":"text","type":"choices","origin":"manual"}],"was_cancelled":false,"ground_truth":false,"created_at":"2026-06-25T07:10:56.177683Z","updated_at":"2026-06-25T07:10:56.177702Z","draft_created_at":"2026-06-25T07:10:40.725815Z","lead_time":4.317,"prediction":{},"result_count":1,"unique_id":"ec19a0ee-0062-4fcb-be7e-3885bbb0e630","import_id":null,"last_action":null,"bulk_created":false,"task":33,"project":12,"updated_by":1,"parent_prediction":null,"parent_annotation":null,"last_created_by":null}],"file_upload":"9ce97d9b-reviews.json","drafts":[],"predictions":[],"data":{"id":2,"text":"等了一个半小时才送到,汤全洒了,面坨成一坨,联系客服也不回,太让人失望了。"},"meta":{},"created_at":"2026-06-25T07:10:04.754256Z","updated_at":"2026-06-25T07:10:56.235324Z","allow_skip":true,"inner_id":2,"total_annotations":1,"cancelled_annotations":0,"total_predictions":0,"comment_count":0,"unresolved_comment_count":0,"last_comment_updated_at":null,"project":12,"updated_by":1,"comment_authors":[]},{"id":34,"annotations":[{"id":22,"completed_by":1,"result":[{"value":{"choices":["正面"]},"id":"TEYHnwsP5p","from_name":"sentiment","to_name":"text","type":"choices","origin":"manual"}],"was_cancelled":false,"ground_truth":false,"created_at":"2026-06-25T07:11:02.709432Z","updated_at":"2026-06-25T07:11:02.709456Z","draft_created_at":"2026-06-25T07:10:44.119328Z","lead_time":5.162,"prediction":{},"result_count":1,"unique_id":"e99cc8d6-b0ec-4ba7-95c9-75e600010972","import_id":null,"last_action":null,"bulk_created":false,"task":34,"project":12,"updated_by":1,"parent_prediction":null,"parent_annotation":null,"last_created_by":null}],"file_upload":"9ce97d9b-reviews.json","drafts":[],"predictions":[],"data":{"id":3,"text":"奶茶是用料很扎实的现煮茶,珍珠Q弹有嚼劲,配送员态度也好,下次还会再点。"},"meta":{},"created_at":"2026-06-25T07:10:04.754319Z","updated_at":"2026-06-25T07:11:02.772529Z","allow_skip":true,"inner_id":3,"total_annotations":1,"cancelled_annotations":0,"total_predictions":0,"comment_count":0,"unresolved_comment_count":0,"last_comment_updated_at":null,"project":12,"updated_by":1,"comment_authors":[]},{"id":35,"annotations":[{"id":23,"completed_by":1,"result":[{"value":{"choices":["正面"]},"id":"WSCVtAKzJa","from_name":"sentiment","to_name":"text","type":"choices","origin":"manual"}],"was_cancelled":false,"ground_truth":false,"created_at":"2026-06-25T07:11:05.544799Z","updated_at":"2026-06-25T07:11:05.544821Z","draft_created_at":"2026-06-25T07:10:46.400039Z","lead_time":3.517,"prediction":{},"result_count":1,"unique_id":"b72bb1ff-9547-471d-8edd-58f5277dca12","import_id":null,"last_action":null,"bulk_created":false,"task":35,"project":12,"updated_by":1,"parent_prediction":null,"parent_annotation":null,"last_created_by":null}],"file_upload":"9ce97d9b-reviews.json","drafts":[],"predictions":[],"data":{"id":4,"text":"配送速度一般,但披萨味道不错,芝士拉丝效果好,性价比高,值得推荐。"},"meta":{},"created_at":"2026-06-25T07:10:04.754376Z","updated_at":"2026-06-25T07:11:05.601129Z","allow_skip":true,"inner_id":4,"total_annotations":1,"cancelled_annotations":0,"total_predictions":0,"comment_count":0,"unresolved_comment_count":0,"last_comment_updated_at":null,"project":12,"updated_by":1,"comment_authors":[]},{"id":36,"annotations":[{"id":24,"completed_by":1,"result":[{"value":{"choices":["负面"]},"id":"qLglBmVk44","from_name":"sentiment","to_name":"text","type":"choices","origin":"manual"}],"was_cancelled":false,"ground_truth":false,"created_at":"2026-06-25T07:11:08.643150Z","updated_at":"2026-06-25T07:11:08.643172Z","draft_created_at":"2026-06-25T07:10:48.222193Z","lead_time":2.6100000000000003,"prediction":{},"result_count":1,"unique_id":"7e05411c-9818-4359-872a-ed1e588d0687","import_id":null,"last_action":null,"bulk_created":false,"task":36,"project":12,"updated_by":1,"parent_prediction":null,"parent_annotation":null,"last_created_by":null}],"file_upload":"9ce97d9b-reviews.json","drafts":[],"predictions":[],"data":{"id":5,"text":"点的麻辣烫食材不新鲜,有股怪味,吃完拉肚子,商家推卸责任,再也不点了。"},"meta":{},"created_at":"2026-06-25T07:10:04.754432Z","updated_at":"2026-06-25T07:11:08.699069Z","allow_skip":true,"inner_id":5,"total_annotations":1,"cancelled_annotations":0,"total_predictions":0,"comment_count":0,"unresolved_comment_count":0,"last_comment_updated_at":null,"project":12,"updated_by":1,"comment_authors":[]}] \ No newline at end of file diff --git a/q3/q3_3_质量自评.md b/q3/q3_3_质量自评.md new file mode 100644 index 0000000..6f2ef33 --- /dev/null +++ b/q3/q3_3_质量自评.md @@ -0,0 +1,9 @@ +标注质量自评 +1.标注前准备 +本次文本情感分类标注规范:仅设置(正面)(负面)俩类标签,一条评论只能标注单一情感,无中立标签;判断依据以用户核心情绪、评价倾向为准,忽略无关客观描述。标注前查看2条示例评论,明确极端好评、差评的区别边界,统一判定标准。 + +2.标注过程 +困难:部分评论同时存在优缺点混合描述,情感倾向模糊。解决方法:抓取句子核心总结、整体推荐态度,以最终情绪导向判定;不确定样本反复重读全文,不随意打标签,保证标签唯一。 + +3.标注后检查 +逐条核对5条评论,确保每条都存在id、text、sentiment字段,无漏标数据;没有导入重复数据,全部素材仅标注一次,导出JSON校验字段格式符合作业要求,无缺失、错标。 \ No newline at end of file diff --git a/q4/q4_1/q4_1.py b/q4/q4_1/q4_1.py new file mode 100644 index 0000000..e371b3e --- /dev/null +++ b/q4/q4_1/q4_1.py @@ -0,0 +1,40 @@ +import matplotlib.pyplot as plt +import json + + +with open(r'D:\桌面\期末考试\simulated-examination\q2_1_crawler\movie.json', 'r', encoding='utf-8') as f: + data=json.load(f) + # print(data) + +genre_shu={} +for g in data: + ge=g["genre"] + if ge in genre_shu: + genre_shu[ge]+=1 + else: + genre_shu[ge]=1 +# print("各类型的电影数量",genre_shu) + +genre_lei=list(genre_shu.keys()) +genre_liang=list(genre_shu.values()) + +print(genre_liang) + +plt.figure(figsize=(14, 12)) +plt.bar(genre_lei, genre_liang, + width=0.6) + +# 标题和标签 +plt.title('类型电影数量分布', fontsize=14) +plt.xlabel('类型名称', fontsize=12) +plt.ylabel('电影数量', fontsize=12) + + +import os +save_path = os.path.join(os.path.dirname(__file__),"q4_1_bar.png") +plt.savefig(save_path, dpi=150,format='png') +plt.show() + + + +#with open('movie.json', 'r', encoding='utf-8') as f: \ No newline at end of file diff --git a/q4/q4_1/q4_1_bar.png b/q4/q4_1/q4_1_bar.png new file mode 100644 index 0000000..553bf7d Binary files /dev/null and b/q4/q4_1/q4_1_bar.png differ diff --git a/q4/q4_2/q4_2.py b/q4/q4_2/q4_2.py new file mode 100644 index 0000000..cf32d5b --- /dev/null +++ b/q4/q4_2/q4_2.py @@ -0,0 +1,28 @@ +import matplotlib.pyplot as plt +import json + +rating=[] +duration=[] + +with open(r'D:\桌面\期末考试\simulated-examination\q2_1_crawler\movie.json', 'r', encoding='utf-8') as f: + data=json.load(f) + # print(data) + for i in data: + rating.append(i["rating"]) + duration.append(i["duration"]) +plt.figure(figsize=(12, 8)) +plt.scatter(duration, rating, + c='red', + s=80, # 点的大小 + alpha=0.6, # 透明度 + edgecolors='white') # 点的边框 +plt.title('时长与评分关系散点图', fontsize=14) +plt.xlabel('时长', fontsize=12) +plt.ylabel('评分', fontsize=12) +plt.grid(True, linestyle='--', alpha=0.5) + + +import os +save_path = os.path.join(os.path.dirname(__file__),"q4_2_scatter.png") +plt.savefig(save_path, dpi=150,format='png') +plt.show() \ No newline at end of file diff --git a/q4/q4_2/q4_2_scatter.png b/q4/q4_2/q4_2_scatter.png new file mode 100644 index 0000000..dc99ee4 Binary files /dev/null and b/q4/q4_2/q4_2_scatter.png differ diff --git a/q4/q4_3a/q4_3a.py b/q4/q4_3a/q4_3a.py new file mode 100644 index 0000000..c8ed2f4 --- /dev/null +++ b/q4/q4_3a/q4_3a.py @@ -0,0 +1,24 @@ +import matplotlib.pyplot as plt +import json + +rating=[] + +with open(r'D:\桌面\期末考试\simulated-examination\q2_1_crawler\movie.json', 'r', encoding='utf-8') as f: + data=json.load(f) + # print(data) + for i in data: + rating.append(i["rating"]) + +plt.figure(figsize=(12,8)) +plt.hist(rating, # 数据 + bins=10, # 分成几个柱子 + color='#3498DB', # 颜色 + edgecolor='white') # 柱子边框颜色 +plt.title('评分分布', fontsize=14) +plt.xlabel('评分', fontsize=13) +plt.grid(True, linestyle='--', alpha=0.5, axis='y') + +import os +save_path = os.path.join(os.path.dirname(__file__),"q4_3a_hist.png") +plt.savefig(save_path, dpi=150,format='png') +plt.show() \ No newline at end of file diff --git a/q4/q4_3a/q4_3a_hist.png b/q4/q4_3a/q4_3a_hist.png new file mode 100644 index 0000000..a6e41a1 Binary files /dev/null and b/q4/q4_3a/q4_3a_hist.png differ diff --git a/q4/q4_3b/q4_3b.py b/q4/q4_3b/q4_3b.py new file mode 100644 index 0000000..7b2480a --- /dev/null +++ b/q4/q4_3b/q4_3b.py @@ -0,0 +1,24 @@ +import matplotlib.pyplot as plt +import json +duration=[] + +with open(r'D:\桌面\期末考试\simulated-examination\q2_1_crawler\movie.json', 'r', encoding='utf-8') as f: + data=json.load(f) + # print(data) + for i in data: + duration.append(i["duration"]) + +plt.figure(figsize=(12,8)) +plt.hist( duration, # 数据 + bins=10, # 分成几个柱子 + color='#3498DB', # 颜色 + edgecolor='white') # 柱子边框颜色 +plt.title('时长分布', fontsize=14) +plt.xlabel('时长(分钟)', fontsize=13) +plt.grid(True, linestyle='--', alpha=0.5, axis='y') + + +import os +save_path = os.path.join(os.path.dirname(__file__),"q4_3b_hist.png") +plt.savefig(save_path, dpi=150,format='png') +plt.show() \ No newline at end of file diff --git a/q4/q4_3b/q4_3b_hist.png b/q4/q4_3b/q4_3b_hist.png new file mode 100644 index 0000000..0dfef73 Binary files /dev/null and b/q4/q4_3b/q4_3b_hist.png differ