Repository navigation
Expand file tree
/
Copy pathtest1.py
More file actions
121 lines (103 loc) · 5.29 KB
/
Copy pathtest1.py
File metadata and controls
121 lines (103 loc) · 5.29 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
'''''
문제. step11 의 매출 데이터로 한 장짜리 대시보드를 만드세요.
배치 (2행 3열, 왼쪽 위는 2칸 합쳐서)
┌─────────────────────┬──────────┐
│ 월별 매출 추이 │ 분류 비중 │
│ (최고·최저 강조) ├──────────┤
├─────────────────────┤ 채널 비교 │
│ 지역별 매출 │ │
└─────────────────────┴──────────┘
이 파일은 step10~12 를 한 번에 쓰는 종합 연습입니다.
step10 : argmax / 배열 연산
step11 : read_csv / merge / groupby / pivot_table
step12 : gridspec / 주석 / 강조
'''''
import matplotlib.pyplot as plt
import pandas as pd
from pathlib import Path
import krfont
krfont.apply()
SHOW = False
DATA = Path('..') / 'step11_pandas' / 'data'
OUT = Path('output')
OUT.mkdir(exist_ok=True)
# ── 1. 데이터 준비 (step11 그대로) ────────────────────────────
sales = pd.read_csv(DATA / 'sales.csv', parse_dates=['날짜'])
products = pd.read_csv(DATA / 'products.csv')
before = len(sales)
sales = sales.drop_duplicates().fillna({'지역': '미상'}).reset_index(drop=True)
df = sales.merge(products, on='상품코드', how='left')
assert len(df) == len(sales), 'merge 후 행 수가 달라졌습니다'
df['매출'] = df['수량'] * df['단가']
df['월'] = df['날짜'].dt.month
print(f'정제 : {before}행 → {len(df)}행 / 총매출 {df["매출"].sum():,}원')
monthly = df.groupby('월')['매출'].sum()
by_cat = df.groupby('분류')['매출'].sum().sort_values(ascending=False)
by_region = df.groupby('지역')['매출'].sum().sort_values()
by_channel = df.pivot_table(index='분류', columns='채널', values='매출',
aggfunc='sum', fill_value=0)
# ── 2. 대시보드 ──────────────────────────────────────────────
fig = plt.figure(figsize=(14, 8))
gs = fig.add_gridspec(2, 3, hspace=0.35, wspace=0.28)
# (1) 월별 추이 — 최고·최저 강조
ax1 = fig.add_subplot(gs[0, :2])
ax1.plot(monthly.index, monthly.values, marker='o',
color='lightsteelblue', linewidth=2, zorder=1)
hi, lo = monthly.idxmax(), monthly.idxmin() # step11 의 idxmax
ax1.scatter([hi, lo], [monthly[hi], monthly[lo]],
color=['crimson', 'steelblue'], s=110, zorder=3)
ax1.annotate(f'최고 {monthly[hi]:,}', xy=(hi, monthly[hi]),
xytext=(hi + 0.6, monthly[hi] + 20000), color='crimson',
arrowprops=dict(arrowstyle='->', color='crimson'))
ax1.annotate(f'최저 {monthly[lo]:,}', xy=(lo, monthly[lo]),
xytext=(lo + 0.6, monthly[lo] - 45000), color='steelblue',
arrowprops=dict(arrowstyle='->', color='steelblue'))
ax1.axhline(monthly.mean(), color='gray', linestyle='--', alpha=0.6)
# 라벨을 오른쪽 끝에 붙인다. 왼쪽에 두면 y축 눈금과 겹친다.
ax1.text(12.3, monthly.mean(), f'평균 {monthly.mean():,.0f}',
color='gray', fontsize=9, va='center', ha='left')
ax1.set_title('월별 매출 추이', fontsize=13)
ax1.set_xticks(range(1, 13))
ax1.set_xticklabels([f'{m}월' for m in range(1, 13)], fontsize=9)
ax1.set_ylim(0, monthly.max() * 1.25)
ax1.set_ylabel('매출 (원)')
ax1.grid(axis='y', alpha=0.25)
ax1.spines[['top', 'right']].set_visible(False)
# (2) 분류 비중 — 파이 대신 가로 막대 (chartEx2.py 참고)
ax2 = fig.add_subplot(gs[0, 2])
share = by_cat / by_cat.sum() * 100
bars = ax2.barh(share.index[::-1], share.values[::-1], color='#4C72B0')
for bar, v in zip(bars, share.values[::-1]):
ax2.text(v + 1, bar.get_y() + bar.get_height() / 2,
f'{v:.1f}%', va='center', fontsize=9)
ax2.set_xlim(0, 60)
ax2.set_title('분류별 비중', fontsize=13)
ax2.spines[['top', 'right']].set_visible(False)
# (3) 지역별 매출
ax3 = fig.add_subplot(gs[1, :2])
colors = ['#DD8452' if r == '미상' else '#4C72B0' for r in by_region.index]
ax3.bar(by_region.index, by_region.values, color=colors)
for x, v in enumerate(by_region.values):
ax3.text(x, v + 8000, f'{v/10000:.0f}만', ha='center', fontsize=9)
ax3.set_title('지역별 매출 (주황 = 지역 미입력)', fontsize=13)
ax3.set_ylim(0, by_region.max() * 1.18)
ax3.set_ylabel('매출 (원)')
ax3.grid(axis='y', alpha=0.25)
ax3.spines[['top', 'right']].set_visible(False)
# (4) 분류 × 채널
ax4 = fig.add_subplot(gs[1, 2])
by_channel.plot(kind='bar', stacked=True, ax=ax4, legend=True) # pandas 로 그리고
ax4.set_title('분류 × 채널', fontsize=13) # matplotlib 으로 다듬기
ax4.set_xlabel('')
ax4.tick_params(axis='x', rotation=0, labelsize=9)
ax4.legend(fontsize=8, title='채널')
ax4.spines[['top', 'right']].set_visible(False)
fig.suptitle(f'2019년 매출 대시보드 | {len(df)}건 · '
f'{df["매출"].sum():,}원', fontsize=16, y=0.98)
fig.savefig(OUT / '60_dashboard.png', dpi=120, bbox_inches='tight')
if SHOW:
plt.show()
plt.close(fig)
print('60_dashboard.png 저장 완료')
print(f' 최고 {hi}월 {monthly[hi]:,}원 / 최저 {lo}월 {monthly[lo]:,}원')
print(f' 1위 분류 {by_cat.index[0]} ({share.iloc[0]:.1f}%)')