معاينة مختبر آمنة
notebook
هذي معاينة منقّحة للقراءة فقط؛ ما فيه أي شيء يشتغل داخل الصفحة.
قراءة فقط
معاينة الدفتر
notebook
> **ملاحظة بيئة التشغيل المدمجة:** هالمعاينة تستخدم عيّنة صغيرة وثابتة وآمنة من ناحية الحقوق عشان تكون النتايج قابلة للتكرار. النتايج بالحجم الكامل تحتاج مجموعة البيانات أو النموذج الموثّق بالدرس داخل بيئة خارجية معتمدة.
## تزلّج CartPole
> **المسألة**: إذا كان بيتر يبي يهرب من الذئب، لازم يتحرك أسرع منه. بنشوف كيف يقدر بيتر يتعلم التزلج، وبالذات كيف يحافظ على توازنه، باستخدام Q-Learning.
أول شيء، خلونا نثبّت Gym ونستورد المكتبات المطلوبة:
import sys
# Dependencies are provisioned by this course edition's pinned runtime profile.
import gymnasium as gym
import matplotlib.pyplot as plt
import numpy as np
import random
np.random.seed(2026)
random.seed(2026)## إنشاء بيئة CartPole
env = gym.make("CartPole-v1", render_mode="rgb_array")
print(env.action_space)
print(env.observation_space)
print(env.action_space.sample())عشان نشوف كيف تشتغل البيئة، خلونا نشغّل محاكاة قصيرة من 100 خطوة.
env.reset()
for _ in range(100):
env.render()
_, _, terminated, truncated, _ = env.step(env.action_space.sample())
if terminated or truncated:
breakأثناء المحاكاة نحتاج نستقبل الملاحظات عشان نقرر وش الإجراء اللي نسويه. دالة `step` ترجع لنا الملاحظات الحالية وقيمة المكافأة وعلامة `done`، وهي اللي توضّح هل من المنطقي نكمّل المحاكاة أو نوقف:
env.reset()
done = False
while not done:
env.render()
obs, rew, terminated, truncated, info = env.step(env.action_space.sample())
done = terminated or truncated
print(f"{obs} -> {rew}")نقدر نطلع أصغر قيمة وأكبر قيمة لهالأرقام:
print(env.observation_space.low)
print(env.observation_space.high)## تقسيم الحالة إلى قيم متقطعة
def discretize(x):
return tuple((x/np.array([0.25, 0.25, 0.01, 0.1])).astype(np.int64))خلونا نستكشف بعد طريقة ثانية للتحويل إلى قيم متقطعة باستخدام الحاويات (bins):
def create_bins(i,num):
return np.arange(num+1)*(i[1]-i[0])/num+i[0]
print("Sample bins for interval (-5,5) with 10 bins\n",create_bins((-5,5),10))
ints = [(-5,5),(-2,2),(-0.5,0.5),(-2,2)] # intervals of values for each parameter
nbins = [20,20,10,10] # number of bins for each parameter
bins = [create_bins(ints[i],nbins[i]) for i in range(4)]
def discretize_bins(x):
return tuple(np.digitize(x[i],bins[i]) for i in range(4))الحين خلونا نشغّل محاكاة قصيرة ونراقب قيم البيئة المتقطعة.
env.reset()
done = False
while not done:
#env.render()
obs, rew, terminated, truncated, info = env.step(env.action_space.sample())
done = terminated or truncated
#print(discretize_bins(obs))
print(discretize(obs))## بنية Q-Table
Q = {}
actions = (0,1)
def qvalues(state):
return [Q.get((state,a),0) for a in actions]## خلونا نبدأ Q-Learning!
# hyperparameters
alpha = 0.3
gamma = 0.9
epsilon = 0.90def probs(v,eps=1e-4):
v = v-v.min()+eps
v = v/v.sum()
return v
Qmax = 0
cum_rewards = []
rewards = []
for epoch in range(200):
obs, _ = env.reset(seed=2026 + epoch)
done = False
cum_reward=0
# == do the simulation ==
while not done:
s = discretize(obs)
if random.random()<epsilon:
# exploitation - chose the action according to Q-Table probabilities
v = probs(np.array(qvalues(s)))
a = random.choices(actions,weights=v)[0]
else:
# exploration - randomly chose the action
a = np.random.randint(env.action_space.n)
obs, rew, terminated, truncated, info = env.step(a)
done = terminated or truncated
cum_reward+=rew
ns = discretize(obs)
bootstrap = 0.0 if terminated else gamma * max(qvalues(ns))
Q[(s,a)] = (1 - alpha) * Q.get((s,a), 0) + alpha * (rew + bootstrap)
cum_rewards.append(cum_reward)
rewards.append(cum_reward)
# == Periodically print results and calculate average reward ==
if epoch % 50==0:
print(f"{epoch}: {np.average(cum_rewards)}, alpha={alpha}, epsilon={epsilon}")
if np.average(cum_rewards) > Qmax:
Qmax = np.average(cum_rewards)
Qbest = Q
cum_rewards=[]## رسم تقدّم التدريب
plt.plot(rewards)ما نقدر نستنتج شيء واضح من هالرسم؛ لأن طول جلسات التدريب يتفاوت كثير بسبب طبيعة التدريب العشوائية. عشان يصير الرسم أوضح، نقدر نحسب **المتوسط المتحرك** عبر سلسلة من التجارب، ولنقل 100 تجربة. ونسوي هالشي بسهولة باستخدام `np.convolve`:
def running_average(x,window):
return np.convolve(x,np.ones(window)/window,mode='valid')
plt.plot(running_average(rewards,100))## تغيير المعاملات الفائقة ومشاهدة النتيجة
الحين صار ممتع نشوف كيف يتصرف النموذج بعد التدريب. خلونا نشغّل المحاكاة ونتبع استراتيجية اختيار الإجراءات نفسها اللي استخدمناها أثناء التدريب: نأخذ عينة على حسب توزيع الاحتمالات في Q-Table:
obs, _ = env.reset()
done = False
while not done:
s = discretize(obs)
env.render()
v = probs(np.array(qvalues(s)))
a = random.choices(actions,weights=v)[0]
obs, _, terminated, truncated, _ = env.step(a)
done = terminated or truncated## حفظ النتيجة في ملف GIF متحرك
لو ودّكم تعرضون النتيجة لأصحابكم، تقدرون ترسلون لهم صورة GIF متحركة للعمود وهو يتوازن. عشان نسويها، نستدعي `env.render` لإنتاج إطار صورة، ثم نحفظ الإطارات في ملف GIF متحرك باستخدام مكتبة PIL:
from pathlib import Path
from PIL import Image
output_dir = Path("outputs")
output_dir.mkdir(parents=True, exist_ok=True)
obs, _ = env.reset()
done = False
i = 0
ims = []
while not done:
s = discretize(obs)
img = env.render()
if img is None:
raise RuntimeError("CartPole rgb_array rendering returned no frame")
ims.append(Image.fromarray(img))
v = probs(np.array([Qbest.get((s,a),0) for a in actions]))
a = random.choices(actions,weights=v)[0]
obs, _, terminated, truncated, _ = env.step(a)
done = terminated or truncated
i += 1
env.close()
if not ims:
raise RuntimeError("CartPole produced no frames")
output_path = output_dir / "cartpole-balance.gif"
ims[0].save(output_path,save_all=True,append_images=ims[1::2],loop=0,duration=5)
print(f"Saved {i} frames to {output_path}")حذفنا المخرجات وعدّادات التشغيل والودجات والمحتوى النشط وقت الاستيراد. شغّل الدفاتر بس في بيئة خارجية تثق فيها.
سجّل تطبيقك
التسجيل اختياري، يفيدك تتذكر وش طبّقت، ولا يمنع إكمال الدورة.