C#/수업내용

0824 커리큘럼 머신러닝

버스창문 2022. 8. 24. 16:50

 

이거 아래부터 커리큘럼임

 

behaviors:
  BigWallJump:
    trainer_type: ppo
    hyperparameters:
      batch_size: 128
      buffer_size: 2048
      learning_rate: 0.0003
      beta: 0.005
      epsilon: 0.2
      lambd: 0.95
      num_epoch: 3
      learning_rate_schedule: linear
    network_settings:
      normalize: false
      hidden_units: 256
      num_layers: 2
      vis_encode_type: simple
    reward_signals:
      extrinsic:
        gamma: 0.99
        strength: 1.0
    keep_checkpoints: 5
    max_steps: 20000000
    time_horizon: 128
    summary_freq: 20000
  SmallWallJump:
    trainer_type: ppo
    hyperparameters:
      batch_size: 128
      buffer_size: 2048
      learning_rate: 0.0003
      beta: 0.005
      epsilon: 0.2
      lambd: 0.95
      num_epoch: 3
      learning_rate_schedule: linear
    network_settings:
      normalize: false
      hidden_units: 256
      num_layers: 2
      vis_encode_type: simple
    reward_signals:
      extrinsic:
        gamma: 0.99
        strength: 1.0
    keep_checkpoints: 5
    max_steps: 5000000
    time_horizon: 128
    summary_freq: 20000
environment_parameters:
  big_wall_height:
    curriculum:
      - name: Lesson0 # The '-' is important as this is a list
        completion_criteria:
          measure: progress
          behavior: BigWallJump
          signal_smoothing: true
          min_lesson_length: 100
          threshold: 0.1
        value:
          sampler_type: uniform
          sampler_parameters:
            min_value: 0.0
            max_value: 4.0
      - name: Lesson1 # This is the start of the second lesson
        completion_criteria:
          measure: progress
          behavior: BigWallJump
          signal_smoothing: true
          min_lesson_length: 100
          threshold: 0.3
        value:
          sampler_type: uniform
          sampler_parameters:
            min_value: 4.0
            max_value: 7.0
      - name: Lesson2
        completion_criteria:
          measure: progress
          behavior: BigWallJump
          signal_smoothing: true
          min_lesson_length: 100
          threshold: 0.5
        value:
          sampler_type: uniform
          sampler_parameters:
            min_value: 6.0
            max_value: 8.0
      - name: Lesson3
        value: 8.0
  small_wall_height:
    curriculum:
      - name: Lesson0
        completion_criteria:
          measure: progress
          behavior: SmallWallJump
          signal_smoothing: true
          min_lesson_length: 100
          threshold: 0.1
        value: 1.5
      - name: Lesson1
        completion_criteria:
          measure: progress
          behavior: SmallWallJump
          signal_smoothing: true
          min_lesson_length: 100
          threshold: 0.3
        value: 2.0
      - name: Lesson2
        completion_criteria:
          measure: progress
          behavior: SmallWallJump
          signal_smoothing: true
          min_lesson_length: 100
          threshold: 0.5
        value: 2.5
      - name: Lesson3
        value: 4.0

 

 

min_lesson_length

 <- 내가 설정해놓는 교육의 목표치 > 저게 달성되면 다음 교육으로 넘어가짐

using System.Collections;
using System.Collections.Generic;
using UnityEngine;
using Unity.MLAgents;
using Unity.MLAgents.Actuators;
using Unity.MLAgents.Sensors;

public class HeroAgent : Agent
{
    public float jumpTime = 0.2f;
    public float jumpPower = 70;

    public GameObject shortBlock;
    public GameObject groundGO;
    public GameObject wallGO;


    private EnvironmentParameters envParams;

    //private Rigidbody rBody;

    public override void Initialize()
    {
        this.envParams = Academy.Instance.EnvironmentParameters;
    }

    private void Start()
    {
        //this.rBody = this.GetComponent<Rigidbody>();
    }

    public override void CollectObservations(VectorSensor sensor)
    {
         //var dir = rBody.position - groundGO.transform.position;
    }
    public override void OnEpisodeBegin()
    {
        this.shortBlock.transform.localPosition = new Vector3(Random.Range(-2f,2f), 0, -2.07999992f);
        this.transform.localPosition = new Vector3(0, 0, -3.86999989f);
    }

    public override void OnActionReceived(ActionBuffers actions)
    {
        //매 틱마다 행동

        Vector3 dir = Vector3.zero;

        dir.x = actions.DiscreteActions[0] - 1;
        dir.z = actions.DiscreteActions[1] - 1;
        //dir.y = actions.DiscreteActions[2];

        dir = dir.normalized;

        if (dir.x != 0 || dir.z != 0)
        {
            this.transform.rotation = Quaternion.LookRotation(dir);
            //this.modelGO.GetComponent<Animator>().SetInteger("state", 1);
        }
        else
        {
            //this.modelGO.GetComponent<Animator>().SetInteger("state", 0);
        }
        dir = dir.normalized;
        this.transform.Translate(dir * Time.deltaTime * 5, Space.World);
        //Debug.Log(this.CheckGround());
        if (actions.DiscreteActions[2] == 1 && this.CheckGround() && this.jumpTime < 0f)
        {
            this.jumpTime = 0.2f;
            //Debug.Log(this.CheckGround());
            this.GetComponent<Rigidbody>().AddForce(dir + Vector3.up * this.jumpPower);
        }
        this.jumpTime -= Time.fixedDeltaTime;

        if (this.transform.localPosition.y < -1f
            || this.shortBlock.transform.localPosition.y < -1f)
        {
            SetReward(-1f);
            this.shortBlock.transform.localPosition = new Vector3(Random.Range(-2f, 2f), 0, -2.07999992f);
            EndEpisode();
        }

    }
    public override void Heuristic(in ActionBuffers actionsOut)
    {
        var discreteAction = actionsOut.DiscreteActions;
        discreteAction[0] = (int)Input.GetAxis("Horizontal") + 1;
        discreteAction[1] = (int)Input.GetAxis("Vertical") + 1;
        discreteAction[2] = Input.GetKey(KeyCode.Space) ? 1:0;
    }
    private bool CheckGround()
    {
        RaycastHit hit;
        Physics.Raycast(transform.position + new Vector3(0, -0.05f, 0), -Vector3.up, out hit,
            0.6f);

        //            && hit.normal.y > 0.95f

        if (hit.collider != null &&
            (hit.collider.CompareTag("Ground") ||
             hit.collider.CompareTag("Box")))
        {
            return true;
        }

        return false;
    }
    private void OnTriggerStay(Collider other)
    {
        if (other.gameObject.tag == "Goal")
        {
            SetReward(1f);
            EndEpisode();
        }
    }
        private void FixedUpdate()
    {
        var localScale = wallGO.transform.localScale;

        localScale = new Vector3(
                localScale.x,
                this.envParams.GetWithDefault("small_wall_height", 4),
        localScale.z);
        wallGO.transform.localScale = localScale;

    }
}
behaviors:
  HeroJump:
    trainer_type: ppo
    hyperparameters:
      batch_size: 128
      buffer_size: 2048
      learning_rate: 0.0003
      beta: 0.005
      epsilon: 0.2
      lambd: 0.95
      num_epoch: 3
      learning_rate_schedule: linear
    network_settings:
      normalize: false
      hidden_units: 256
      num_layers: 2
      vis_encode_type: simple
    reward_signals:
      extrinsic:
        gamma: 0.99
        strength: 1.0
    keep_checkpoints: 5
    max_steps: 20000000
    time_horizon: 128
    summary_freq: 20000
environment_parameters:
  small_wall_height:
    curriculum:
      - name: Lesson0
        completion_criteria:
          measure: progress
          behavior: SmallWallJump
          signal_smoothing: true
          min_lesson_length: 100
          threshold: 0.1
        value: 1.5
      - name: Lesson1
        completion_criteria:
          measure: progress
          behavior: SmallWallJump
          signal_smoothing: true
          min_lesson_length: 100
          threshold: 0.3
        value: 2.0
      - name: Lesson2
        completion_criteria:
          measure: progress
          behavior: SmallWallJump
          signal_smoothing: true
          min_lesson_length: 100
          threshold: 0.5
        value: 2.5
      - name: Lesson3
        value: 4.0

동영상 서비스가 종료되어 해당 콘텐츠를 재생할 수 없습니다.

1차 교육 (벽높이 1.5)

 

 

동영상 서비스가 종료되어 해당 콘텐츠를 재생할 수 없습니다.

 

 

2차교육 > 벽높이 3.0  (그냥도 넘어갈수는 있지만, 블럭이 있으면 더 편하게 넘어가도록)




시간이 없어서 3차는 못찍지만 3차교육 > 벽높이 4.0 (블럭없이 못넘어감, 블럭깔면 편히 넘어가짐)

 

'C# > 수업내용' 카테고리의 다른 글

0830 좀비 시민 구분해서 쏘기  (0) 2022.08.30
0825 전방 사물 인식해서 그에 맞게 찾아가기_2  (0) 2022.08.25
0823 머신러닝  (0) 2022.08.23
0822 머신러닝  (0) 2022.08.22
0816 polybrush 안되는 거 해결방법  (0) 2022.08.16