{
  "$type": "site.standard.document",
  "coverImage": {
    "$type": "blob",
    "ref": {
      "$link": "bafkreibgaksrrcxpnmnkizdrhypj4ov44imzw6vlgkamakhoidf3ztspj4"
    },
    "mimeType": "image/png",
    "size": 89770
  },
  "description": "According to one embodiment, a learning method, comprises receiving a first signal including a previous auxiliary variable value, previous action information regarding a previous action, or a set of previous scores, receiving current sensor data, selecting a current action of the control target…",
  "path": "/patents/1279131",
  "publishedAt": "2020-12-17T00:00:00.000Z",
  "site": "at://did:plc:oql6ds5vnff4ugar6rruliwd/site.standard.publication/3mn3ohu7oxx5w",
  "tags": [
    "G06N20/00",
    "KABUSHIKI KAISHA TOSHIBA"
  ],
  "textContent": "According to one embodiment, a learning method, comprises receiving a first signal including a previous auxiliary variable value, previous action information regarding a previous action, or a set of previous scores, receiving current sensor data, selecting a current action of the control target based on the first signal, the current sensor data, and a parameter for obtaining a score from sensor data, causing the control target to execute the current action, receiving next sensor data and a reward, and updating the parameter based on the current sensor data, current action information regarding the current action, the next sensor data, and the reward. A degree of selecting a previous action as the current action is increased.",
  "title": "LEARNING METHOD AND PROGRAM"
}