diff --git a/_data/navigation.yml b/_data/navigation.yml index 3821c1686..d88252d28 100644 --- a/_data/navigation.yml +++ b/_data/navigation.yml @@ -11,7 +11,13 @@ dropdown: - name: Overview group: research - link: "/research" + dropdown: + - name: Control Problems + group: research + link: "/research/control_problems/00_future_mobility_control_landscape" + - name: Research Themes + group: research + link: "/research/research_themes/00_online_learning_based_optimal_control" # - name: Learning + Control Theory # group: research # link: "/research/control_learning" @@ -55,4 +61,3 @@ - name: Contact link: "/contact" group: contact - diff --git a/_includes/footer.html b/_includes/footer.html index d95a69119..b3e401124 100644 --- a/_includes/footer.html +++ b/_includes/footer.html @@ -13,6 +13,28 @@ + diff --git a/_includes/header.html b/_includes/header.html index c1eeff5ad..f96bf3209 100644 --- a/_includes/header.html +++ b/_includes/header.html @@ -19,6 +19,22 @@ + + {% if page.math %} + + + {% endif %} @@ -51,7 +67,20 @@ diff --git "a/_projects/2020-\354\206\214\355\230\225.md" "b/_projects/2020-\354\206\214\355\230\225.md" index b58962f4b..011d6c2d4 100644 --- "a/_projects/2020-\354\206\214\355\230\225.md" +++ "b/_projects/2020-\354\206\214\355\230\225.md" @@ -12,5 +12,5 @@ amount: '₩ 91,000,000' startdate: '2020.07.01' enddate: '2021.06.30' pmid: '2020-소형' -image: '/static/img/research/projects/2020-소형.jpg' +image: '/static/projects/img/2020-소형.jpg' --- diff --git "a/_projects/2021-\354\210\230\354\206\214\353\262\204\354\212\244.md" "b/_projects/2021-\354\210\230\354\206\214\353\262\204\354\212\244.md" index e47e13145..74f3a54cf 100644 --- "a/_projects/2021-\354\210\230\354\206\214\353\262\204\354\212\244.md" +++ "b/_projects/2021-\354\210\230\354\206\214\353\262\204\354\212\244.md" @@ -12,5 +12,5 @@ amount: '₩ 50,000,000' startdate: '2021.01.01' enddate: '2021.12.31' pmid: '2021-수소버스' -image: '/static/img/research/projects/2021-수소버스.jpg' +image: '/static/projects/img/2021-수소버스.jpg' --- diff --git "a/_projects/2022-\353\217\231\352\270\260\354\240\204\353\217\231\352\270\260\353\245\274.md" "b/_projects/2022-\353\217\231\352\270\260\354\240\204\353\217\231\352\270\260\353\245\274.md" index d8cb2c063..d626a3198 100644 --- "a/_projects/2022-\353\217\231\352\270\260\354\240\204\353\217\231\352\270\260\353\245\274.md" +++ "b/_projects/2022-\353\217\231\352\270\260\354\240\204\353\217\231\352\270\260\353\245\274.md" @@ -12,5 +12,5 @@ amount: '₩ 70,000,000' startdate: '2023.03.01' enddate: '2024.02.29' pmid: '2022-동기전동기를' -image: '/static/img/research/projects/2022-동기전동기를.jpg' +image: '/static/projects/img/2022-동기전동기를.jpg' --- diff --git "a/_projects/2022-\353\257\270\353\236\230\355\230\225\354\236\220\353\217\231\354\260\250.md" "b/_projects/2022-\353\257\270\353\236\230\355\230\225\354\236\220\353\217\231\354\260\250.md" index 32c0a41d1..b516e8d9e 100644 --- "a/_projects/2022-\353\257\270\353\236\230\355\230\225\354\236\220\353\217\231\354\260\250.md" +++ "b/_projects/2022-\353\257\270\353\236\230\355\230\225\354\236\220\353\217\231\354\260\250.md" @@ -12,7 +12,7 @@ amount: startdate: '2022.03.01' enddate: '2025.02.28' pmid: '2022-미래형자동차' -image: '/static/img/research/projects/2022-미래형자동차.jpg' +image: '/static/projects/img/2022-미래형자동차.jpg' link: - name: '미래형자동차 핵심기술 R&D 전문인력양성' url: 'https://i4ft.yonsei.ac.kr/index.php' diff --git "a/_projects/2022-\354\236\220\354\234\250\354\243\274\355\226\211.md" "b/_projects/2022-\354\236\220\354\234\250\354\243\274\355\226\211.md" index b3dcfca81..e72493d08 100644 --- "a/_projects/2022-\354\236\220\354\234\250\354\243\274\355\226\211.md" +++ "b/_projects/2022-\354\236\220\354\234\250\354\243\274\355\226\211.md" @@ -12,5 +12,5 @@ amount: '₩ 50,000,000' startdate: '2022.06.01' enddate: '2023.05.31' pmid: '2022-자율주행' -image: '/static/img/research/projects/2022-자율주행.jpg' +image: '/static/projects/img/2022-자율주행.jpg' --- diff --git "a/_projects/2022-\354\240\204\352\270\260\354\236\220\353\217\231\354\260\250\354\235\230.md" "b/_projects/2022-\354\240\204\352\270\260\354\236\220\353\217\231\354\260\250\354\235\230.md" index b319967fb..e1fbf9b0e 100644 --- "a/_projects/2022-\354\240\204\352\270\260\354\236\220\353\217\231\354\260\250\354\235\230.md" +++ "b/_projects/2022-\354\240\204\352\270\260\354\236\220\353\217\231\354\260\250\354\235\230.md" @@ -12,5 +12,5 @@ amount: '₩ 40,000,000' startdate: '2022.05.01' enddate: '2023.04.30' pmid: '2022-전기자동차의' -image: '/static/img/research/projects/2022-전기자동차의.jpg' +image: '/static/projects/img/2022-전기자동차의.jpg' --- diff --git "a/_projects/2022-\354\264\210\354\213\244\352\260\220.md" "b/_projects/2022-\354\264\210\354\213\244\352\260\220.md" index 7b71419bf..2e0d3d33c 100644 --- "a/_projects/2022-\354\264\210\354\213\244\352\260\220.md" +++ "b/_projects/2022-\354\264\210\354\213\244\352\260\220.md" @@ -15,5 +15,5 @@ amount: '₩ 500,000,000' startdate: '2022.12.01' enddate: '2027.11.30' pmid: '2022-초실감' -image: '/static/img/research/projects/2022-초실감.jpg' +image: '/static/projects/img/2022-초실감.jpg' --- diff --git "a/_projects/2023-\354\236\220\354\234\250\354\243\274\355\226\211\354\235\204.md" "b/_projects/2023-\354\236\220\354\234\250\354\243\274\355\226\211\354\235\204.md" index 9ed0297cb..596c48115 100644 --- "a/_projects/2023-\354\236\220\354\234\250\354\243\274\355\226\211\354\235\204.md" +++ "b/_projects/2023-\354\236\220\354\234\250\354\243\274\355\226\211\354\235\204.md" @@ -12,5 +12,5 @@ amount: '₩ 30,000,000' startdate: '2023.08.01' enddate: '2023.11.30' pmid: '2023-자율주행을' -image: '/static/img/research/projects/2023-자율주행을.jpg' +image: '/static/projects/img/2023-자율주행을.jpg' --- diff --git "a/_projects/2023-\354\240\204\352\270\260\354\260\250.md" "b/_projects/2023-\354\240\204\352\270\260\354\260\250.md" index e2b804723..4fa9f1378 100644 --- "a/_projects/2023-\354\240\204\352\270\260\354\260\250.md" +++ "b/_projects/2023-\354\240\204\352\270\260\354\260\250.md" @@ -12,5 +12,5 @@ amount: '₩ 22,000,000' startdate: '2023.10.18' enddate: '2024.01.31' pmid: '2023-전기차' -image: '/static/img/research/projects/2023-전기차.jpg' +image: '/static/projects/img/2023-전기차.jpg' --- diff --git "a/_projects/2024-\352\270\211\353\263\200\353\266\200\355\225\230.md" "b/_projects/2024-\352\270\211\353\263\200\353\266\200\355\225\230.md" index 98b1bb0c5..3df27c600 100644 --- "a/_projects/2024-\352\270\211\353\263\200\353\266\200\355\225\230.md" +++ "b/_projects/2024-\352\270\211\353\263\200\353\266\200\355\225\230.md" @@ -12,5 +12,5 @@ amount: '₩ 20,000,000' startdate: '2024.05.20' enddate: '2024.08.18' pmid: '2024-급변부하' -image: '/static/img/research/projects/2024-급변부하.jpg' +image: '/static/projects/img/2024-급변부하.jpg' --- diff --git a/_projects/2025-RWS.md b/_projects/2025-RWS.md index ca83999ce..a7d22600f 100644 --- a/_projects/2025-RWS.md +++ b/_projects/2025-RWS.md @@ -12,5 +12,5 @@ amount: # not need to fill (will be not displayed) startdate: '2025.07.01' enddate: '2026.07.01' pmid: '2025-RWS' -image: '/static/img/research/projects/2025-RWS.jpg' +image: '/static/projects/img/2025-RWS.jpg' --- diff --git "a/_projects/2025-\354\213\254\354\270\265\355\225\231\354\212\265\354\235\230.md" "b/_projects/2025-\354\213\254\354\270\265\355\225\231\354\212\265\354\235\230.md" index 8308dcc17..9d083432c 100644 --- "a/_projects/2025-\354\213\254\354\270\265\355\225\231\354\212\265\354\235\230.md" +++ "b/_projects/2025-\354\213\254\354\270\265\355\225\231\354\212\265\354\235\230.md" @@ -12,5 +12,5 @@ amount: '₩ 250,000,000' startdate: '2025.05.12' enddate: '2028.12.31' pmid: '2025-심층학습의' -image: '/static/img/research/projects/2025-심층학습의.jpg' +image: '/static/projects/img/2025-심층학습의.jpg' --- diff --git "a/_projects/2025-\354\225\210\354\240\225\354\204\261.md" "b/_projects/2025-\354\225\210\354\240\225\354\204\261.md" index b760a3c84..4feca33d5 100644 --- "a/_projects/2025-\354\225\210\354\240\225\354\204\261.md" +++ "b/_projects/2025-\354\225\210\354\240\225\354\204\261.md" @@ -15,5 +15,5 @@ amount: '₩ 1,056,556,000' startdate: '2025.03.01' enddate: '2029.02.28' pmid: '2025-안정성' -image: '/static/img/research/projects/2025-안정성.jpg' +image: '/static/projects/img/2025-안정성.jpg' --- diff --git a/research/index.md b/research/_index.md similarity index 100% rename from research/index.md rename to research/_index.md diff --git a/research/control_problems/00_future_mobility_control_landscape.md b/research/control_problems/00_future_mobility_control_landscape.md new file mode 100644 index 000000000..48a61ac64 --- /dev/null +++ b/research/control_problems/00_future_mobility_control_landscape.md @@ -0,0 +1,209 @@ +--- +title: Future Mobility Control Landscape +layout: default +group: research +--- + +
+
+
+ +# Future Mobility Control Landscape + +> **Core idea.** Connected, automated, and electrified vehicles expand the +> information and decision scope of mobility control while adding coupled +> control and actuation degrees of freedom (DOFs) at mobility-system, vehicle, +> and component levels. Realizing this performance potential is naturally +> formulated as an optimal-control problem, but computational and information +> limitations often prevent its direct implementation. These shared barriers +> motivate learning-based optimal control. + +
+ qFit +
+ Future-mobility Control Problems at three levels lead through two representative barriers + and online learning-based optimal control to the MIC Lab Research Themes. +
+
+ +*The two main axes organize where and what is controlled (Control Problems) +and what is learned or adapted (Research Themes). Computational and +information/formulation barriers motivate online learning-based optimal +control as the bridge between them.* + +> **How to read this overview.** Sections 1–3 establish the CAEV context, the +> common optimal-control principle, and the three control domains. Section 4 +> explains their shared implementation barriers and why they motivate the +> laboratory's online learning-based control program. Section 5 maps the two +> complementary document families. + + +## 1. CAEVs and the need for multilevel optimal control + +Future mobility is often described by **CASE**: connected, automated, shared, +and electrified mobility. This control landscape focuses on **connected, +automated, and electrified vehicles (CAEVs)**. Shared mobility remains relevant +when demand, fleets, or shared resources enter a mobility-system problem, but +it does not define a separate physical control level here. + +The three CAEV characteristics change control in complementary ways: + +- **Connected:** broadens the available information to include route, traffic, + infrastructure, other agents, pedestrians, and cloud services. +- **Automated:** transfers more planning, coordination, and actuation + decisions to feedback controllers. +- **Electrified:** adds coupled energy and thermal states, multiple power + sources, distributed electric actuators, and fast electrical commands. + +These characteristics create control opportunities at three levels: + +- **Mobility-system level:** use route, traffic, infrastructure, mission, + or connected-agent information for trip-scale energy and thermal management + or multi-vehicle coordination. +- **Vehicle level:** coordinate onboard drive, brake, steering, or suspension + actuators. +- **Component level:** control device-level variables such as motor + voltage, current, flux, and torque. + +Connected, automated, and electrified functions contribute across all three +levels rather than mapping one-to-one to them. Across the levels, CAEVs broaden +the information available to the controller, enlarge the set of coupled +decisions, and add control and actuation degrees of freedom (DOFs). + +For example, a connected electrified vehicle (xEV) can use previews of road +grade, signal timing, traffic, weather, or charging opportunities to improve +present energy and thermal decisions over the trip. Multiple connected +vehicles can additionally coordinate speed, spacing, lane selection, or +intersection passage. Optimal control provides a common formulation for using +this information and the available control DOFs to balance performance and +constraints. + + + + + +## 2. Optimal control as a common decision principle + +The details change across applications, but the basic optimal-control question +is the same: + +| | Conceptual problem definition | +|---|---| +| **Find** | An admissible control input, control sequence, or feedback policy | +| **To minimize** | A performance index encoding energy use, travel time, safety penalties, motion error, discomfort, or component loss | +| **Subject to** | System dynamics, physical and safety constraints, actuator limits, environmental conditions, and the information available to the controller | + +The exact states, inputs, costs, models, and constraints depend on the control +level and application; Section 3 specifies those differences. + +## 3. Three mobility control problem domains + +The three levels are classified first by the primary performance objective and +the boundary of the coupled decisions and information—not merely by where an +actuator is physically located. Physical location and time scale are secondary +descriptors. The same electric motor can therefore support component-level +torque control, vehicle-level actuator allocation, and mobility-system-level +energy management. + +#### [Mobility-System-Level Optimal Control](/research/control_problems/01_mobility_system_level_optimal_control) + +This domain uses route, traffic, infrastructure, ambient, mission, or +connected-agent information beyond the local vehicle state. It includes a +single xEV using such context for predictive or infinite-horizon energy and +thermal management, as well as multiple vehicles coordinating through shared +information. A traffic network is therefore one information source within the +broader mobility system; multiple vehicles are an important case, but not the +definition of this level. Energy management belongs here when route or +mobility context and long-horizon coupling define the problem; a local +power-split problem using only onboard variables can instead be vehicle-level. + +#### [Vehicle-Level Optimal Control](/research/control_problems/02_vehicle_level_optimal_control) + +This domain coordinates control and actuation DOFs contained in the whole +vehicle. Representative decisions include propulsion and braking allocation, +torque vectoring, four-wheel independent drive or steering, and +steering–suspension coordination. The objectives may combine motion, +stability, safety, comfort, and efficiency subject to tire, actuator, power, +and vehicle-dynamics constraints. + +A related deployment problem is **Automatic Controller Calibration**: gains, +maps, cost weights, filters, thresholds, or learned parameters of +vehicle-motion controllers are adjusted to reproduce the intended behavior +across maneuvers and operating conditions. The vehicle-level page treats this +as an outer-loop workflow surrounding the physical control problem. + +#### [Component-Level Optimal Control](/research/control_problems/03_component_level_optimal_control) + +This domain exploits fast device-level control DOFs. A representative problem +is optimal torque production in a synchronous machine through voltage or +inverter-switching decisions that shape current and flux while respecting +electrical, magnetic, thermal, inverter, and sampling constraints. + +The same workflow appears at the component level when current-, torque-, +speed-, position-, estimator-, or solver-related parameters must be adjusted +across machine and operating conditions. In both the vehicle and component +domains, calibration is a deployment and lifecycle workflow rather than a +fourth physical control level. + +## 4. Why direct optimal control is difficult — and what learning contributes + +Implementing the principle in Section 2 requires both finding the optimal +input or policy within the available computation time and having the model, +state, context, objective, constraints, and transition information needed to +define and evaluate the decision. Two difficulties recur across all three +control levels: + +- **Computational difficulty:** high-dimensional state and action spaces, + long horizons, hybrid decisions, nonlinear constraints, coupled agents, and + short control deadlines can make exact policy computation or repeated online + optimization impractical. +- **Information and formulation limitations:** the model, state, context, + objective, constraints, or transition law needed to define and evaluate the + decision may be incomplete, uncertain, indirectly measured, or changing. + Limited real-world data can further restrict identification and validation. + +The first difficulty means that an exact optimal decision may be too expensive +even when the required information is available. The second means that the +decision problem or its evaluation is incomplete, unreliable, or +nonstationary. Many mobility problems contain both. + +These barriers motivate **learning-based optimal control**. When exact +decision computation is too expensive, a learned value or policy—or a learned +object combined with tractable online optimization—can approximate the +optimal decision. When the required information is incomplete or changing, +the relevant model, state, context, or representation can instead be +estimated, learned, or updated. + +Within this broad class, the MIC Lab program emphasizes the **online +realization of learning-based optimal control**, including online learning when +deployed learned objects must adapt. Current measurements, context, forecasts, +and constraints may support online estimation, decision improvement, or +parameter learning; *online learning* is reserved for the last case. The +common formulation and eight Research Themes are developed in [Online +Learning-Based Optimal +Control](/research/research_themes/00_online_learning_based_optimal_control). + +## 5. Document map — two complementary views + +The two views are complementary rather than one-to-one. A project may involve +one or more physical Control Problem domains and draw on one or more Research +Themes. + +| View | Organizing question | Overview and child pages | +|---|---|---| +| **Mobility Control Problem Domains** | Where and what is controlled? | **Overview (00):** [Future Mobility Control Landscape](/research/control_problems/00_future_mobility_control_landscape)
**Domain pages (01–03):**
[01 · Mobility-System-Level Optimal Control](/research/control_problems/01_mobility_system_level_optimal_control)
[02 · Vehicle-Level Optimal Control](/research/control_problems/02_vehicle_level_optimal_control)
[03 · Component-Level Optimal Control](/research/control_problems/03_component_level_optimal_control) | +| **Learning-Based Control Research Themes** | Which bottleneck and learned object are addressed, and how are they used online? | **Overview (00):** [Online Learning-Based Optimal Control](/research/research_themes/00_online_learning_based_optimal_control)
**Theme pages (01–08):**
[01 · Real-World RL](/research/research_themes/01_real_world_rl)
[02 · Online Multistep Lookahead](/research/research_themes/02_online_multistep_lookahead)
[03 · Semantic Critic Learning](/research/research_themes/03_semantic_critic_learning)
[04 · Nonstationary Infinite-Horizon OCP](/research/research_themes/04_nonstationary_infinite_horizon_ocp)
[05 · Continual Model Learning](/research/research_themes/05_continual_model_learning)
[06 · Constrained PINN](/research/research_themes/06_constrained_pinn)
[07 · Neuro-Adaptive Control](/research/research_themes/07_neuro_adaptive_control)
[08 · Structured Critic Adaptation](/research/research_themes/08_structured_critic_adaptation) | + + +
+
+
diff --git a/research/control_problems/01_mobility_system_level_optimal_control.md b/research/control_problems/01_mobility_system_level_optimal_control.md new file mode 100644 index 000000000..46e3e7859 --- /dev/null +++ b/research/control_problems/01_mobility_system_level_optimal_control.md @@ -0,0 +1,433 @@ +--- +title: Mobility-System-Level Optimal Control +layout: default +group: research +math: true +--- + +
+
+
+ +# Mobility-System-Level Optimal Control + +> **Core idea.** Mobility-system-level optimal control uses information and +> decision couplings beyond the local vehicle boundary. It includes both a +> single vehicle using traffic-network context for long-horizon operation and +> multiple vehicles coordinating through shared information. + + +
+ qFit +
+ Route, traffic, infrastructure, ambient, and vehicle-to-everything (V2X) information + form a mobility-system decision state for energy, thermal, and coordination control. +
+
+ +*The number of controlled vehicles does not define the level by itself. The +defining feature is the information and decision scope used by the +controller.* + +## 1. Scope: control beyond the local vehicle boundary + +[Future Mobility Control Landscape](/research/control_problems/00_future_mobility_control_landscape) +defines mobility-system-level control by an information or decision scope that +extends beyond the local vehicle boundary. In addition to onboard +measurements, a controller may use road grade, signal phase and timing, +traffic flow, route or mission information, weather and ambient conditions, +charging or refueling availability, and vehicle-to-everything (V2X) messages +from vehicles or infrastructure. The relevant agents can include connected and +automated vehicles, human-driven vehicles, infrastructure, fleets, and shared +energy resources. + +This broader information enables a wider set of coupled decisions: current and +future power or thermal allocation, speed and trajectory planning, route and +charging choices, and coordination among vehicles. It also enlarges the +prediction horizon, state and action spaces, and set of interaction and +resource constraints. The control problem becomes mobility-system-level when +these external information sources, cross-agent coupling, or long-run +consequences across repeated trips materially change the objective, dynamics, +or constraints. + +Energy and thermal management illustrate the boundary. They are +mobility-system-level when route, traffic, infrastructure, ambient forecasts, +or long-run network operation determine the decision. The same actuators can +form a vehicle-level problem when the controller uses only onboard states and +the current local power or thermal demand. + +## 2. Representative control problems + +| Problem | Representative decisions | Objectives | Main constraints | +|---|---|---|---| +| **Electrified-vehicle (xEV) predictive energy and/or thermal management** | power-source allocation, battery use, cooling or heating, charging or refueling decisions | energy or fuel use, thermal performance, degradation, and long-run reserve | power balance, state of charge, temperatures, actuator limits, and route feasibility | +| **Multi-vehicle coordination** | speed, spacing, trajectory, merging, or shared-resource decisions | safety, throughput, travel time, energy, and comfort | collision avoidance, road geometry, actuator limits, communication, and distributed information | + +Multi-vehicle coordination is mobility-system-level because vehicle decisions +are coupled through safety, traffic flow, or shared resources. It is distinct +from single-vehicle predictive energy and thermal management and is not +required for every mobility-system-level problem. + +## 3. xEV predictive energy management + +Suppose an xEV has an intended trip or predicted driving segment covering +physical sampling steps $k=0,\ldots,T$. Let $x_k$ collect its energy states, +$u_k$ denote power-source allocation, and $w_k$ contain predicted +route-dependent traction and auxiliary demand. A generic predictive +energy-management problem is + +$$ +\begin{aligned} +\min_{u_0,\ldots,u_{T-1}}\quad& +\sum_{k=0}^{T-1} +g_{\mathrm E}(x_k,u_k,w_k) ++ +\Phi_T(x_T)\\ +\mathrm{s.t.}\quad& +x_{k+1}=f(x_k,u_k,w_k),\\ +& +(x_k,u_k)\in\mathcal Z_k,\qquad +x_T\in\mathcal X_T . +\end{aligned} +$$ + +$g_{\mathrm E}$ represents fuel or electrical-energy use and possibly +degradation; $\Phi_T$ and $\mathcal X_T$ describe the terminal energy +requirement. Power balance, source and battery limits, and state-of-charge +constraints are included in $f$ and $\mathcal Z_k$. + +A fuel-cell electric vehicle (FCEV) is one concrete instance: $x_k$ can be +battery state of charge (SOC), $u_k$ fuel-cell power, and, for sampling +interval $\Delta t$, let $g_{\mathrm E}(x_k,u_k,w_k)=\Delta t\,\dot m_{\mathrm{fc}}(u_k)$ denote the hydrogen consumed over one step. Equivalently, +$\Delta t$ may be absorbed into the stage-cost definition. The battery +supplies the residual traction demand through the power balance. When the +mission requires a prescribed terminal energy state, it can be imposed through +the hard terminal set $\mathcal X_T=\{x_{\mathrm{tar}}\}$. + +The physical model need not be written as $f_k$. A time-varying preview +$w_k$ induces the effective dynamics and cost + +$$ +F_k(x,u):=f(x,u,w_k), +\qquad +G_k(x,u):=g_{\mathrm E}(x,u,w_k), +$$ + +so the problem is nonstationary in physical time when represented by $x$ +alone, even if the underlying plant function $f$ is time invariant. This time +dependence does not by itself imply uncertainty: the problem is nonstationary +even when the complete sequence $\{w_k\}$ is known. Uncertainty arises because +future power demand, traffic, route, and ambient context are usually predicted +imperfectly. + +If the mission truly ends at $T$, the finite-horizon formulation can define +the correct optimal-control problem. For a vehicle operating continuously +across trips, however, the performance horizon is effectively infinite. A +finite trip or prediction horizon is then only an approximation and requires +a terminal value that represents consequences beyond $T$. + +## 4. Coupled predictive extensions + +### Multi-vehicle coordination within the predictable horizon + +The full multi-vehicle coordination domain includes safety, traffic flow, and +shared-resource objectives beyond energy management. The following formulation +is one coupled example in which coordination forms the short-horizon part of +predictive energy and thermal management. Let +$Z_k$ collect the joint vehicle and traffic states, $U_k$ the speed, +trajectory, or coordination decisions, $S_k$ the vehicles' energy and thermal +states, and $V_k$ their energy and thermal controls. The coordination +decisions determine the predicted traction and thermal demands: + +$$ +w_{j\mid k}^i += +\mathcal D^i +\!\left(Z_{j\mid k},U_{j\mid k}\right), +\qquad i=1,\ldots,N_{\mathrm{veh}} . +$$ + +Here $j\mid k$ denotes a $j$-step-ahead prediction made at decision time $k$, +$H$ is the prediction horizon, $N_{\mathrm{veh}}$ is the number of vehicles, +and $\mathcal D^i$ maps the joint motion prediction to vehicle $i$'s demand. +Let $w_{j\mid k}$ collect these per-vehicle predicted demands. The symbols +$U$ and $V$ below denote the corresponding decision sequences over the +horizon. + +A coupled short-horizon problem can then be written as + +$$ +\begin{aligned} +\min_{U,V}\quad& +\sum_{j=0}^{H-1} +\alpha^j +\Big[ +g_{\mathrm{coord}}(Z_{j\mid k},U_{j\mid k}) ++ +g_{\mathrm{E/T}}(S_{j\mid k},V_{j\mid k},w_{j\mid k}) +\Big] ++ +\alpha^H +J_{\mathrm{long}}(S_{H\mid k},Z_{H\mid k})\\ +\mathrm{s.t.}\quad& +\text{vehicle and energy--thermal dynamics,}\\ +& +\text{road, collision, actuator, and information constraints.} +\end{aligned} +$$ + +Here $g_{\mathrm{coord}}$ and $g_{\mathrm{E/T}}$ are the coordination and +aggregate per-vehicle energy--thermal stage costs, $J_{\mathrm{long}}$ is the +beyond-horizon value, and $0<\alpha\le 1$ is the finite-horizon discount +factor. Its argument $Z_{H\mid k}$ supplies the mobility context needed to +evaluate long-run consequences beyond the energy and thermal state +$S_{H\mid k}$. + +The explicit horizon uses predictable traffic interactions to optimize +coordination, energy, and thermal performance while enforcing immediate safety +and physical constraints. The terminal value +$J_{\mathrm{long}}$ evaluates energy and thermal consequences beyond the +reliable prediction horizon. Thus, coordination changes the near-term demand +trajectory, whereas the terminal cost supplies the longer-run perspective. + +### Coupled energy–thermal extension + +Thermal management has an analogous structure after augmenting the energy +state and control. For example, + +$$ +s_k= +\begin{bmatrix} +x_k & T_{\mathrm{bat},k} & T_{\mathrm{fc},k} +\end{bmatrix}^{\!\top}, +\qquad +v_k= +\begin{bmatrix} +u_k & P_{\mathrm{pump},k} & P_{\mathrm{fan},k} +\end{bmatrix}^{\!\top}. +$$ + +Here $x_k$ is the energy state introduced in Section 3, such as battery SOC; +$T_{\mathrm{bat},k}$ and $T_{\mathrm{fc},k}$ are the battery and fuel-cell +temperatures; and $u_k$ is the power-source allocation. The variables +$P_{\mathrm{pump},k}$ and $P_{\mathrm{fan},k}$ are the commanded electrical +powers of the coolant pump and radiator fan. Thus, $T_{\cdot,k}$ denotes +temperature, whereas $H$ below is the prediction-horizon length. Let $w_k$ +collect the predicted traction demand and thermal context, including ambient +conditions. + +A representative problem is + +$$ +\begin{aligned} +\min_{v_0,\ldots,v_{H-1}}\quad& +\sum_{k=0}^{H-1} +\left[ +\ell_{\mathrm{energy}}(s_k,v_k,w_k) ++ +\rho_{\mathrm{th}}\ell_{\mathrm{thermal}}(s_k) +\right] ++ +J_{\mathrm{E/T}}^{\mathrm{tail}}(s_H)\\ +\mathrm{s.t.}\quad& +s_0=s_{\mathrm{init}},\\ +& +s_{k+1}=F_k^{\mathrm{E/T}}(s_k,v_k,w_k), +\qquad k=0,\ldots,H-1,\\ +& +\underline s_k\le s_k\le\overline s_k, +\qquad k=0,\ldots,H,\\ +& +\underline v_k\le v_k\le\overline v_k, +\qquad k=0,\ldots,H-1 . +\end{aligned} +$$ + +Here $\ell_{\mathrm{energy}}$ accounts for fuel or electrical-energy use and, +when relevant, degradation; $\ell_{\mathrm{thermal}}$ penalizes undesirable +temperature operation; and $\rho_{\mathrm{th}}\ge 0$ sets their relative +weight. $J_{\mathrm{E/T}}^{\mathrm{tail}}$ evaluates energy and thermal +consequences beyond the prediction horizon. $s_{\mathrm{init}}$ is the +measured or estimated initial augmented state, +$F_k^{\mathrm{E/T}}$ is the context-dependent coupled energy--thermal +transition map, and the underlined and overlined quantities are the state and +control bounds. + +The problem is mobility-system-level when route, traffic, charging, or ambient +forecasts materially determine these decisions; with only local states and +demand, the same formulation is vehicle-level. + +## 5. Why these predictive problems are difficult + +- **Information and formulation limitations:** future traction-power demand + $w_k$, route, traffic, ambient conditions, and other agents evolve with + physical time and location. Consequently, the dynamics, cost, constraints, + or disturbance distribution can be time- or context-dependent. One + stationary policy or value function (critic) need not represent the exact + optimum unless sufficient context is included or the problem is + reformulated. An approximate or robust stationary policy may still be useful + under stated conditions. Missing transition and context models, prediction + uncertainty, and estimation of SOC or temperature also limit the information + available to the controller. + +- **Computational difficulty:** long horizons, nonlinear battery, fuel-cell, + and thermal dynamics, hybrid operating modes, coupled vehicle or resource + decisions, and numerous state, actuator, safety, and collision constraints + produce large optimization or dynamic-programming problems. The solution + must still be updated within the relevant online control deadline. + +The information and formulation barrier determines which representation, +prediction, estimation, or reformulation is needed to define the decision +problem; computational difficulty limits how accurately it can be solved +online. + +## 6. Two research questions + +The application problems above lead to two research questions: + +1. How can a time- and context-dependent mobility-system problem be converted + into a reusable long-run optimal-control problem despite uncertain future + context? +2. How can its value and current decisions be computed or approximated within + an online deadline, especially for nonlinear energy–thermal systems or + coupled multi-vehicle decisions? + +## 7. MIC Lab approach and connections to research themes + +The xEV formulation is one concrete MIC Lab pathway. The cited hybrid electric +vehicle (HEV) and FCEV studies document successive steps in that pathway; they +do not prescribe how every mobility-system problem must be solved. + +### 7.1 Nonstationarity and context uncertainty: construct a reusable long-run critic + +[**Nonstationary Infinite-Horizon +OCP**](/research/research_themes/04_nonstationary_infinite_horizon_ocp) addresses +the first research question. In this application, the physical-time problem +is first aggregated over road links so that repeated operation can be +represented by a reusable network state rather than by absolute physical +time. + +Let $k$ denote a physical sampling step, $n$ a link step, $q_n$ the current +road link, $\bar x_n$ the energy state at the entry boundary of $q_n$, and +$a_n$ a link-level decision parameter. The time-domain dynamics, costs, and +preview within link $q_n$ are reduced to a link-domain transition and cost: + +$$ +\left\{ +F_k,\ G_k,\ w_k +\right\}_{k\ \mathrm{within}\ q_n} +\quad\longrightarrow\quad +\begin{cases} +\bar x_{n+1}=F_{q_n}(\bar x_n,a_n),\\ +\ell_{q_n}(\bar x_n,a_n). +\end{cases} +$$ + +The [2024 T-ITS +study](https://doi.org/10.1109/TITS.2024.3384358) provides the within-route +reduction: time-sampled demand over a known finite route is summarized by +link-level energy and duration information, while the route horizon and +terminal energy requirement remain explicit. The [2026 +study](https://kaist-mic-lab.github.io/publications/2026-traffic-network/) +extends the same link-domain foundation to repeated operation over a traffic +network. It adds stochastic transitions among links and treats +$(\bar x_n,q_n)$ as a persistent network state. + +If $p_{qq'}=\Pr(q_{n+1}=q'\mid q_n=q)$ is the link-transition probability, the +reusable network value satisfies + +$$ +J_{\mathrm{net}}^\star(\bar x,q) +:= +\min_{a\in\mathcal A(\bar x,q)} +\left\{ +\ell_q(\bar x,a) ++ +\gamma +\sum_{q'}p_{qq'} +J_{\mathrm{net}}^\star +\!\left(F_q(\bar x,a),q'\right) +\right\}. +$$ + +Here $\mathcal A(\bar x,q)$ is the admissible set of link-level decisions, +$F_q$ and $\ell_q$ are the aggregated transition and cost, and +$0<\gamma<1$ is the long-run discount factor. Value iteration computes this +long-run value function, or critic, from the link-level model and transition +statistics. During operation, its expected terminal value is appended to the +finite predicted-link problem. +The optimizer can then select the terminal energy state by balancing the +current route against subsequent network operation, rather than imposing a +fixed terminal-energy target. + +### 7.2 Computational difficulty: approximate DP and online multistep lookahead + +When the state or context space is too large for tabular value iteration, +[Online Learning-Based Optimal +Control](/research/research_themes/00_online_learning_based_optimal_control) +uses approximate dynamic programming (ADP) to represent the long-run value or +policy. Let $\zeta=(\bar x,q)$ denote the link-domain state and $\theta$ the +critic parameters; then + +$$ +\widehat J_{\theta}(\zeta) +\approx +J_{\mathrm{net}}^\star(\bar x,q). +$$ + +For the current prediction, [**Online Multistep +Lookahead**](/research/research_themes/02_online_multistep_lookahead) holds this +critic fixed. More generally, let $\zeta_{j\mid k}$ denote the predicted +application state, $a_{j\mid k}$ its decision, +$A_k=(a_{0\mid k},\ldots,a_{H-1\mid k})$ the decision sequence, and +$g_{\mathrm{use}}$ the declared stage cost. The online problem has the form + +$$ +\widehat A_k^{(H)} +\in +\arg\min_{A_k} +\left\{ +\sum_{j=0}^{H-1} +\alpha^j +g_{\mathrm{use}}(\zeta_{j\mid k},a_{j\mid k}) ++ +\alpha^H +\widehat J_{\theta}(\zeta_{H\mid k}) +\right\}. +$$ + +For link-domain energy management, +$\zeta=(\bar x,q)$ and $a$ is a link decision. For the coupled formulation in +Section 4, $\zeta$ collects $(S,Z)$ and $a$ collects $(V,U)$. The resulting +multistep solution can be executed directly or used to train a policy network +online. The 2026 FCEV study uses link-based tabular value iteration and direct +finite-horizon optimization; critic approximation and online policy training +are broader research extensions. + +### 7.3 Additional learning-based connections + +Other themes can support different missing pieces: + +| Research theme | Possible role in this control domain | +|---|---| +| **[Continual Model Learning](/research/research_themes/05_continual_model_learning)** | Update power-demand, traffic-transition, degradation, or thermal models while retaining behavior learned in earlier operating regions. | +| **[Structured Critic Adaptation](/research/research_themes/08_structured_critic_adaptation)** | Reconfigure a prelearned long-run critic using identified traffic, demand, weather, or degradation parameters. | +| **[Semantic Critic Learning](/research/research_themes/03_semantic_critic_learning)** | Incorporate incident, map, rule, mission, or scene semantics that are not captured by a compact numerical network state. | +| **[Real-World RL](/research/research_themes/01_real_world_rl)** | Learn critic and policy from real streaming operation when simulation or explicit network models omit important effects. | + +## 8. References + +- K. Choi, G. Park, and D. Kum, + “[An Analytical Approach to the Predictive Energy Management of Connected + HEVs: What Information Do We Need to Guarantee Global + Optimality?](https://doi.org/10.1109/TITS.2024.3384358),” + *IEEE Transactions on Intelligent Transportation Systems*, vol. 25, no. 9, + pp. 12749–12761, 2024. +- K. Choi, + “[Traffic Network-Aware Energy Management for FCEVs: Integrating Trip-Specific Control and Long-Run Optimality](https://kaist-mic-lab.github.io/publications/2026-traffic-network/),” + *Asian Control Conference*, 2026. + [[Full text](https://kaist-mic-lab.github.io/static/pub/2026-traffic-network.pdf)] + +
+
+
diff --git a/research/control_problems/02_vehicle_level_optimal_control.md b/research/control_problems/02_vehicle_level_optimal_control.md new file mode 100644 index 000000000..b59f750db --- /dev/null +++ b/research/control_problems/02_vehicle_level_optimal_control.md @@ -0,0 +1,397 @@ +--- +title: Vehicle-Level Optimal Control +layout: default +group: research +math: true +--- + +
+
+
+ + +# Vehicle-Level Optimal Control + +> **Core idea.** Vehicle-level optimal control maps driver or automated-driving +> commands into coordinated propulsion, braking, steering, and suspension +> actions. Automatic calibration is a coupled deployment workflow that adjusts +> gains, maps, cost weights, and learned parameters to reproduce the intended +> agility–stability balance across real operating conditions under declared +> safety checks. + +
+ qFit +
+ A motion command is realized through coordinated drive, brake, steering, and suspension actuation, + while measured vehicle response supports automatic calibration. +
+
+ + +*Vehicle-level control coordinates onboard actuators to shape whole-vehicle +motion. Automatic calibration is a coupled deployment workflow rather than a +separate physical control level.* + +## 1. Scope: coordinated control within the vehicle + +[Future Mobility Control Landscape](/research/control_problems/00_future_mobility_control_landscape) +defines vehicle-level control by where the primary performance objective and +coupled decisions are formed. The boundary is the whole vehicle: driver or +automated-driving commands and onboard measurements enter the controller, and +the controller coordinates onboard actuators to shape the resulting motion. + +Representative elements are + +| Role | Examples | +|---|---| +| **Commands and context** | desired speed, path, yaw or acceleration response, vehicle mode, payload, and estimated road condition | +| **Decision state** | longitudinal and lateral motion, yaw, sideslip, roll or heave, wheel motion, tire-force utilization, and actuator states | +| **Control inputs** | distributed drive or brake torque, front/rear or individual-wheel steering, and active-suspension force | +| **Performance** | path and motion tracking, agility, stability, safety, ride comfort, efficiency, and actuator smoothness | +| **Constraints** | tire-force capacity, vehicle stability envelope, actuator magnitude/rate, power, comfort, and electronic control unit (ECU) execution limits | + +This scope includes controllers acting through one or several actuator +families. At its broadest realization, it includes **integrated chassis +control**, which coordinates longitudinal, lateral, and vertical vehicle +motion through propulsion or braking, steering, and suspension actuators. The +vehicle-level domain is not restricted to such a fully integrated +architecture: a steering-only or torque-vectoring controller also belongs +when its objective is whole-vehicle motion. + +The subject is established, but expanding vehicle actuation changes the +research question. Distributed electric drive, independent steering, braking, +and active suspension provide more ways to generate the same net force and +moment, while all of them compete for coupled tire and actuator capacity. The +challenge is therefore not merely to add another feedback loop. It is to use +the available actuation degrees of freedom coherently under uncertain +tire–road interaction and to realize a desired vehicle character on the real +vehicle. + +Local power-source allocation using only onboard state and demand is formally +vehicle-level. MIC Lab's main energy-management work is nevertheless treated +under mobility-system-level control because route, traffic, infrastructure, +mission, and long-horizon context materially determine the decision. This page +therefore focuses on whole-vehicle motion and its calibration workflow. + +## 2. Representative control problem and deployment workflow + +| Role | Main decisions | Main objective | +|---|---|---| +| **Physical control problem — Vehicle motion control** | coordinate the available traction, braking, steering, and suspension actions over the vehicle | produce the commanded motion while balancing agility, stability, comfort, efficiency, and tire/actuator margins | +| **Cross-cutting deployment workflow — Automatic calibration of vehicle-motion controllers** | select gains, maps, cost weights, filters, thresholds, or critic/policy parameters | reproduce the intended behavior across speed, maneuver, road, load, tire, and vehicle conditions with fewer manual real-vehicle iterations | + +These are coupled stages in one deployment workflow, not separate physical +control levels. Motion control determines the action for a fixed design and +calibration; the outer-loop calibration workflow in Section 3.2 adjusts only +declared parameters against measured vehicle response and an evaluable target +behavior. + +## 3. Vehicle motion control and its calibration workflow + +### 3.1 Vehicle motion control + +Let $x_k$ denote the vehicle, wheel, and actuator state; $u_k$ the coordinated +actuator command; $\xi_k^{\mathrm{cmd}}$ the driver or automated-driving +command; and $\eta_k$ the control-relevant operating condition. The latter may +include road friction, tire condition, mass and load distribution, or another +slowly changing physical context: + +$$ +\begin{aligned} +x_{k+1} +&= +f(x_k,u_k;\eta_k)+w_k,\\ +y_k +&= +h(x_k)+v_k . +\end{aligned} +$$ + +Here $y_k$ is the measured output, $h$ the measurement map, and $w_k$ and +$v_k$ process and measurement disturbances, respectively. +Tire force, sideslip, friction, and some vertical-load or actuator states may +not be measured directly, so the controller generally uses +$\widehat x_k$ and $\widehat\eta_k$ supplied by an estimator or identifier. + +At physical time $k$, a representative predictive motion-control problem is + +$$ +\begin{aligned} +U_k^\star +\in +\arg\min_{U_k}\quad& +\sum_{j=0}^{H-1} +\alpha^j +\ell_{\mathrm{mot}} +\!\left( +x_{j\mid k}, +u_{j\mid k}, +\xi_{j\mid k}^{\mathrm{cmd}};q +\right) ++ +\alpha^H V_{\mathrm f}(x_{H\mid k})\\ +\mathrm{s.t.}\quad& +x_{0\mid k}=\widehat x_k,\\ +& +x_{j+1\mid k} + = + f_{\mathrm{use}} + \!\left( + x_{j\mid k},u_{j\mid k};\widehat\eta_k + \right), + \quad j=0,\ldots,H-1,\\ +& + x_{j\mid k} + \in + \mathcal X + \!\left(\widehat\eta_k\right), + \quad j=0,\ldots,H,\\ +& + u_{j\mid k} + \in + \mathcal U + \!\left(x_{j\mid k};\widehat\eta_k\right), + \quad j=0,\ldots,H-1. +\end{aligned} +$$ + +$U_k=(u_{0\mid k},\ldots,u_{H-1\mid k})$ is the candidate actuator +sequence, $H$ is the horizon, and $0<\alpha\le 1$ is the finite-horizon +discount factor. $f_{\mathrm{use}}$ is the declared prediction model, +$V_{\mathrm f}$ the terminal value, and $q$ the stage-cost weight vector. +The feasible sets $\mathcal X$ and $\mathcal U$ can encode a tire-force or +friction envelope, vehicle-stability bounds, actuator magnitude and rate +limits, power limits, and ride or safety constraints. $H$ may be one for a +static allocation or longer for predictive motion control. + +A schematic stage cost is + +$$ +\ell_{\mathrm{mot}} += +q_{\mathrm{trk}}\ell_{\mathrm{trk}} ++ +q_{\mathrm{ag}}\ell_{\mathrm{agility}} ++ +q_{\mathrm{st}}\ell_{\mathrm{stability}} ++ +q_{\mathrm{com}}\ell_{\mathrm{comfort}} ++ +q_u\ell_{\mathrm{effort}} . +$$ + +This decomposition is not a universal definition of vehicle feel. The loss +terms select measurable proxies, while the weight vector $q$ determines their +trade-off. Both the model used by the optimizer and the cost structure used to +judge the response must therefore be validated on the real vehicle. + +The deployed controller may be an online optimizer, a conventional controller +with calibrated maps, or a learned policy. They can be represented uniformly +as + +$$ +u_k += +\mu_{\rho} +\!\left( +\widehat x_k, +\xi_k^{\mathrm{cmd}}, +\widehat\eta_k +\right), +$$ + +where $\rho$ collects the quantities exposed to calibration: gains, maps, +filters, thresholds, optimal-control weights, or policy parameters. + +### 3.2 Automatic calibration during implementation and validation + +A representative model-based workflow produces an initial calibration +$\rho_0$. After deployment to a vehicle ECU, engineers repeat real-vehicle +tests, quantitative evaluation, driver assessment, and manual calibration: + +$$ +\text{model-based design} +\rightarrow +\text{ECU deployment} +\rightarrow +\text{vehicle test} +\rightarrow +\text{evaluation} +\rightarrow +\text{manual recalibration}. +$$ + +Automatic calibration seeks to make this outer loop systematic and +sample-efficient. Let $n$ index a real or accepted high-fidelity calibration +trial, $\chi_n$ its operating context, and +$\tau_n(\rho_n)$ the measured closed-loop trajectory under calibration +$\rho_n$. The trial produces quantitative motion metrics $m_n$ and, when +used, a declared driver or test-engineer assessment $s_n$. Define + +$$ +d_n^{\mathrm{cal}} +:= +\left( +\chi_n,\rho_n,\tau_n,m_n,s_n +\right), +\qquad +\mathcal D_{0:n}^{\mathrm{cal}} +:= +\left\{ +d_i^{\mathrm{cal}} +\right\}_{i=0}^{n}. +$$ + +The ideal calibration objective is the expected evaluation loss over the +declared operating-condition distribution $\mathcal P_{\mathrm{op}}$: + +$$ +\mathcal L_{\mathrm{cal}}(\rho) +:= +\mathbb E_{\chi\sim\mathcal P_{\mathrm{op}}} +\left[ +\mathcal L_{\mathrm{eval}} +\!\left( +\tau(\mu_\rho;\chi), +s_{\mathrm{des}} +\right) +\right], +$$ + +where $s_{\mathrm{des}}$ denotes the intended vehicle character or preference. +Because this expectation cannot be evaluated freely on a real vehicle, a +data-driven tuning step uses the accumulated trials to select the next +admissible calibration: + +$$ +\rho_{n+1} +\in +\arg\min_{\rho\in\mathcal P_n^{\mathrm{adm}}} +\widehat{\mathcal L}_{\mathrm{cal},n} +\!\left( +\rho;\mathcal D_{0:n}^{\mathrm{cal}} +\right). +$$ + +$\widehat{\mathcal L}_{\mathrm{cal},n}$ is a data-supported estimate or +surrogate of $\mathcal L_{\mathrm{cal}}$, and +$\mathcal P_n^{\mathrm{adm}}$ is the candidate set admitted at trial $n$ by +declared parameter bounds, pre-test safety checks, and controller acceptance +logic. $\mathcal L_{\mathrm{eval}}$ can combine quantitative motion metrics +with a preference error only after the latter has been converted into an +evaluable rating, label, comparison, or learned surrogate. Actuator and state +constraints, worst-case ECU execution time, fallback behavior, and the number +and coverage of safe real trials remain part of the calibration protocol even +when they are not all written inside this compact argmin. + +This formulation does not require every production parameter to be learned. +The calibration object $\rho$ should include only parameters whose role, +admissible range, and verification procedure have been declared. + +## 4. Why vehicle motion control and calibration are difficult + +- **Information and formulation limitations:** tire forces are nonlinear, + saturating, coupled, and only partly observed; friction, tire condition, + temperature, payload, and maneuver further change the relevant dynamics. + Agility and stability can be quantified through selected proxies, but their + relation to perceived handling quality or the intended vehicle character is + not unique. The controller must therefore identify the required state and + model while expressing subjective assessment through measurable metrics, + ratings, comparisons, or another explicit teaching signal. Real calibration + data arrive sequentially and cannot be explored freely. + +- **Computational difficulty:** a moderate-order, fixed-model model predictive + controller (MPC) can be practical. The burden grows with nonlinear coupled + dynamics, multiple actuators, tire and safety constraints, longer-horizon + effects, and online learning. These operations must fit within ECU memory, + latency, and verification limits. + +The information and formulation barrier determines what must be identified +and what performance should mean; computational difficulty limits how much of +the resulting control and learning problem can be performed online. + +## 5. Two research questions + +1. **Representation under uncertainty:** How can control-oriented vehicle and + tire dynamics, latent motion states, and the desired agility–stability + behavior be represented accurately across changing road, tire, load, + maneuver, and vehicle conditions? +2. **Real-time policy and calibration:** How can the coupled motion policy and + its calibration be computed or learned within ECU and real-test limits + while preserving safety, prior knowledge, and performance over the + declared operating domain? + +## 6. MIC Lab approach and connections to research themes + +The two research questions can be approached through model/state learning, +approximate optimal control, and safe real-world policy improvement. These +mechanisms may be used separately or combined; no single research theme is +assumed to solve the complete vehicle problem. + +### 6.1 Model and state uncertainty: continual control-oriented model learning + +Fast observers and adaptive identifiers can estimate the current condition, +but minimizing the latest residual does not ensure that the learned model +remains accurate across the vehicle's operating domain. +[**Continual Model +Learning**](/research/research_themes/05_continual_model_learning) instead adds new +tire and vehicle behavior while retaining control-relevant behavior learned +in earlier conditions. When the required state is latent, state and model +estimation can be coupled while a retention term preserves selected prior +function behavior. The result must be evaluated by state-estimation quality, +multistep tire/vehicle prediction, and downstream control performance—not only +one-step prediction error. + +### 6.2 Computational difficulty: approximate DP and structured critic reuse + +[Online Learning-Based Optimal +Control](/research/research_themes/00_online_learning_based_optimal_control) +connects the vehicle problem to approximate dynamic programming (ADP): an +expensive long-horizon solution can be represented by an approximate critic +or policy and reused online. + +When a compact operating parameter $\eta$ captures changes such as road +friction, load, or tire condition, [**Structured Critic +Adaptation**](/research/research_themes/08_structured_critic_adaptation) provides a +candidate bridge between identification and policy improvement. The identified +condition reconfigures a stored value structure before policy improvement, +avoiding repeated full critic or policy learning across recurring conditions. +This does not make every fixed MPC solve faster by itself. + +[**Online Multistep +Lookahead**](/research/research_themes/02_online_multistep_lookahead) can then use +the fixed or reconfigured critic as a terminal value while a short online +horizon resolves the current nonlinear dynamics, tire limits, and actuator +constraints. If repeated optimization remains too expensive, its +state-to-solution map can be learned as a policy for faster ECU execution. + +### 6.3 Automatic calibration: constrained Real-World RL + +[**Real-World RL**](/research/research_themes/01_real_world_rl) is a candidate +method for the automatic-calibration problem in Section 3. It can use streaming +real transitions to improve a critic and, when declared, policy or controller +parameters, reducing dependence on a simulator that omits control-relevant +tire, sensing, actuator, or driver effects. The intended goal is reusable +task-domain learning across the declared operating conditions, not only a +trajectory-local adaptive fit. + +The reward or cost must still be specified. Quantitative agility, stability, +comfort, effort, and safety metrics can provide its primary terms; driver or +test-engineer assessment can be used only after it is converted into a +declared learning signal. Explicit constraints, guarded updates, fallback +control, test acceptance, and closed-loop evidence remain necessary for +real-vehicle learning. Real-World RL is therefore a possible implementation +of automatic calibration, not a synonym for the calibration problem itself. + +### 6.4 Supporting research connections + +| Research theme | Possible role in this control domain | +|---|---| +| **[Semantic Critic Learning](/research/research_themes/03_semantic_critic_learning)** | Convert scenario descriptions, driver comparisons, or other contextual feedback into an auxiliary critic-learning signal. This is a candidate extension, not a substitute for a declared control objective. | +| **[Neuro-Adaptive Control](/research/research_themes/07_neuro_adaptive_control)** | Approximate an uncertain ideal motion-control law or residual directly and adapt it under closed-loop and input constraints. Its trajectory-local adaptation role should be distinguished from global model retention and task-domain calibration. | +| **[Constrained PINN](/research/research_themes/06_constrained_pinn)** | Learn a physics-consistent tire or vehicle field/model while imposing selected boundary, initial, constitutive, or feasibility conditions explicitly. | +| **[Nonstationary Infinite-Horizon OCP](/research/research_themes/04_nonstationary_infinite_horizon_ocp)** | Address cases in which changing operating context, constraints, or desired behavior can prevent one stationary critic from representing the exact long-run optimum. An approximate or robust stationary critic may remain useful under stated conditions, so this Theme is not required for every motion controller. | + +
+
+
\ No newline at end of file diff --git a/research/control_problems/03_component_level_optimal_control.md b/research/control_problems/03_component_level_optimal_control.md new file mode 100644 index 000000000..c1e67a8f6 --- /dev/null +++ b/research/control_problems/03_component_level_optimal_control.md @@ -0,0 +1,525 @@ +--- +title: Component-Level Optimal Control +layout: default +group: research +math: true +--- + +
+
+
+ +# Component-Level Optimal Control + +> **Core idea.** Component-level optimal control converts a torque command +> into fast voltage or inverter-switching decisions while minimizing drive +> loss or another performance index and satisfying current and voltage limits. +> Automatic calibration is a coupled deployment workflow that seeks to +> reproduce this behavior across the real electric drive's operating domain +> under declared verification checks. + +
+ qFit +
+ A torque command is converted into constrained voltage or inverter-switching actions + that shape current and flux to produce torque. +
+
+ +*The component-level controller realizes the requested torque through fast +electrical actuation, while current feedback and flux estimation support the +drive-level loop.* + +## 1. Scope: optimal control at the electric-drive component + +[Future Mobility Control Landscape](/research/control_problems/00_future_mobility_control_landscape) +defines component-level control first by a device-level performance objective +and coupled decision boundary. The physical-device timescale is a secondary +descriptor. For an electric drive, the +representative component is a synchronous machine and its inverter. The +controller acts on electrical variables much faster than vehicle or +mobility-system controllers act on motion, energy, or traffic decisions. + +Representative elements are + +| Role | Examples | +|---|---| +| **Commands and context** | torque command, rotor speed or position, DC-link voltage, machine temperature, and operating mode | +| **Decision state** | direct–quadrature ($dq$) stator current, stator flux linkage, inverter state, and selected thermal or loss-related states | +| **Control inputs** | continuous stator-voltage command or discrete inverter switching action | +| **Performance** | torque tracking, copper/iron/inverter loss, torque and current ripple, switching behavior, and transient response | +| **Constraints** | current, voltage, inverter switching, thermal, sampling-time, and actuator-feasibility limits | + +Torque is the mechanical interface between the electric drive and the system +it actuates. Rotor speed and position evolve through the balance between +electromagnetic torque and the mechanical load, and propulsion, steering, +robotic, and industrial systems are driven through this torque. Accurately +producing a requested torque while using the available electrical degrees of +freedom efficiently is therefore a central component-level control problem. + +The boundary is determined by the objective. A higher-level vehicle or motion +controller may decide the required wheel or shaft torque; the component-level +controller realizes that command by choosing voltage or inverter-switching +inputs that shape the resulting current and flux trajectories. If the primary +objective instead coordinates several vehicle actuators or trip-scale energy, +the problem moves to the vehicle or mobility-system level even though an +electric motor executes the final action. + +This page uses synchronous-machine torque control as the representative +problem. The same control logic—fast constrained actuation under uncertain +device physics—can extend to other electric drives and mechatronic components. + +## 2. Representative control problem and deployment workflow + +| Role | Main decisions | Main objective | +|---|---|---| +| **Physical control problem — Optimal torque control** | choose voltage or inverter switching actions to shape the resulting current/flux trajectory | track the commanded torque while minimizing drive loss, ripple, switching, or another performance index under current and voltage limits | +| **Cross-cutting deployment workflow — Automatic calibration of electric-drive controllers** | select and adjust current-, torque-, speed-, position-, estimator-, and solver-related parameters | reproduce the intended tracking, efficiency, robustness, and constraint behavior across speed, torque, temperature, DC-link, load, and machine conditions | + +These are two coupled stages in the controller deployment workflow, not two +physical control levels. Optimal torque control determines the electrical +action for a fixed model, controller structure, and calibration. Automatic +calibration then adjusts gains, maps, loss weights, observer parameters, +limits, filters, and numerical-solver parameters so that the measured hardware +response satisfies the intended performance and feasibility objectives +throughout the declared operating domain. + +The calibration problem can include the surrounding drive loops. A torque or +current controller is usually embedded inside speed and position loops, while +state or parameter estimators supply quantities not measured directly. +Calibrating each block independently may not reproduce the desired behavior of +the complete interconnected drive. + +## 3. Optimal torque control and its calibration workflow + +### 3.1 Optimal torque control + +Let $x_k$ collect the controller-relevant electrical and inverter state, +$u_k$ denote a continuous voltage vector or discrete switching action, +$T_{e,k}$ the electromagnetic torque, and $\eta_k$ the control-relevant +operating condition. A control-usable model has the form + +$$ +\begin{aligned} +x_{k+1} +&= +f +\!\left( +x_k, +u_k; +\eta_k +\right),\\ +T_{e,k} +&= +\mathcal T +\!\left( +x_k; +\eta_k +\right),\\ +y_k +&= +h(x_k)+\nu_k . +\end{aligned} +$$ + +Here $y_k$ is the measured output, $h$ the measurement map, and $\nu_k$ the +measurement disturbance. +$\eta_k$ may contain rotor speed, DC-link voltage, temperature, +resistance, or magnetic-model information. In particular, nonlinear stator +flux linkage is a function of current and operating condition: + +$$ +\boldsymbol\lambda_{dq,k} += +\Lambda +\!\left( +\boldsymbol i_{dq,k}; +\eta_k +\right). +$$ + +Depending on the chosen machine representation, +$\boldsymbol\lambda_{dq,k}$ can be included explicitly in $x_k$ or reconstructed +from current and operating condition through a known or learned map +$\Lambda$. The latter case makes flux-model learning and flux estimation part +of the information needed for torque control. + +At physical time $k$, a representative finite-horizon predictive +torque-control problem with horizon $H<\infty$ is written over +$U_k=(u_{0\mid k},\ldots,u_{H-1\mid k})$: + +$$ +\begin{aligned} +U_k^\star +\in +\arg\min_{U_k}\quad& +\sum_{j=0}^{H-1} +\alpha^j +\Big[ +q_T +\left\| +T_{e,j+1\mid k}-T_{j+1\mid k}^{\mathrm{cmd}} +\right\|^2 ++ +\ell_{\mathrm{perf}} +\!\left( +x_{j+1\mid k}, +u_{j\mid k} +\right) +\Big]\\ +&+ +\alpha^H V_{\mathrm f}(x_{H\mid k})\\ +\mathrm{s.t.}\quad& +x_{0\mid k} += +\widehat x_k,\\ +& +x_{j+1\mid k} + = + f_{\mathrm{use}} + \!\left( + x_{j\mid k}, + u_{j\mid k}; + \widehat\eta_k + \right), + \quad j=0,\ldots,H-1,\\ +& + \left\| + \boldsymbol i_{dq,j+1\mid k} + \right\| + \le I_{\max}, + \quad j=0,\ldots,H-1,\\ +& + u_{j\mid k} + \in + \mathcal U_{\mathrm{inv}}, + \quad j=0,\ldots,H-1. +\end{aligned} +$$ + +$T_{j+1\mid k}^{\mathrm{cmd}}$ is the predicted torque command, +$q_T\ge0$ its tracking weight, and $0<\alpha\le1$ the finite-horizon discount +factor. The vectors $\boldsymbol i_{dq}$ and $\boldsymbol v_{dq}$ are the +$dq$ current and voltage, and $I_{\max}$ is the allowable current magnitude. +$f_{\mathrm{use}}$ is the declared known or learned prediction model. +$V_{\mathrm f}$ is the finite-horizon terminal value, possibly supplied by a +learned value-function approximation (critic). +$\ell_{\mathrm{perf}}$ can represent copper, iron, or inverter loss, +switching activity, ripple, thermal stress, or another declared drive-level +objective. For a two-level voltage-source inverter, +$\mathcal U_{\mathrm{inv}}$ is the DC-link-dependent voltage hexagon in a +continuous-input implementation or the finite set of admissible inverter +voltage vectors in direct finite-control-set switching. + +The [Generalized Model Predictive Torque Control +(GMPTC)](https://doi.org/10.1109/TMECH.2024.3461209) study provides a concrete +one-step instance. It enforces torque, current, and voltage feasibility while +allowing a declared combination of copper, iron, and inverter loss. When the +requested torque is infeasible, it instead maximizes achievable torque in the +commanded direction. GMPTC supports continuous or finite inverter control sets +and is a practical constrained short-horizon realization, not an +infinite-horizon solution. + +The preceding problems are finite-horizon formulations: GMPTC uses $H=1$, +while a longer-horizon model predictive controller (MPC) uses a finite $H>1$. +Receding-horizon execution may +continue indefinitely, but each online optimization explicitly evaluates only +the next $H$ stages and represents everything beyond them through +$V_{\mathrm f}$. If that terminal value does not accurately represent the +long-run consequence, near-term torque and loss optimization can remain +suboptimal over continuous operation. + +The ideal long-run decision problem can instead be defined directly as an +infinite-horizon optimal-control problem (OCP). For this definition, $x_k$ is +understood to be augmented with the torque command, rotor speed, DC-link +condition, temperature, and any other context required to make the decision +state Markov. This augmentation is valid only when the context dynamics or +transition law and the probability law underlying the expectation are also +specified: + +$$ +\mu^\star +\in +\arg\min_{\mu} +\mathbb E +\left[ +\sum_{j=0}^{\infty} +\alpha^j +g +\!\left( +x_{k+j}, +\mu(x_{k+j}) +\right) +\right], +\qquad +0<\alpha<1, +$$ + +subject to the electric-drive dynamics and current, voltage, thermal, and +switching constraints, where $g$ is the declared long-run stage cost. Solving +this problem would minimize the declared long-run cost directly. Because its +exact solution is generally impractical, the research objective is to +approximate its optimal value and policy through approximate dynamic +programming (ADP). Finite-horizon MPC remains a useful online implementation, +especially when $V_{\mathrm f}$ is supplied by an approximate infinite-horizon +critic rather than by an arbitrary short-horizon terminal penalty. + +### 3.2 Automatic calibration during implementation and validation + +Automatic calibration is not limited to the fast torque controller. A deployed +electric-drive control system can contain current or torque control, outer +speed and position loops, flux and rotor-state estimators, inverter logic, and +the driven mechanical plant. Their gains, maps, filters, limits, objective +weights, and solver settings interact through the complete cascaded, +multi-rate closed loop. + +Let $r_k$ denote the command presented to this stack—such as a torque, speed, +or position command—and let $\mu_\rho$ denote the resulting composite +controller: + +$$ +u_k += +\mu_\rho +\!\left( +\widehat x_k, +r_k +\right). +$$ + +Here, $\widehat x_k$ contains the available electrical, mechanical, estimated, +and operating-context variables. The deployable calibration vector $\rho$ +collects parameters across the inner and outer controllers, estimators, +constraint handling, filters, and numerical solver. Calibration can therefore +evaluate the complete response from $r_k$ to torque, speed, or position rather +than assuming that every inner loop is ideal. + +Model-based design supplies an initial controller and calibration $\rho_0$. +After implementation on the drive electronic control unit (ECU), repeated +simulation, dynamometer, or system tests evaluate tracking, loss, ripple, +constraint activity, robustness, and execution time: + +$$ +\text{model-based design} +\rightarrow +\text{ECU/inverter implementation} +\rightarrow +\text{drive or system test} +\rightarrow +\text{evaluation} +\rightarrow +\text{manual recalibration}. +$$ + +Let $n$ index an accepted calibration trial, $\chi_n$ its machine, load, and +operating context, $\tau_n(\rho_n)$ the measured trajectory of the complete +cascaded closed loop, and $m_n$ the resulting performance and feasibility +metrics. Define + +$$ +d_n^{\mathrm{cal}} +:= +\left( +\chi_n,\rho_n,\tau_n,m_n +\right), +\qquad +\mathcal D_{0:n}^{\mathrm{cal}} +:= +\left\{ +d_i^{\mathrm{cal}} +\right\}_{i=0}^{n}. +$$ + +A data-driven tuning step can select the next admissible calibration through + +$$ +\rho_{n+1} +\in +\arg\min_{\rho\in\mathcal P_n^{\mathrm{adm}}} +\widehat{\mathcal L}_{\mathrm{cal},n} +\!\left( +\rho;\mathcal D_{0:n}^{\mathrm{cal}} +\right). +$$ + +$\widehat{\mathcal L}_{\mathrm{cal},n}$ estimates the calibration objective +over the declared speed, torque, temperature, DC-link, load, and machine +domain. $\mathcal P_n^{\mathrm{adm}}$ is the candidate set admitted by +parameter bounds, pre-test checks, current/voltage and thermal constraints, +fallback logic, and ECU timing requirements. The algorithm must learn from +limited safe experiments without treating constraint violations as freely +available exploration. + +## 4. Why electric-drive optimal control and calibration are difficult + +- **Information and formulation limitations:** the nonlinear stator + flux-linkage map is central to torque prediction and constraint handling but + is not measured directly. Resistance, inverter nonlinearity, iron loss, + temperature, aging, and sensorless rotor-state estimation add uncertainty. + The surrounding current, torque, speed, and position loops also interact + through saturation, finite bandwidth, estimators, and shared model errors, + so isolated block behavior does not determine the complete closed loop. + Hardware calibration data remain limited because unsafe electrical, thermal, + or unstable controller parameters cannot be explored freely. + +- **Computational difficulty:** one-step MPC can be practical. The burden grows + with longer horizons, nonlinear magnetic and loss models, constrained + continuous optimization, or branching over switching actions, while + electrical sampling periods remain short. Exact infinite-horizon dynamic + programming and online model, critic, or policy updates must also fit within + embedded memory, computation, and verification limits. + +## 5. Three research questions + +1. **Model and state learning:** How can a physically consistent flux-linkage + model and the unmeasured control state be learned online across changing + electrical and thermal conditions while retaining behavior learned in + earlier operating regions? +2. **Interconnected control and calibration:** How can controller and + estimator parameters across the torque/current, speed, and position loops + be calibrated from the behavior of the interconnected + inverter–machine–mechanical closed loop rather than through isolated + ideal-loop assumptions? +3. **Long-run embedded optimality:** How can an infinite-horizon + electric-drive value or policy be approximated and improved within the + drive ECU's short control deadline while preserving current, voltage, + thermal, and switching feasibility? + +## 6. MIC Lab approach and connections to research themes + +The two cited papers address complementary parts of the component-level +problem. GMPTC supplies a constrained short-horizon torque-control baseline +under a known magnetic model. Physics-Informed Online Learning (PIOL) addresses +the type of flux-model uncertainty treated as known in GMPTC. Infinite-horizon +ADP, continual retention, and automatic calibration are research extensions +rather than achieved claims of those papers. + +### 6.1 Flux-model and state uncertainty: physics-constrained online learning and continual retention + +The [PIOL +study](https://doi.org/10.1109/IECON58223.2025.11221587) treats stator flux +linkage as a nonlinear function of measured $dq$ currents. Applying the chain +rule converts the measured electrical ordinary differential equations into +current-domain partial differential constraints. The neural flux model +minimizes this physics residual while explicitly bounding physically +meaningful self-differential inductances. PIOL estimates the current derivative +by finite differences and treats stator resistance as known. +This is a concrete component-level instance of a [**Constrained +physics-informed neural network +(PINN)**](/research/research_themes/06_constrained_pinn): physics residuals define +the objective, physical inequalities remain explicit constraints, and +primal–dual updates learn the model parameters and constraint multipliers. +The learned model can also act as an online flux-linkage estimator. + +PIOL demonstrated simulation-level feasibility on an interior permanent-magnet +synchronous machine (IPMSM) by updating the output weights of a +single-hidden-layer network. It assumes measured currents and rotor speed, +treats the applied voltage as available and the stator resistance as known, +and uses an ideal-inverter model while neglecting iron-loss effects. Its +first-order Karush–Kuhn–Tucker (KKT) conditions do not establish global +learning optimality, and the study does not provide hardware evidence or +prior-function retention. + +[**Continual Model +Learning**](/research/research_themes/05_continual_model_learning) is one possible +extension from online trajectory adaptation toward a globally reusable +magnetic model. New temperature, saturation, aging, and machine behavior +should be added without overwriting useful flux behavior learned in earlier +regions. Replay anchors, local activation, or previous-model pseudo-data can +supply the retention mechanism; their benefit must be evaluated through +return-to-prior-region model error and downstream torque-control performance. + +### 6.2 Cascaded interaction and automatic calibration: Real-World RL + +[**Real-World RL**](/research/research_themes/01_real_world_rl) provides a +candidate method for the automatic-calibration problem. Instead of calibrating +each loop only against an isolated local response, the real transition stream +can evaluate the full task: torque production, speed or position response, +drive loss, ripple, estimator behavior, saturation, and constraint activity. +Critic and policy or controller parameters can then be improved over the +declared operating domain. + +The control objective must still be numerical and auditable. Torque error, +loss, ripple, thermal response, settling, robustness, and violations can form +the primary cost and constraints. Real-World RL does not itself guarantee safe +calibration; guarded experiments, admissible parameter sets, fallback +controllers, update acceptance, and closed-loop evidence remain necessary. +Its intended distinction from conventional auto-tuning is task-domain +learning and reuse rather than repeated local fitting at the latest operating +point. + +### 6.3 Computational difficulty: from one-step GMPTC to long-run ADP + +GMPTC demonstrates that one-step constrained optimal torque control can be +implemented using either a continuous or finite control set and a configurable +drive-performance index. The paper reports numerical validation on a +synchronous reluctance machine (SynRM) and experimental validation on an +IPMSM. It assumes the nonlinear flux map is known, requires empirical +solver-parameter tuning, and provides numerical rather than general analytical +closed-loop stability evidence. It should therefore be read as a practical +short-horizon baseline. + +For continuous operation, [Online Learning-Based Optimal +Control](/research/research_themes/00_online_learning_based_optimal_control) +suggests an ADP extension in which an approximate critic represents +consequences beyond the immediate switching or electrical horizon. +[**Online Multistep +Lookahead**](/research/research_themes/02_online_multistep_lookahead) can hold this +critic fixed as the terminal value of the $H$-step constrained problem in +Section 3.1: + +$$ +f_{\mathrm{use}} +\leftarrow +\widehat f_{\psi}, +\qquad +V_{\mathrm f} +\leftarrow +\widehat J_{\theta}, +\qquad +u_k += +\left[ +U_k^\star +\right]_0 . +$$ + +Here $\widehat f_\psi$ and $\widehat J_\theta$ are the learned model and critic, +respectively, and $[\,\cdot\,]_0$ selects the first action of the optimized +sequence. +The online short horizon resolves the electrical dynamics, current/voltage +limits, and inverter actions; the fixed critic supplies the long-run tail. +If the online solve is still too expensive, its solution map can be learned as +a policy for fast execution. + +[**Structured Critic +Adaptation**](/research/research_themes/08_structured_critic_adaptation) is a +candidate reusable mechanism. A compact identified condition—such +as temperature, DC-link voltage, resistance, magnetic state, or load—can +reconfigure a stored critic before policy improvement, rather than requiring +full critic relearning whenever the operating condition changes. + +### 6.4 Supporting research connections + +| Research theme | Possible role in this control domain | +|---|---| +| **[Neuro-Adaptive Control](/research/research_themes/07_neuro_adaptive_control)** | Approximate an uncertain ideal voltage, torque-control, or residual law directly and adapt it online under weight and input constraints. | +| **[Nonstationary Infinite-Horizon OCP](/research/research_themes/04_nonstationary_infinite_horizon_ocp)** | Represent long-run drive operation when speed, torque demand, temperature, DC-link condition, or operating mode changes the relevant dynamics, cost, or constraints. One stationary critic need not represent the exact optimum unless sufficient context is included or the problem is reformulated; an approximate or robust stationary critic may still be useful under stated conditions. | +| **[Semantic Critic Learning](/research/research_themes/03_semantic_critic_learning)** | Incorporate maintenance, fault, mission, or operating-mode semantics into a critic when they are not captured by the numerical electrical state. This is a secondary connection rather than a requirement for the fast torque loop. | + +## 7. References + +- K. Choi, J. Kim, and K.-B. Park, + “[Generalized Model Predictive Torque Control of Synchronous + Machines](https://doi.org/10.1109/TMECH.2024.3461209),” + *IEEE/ASME Transactions on Mechatronics*, vol. 30, no. 4, + pp. 2643–2653, 2025. +- S. Jang, M. Ryu, and K. Choi, + “[Physics-Informed Online Learning of Flux Linkage Model for Synchronous + Machines](https://doi.org/10.1109/IECON58223.2025.11221587),” + *IECON 2025 – 51st Annual Conference of the IEEE Industrial Electronics + Society*, 2025. + +
+
+
\ No newline at end of file diff --git a/research/control_problems/assets/electric_drive_torque_control.png b/research/control_problems/assets/electric_drive_torque_control.png new file mode 100644 index 000000000..03a81a35b Binary files /dev/null and b/research/control_problems/assets/electric_drive_torque_control.png differ diff --git a/research/control_problems/assets/future_mobility_control_landscape.png b/research/control_problems/assets/future_mobility_control_landscape.png new file mode 100644 index 000000000..500170f2e Binary files /dev/null and b/research/control_problems/assets/future_mobility_control_landscape.png differ diff --git a/research/control_problems/assets/network_aware_mobility_control.png b/research/control_problems/assets/network_aware_mobility_control.png new file mode 100644 index 000000000..c8589b1c1 Binary files /dev/null and b/research/control_problems/assets/network_aware_mobility_control.png differ diff --git a/research/control_problems/assets/vehicle_motion_control.png b/research/control_problems/assets/vehicle_motion_control.png new file mode 100644 index 000000000..54a50e73e Binary files /dev/null and b/research/control_problems/assets/vehicle_motion_control.png differ diff --git a/research/research_themes/00_online_learning_based_optimal_control.md b/research/research_themes/00_online_learning_based_optimal_control.md new file mode 100644 index 000000000..837d518a1 --- /dev/null +++ b/research/research_themes/00_online_learning_based_optimal_control.md @@ -0,0 +1,450 @@ +--- +title: Online Learning-Based Optimal Control +layout: default +group: research +math: true +--- + +
+
+
+ +# Online Learning-Based Optimal Control + +> **Core idea.** MIC Lab starts from one optimal-control problem. Bellman +> optimality gives the exact decision principle. Learning makes that principle +> implementable when exact computation, the dynamics, or the physical state is +> unavailable. + +
+ MIC Lab learning-based control research themes: a common online learning-based optimal-control framework surrounded by eight research themes. +
+ MIC Lab learning-based control research themes: + a common online learning-based optimal-control framework surrounded by eight research themes +
+
+ +*MIC Lab Learning-Based Control Research Themes. The center presents the +common optimal-control, environment, identification, and estimation framework. +The surrounding cards organize eight research themes by the learned object and +its role in the closed loop.* + +> **How to read this overview.** Sections 1–3 establish the common +> optimal-control and Bellman reference. Sections 4–6 organize the three +> learning roles. Section 7 summarizes the boundary among all eight Research +> Themes. Section 8 states the scope and evidence requirements. + +## 1. The optimal-control problem + +The environment supplies dynamics, measurements, and stage cost: + +$$ +\begin{aligned} +x_{k+1} +&=f(x_k,u_k,w_k),\qquad +y_k=h(x_k)+v_k,\\ +g_k&:=g(x_k,u_k). +\end{aligned} +$$ + +$x_k$, $u_k$, $w_k$, $y_k$, and $v_k$ denote the state, control input, +process disturbance, measurement, and measurement noise; $f$, $h$, and $g$ +are the dynamics, measurement map, and stage cost. Thus $g_k$ is the realized +scalar stage cost at step $k$. + +To obtain the best long-horizon performance in this environment—equivalently, +to minimize the stated cost—the controller is determined by solving + +$$ +\begin{aligned} +\mu^* +&\in +\arg\min_{\mu}\; +\mathbb E_{\mu,\,x_0\sim p_0}\!\left[ +\sum_{k=0}^{\infty} +\alpha^k g\!\left(x_k,\mu(x_k)\right) +\right],\\ +\text{s.t.}\qquad +u_k&=\mu(x_k)\in\mathcal U(x_k), +\qquad 0<\alpha<1. +\end{aligned} +$$ + +Here $p_0$ is the declared initial-state distribution; a fixed initial state +is represented by a point mass at that state. The expectation is over this +initial condition and any declared stochastic dynamics or disturbances. + +When $x_k$ is not measured, an estimated state or another sufficient decision +state must replace it in the implemented controller; Section 5 explains that +part of the loop. + +## 2. Bellman optimality gives the solution principle + +The optimal value function is the minimum cost-to-go from state $x$: + +$$ +J^*(x) +:= +\inf_{\mu} +\mathbb E_{\mu}\!\left[ +\left. +\sum_{j=0}^{\infty} +\alpha^j g\!\left(x_j,\mu(x_j)\right) +\right|x_0=x +\right]. +$$ + +For a stationary discounted problem with a valid Markov decision state, it +satisfies + +$$ +\begin{aligned} +J^*(x) +&= +\min_{u\in\mathcal U(x)} +\left\{ +g(x,u)+\alpha\, +\mathbb E[J^*(x^+)\mid x,u] +\right\},\\ +\mu^*(x) +&\in +\arg\min_{u\in\mathcal U(x)} +\left\{ +g(x,u)+\alpha\, +\mathbb E[J^*(x^+)\mid x,u] +\right\}. +\end{aligned} +$$ + +$x^+$ denotes the successor state associated with the current state and +action. + +Dynamic programming (DP) uses this recursion to evaluate a policy and improve +the decision rule. Thus, $J^*$ is the exact **value function**, and the +minimizing decision defines the exact optimal policy. The next section +explains how learning makes these Bellman objects—and the information required +to compute them—available in practice. + +## 3. Why learning-based control is needed + +- **Computational difficulty:** even with usable state and dynamics, + high-dimensional DP, long-horizon optimization, and constraint handling may + be too expensive for exact offline or online solution. +- **Information and formulation limitations:** the dynamics $f$, current state + $x_k$, disturbance law, operating context, objective, or constraints needed + by the Bellman problem may be incomplete, only indirectly measured, or + changing during operation. + +Learning-based control makes the Bellman principle implementable by learning +the objects or information that its exact solution requires. Within this +broader framework, the **Controller & Policy** lane is naturally viewed as +**approximate dynamic programming (ADP)**: it approximates the Bellman value +or Q-factor and/or the policy produced by Bellman improvement. + +Learning a dynamics model or estimating the state is not, by itself, ADP. The +**Identifier & Estimator** lane supplies missing decision-state or successor +information that ADP—or another control optimizer—can use. The **Joint +Control × Identification** lane couples these two roles. + +
+ Bellman optimality, the environment, and joint online model learning and state estimation form the learning-based control loop. +
+ Bellman optimality, the environment, and joint online model learning and state estimation form the learning-based control loop. +
+
+ +*The diagram suppresses some function-parameter subscripts for readability. +Section 4 writes the learned critic, policy, and model as +$\widehat J_{\theta_t}$, +$\widehat\mu_{\phi_t}$, and $\widehat f_{\psi_t}$.* + +## 4. Controller & policy — learn a critic, a policy, or both + +Section 2 gives the exact Bellman recursion, and Section 3 identifies its +control-side approximation as ADP. This lane implements that idea by learning +a critic, a policy, or both, while treating state and successor information as +measured, estimated in Section 5, or sampled from the environment. The Themes +below differ mainly in which Bellman object is learned online and which object +is fixed or reformulated. + +A representative model-based realization trains a policy to reproduce the +action obtained from a learned critic and model. Physical time is indexed by +$k$, optimizer updates by $t$, and $t(k)$ denotes the parameter snapshot used +at step $k$: + +$$ +\widehat\mu_{\phi_t}(\widehat x) +\approx +\arg\min_{u\in\mathcal U(\widehat x)} +\left\{ +g(\widehat x,u) ++\alpha\widehat J_{\theta_t} +\!\left(\widehat f_{\psi_t}(\widehat x,u)\right) +\right\}, +\qquad +u_k=\widehat\mu_{\phi_{t(k)}}(\widehat x_k). +$$ + +$\widehat J_{\theta_t}$ is the learned critic, +$\widehat\mu_{\phi_t}$ the policy, and $\widehat f_{\psi_t}$ the successor +model. This is the model-based form of a conceptual Bellman-improvement +target, not the definition of every actor–critic algorithm. + +If a successor model is unavailable, an observed transition provides a +sample-based Q-factor target directly: + +$$ +\widehat Q_t^{\mathrm{tar}}(\widehat x_k,u_k) +:= +g(\widehat x_k,u_k)+ +\alpha\widehat J_{\theta_t}(\widehat x_{k+1}). +$$ + +A learned Q-critic fits such targets over state–action pairs and supports +policy improvement by minimizing the learned Q-factor—the action-conditioned +cost-to-go—over $u$. This is the sample-based, model-free form of the same ADP +logic. + +- **[Real-World RL](/research/research_themes/01_real_world_rl) — update critic and, when explicitly + parameterized, policy:** selected streaming transitions $\mathcal S_t$ can + train both function approximations in an actor–critic realization, rather + than only tracking a local adaptive parameter: + + $$ + \left(\widehat J_{\theta_t},\widehat\mu_{\phi_t}\right) + \xrightarrow[\mathcal S_t]{\mathrm{online\ training}} + \left(\widehat J_{\theta_{t+1}},\widehat\mu_{\phi_{t+1}}\right). + $$ + + $\mathcal S_t$ may contain only the latest real transition or selected + retained transitions. Replay and minibatch training are optional mechanisms, + not part of the Research Theme definition. + +- **[Online Multistep Lookahead](/research/research_themes/02_online_multistep_lookahead) — fix the + critic, improve the policy:** an + $H$-step minimization uses a fixed terminal critic + $\widehat J_{\bar\theta}$ to generate a better action; the policy is then + trained online to reproduce that action: + + $$ + \widehat u_k^{(H)} + = + \left[ + \arg\min_U + \left\{ + \sum_{j=0}^{H-1}\alpha^j g_{k+j\mid k} + +\alpha^H\widehat J_{\bar\theta} + (\widehat x_{k+H\mid k}) + \right\} + \right]_0, + \qquad + \phi_{t+1} + \leftarrow + \operatorname{Train}_{\mathrm{online}} + \!\left(\phi_t;\widehat x_k\mapsto\widehat u_k^{(H)}\right). + $$ + + Here $H$ is the lookahead horizon, $U$ is the candidate input sequence, + $g_{k+j\mid k}$ is its predicted stage cost, and $[\cdot]_0$ selects the + first action. + +- **[Semantic Critic Learning](/research/research_themes/03_semantic_critic_learning) — use semantics + to update the critic:** a VLM or + LLM extracts task-relevant semantic information from observations; that + information guides critic targets or updates, followed by policy + improvement: + + $$ + z_k^{\mathrm{sem}} + = + \operatorname{VLM/LLM}(o_k) + \;\longrightarrow\; + \widehat J_{\theta_{t+1}} + \;\longrightarrow\; + \widehat\mu_{\phi_{t+1}}. + $$ + +- **[Nonstationary Infinite-Horizon OCP](/research/research_themes/04_nonstationary_infinite_horizon_ocp) + — reformulate the reference problem:** + if $\widetilde f_k$, $\widetilde g_k$, constraints, or disturbance laws + depend on physical time $k$, then the exact reference is $J_k^*$ and + $\mu_k^*$. A single stationary critic and policy require a valid + reformulation or state augmentation: + + $$ + J_k^*(x) + = + \min_u + \left\{ + \widetilde g_k(x,u)+\alpha J_{k+1}^* + \!\left(\widetilde f_k(x,u)\right) + \right\} + \quad\xRightarrow{\mathrm{reformulate}}\quad + \overline J^*(z) + = + \min_u + \left\{ + \overline g(z,u) + +\alpha\, + \mathbb E[ + \overline J^*(z^+)\mid z,u] + \right\}. + $$ + + Here $z=(x,c)$ contains sufficient context $c$ and has a declared + time-homogeneous transition law. This stationary form is available only + when that augmented process is Markov and preserves the original objective + and constraints. + +## 5. Identifier & estimator — incomplete model or state information + +When the state and model are both incomplete, the general principle is to find +a trajectory and model that are simultaneously consistent with measurements +and dynamics: + +$$ +\left(x_{0:k}^*,\psi^*\right) +\in +\arg\min_{x_{0:k},\psi} +\left\{ +\mathcal L_{\mathrm{meas}}(x_{0:k};y_{0:k}) ++ +\mathcal L_{\mathrm{dyn}}(x_{0:k},\psi;u_{0:k-1}) +\right\}. +$$ + +$\psi$ parameterizes the learned dynamics model, while +$\mathcal L_{\mathrm{meas}}$ and $\mathcal L_{\mathrm{dyn}}$ measure +measurement and dynamics consistency. + +An online realization can restrict this fit to a recent data window or use a +recursive observer. Its convergence and information requirements remain +method dependent. + +- **[Continual Model Learning](/research/research_themes/05_continual_model_learning):** jointly fit + recent state and model information while retaining prior function behavior. + At learner update $t$, let $k_t:=k(t)$ be the newest available physical + sample and $M$ the estimation-window length: + + $$ + \begin{aligned} + \left(\widehat x_{k_t-M:k_t},\widehat\psi_{t+1}\right) + \approx + \arg\min_{x_{k_t-M:k_t},\psi} + \Big\{& + \widehat{\mathcal L}_{\mathrm{meas},t}(x) + +\widehat{\mathcal L}_{\mathrm{dyn},t}(x,\psi)\\ + &+\lambda_{\mathrm{retain}} + \mathcal L_{\mathrm{retain}} + \!\left( + \widehat f_\psi,\widehat f_{\widehat\psi_t};\mathcal M_t + \right) + \Big\}. + \end{aligned} + $$ + + $\widehat\psi_{t+1}$ is the updated model-parameter estimate, + $\mathcal M_t$ is a retained reference set, and + $\lambda_{\mathrm{retain}}\ge0$ weights retention. The first two terms fit + current measurements and dynamics; the last retains selected behavior of + the previously learned model. + +- **[Constrained PINN](/research/research_themes/06_constrained_pinn):** learn complex + physics-consistent fields or models, then + express boundary, initial, or other physical information as explicit + constraints instead of manually weighted loss terms: + + $$ + \min_{\psi}\quad\mathcal L_{\mathrm{physics}}(\psi) + \qquad + \mathrm{s.t.}\quad + c_{\mathrm{BC/IC}}(\psi)=0,\qquad + c_{\mathrm{phys}}(\psi)\le 0. + $$ + +These are alternative or composable mechanisms, not a required causal +sequence. + +## 6. Joint control × identification + +The two paths can also be coupled so that learning about uncertainty directly +changes the control object. + +- **[Neuro-Adaptive Control](/research/research_themes/07_neuro_adaptive_control):** instead of first + producing a reusable model, a neural policy approximates the uncertain + ideal feedback law: + + $$ + \begin{aligned} + \mu_{\mathrm{ideal}}(x) + &= + \mu_{\phi_{\mathrm{ideal}}}(x) + +\varepsilon_{\mathrm{app}}(x),\\ + u_k + &= + \mu_{\phi_{t(k)}}(\widehat x_k),\\ + \phi_t + &\xrightarrow{\mathrm{online\ adaptation}} + \phi_{t+1}. + \end{aligned} + $$ + + $\mu_{\mathrm{ideal}}$ is a declared ideal feedback law and need not be the + Bellman-optimal policy $\mu^*$. $\phi_{\mathrm{ideal}}$ is its ideal + parameter in the chosen neural class, + $\varepsilon_{\mathrm{app}}$ is its approximation error, and $\phi_t$ is + the deployed parameter adapted from closed-loop data. The specific update, + admissible parameter set, and stability conditions remain method dependent. + +- **[Structured Critic Adaptation](/research/research_themes/08_structured_critic_adaptation):** an + identifier estimates the current + environment parameter, reconfigures a prelearned critic, and then improves + the policy: + + $$ + \mathcal S_t + \xrightarrow{\mathrm{ID}} + \widehat\eta_t,\qquad + \widehat J_t(x) + = + \beta_{\omega}(x,\widehat\eta_t)J_0(x), + \qquad + \widehat\mu_{t+1} + \leftarrow + \operatorname{Improve} + [\widehat J_t;\widehat\eta_t]. + $$ + + Here $\mathcal S_t$ is the selected history available at learner update + $t$, $J_0$ is a prelearned reference critic, and $\beta_\omega$ reconfigures + that critic from the identified condition. The improvement operator must + declare any model or action-value information it requires. + +## 7. Boundaries among the eight Research Themes + +| Research Theme | Primary object changed or formed | Defining role | +|---|---|---| +| **[Real-World RL](/research/research_themes/01_real_world_rl)** | critic, and policy when explicitly parameterized | learn task-domain value and decision structure from selected real transitions | +| **[Online Multistep Lookahead](/research/research_themes/02_online_multistep_lookahead)** | online action sequence and progressively learned solution policy | hold a terminal critic fixed while online lookahead improves the current action | +| **[Semantic Critic Learning](/research/research_themes/03_semantic_critic_learning)** | semantic supervision and critic | use accepted semantic information to teach the critic before policy improvement | +| **[Nonstationary Infinite-Horizon OCP](/research/research_themes/04_nonstationary_infinite_horizon_ocp)** | context-augmented reference problem | recover a time-homogeneous Bellman formulation when sufficient context exists | +| **[Continual Model Learning](/research/research_themes/05_continual_model_learning)** | dynamics model and, when needed, state trajectory | fit new operating information while retaining prior control-relevant model behavior | +| **[Constrained PINN](/research/research_themes/06_constrained_pinn)** | physics field or model and dual variables | represent selected physical requirements as explicit learning constraints | +| **[Neuro-Adaptive Control](/research/research_themes/07_neuro_adaptive_control)** | policy parameter | adapt an uncertain ideal feedback-law approximation directly from closed-loop signals | +| **[Structured Critic Adaptation](/research/research_themes/08_structured_critic_adaptation)** | environment estimate and critic configuration | reuse a critic family by identifying a compact condition rather than relearning the full critic | + +These Themes are not pairwise opposites. They identify dominant mechanisms +that can be composed in one control problem; the table prevents adjacent +mechanisms from being mistaken for the same contribution. + +## 8. Scope and evidence + +This framework does not claim that every learned controller is optimal, +stable, or safe. Each Theme must state its decision problem, learned object, +information assumptions, offline and online roles, guarantees, and evidence. +Evaluation includes physical performance, constraints, shift, samples, +latency, baselines, and provenance—not training loss alone. + +
+
+
\ No newline at end of file diff --git a/research/research_themes/01_real_world_rl.md b/research/research_themes/01_real_world_rl.md new file mode 100644 index 000000000..10a757559 --- /dev/null +++ b/research/research_themes/01_real_world_rl.md @@ -0,0 +1,304 @@ +--- +title: Real-World RL +layout: default +group: research +math: true +--- + +
+
+
+ +# Real-World RL + +> **Core idea.** Real-World RL targets critic and policy learning from +> streaming interactions with the real system. The goal is to learn +> task-level value and decision structure—not merely to fit the latest +> closed-loop trajectory. The current BOCO evidence is a critic-learning and +> primal-control precursor within this broader Theme. + +
+ Real interactions generate streaming transitions that train both the critic and policy online. +
+ Real interactions generate streaming transitions that train both the critic and policy online +
+
+ +*The environment returns measured transition records; an assumed state +estimator converts the measurements into $\widehat x_k$ before the +state-based transition enters $\mathcal S_t$. Without an estimator, +$\widehat x_k$ is replaced by the information state $o_k$ defined below. +Physical interaction is indexed by $k$, whereas network updates are indexed +by $t$; the two clocks need not advance at the same rate. Here $k(t)$ is the +latest physical-state sample available at learner update $t$, and $t(k)$ is +the latest learner update available at physical step $k$. The figure +abbreviates the available transition history for readability.* + +## 1. Problem definition + +[Online Learning-Based Optimal Control](/research/research_themes/00_online_learning_based_optimal_control) +defines the desired policy through Bellman optimality. The broader Real-World +RL Theme addresses the case in which its critic and, when explicitly +parameterized, policy are learned from real closed-loop data. A representative +approximate policy is + +$$ +\widehat\mu_{\phi_t}(\widehat x) +\approx +\arg\min_{u\in\mathcal U(\widehat x)} +\left\{ +g(\widehat x,u) ++\alpha\widehat J_{\theta_t} +\!\left(\widehat f_{\psi_t}(\widehat x,u)\right) +\right\}. +$$ + +$\widehat J_{\theta_t}$ is the critic and +$\widehat\mu_{\phi_t}$ is the policy network, +$\widehat f_{\psi_t}$ is a successor model with parameter $\psi_t$, and +$0<\alpha<1$ is the discount factor. The expression above is model-based. +Without a successor model, an observed transition provides a sample-based +Q-factor target directly: + +$$ +\widehat Q_t^{\mathrm{tar}}(\widehat x_k,u_k) +:= +g(\widehat x_k,u_k)+ +\alpha\widehat J_{\theta_t}(\widehat x_{k+1}). +$$ + +A learned Q-critic fits such targets over state–action pairs; model-free +policy improvement then minimizes the learned Q-factor over $u$. + +For the primary notation, a state estimator is assumed to supply the +controller-facing state estimate. The real system then supplies a stream + +$$ +d_k=(\widehat x_k,u_k,g_k,\widehat x_{k+1}), +\qquad +\mathcal D_{0:k(t)-1} +:= +(d_0,\ldots,d_{k(t)-1}), +\qquad +\mathcal S_t\subseteq\mathcal D_{0:k(t)-1}. +$$ + +Here $g_k:=g(\widehat x_k,u_k)$ is the realized stage cost. Because a complete +transition ending at the latest state sample $k(t)$ is indexed by $k(t)-1$, +$\mathcal D_{0:k(t)-1}$ is the ordered collection of all completed transition +records available to the learner. The update set $\mathcal S_t$ may contain +the latest transition alone or selected prior transitions. Replay or +minibatch training is optional rather than a defining assumption. + +Without a state estimator, the problem is a **partially observed setting**. +Then $\widehat x_k$ should be replaced by an observation or information state +$o_k$: $o_k=y_k$ only when the current measurement is sufficient for control; +otherwise $o_k$ should summarize an observation history, belief state, or +learned latent state. The corresponding record is +$d_k=(o_k,u_k,g_k,o_{k+1})$. + +In an actor–critic realization, the selected update data train both learned +objects: + +$$ +\left( +\widehat J_{\theta_t}, +\widehat\mu_{\phi_t} +\right) +\xrightarrow[\mathcal S_t]{\mathrm{online\ training}} +\left( +\widehat J_{\theta_{t+1}}, +\widehat\mu_{\phi_{t+1}} +\right), +\qquad +u_k=\widehat\mu_{\phi_{t(k)}}(\widehat x_k). +$$ + +## 2. Why this problem matters + +Many control-oriented RL methods are trained and evaluated primarily in +simulation. Their deployed performance can then be limited by the +**sim-to-real gap**: unmodeled dynamics, sensing imperfections, disturbances, +and operating conditions absent from the simulator. + +Learning directly from the real system can capture this missing information +and continue improving when deployment reveals new operating conditions. Here, +**adaptability** means accumulating and reusing experience to learn a critic +and policy that remain useful over a declared task domain. It does not mean +only tracking a local parameter change along the currently excited trajectory, +as in an adaptive-control-like update. + +## 3. Key challenges + +- **Limited and policy-dependent data:** real transitions arrive sequentially + and are correlated. Unlike simulator rollouts, they cannot be generated + freely over the entire state–action domain. +- **Local fit versus task-level learning:** reducing loss on the latest + trajectory can produce a local fit, forget earlier regimes, or repeatedly + relearn the same task. The learned functions must retain broader + task-relevant structure. +- **Learning inside the closed loop:** critic and policy change while the + system is operating. Update time, action latency, stability, constraints, + fallback behavior, and safety evidence must therefore be part of the + learning problem. + +## 4. A conventional approach — TD-error-based approximate policy iteration + +One conventional approach is TD-error-based approximate policy iteration: +fix the current policy, reduce its one-step temporal-difference error, and +then perform policy improvement. For notational simplicity, this section +writes $x_k$ as the state available to the controller; when an estimator is +used, it is replaced by $\widehat x_k$. Here $\pi_i$ is the policy at +policy-iteration step $i$. + +$$ +\begin{aligned} +\delta_k^{\pi_i} +&= +g(x_k,\pi_i(x_k)) ++\alpha\widehat J^{\pi_i}(x_{k+1}) +-\widehat J^{\pi_i}(x_k),\\ +\widehat J^{\pi_i} +&\xleftarrow{\mathrm{TD\ error\ reduction}} +\operatorname{Fit}\!\left(\delta_k^{\pi_i}\right), +\qquad +\pi_{i+1} +\leftarrow +\operatorname{Greedy}\!\left(\widehat J^{\pi_i}\right). +\end{aligned} +$$ + +This is a valid approximate-DP mechanism, but streaming data may make it +adaptation-like: a small TD error on visited samples establishes sampled +policy consistency, not task-wide Bellman optimality. Limited coverage can +therefore yield episode- or trajectory-dependent critic parameters. + +## 5. Current precursor — BOCO critic and constrained control learning + +**BOCO (Bellman Optimality via Constrained Optimization)** first parameterizes +the unknown optimal critic, + +$$ +J^*(x)\approx\widehat J_\theta(x), +$$ + +and, at each selected update state, expresses the local Bellman minimization +and equality as a finite-dimensional constrained optimization problem. This +makes the local condition amenable to numerical KKT or primal–dual +implementation: + +$$ +\begin{aligned} +F_k(u,\theta) +&:= +g(x_k,u)+ +\alpha\widehat J_\theta\!\left(x_{k+1}(u)\right),\\ +\delta_k(u,\theta) +&:= +F_k(u,\theta)-\widehat J_\theta(x_k), +\end{aligned} +$$ + +$$ +\begin{aligned} +\min_{u,\theta}\quad +&F_k(u,\theta)\\ +\mathrm{s.t.}\quad +&\delta_k(u,\theta)=0,\\ +&c_i(x_k,u)\le 0,\qquad i=1,\ldots,n_c. +\end{aligned} +$$ + +Here $x_{k+1}(u)$ denotes the successor associated with candidate input $u$. +Evaluating this quantity for arbitrary candidate inputs requires a declared +control-usable model, simulator, or other counterfactual mechanism; one +realized transition supplies the successor only for the input that was +actually applied. The set $\mathcal U(x_k)$ contains admissible inputs, +$n_c$ is the number of explicit inequality constraints, and $c_i$ is the +$i$th such constraint. It may represent an actuator limit, a state-dependent +feasibility condition, or a safety-related operating bound. These constraints +can be added explicitly instead of being handled only by a penalty. + +A global critic still requires accumulated streaming constraints, a declared +collocation or update set, or another function-approximation condition; one +local equality does not identify the complete value function. + +The Lagrangian is + +$$ +\mathcal L_k(u,\theta,\lambda) += +F_k(u,\theta) ++\lambda_\delta\delta_k(u,\theta) ++\sum_{i=1}^{n_c}\lambda_i c_i(x_k,u), +\qquad +\lambda_i\ge 0. +$$ + +At online update $t$, the primal control, critic parameter, and multipliers +can be updated conceptually as + +$$ +\begin{aligned} +u_k^{\mathrm{BOCO}} +&\in +\arg\min_{u\in\mathcal U(x_k)} +\mathcal L_k(u,\theta_t,\lambda_t),\\ +\theta_{t+1} +&= +\theta_t-\eta_{\theta,t} +\nabla_\theta\mathcal L_k +\!\left(u_k^{\mathrm{BOCO}},\theta_t,\lambda_t\right),\\ +\lambda_{\delta,t+1} +&= +\lambda_{\delta,t} ++\eta_{\delta,t}\, +\delta_k\!\left(u_k^{\mathrm{BOCO}},\theta_t\right),\\ +\lambda_{i,t+1} +&= +\left[ +\lambda_{i,t} ++\eta_{\lambda_i,t}\, +c_i\!\left(x_k,u_k^{\mathrm{BOCO}}\right) +\right]_+ . +\end{aligned} +$$ + +Here $\lambda_t=(\lambda_{\delta,t},\lambda_{1,t},\ldots,\lambda_{n_c,t})$ +collects the current multipliers, the $\eta$ terms are positive update step +sizes, and $[\cdot]_+$ projects each inequality multiplier onto the +nonnegative orthant. The equality multiplier $\lambda_\delta$ is unrestricted +in sign. The physical state $x_k$ is embedded in $F_k$, $\delta_k$, $c_i$, and +therefore $\mathcal L_k$. + +Unlike TD-error reduction alone, BOCO keeps control minimization, critic +consistency, and explicit constraints together. In the current formulation, +$u_k^{\mathrm{BOCO}}$ is the primal actor decision and a separate +policy-network parameter $\phi$ is not required. A future amortized policy +network could be trained to reproduce this solution, but that is a distinct +extension. + +The current BOCO paper provides this formulation and constrained nonlinear +simulation evidence. Formal stability, safety, broader task-domain learning, +separate policy-network training, and real-system validation require +additional assumptions and evidence. + +## 6. Application domains + +- **Automatic tuning and calibration of control systems for electric drives + and vehicle or mobility systems**, using operational data to improve the + deployed critic and controller. +- **Controller learning in complex, unstructured real-world environments** + where a faithful simulator and exhaustive offline training data are + difficult to obtain. + +## References + +- H. Lee and K. Choi, “Constrained Optimization Formulation of Bellman + Optimality Equation for Online Reinforcement Learning,” TechRxiv preprint, + v1, 2025. + [Full text](https://www.techrxiv.org/doi/full/10.36227/techrxiv.175790827.72477252/v1) + +
+
+
\ No newline at end of file diff --git a/research/research_themes/02_online_multistep_lookahead.md b/research/research_themes/02_online_multistep_lookahead.md new file mode 100644 index 000000000..7d6b13d8c --- /dev/null +++ b/research/research_themes/02_online_multistep_lookahead.md @@ -0,0 +1,241 @@ +--- +title: Online Multistep Lookahead +layout: default +group: research +math: true +--- + +
+
+
+ +# Online Multistep Lookahead + +> **Core idea.** Online Multistep Lookahead holds an approximate critic fixed +> and uses it as the terminal value of a short online optimal-control problem. +> The resulting action refines the policy target, while a policy network +> progressively learns the state-to-solution map for fast deployment. + +
+ A fixed critic closes a finite lookahead problem, whose first optimized action is progressively represented by an online-learned policy. +
+ A fixed critic closes a finite lookahead problem, whose first optimized action is progressively represented by an online-learned policy +
+
+ +*Physical interaction is indexed by $k$, whereas policy-network updates are +indexed by $t$. Here $t(k)$ is the latest policy-network update available at +physical step $k$, and $k(t)$ is the latest physical sample available at +learner update $t$. The critic parameter $\bar\theta$ is deliberately fixed +in this Theme. A state estimate $\widehat x_k$ and a control-usable prediction +model $f_{\mathrm{use}}$ are assumed available.* + +## 1. Problem definition + +[Online Learning-Based Optimal Control](/research/research_themes/00_online_learning_based_optimal_control) +connects an approximate critic to policy improvement through Bellman +minimization. Here the critic is already available but imperfect. Rather than +using only a one-step greedy action, solve an $H$-step problem whose terminal +cost is the fixed critic: + +$$ +\begin{aligned} +\widehat U_k^{(H)} +\in\arg\min_{U_k}\quad& +\sum_{j=0}^{H-1} +\alpha^j g(\widehat x_{j|k},u_{j|k}) ++ +\alpha^H\widehat J_{\bar\theta}(\widehat x_{H|k})\\ +\mathrm{s.t.}\quad& +\widehat x_{0|k}=\widehat x_k,\\ +& +\widehat x_{j+1|k} += +f_{\mathrm{use}}(\widehat x_{j|k},u_{j|k}), +\qquad j=0,\ldots,H-1,\\ +& +u_{j|k}\in\mathcal U(\widehat x_{j|k}), +\qquad j=0,\ldots,H-1,\\ +& +\widehat x_{j|k}\in\mathcal X, +\qquad j=0,\ldots,H, +\end{aligned} +$$ + +$$ +\widehat u_k^{(H)} += +\left[\widehat U_k^{(H)}\right]_0 . +$$ + +$H$ is a positive lookahead horizon, $0<\alpha<1$ is the discount factor, and +$U_k=(u_{0|k},\ldots,u_{H-1|k})$ is the candidate input sequence. +$f_{\mathrm{use}}$ is the declared prediction model, +$\mathcal U(x)$ and $\mathcal X$ are the admissible input and state sets, and +$\widehat x_{j|k}$ is the state predicted $j$ steps from the current estimate. +$\widehat u_k^{(H)}$ selects the first optimized action. For $H=1$, the +formulation reduces to one-step critic-based policy improvement. + +The online-learned policy represents the solution map: + +$$ +\widehat\mu_{\phi_t}(\widehat x_k) +\approx +\widehat u_k^{(H)}, +\qquad +u_k^{\mathrm{NN}} += +\widehat\mu_{\phi_{t(k)}}(\widehat x_k). +$$ + +The critic is not updated in this Theme. The learned object is the policy +induced by the fixed critic and multistep optimization. The model can be known +or a declared snapshot of a learned model, but it is held fixed while each +lookahead problem is solved. Changes in that snapshot, preview convention, or +constraint set between solves move the teacher solution map and must be +represented in the training data or policy input. + +## 2. Why this problem matters + +An approximate critic summarizes the infinite-horizon tail, but local critic +error can lead to a poor one-step greedy action. Multistep lookahead evaluates +the current action through $H$ explicit stages before applying the critic. +Under suitable discounted-DP and rollout assumptions, this can improve the +policy even while the critic remains fixed, and a longer horizon can reduce +the influence of terminal-critic error. Improvement is possible rather than +automatic: model error, incomplete optimization, and policy approximation can +offset the benefit. + +Repeatedly solving the $H$-step problem may still be too expensive for an +embedded controller. Learning its state-to-solution map amortizes the +computation across operating points: lookahead supplies improved action +targets during learning, whereas the converged policy network can return an +action immediately without a full optimization at every control step. + +## 3. Key challenges + +- **Horizon and model trade-off:** increasing $H$ can reduce dependence on the + terminal critic but raises computation and accumulated model error. +- **Constrained real-time optimization:** nonlinear dynamics and constraints + create feasibility, local-solution, warm-start, and worst-case latency + issues. +- **Solution-policy learning:** targets are concentrated on visited states, + and a small action-fitting error does not by itself imply small closed-loop + performance loss. +- **Safe deployment:** approximating a feasible optimizer does not + automatically preserve feasibility or stability; a correction solve, + projection, safety filter, or fallback may remain necessary. + +## 4. A conventional approach — repeated online lookahead + +A conventional realization solves the problem in Section 1 at every physical +control step and applies only the first action: + +$$ +u_k=\widehat u_k^{(H)}, +\qquad +k=0,1,\ldots . +$$ + +This receding-horizon implementation uses the current state, model, critic, +and constraints directly. Its main limitation is repeated numerical +optimization within the control deadline. + +This mechanism is MPC-like, but its defining role here is more specific: the +fixed approximate infinite-horizon critic closes the finite lookahead problem. +Learning a conventional finite-horizon MPC solution without that terminal +critic remains a useful amortized-MPC baseline, not this Theme's defining +mechanism. + +## 5. Research direction — online learning of the lookahead solution + +Let selected online lookahead targets be + +$$ +\mathcal S_t^{\mathrm{LA}} +\subseteq +\left\{ +\left(\widehat x_i,\widehat u_i^{(H)}\right) +\right\}_{i=0}^{k(t)} . +$$ + +A conceptual policy-learning problem is + +$$ +\phi_{t+1} +\approx +\arg\min_{\phi} +\sum_{(x,u^{(H)})\in\mathcal S_t^{\mathrm{LA}}} +\left\| +\widehat\mu_{\phi}(x)-u^{(H)} +\right\|_{W_u}^{2}. +$$ + +Here $W_u\succeq 0$ is the declared action-fitting weight matrix. + +Equivalently, the research loop can be summarized as + +$$ +\widehat J_{\bar\theta}\ \text{fixed}, +\qquad +\phi_t +\xrightarrow[\mathcal S_t^{\mathrm{LA}}] +{\text{online solution learning}} +\phi_{t+1}, +\qquad +u_k^{\mathrm{NN}} +=\widehat\mu_{\phi_{t(k)}}(\widehat x_k). +$$ + +The argmin states the learning objective; a real-time implementation may take +only one or a few incremental updates. Training signals may come from a +completed lookahead solve or a corrected warm start. If necessary, deployment +can retain a real feasibility mechanism, + +$$ +u_k += +\mathcal C_{\mathrm{feas}} +\left( +\widehat\mu_{\phi_{t(k)}}(\widehat x_k), +\widehat x_k +\right), +$$ + +where $\mathcal C_{\mathrm{feas}}$ must denote an implemented projection, +filter, or correction procedure rather than an assumed guarantee. + +Evidence should compare the network action and objective with the online +optimizer, constraint violations before and after correction, closed-loop +cost, worst-case latency, model and critic perturbations, and performance on +states not represented in the update stream. This page states the research +direction and evaluation criteria; it does not claim a validated MIC Lab +result. + +## 6. Application domains + +- **Electric-drive current or torque control:** combine a learned terminal + critic with short online lookahead to make an approximate infinite-horizon + MPC realization practical. +- **Coordination of multiple connected and automated vehicles and other + complex decision problems:** use multistep optimization to generate a + high-quality policy target, amortize that target into a fast policy network, + or use the lookahead action as an online policy-improvement target within an + RL framework. +- **Vehicle energy and thermal management:** represent long-horizon + consequences with the terminal critic while multistep lookahead resolves + short-horizon dynamics, constraints, and current operating information. +- **Embedded motion control and MPC-like systems** where a reliable optimizer + is available during learning but too costly for permanent execution. + +## References + +- D. P. Bertsekas, *Reinforcement Learning and Optimal Control*, Athena + Scientific, 2019. +- D. P. Bertsekas, *A Course in Reinforcement Learning*, 2nd ed., Athena + Scientific, 2025. + + +
+
+
\ No newline at end of file diff --git a/research/research_themes/03_semantic_critic_learning.md b/research/research_themes/03_semantic_critic_learning.md new file mode 100644 index 000000000..b6ada5137 --- /dev/null +++ b/research/research_themes/03_semantic_critic_learning.md @@ -0,0 +1,412 @@ +--- +title: Semantic Critic Learning +layout: default +group: research +math: true +--- + +
+
+
+ +# Semantic Critic Learning + +> **Core idea.** Semantic Critic Learning uses a VLM or LLM to convert +> observations into semantic annotations that supplement Bellman or TD +> critic learning. The updated critic then guides policy improvement; the +> semantic model is neither the low-level controller nor assumed to be an +> oracle. If semantic context is needed at action time or changes the desired +> return, the decision state or cost must be redefined explicitly. + +
+ Multimodal observations produce semantic annotations that provide auxiliary critic supervision alongside Bellman data, and the updated critic guides policy improvement. +
+ Multimodal observations produce semantic annotations + that provide auxiliary critic supervision alongside Bellman data, and the updated critic guides policy improvement +
+
+ +*Physical interaction is indexed by $k$, whereas critic and policy updates are +indexed by $t$. Semantic annotations may be generated offline, +asynchronously, or during operation; their timing and delay must be declared. +The figure shows the defining training-only route. Semantics required at +action time must instead enter the deployed information state.* + +## 1. Problem definition + +[Online Learning-Based Optimal Control](/research/research_themes/00_online_learning_based_optimal_control) +defines a policy through a value or Q-function. Semantic Critic Learning +addresses cases in which multimodal or textual observations can provide +evaluation-relevant supervision not readily expressed by the standard +numerical transition and stage-cost record. + +Let the controller receive a state estimate $\widehat x_k$, while $o_k$ +contains observations such as images, scene relations, task descriptions, or +operator-provided rules. A VLM is suitable when visual and language context +must be interpreted together; an LLM can be used when the relevant context is +represented textually. Denote the selected high-level semantic interpreter by +$G_{\mathrm{sem}}$. It produces + +$$ +z_k^{\mathrm{sem}} += +G_{\mathrm{sem}}(o_k,\chi_k), +$$ + +where $\chi_k$ is optional task, prompt, or rule context. +$z_k^{\mathrm{sem}}$ is one semantic annotation, such as a label, score, +preference, explanation, or feature; it is not itself a control input. Before +entering a critic loss, a raw annotation must be mapped to a declared target, +comparison, constraint, or feature. In the notation below, +$G_{\mathrm{sem}}$ includes any required mapping and +$z_k^{\mathrm{sem}}$ denotes the resulting critic-usable annotation. + +The defining case uses $z_k^{\mathrm{sem}}$ only as training supervision. If +semantic context is needed to distinguish deployed values or actions, it must +enter the controller-facing information state: + +$$ +\xi_k=(\widehat x_k,z_k^{\mathrm{sem}}),\qquad +\widehat Q_\theta(\xi_k,u),\qquad +\widehat\mu_\phi(\xi_k). +$$ + +Here $z_k^{\mathrm{sem}}$ is a declared control-usable encoding available at +decision time, not unrestricted text. Training-only semantic targets must be +representable from the deployed information state; otherwise identical +deployed inputs can receive conflicting targets. Semantics that changes the +desired return requires an explicitly redefined stage cost and Bellman +object. It also does not by itself make a partially observed process Markov: +the controller still needs a sufficient observation history, belief state, or +learned information state. + +Define a physical transition and the data selected at learning update $t$ as + +$$ +d_k += +(\widehat x_k,u_k,g_k,\widehat x_{k+1}), +\qquad +g_k +:= +g(\widehat x_k,u_k), +\qquad +\mathcal S_t += +\{d_k:k\in\mathcal I_t\}, +$$ + +where $\mathcal I_t$ is the set of physical transition indices selected for +update $t$. For an available annotation, let a predeclared rule decide whether +it is admitted: + +$$ +a_k^{\mathrm{acc}} +:= +A_{\mathrm{sem}} +\!\left( +d_k,z_k^{\mathrm{sem}};\mathcal R_{\mathrm{acc}} +\right) +\in\{0,1\}. +$$ + +Set $a_k^{\mathrm{acc}}=0$ when no annotation is available, and define + +$$ +\mathcal I_t^{\mathrm{sem}} += +\left\{ +k\in\mathcal I_t: +a_k^{\mathrm{acc}}=1 +\right\}. +$$ + +Here $\mathcal R_{\mathrm{acc}}$ is the declared acceptance rule. It may check +schema and provenance, temporal assignment, consistency with measurements and +hard rules, and a calibrated confidence or agreement test. Raw VLM or LLM +self-confidence is not assumed to be calibrated. The gate +$a_k^{\mathrm{acc}}$ is determined before critic training; it is not optimized +with $\theta$. + +The accepted paired dataset is + +$$ +\mathcal Z_t += +\left\{ +(d_k,z_k^{\mathrm{sem}}) +: +k\in\mathcal I_t^{\mathrm{sem}} +\right\}. +$$ + +Rejected or unavailable annotations do not remove $d_k$ from +$\mathcal S_t$; they only omit semantic supervision for that transition. For +compactness, the notation uses a common update batch. An asynchronous +implementation may reselect a retained transition when its annotation +arrives. + +The learning path is represented conceptually as + +$$ +\begin{aligned} +\theta_{t+1} +&\leftarrow +\operatorname{TrainCritic} +\!\left( +\theta_t;\mathcal S_t,\mathcal Z_t +\right),\\ +\phi_{t+1} +&\leftarrow +\operatorname{ImprovePolicy} +\!\left( +\phi_t;\widehat Q_{\theta_{t+1}} +\right), +\end{aligned} +$$ + +where $\theta$ and $\phi$ are the critic and policy parameters. These +operators are conceptual rather than a commitment to one algorithm. +$\mathcal S_t$ supplies stage-cost and successor-state evidence for ordinary +Bellman or TD learning; $\mathcal Z_t$ supplies auxiliary semantic guidance +for the accepted subset. + +The equations use a Q-factor because a sampled transition provides a direct +model-free critic target. A state-value critic $\widehat J_\theta$ can be used +instead when its successor-state and policy dependence are declared, as in +the overview figure. + +## 2. Why this problem matters + +A scalar stage cost provides one scalar evaluation channel, while images, +relations, or rules may support additional labels and comparisons: whether +another agent is yielding, whether a maneuver conflicts with a contextual +rule, or whether two outcomes should be evaluated differently. + +A VLM or LLM may provide such high-level context as an auxiliary learning +signal without being placed directly in the fast control loop. The research +question is whether Bellman-compatible semantic supervision can improve +task-level evaluation by the critic and, through that critic, improve the +policy. Because the semantic model is fallible and distribution-dependent, +richer context is not by itself evidence of safety, optimality, or +generalization. + +## 3. Key challenges + +- **Physical grounding and semantic reliability:** a semantic annotation can + conflict with measurements, dynamics, action feasibility, or operating + constraints. Designing and calibrating the acceptance rule, rejection + behavior, and fallback path is therefore part of the method. +- **Closed-loop timing and causal attribution:** semantic annotations may be + delayed, asynchronous, or stale. They must be assigned to the state–action + transitions that caused the evaluated outcome without future-data leakage. +- **Control-objective and safety consistency:** semantic supervision may + change which decisions the critic prefers. Its role in the objective or + information state must be declared, and closed-loop evidence must separate + its effect from extra data, reward tuning, or imitation while evaluating + performance and constraint violations. + +## 4. Established approaches — semantic rewards and policy guidance + +**Semantic reward specification or shaping.** An LLM can act as a proxy reward +function, while a VLM can score visual observations against a language-defined +goal ([Kwon et al., 2023](https://openreview.net/forum?id=10uNUgI5Kl); +[Rocamonde et al., 2024](https://proceedings.iclr.cc/paper_files/paper/2024/hash/7a7f6cc5dc2a84fb4edf0feb8e5cfd50-Abstract-Conference.html)). +The resulting semantic score modifies the cost used by an otherwise standard +RL algorithm: + +$$ +\widetilde g_k += +g_k+\lambda_{\mathrm{rew}}s_k^{\mathrm{sem}}, +\qquad +\lambda_{\mathrm{rew}}\ge0, +$$ + +so the critic and policy optimize the modified return. This is a reward-level +route, and an arbitrary semantic term can change the control objective. Here +$s_k^{\mathrm{sem}}$ is defined as a semantic **cost** or penalty, so a larger +value is less desirable; if a semantic model produces a reward-like score, +its sign must be converted before it is added to the cost. + +**Preference-derived reward learning.** Instead of requesting a raw scalar +score, human or VLM feedback can compare trajectory segments or observations. +A separate reward model is then learned from those preferences and supplied +to RL ([Christiano et al., 2017](https://papers.nips.cc/paper/7017-deep-reinforcement-learning); +[Wang et al., 2024](https://proceedings.mlr.press/v235/wang24bn.html)). +Semantics still reaches the critic through a scalar learned reward. + +**Direct policy guidance.** A preferred semantic action or demonstration can +supervise the policy without defining a critic-level learning signal: + +$$ +\mathcal L_{\mathrm{IL}}(\phi) += +- +\sum_{k\in\mathcal I_{\mathrm{IL}}} +\log +\pi_\phi +\!\left( +u_k^{\mathrm{sem}}\mid\widehat x_k +\right). +$$ + +$\mathcal I_{\mathrm{IL}}$ indexes semantic demonstrations, and +$u_k^{\mathrm{sem}}$ is the preferred or demonstrated action. +Language-conditioned imitation is an established example of this direct route +([Lynch and Sermanet, 2021](https://www.roboticsproceedings.org/rss17/p047.html)). +The stochastic-policy symbol $\pi_\phi$ is used here because the displayed +loss is a likelihood-based imitation objective; the default control-policy +symbol elsewhere in these pages remains $\mu_\phi$. + +## 5. Research direction — critic-level semantic supervision + +The proposed direction keeps physical transitions and stage costs in an +explicit Bellman loss while adding a separate semantic supervision term. The +equations below use the training-only case defined in Section 1, with deployed +critic and policy inputs $\widehat x$. + +For a nonempty update set $\mathcal S_t$ and a declared nonnegative TD-residual +loss $\ell_{\mathrm{TD}}$, write + +$$ +\begin{aligned} +\mathcal L_{\mathrm{Bellman}} +\!\left( +\theta;\mathcal S_t +\right) +&= +\frac{1}{|\mathcal S_t|} +\sum_{k\in\mathcal I_t} +\ell_{\mathrm{TD}} +\!\left( +\widehat Q_\theta(\widehat x_k,u_k) +- +y_{k,t}^{\mathrm{TD}} +\right),\\ +y_{k,t}^{\mathrm{TD}} +&= +g_k ++ +\alpha +\widehat Q_{\bar\theta_t} +\!\left( +\widehat x_{k+1}, +\widehat\mu_{\bar\phi_t}(\widehat x_{k+1}) +\right), +\end{aligned} +$$ + +where $(\bar\theta_t,\bar\phi_t)$ are fixed target parameters during update +$t$, and $0<\alpha<1$ is the discount factor. This is an approximate one-step +policy-evaluation target associated with the target policy +$\widehat\mu_{\bar\phi_t}$, not an optimal-Q target. Using data generated by +another behavior policy requires the usual coverage or off-policy assumptions. + +The accepted annotations provide a separate critic-level supervision term: + +$$ +\theta_{t+1} +\approx +\arg\min_\theta +\left\{ +\mathcal L_{\mathrm{Bellman}}(\theta;\mathcal S_t) ++ +\lambda_{\mathrm{sem},t} +\mathcal L_{\mathrm{sem}}(\theta;\mathcal Z_t) +\right\}, +\qquad +\lambda_{\mathrm{sem},t}\ge0, +$$ + +$$ +\mathcal L_{\mathrm{sem}}(\theta;\mathcal Z_t) += +\begin{cases} +\displaystyle +\frac{1}{|\mathcal Z_t|} +\sum_{(d_k,z_k^{\mathrm{sem}})\in\mathcal Z_t} +\ell_{\mathrm{sem}} +\!\left( +\widehat Q_\theta; +d_k,z_k^{\mathrm{sem}} +\right), +& |\mathcal Z_t|>0,\\ +0, +& \text{otherwise}, +\end{cases}. +$$ + +The acceptance rule has already removed rejected annotations, so this +overview weights the accepted pairs equally. A later implementation may use +separately calibrated confidence weights, but their source and calibration +must be declared; raw VLM or LLM self-confidence is not treated as a +probability of correctness. + +$\lambda_{\mathrm{sem},t}$ controls the relative contribution of semantic +supervision. To retain $\widehat Q_\theta$ as a critic for the original stage +cost $g$, $\ell_{\mathrm{sem}}$ is limited to accepted value comparisons or +terminal labels that are demonstrably compatible with that return. A +comparison may reference another transition or trajectory segment through +$z_k^{\mathrm{sem}}$. A new preference or risk specification instead requires +an explicitly redefined objective or constraint; conflicting labels that +reveal missing decision context require an augmented deployed information +state. + +The updated critic then guides policy improvement: + +$$ +\phi_{t+1} +\leftarrow +\operatorname{ImprovePolicy} +\!\left( +\phi_t;\widehat Q_{\theta_{t+1}} +\right). +$$ + +A deployable method also needs annotation provenance, delay and cache +handling, rejection, and fallback rules. + +Evaluation should compare a nonsemantic critic, numerical reward shaping, +semantic reward or reward-model baselines, direct semantic imitation, and the +critic-level formulation above. Relevant metrics include Bellman error, +critic calibration, control performance, semantic acceptance and rejection, +constraint violations, inference and update time, annotation sensitivity, +and contextual distribution shift. + +## 6. Application domains + +- **Autonomous-driving decision and motion planning**, where visual scene + relations, road conventions, and interaction context may provide auxiliary + critic supervision or, when available online, augment the decision state. +- **Human–robot and language-conditioned control**, where task semantics + may supervise value evaluation without unrestricted language output driving + actuators. +- **Mobility and energy-management systems with contextual operating rules**, + including exceptional modes difficult to encode in one handcrafted scalar + cost. + +## References + +- M. Kwon, S. M. Xie, K. Bullard, and D. Sadigh, “Reward Design with + Language Models,” *International Conference on Learning Representations + (ICLR)*, 2023. [Paper](https://openreview.net/forum?id=10uNUgI5Kl) +- J. Rocamonde, V. Montesinos, E. Nava, E. Perez, and D. Lindner, + “Vision-Language Models are Zero-Shot Reward Models for Reinforcement + Learning,” *ICLR*, 2024. + [Paper](https://proceedings.iclr.cc/paper_files/paper/2024/hash/7a7f6cc5dc2a84fb4edf0feb8e5cfd50-Abstract-Conference.html) +- P. F. Christiano et al., “Deep Reinforcement Learning from Human + Preferences,” *Advances in Neural Information Processing Systems*, vol. 30, + 2017. + [Paper](https://papers.nips.cc/paper/7017-deep-reinforcement-learning) +- Y. Wang et al., “RL-VLM-F: Reinforcement Learning from Vision Language + Foundation Model Feedback,” *Proceedings of the 41st International + Conference on Machine Learning*, PMLR 235, pp. 51484–51501, 2024. + [Paper](https://proceedings.mlr.press/v235/wang24bn.html) +- C. Lynch and P. Sermanet, “Language Conditioned Imitation Learning Over + Unstructured Data,” *Robotics: Science and Systems XVII*, 2021. + [Paper](https://www.roboticsproceedings.org/rss17/p047.html) + +
+
+
\ No newline at end of file diff --git a/research/research_themes/04_nonstationary_infinite_horizon_ocp.md b/research/research_themes/04_nonstationary_infinite_horizon_ocp.md new file mode 100644 index 000000000..b84496ad1 --- /dev/null +++ b/research/research_themes/04_nonstationary_infinite_horizon_ocp.md @@ -0,0 +1,364 @@ +--- +title: Nonstationary Infinite-Horizon OCP +layout: default +group: research +math: true +--- + +
+
+
+ +# Nonstationary Infinite-Horizon OCP + +> **Core idea.** A nonstationary infinite-horizon problem generally requires +> a time-indexed family of values and policies. This research seeks a +> sufficient context or event-based representation under which the dynamics, +> cost, constraints, and transition law become time homogeneous, allowing one +> context-conditioned critic and policy to represent long-run control. + +
+ A k-dependent Bellman problem is transformed, when a sufficient context exists, into a stationary Bellman problem on a reformulated state. +
+ A k-dependent Bellman problem is transformed, when a sufficient context exists, + into a stationary Bellman problem on a reformulated state +
+
+ +*The main formulation uses a discounted infinite-horizon objective with +$0<\alpha<1$. A stationary reformulation is a research condition to be +established, not an assumption that every nonstationary problem can be +converted.* + +## 1. Problem definition + +[Online Learning-Based Optimal Control](/research/research_themes/00_online_learning_based_optimal_control) +uses a stationary Bellman equation when the decision state, dynamics, stage +cost, constraints, and uncertainty law are time homogeneous. Many mobility +problems are instead written with explicit dependence on physical time, +position within an operating profile, mission phase, operating mode, or +exogenous context: + +$$ +x_{k+1}=f_k(x_k,u_k,w_k), +\qquad +u_k\in\mathcal U_k(x_k), +$$ + +with incurred stage cost $g_k(x_k,u_k)$. Here $w_k$ is the exogenous +disturbance and $\mathcal U_k(x_k)$ is the time-dependent admissible input +set. + +Let $\Pi_{k:\infty}$ be an admissible class of causal state-feedback policy +sequences, with $u_\ell=\mu_\ell(x_\ell)$. The explicit index on $\mu_\ell$ +allows known time dependence. When additional observed context is needed, it +is included in the decision state as in Section 5. The time-indexed optimal +value is + +$$ +J_k^\star(x) +:= +\inf_{\mu_{k:\infty}\in\Pi_{k:\infty}} +\mathbb E +\!\left[ +\left. +\sum_{j=0}^{\infty} +\alpha^j +g_{k+j}(x_{k+j},u_{k+j}) +\right| +x_k=x +\right], +$$ + +with Bellman recursion + +$$ +\begin{aligned} +J_k^\star(x) +&= +\min_{u\in\mathcal U_k(x)} +\left\{ +g_k(x,u) ++ +\alpha +\mathbb E +\!\left[ +J_{k+1}^\star(x^+) +\mid x,u,k +\right] +\right\},\\ +\mu_k^\star(x) +&\in +\arg\min_{u\in\mathcal U_k(x)} +\left\{ +g_k(x,u) ++ +\alpha +\mathbb E +\!\left[ +J_{k+1}^\star(x^+) +\mid x,u,k +\right] +\right\}. +\end{aligned} +$$ + +Here $x^+$ denotes the successor state. The dynamic-programming principle does +not require stationarity: a nonstationary finite-horizon recursion is anchored +by a terminal condition, while the infinite-horizon version is a time-indexed +Bellman family. Its validity requires a well-posed discounted limit—such as +bounded costs with an appropriate boundedness or transversality condition—and +the future sequence of models, costs, constraints, or its probabilistic law. + +Because $f_k$, $g_k$, $\mathcal U_k$, or the disturbance law can change with +$k$, a single $J^\star(x)$ and $\mu^\star(x)$ generally cannot represent the +exact optimum. An approximate or robust stationary policy may still be useful +under declared conditions. The exact recursion is nevertheless a coupled +sequence of Bellman equations, not one stationary fixed-point equation. If +the $k$-dependence is unknown or drifting rather than known in advance or +generated by an observed Markov context, the recursion is not directly +implementable. The research problem is then to determine whether its +control-relevant cause can be represented as part of a sufficient decision +state. + +## 2. Why this problem matters + +In predictive control, the reference, demand, constraints, dynamics, or +disturbance statistics may vary with operating phase and available preview. A +critic trained only on physical state $x$ may then assign the same value to +two situations that share $x$ but face different future opportunities, +constraints, or costs. This **context aliasing** can prevent a stationary +policy defined on $x$ alone from representing the exact optimum, although an +approximate or robust policy may remain useful under stated conditions. + +Finite-preview receding-horizon control can exploit the currently available +prediction, but it does not by itself represent the continuation beyond that +preview or recurring operation under future contexts. A valid stationary +reformulation could support a reusable long-run critic and terminal value +while retaining the context required for correct decisions. + +Some apparent nonstationarity is therefore a state-representation problem: if +a finite-dimensional Markov context captures all relevant $k$-dependence and +evolves under a time-homogeneous law, the augmented problem is stationary. If +no such context exists, or its transition law itself drifts, the original +time-indexed problem remains nonstationary. + +## 3. Key challenges + +- **Sufficient context and equivalence:** the selected reference-generator, + operating-phase, environment, mode, or task variables must preserve feasible + decisions, causality, and accumulated cost without requiring the complete + history. +- **Coordinate or event aggregation:** changing from time samples to phases, + segments, cycles, or decision events can lose within-event information; + aggregated transitions, costs, constraints, and discounting must preserve + the original problem. +- **Dimension and coverage:** a richer augmented state increases + value-learning complexity and requires data across the relevant contexts. +- **Changing context statistics:** a critic built for a fixed transition law + can become invalid under distribution shift; convergence on the + reformulated model does not prove that the reformulation remains correct. + +## 4. A conventional approach — finite-preview predictive control + +An exact treatment retains the $k$ index and computes the family +$\{J_k^\star,\mu_k^\star\}$. Over an unbounded horizon this is rarely reusable +or computationally practical. A common alternative solves an $H$-stage +problem using the model, cost, constraints, and forecasts available at the +current decision time: + +$$ +V_{k,H}^{\mathrm{pred}}(x) += +\inf_{\pi_{k:k+H-1}\in\Pi_{k,H}} +\mathbb E +\!\left[ +\left. +\sum_{j=0}^{H-1} +\alpha^j +g_{k+j}(x_{k+j},u_{k+j}) ++ +\alpha^H V_{f,k+H}(x_{k+H}) +\right| +x_k=x +\right], +$$ + +subject to the predicted $k$-dependent dynamics and constraints. Here +$\Pi_{k,H}$ is a declared class of causal finite-horizon policies; each policy +maps the information available at its prediction stage to $u_{k+j}$. +Deterministic MPC commonly reduces this to an input-sequence optimization. The +controller applies the first action and repeats the problem in receding-horizon +fashion as new state and preview information become available. + +If the task truly terminates after $H$ stages with a valid terminal condition, +the finite-horizon formulation can be exact. For continuing operation, its +quality depends on the forecast, prediction horizon, and terminal +approximation $V_{f,k+H}$; a fixed $V_f$ is a special case. Unless the terminal +term correctly represents the continuation, the controller can remain +short-sighted beyond the preview window. A context-free terminal target can +also be inappropriate when slow, stored, or resource states should be +preserved differently under different future contexts. Finite-preview control +is therefore a practical baseline, not by itself a solution to the original +nonstationary infinite-horizon problem. + +## 5. Research direction — context- or event-based stationary reformulation + +**Context-augmented stationary form.** First preserve the physical decision +index $k$. Suppose an observed or predictable context $c_k$ and augmented +state $z_k=(x_k,c_k)$ can be constructed such that + +$$ +\Pr +\!\left( +z_{k+1}\mid z_{0:k},u_{0:k} +\right) += +\overline P +\!\left( +z_{k+1}\mid z_k,u_k +\right), +$$ + +and the stage cost and feasible set admit time-independent representations: + +$$ +g_k(x,u) += +\overline g\!\left((x,c_k),u\right), +\qquad +\mathcal U_k(x) += +\overline{\mathcal U}\!\left((x,c_k)\right). +$$ + +The context may be an operating phase, mode, reference-generator state, +exogenous Markov state, or another finite sufficient statistic of the +available preview. Merely appending the absolute clock does not provide a +useful reusable representation unless the resulting context and its future +law are compact, time homogeneous, and learnable. + +The stationary Bellman problem on the augmented state is + +$$ +\begin{aligned} +\overline J^\star(z) +&= +\min_{u\in\overline{\mathcal U}(z)} +\left\{ +\overline g(z,u) ++ +\alpha +\mathbb E +\!\left[ +\overline J^\star(z^+)\mid z,u +\right] +\right\},\\ +\overline\mu^\star(z) +&\in +\arg\min_{u\in\overline{\mathcal U}(z)} +\left\{ +\overline g(z,u) ++ +\alpha +\mathbb E +\!\left[ +\overline J^\star(z^+)\mid z,u +\right] +\right\}. +\end{aligned} +$$ + +For an augmented decision state, write $J_k^\star(x;c)$ and +$\mu_k^\star(x;c)$ for the value and policy conditional on observed context +$c$. Under an exact sufficient reformulation, + +$$ +J_k^\star(x;c_k) += +\overline J^\star(x,c_k), +\qquad +\mu_k^\star(x;c_k) += +\overline\mu^\star(x,c_k). +$$ + +The semicolon makes the currently observed context explicit. When $c_k$ is a +deterministic function of $k$, this reduces to the earlier notation +$J_k^\star(x)$ and $\mu_k^\star(x)$. Thus one critic exists on $(x,c)$, not on +$x$ alone. Under the well-posedness conditions in Section 1, value iteration +or approximate DP can target this critic only if the augmented transition law +is stationary and the chosen context is Markov-sufficient over the declared +operating domain. + +**Optional event-coordinate form.** When decisions naturally occur at phases, +tasks, segments, or other events, the problem may instead be re-indexed at +event boundaries. Let $z$ be the augmented boundary state, $a$ an event-level +decision, $G$ the discounted cost accumulated until the next boundary, and +$D$ the corresponding duration in physical steps. The essential Bellman form +is + +$$ +\overline J_{\mathrm{ev}}^\star(z) += +\min_{a\in\mathcal A_{\mathrm{ev}}(z)} +\mathbb E +\!\left[ +G ++ +\alpha^D\overline J_{\mathrm{ev}}^\star(z^+) +\mid z,a +\right]. +$$ + +Here $\mathcal A_{\mathrm{ev}}(z)$ is the admissible event-decision set, and +$a$ may be one action or a declared causal policy within the event. The +conditional law of $(G,D,z^+)$ given $(z,a)$ must be time homogeneous, and the +aggregation must preserve within-event feasibility and path constraints. A +fixed discount per event is not equivalent when $D$ varies. Under exact +aggregation, the boundary value agrees with the original time-indexed value at +the same physical boundary. + +**Use as a terminal value.** Once an exact or approximate stationary long-run +critic is available, it can replace the generic terminal approximation +$V_{f,k+H}$ in the finite-preview predictive controller of Section 4. With an +exact reformulation, model, constraints, terminal critic, and optimization, +its first action is Bellman-consistent with the infinite-horizon optimum. An +approximate critic does not provide that guarantee automatically. This is the +connection to [Online Multistep Lookahead](/research/research_themes/02_online_multistep_lookahead): +the present Theme asks when a reusable stationary terminal-value problem +exists, whereas lookahead addresses online action improvement using a fixed +critic. + +## 6. Application domains + +- **Forecast-driven predictive control:** represent recurring reference, + demand, disturbance, constraint, or environment patterns through a + sufficient context so that a finite-preview controller can use a reusable + continuation value. +- **Periodic or mission-phase control:** augment the physical state with a + finite phase or operating-mode variable when it is Markov-sufficient. This + can replace separate time-indexed policies with one stationary critic and + policy across recurring drive cycles, duty cycles, or mission phases. +- **Event-based and networked decisions:** re-index control at task phases, + service events, road segments, charging nodes, or other decision boundaries. + The event model must preserve accumulated cost, constraints, and physical + duration. +- **xEV energy and thermal management:** the 2026 ASCC FCEV study is one + concrete application in which road links provide event coordinates and a + long-run network-conditioned value supplies continuation beyond the finite + prediction window. This example motivates, but does not define, the general + Research Theme. + +## References + +- [Kyunghwan Choi](https://kaist-mic-lab.github.io/members/khChoi/), + “[Traffic Network-Aware Energy Management for FCEVs: Integrating + Trip-Specific Control and Long-Run + Optimality](https://kaist-mic-lab.github.io/publications/2026-traffic-network/),” + *Asian Control Conference (ASCC)*, 2026. + [Full text](https://kaist-mic-lab.github.io/static/pub/2026-traffic-network.pdf) + +
+
+
\ No newline at end of file diff --git a/research/research_themes/05_continual_model_learning.md b/research/research_themes/05_continual_model_learning.md new file mode 100644 index 000000000..7eef64266 --- /dev/null +++ b/research/research_themes/05_continual_model_learning.md @@ -0,0 +1,276 @@ +--- +title: Continual Model Learning +layout: default +group: research +math: true +--- + +
+
+
+ +# Continual Model Learning + +> **Core idea.** Continual Model Learning incorporates newly observed system +> behavior into a learned dynamics model while retaining control-relevant +> behavior learned in earlier operating regions. When the physical state is +> not directly measured, the state trajectory and model can be learned +> jointly. + +
+ Continual model learning fits new data while retaining prior model behavior. +
+ Continual model learning fits new data while retaining prior model behavior +
+
+ +*Physical interaction is indexed by $k$, whereas learner updates are indexed +by $t$; the two clocks need not advance at the same rate. +$\widehat\psi_t$ denotes the model parameter available before update $t$, and +$\widehat\psi_{t+1}$ denotes the updated parameter.* + +## 1. Problem definition + +[Online Learning-Based Optimal Control](/research/research_themes/00_online_learning_based_optimal_control) +requires decision-state and successor information for Bellman-based control. +Continual Model Learning addresses the identifier/estimator-side problem: +learn a dynamics model from new real measurements without losing earlier +model behavior still needed by the estimator or controller. + +Consider + +$$ +x_{k+1}=f(x_k,u_k,w_k), +\qquad +y_k=h(x_k)+v_k, +$$ + +where $x_k$, $u_k$, and $y_k$ are the state, control input, and measurement; +$f$ and $h$ are the state-transition and measurement maps; and $w_k$ and +$v_k$ represent process and measurement uncertainty. Unless a different +stochastic predictor is declared, the learned one-step target is the +conditional mean + +$$ +\overline f(x,u) +:= +\mathbb E[x_{k+1}\mid x_k=x,u_k=u] +\approx +\widehat f_{\widehat\psi_t}(x,u). +$$ + +The primary continual-learning scope is coverage expansion for a fixed +underlying predictor $\overline f$: new operating regions are added while +selected prior function behavior is retained. Genuine physical drift is a +different case. If the same $(x,u)$ acquires a different successor law because +of temperature, aging, load, or another condition, that condition should be +represented explicitly, for example by $\overline f(x,u;\eta)$, or handled by +a declared tracking/forgetting rule. Blind retention can otherwise preserve +obsolete behavior. + +When both state and model are unknown, the basic identification principle is +measurement–dynamics consistency: + +$$ +\left(x_{0:k}^{\star},\psi^{\star}\right) +\in +\arg\min_{x_{0:k},\psi} +\left\{ +\mathcal L_{\mathrm{meas}}(x_{0:k};y_{0:k}) ++ +\mathcal L_{\mathrm{dyn}}(x_{0:k},\psi;u_{0:k-1}) +\right\}. +$$ + +For example, quadratic measurement and dynamics residuals are + +$$ +\begin{aligned} +\mathcal L_{\mathrm{meas}}(x_{0:k};y_{0:k}) +&= +\sum_{i=0}^{k} +\left\|y_i-h(x_i)\right\|_{W_y}^{2},\\ +\mathcal L_{\mathrm{dyn}}(x_{0:k},\psi;u_{0:k-1}) +&= +\sum_{i=0}^{k-1} +\left\| +x_{i+1}-\widehat f_{\psi}(x_i,u_i) +\right\|_{W_x}^{2}, +\end{aligned} +$$ + +where $W_y\succeq0$ and $W_x\succeq0$ weight the measurement and dynamics +residuals, respectively. An online implementation evaluates these losses over +a recent window rather than the entire trajectory. + +Real measurements form the stream + +$$ +d_k=(y_k,u_k,y_{k+1}), +\qquad +\mathcal D_{0:k(t)-1} +:= +(d_0,\ldots,d_{k(t)-1}), +\qquad +\mathcal S_t\subseteq\mathcal D_{0:k(t)-1}. +$$ + +Here $k(t)$ is the latest measurement/state index available at learner update +$t$. The set $\mathcal S_t$ may contain only the latest completed transition +or a short selected history. If the state is directly measured, replace $y_k$ +by $x_k$ and the problem reduces to model learning alone. Otherwise, the +latent state trajectory and model parameters must be estimated jointly, or a +separate estimator must supply $\widehat x_k$. + +## 2. Why this problem matters + +A model learned once offline may become incomplete as a mobility system visits +new loads, temperatures, road conditions, aging states, or operating regions. +Online system identification can reduce the residual on the newest trajectory, +but the updated function may no longer represent regions learned earlier. + +This distinction matters because a controller does not query the model only at +the latest sample. State estimators, MPC or lookahead optimization, and critic +targets may evaluate it over many control-relevant states and inputs. +Continual learning therefore seeks both **new-region fit** and +**prior-function retention**, rather than only a small current residual. + +## 3. Key challenges + +- **Limited and policy-dependent data:** real transitions arrive sequentially + along the deployed closed-loop trajectory and do not freely cover the full + state–input domain. +- **New learning versus prior-function retention (the stability–plasticity + trade-off):** the learner must incorporate newly observed behavior without + overwriting earlier behavior that remains relevant. +- **State–model coupling:** when $x_k$ is latent, state-estimation error and + model error can explain the same residual and must be separated under + suitable observability and excitation conditions. +- **Control-relevant online learning:** memory, update time, multistep error, + model acceptance, and the effect of each update on downstream control matter + in addition to one-step prediction loss. + +## 4. A conventional approach — recent-data residual minimization + +One conventional online-identification approach fits the model to the latest +transition or a short recent window. For notational simplicity, assume here +that the state is available; $x_j$ is replaced by $\widehat x_j$ when a +separate estimator is used: + +$$ +\begin{aligned} +\widehat\psi_{t+1} +&\approx +\arg\min_{\psi} +\mathcal L_{\mathrm{new}}(\psi;\mathcal S_t),\\ +\mathcal L_{\mathrm{new}}(\psi;\mathcal S_t) +&:= +\sum_{j\in\mathcal I_t} +\left\| +x_{j+1}-\widehat f_{\psi}(x_j,u_j) +\right\|_{W_x}^{2}. +\end{aligned} +$$ + +$\mathcal I_t$ indexes the recent transitions admitted to update $t$. This +update is useful for reducing current residuals, especially when the physical +dynamics truly change. Under limited excitation, however, it becomes a +trajectory-local online-identification update: it fits the currently visited +region but does not +explicitly preserve model behavior learned in earlier regions. + +Tracking a genuinely changing model and retaining knowledge over an expanding +task domain are therefore different objectives and should be evaluated +separately. + +## 5. Research direction — function-retaining continual learning + +A continual formulation augments current measurement–dynamics consistency +with explicit retention of the previously learned function. Let +$k_t:=k(t)$ be the newest physical sample available at learner update $t$, and +let $M$ be the estimation-window length. The set $\mathcal S_t$ supplies the +transitions selected to fit newly observed behavior, whereas +$\mathcal M_t=\{(\bar x_\ell,\bar u_\ell)\}_{\ell=1}^{m_t}\subseteq\mathcal X\times\mathcal U$ is a bounded set of state–input anchors +selected to retain prior model behavior. An anchor may be derived from a +transition in $\mathcal S_t$, but $\mathcal M_t$ can additionally contain +earlier stored points with pseudo-targets generated from the pre-update model: + +$$ +\begin{aligned} +\left( +\widehat x_{k_t-M:k_t}, +\widehat\psi_{t+1} +\right) +\approx +\arg\min_{x_{k_t-M:k_t},\psi} +\Big\{& +\mathcal L_{\mathrm{meas},t}(x;\mathcal S_t) ++ +\mathcal L_{\mathrm{dyn},t}(x,\psi;\mathcal S_t)\\ +&+ +\lambda_{\mathrm{retain}} +\mathcal L_{\mathrm{retain}} +\!\left( +\widehat f_{\psi}, +\widehat f_{\widehat\psi_t}; +\mathcal M_t +\right) +\Big\}. +\end{aligned} +$$ + +Here $\mathcal L_{\mathrm{meas},t}$ and +$\mathcal L_{\mathrm{dyn},t}$ are the residual losses defined in Section 1, +evaluated on $\mathcal S_t$. In a recent-window implementation, +$\mathcal S_t=\{d_j\mid j=k_t-M,\ldots,k_t-1\}$ supports the state and +measurement window ending at $k_t$. The retention weight satisfies +$\lambda_{\mathrm{retain}}\ge0$. + +For example, functional retention over a bounded anchor set +$\mathcal M_t$ can be written as + +$$ +\mathcal L_{\mathrm{retain}} += +\frac{1}{|\mathcal M_t|} +\sum_{(\bar x,\bar u)\in\mathcal M_t} +\left\| +\widehat f_{\psi}(\bar x,\bar u) +- +\widehat f_{\widehat\psi_t}(\bar x,\bar u) +\right\|_{W_r}^{2}. +$$ + +This expression assumes $|\mathcal M_t|>0$ and a positive-semidefinite weight +$W_r\succeq0$. If no anchors are retained, the retention term is omitted. + +The first two losses explain current measurements and dynamics; the last loss +retains selected input–output behavior of the model available before the +update. If $x_k$ is measured, remove the state-trajectory decision and +$\mathcal L_{\mathrm{meas},t}$. The retention weight and anchor set define a +design tradeoff; they do not guarantee global accuracy or eliminate +forgetting. + +Candidate mechanisms include a small representative replay set, local or RBF +activation, pseudo-data generated from the previous model, and functional +regularization. These are candidate directions, not achieved MIC Lab results. + +Future evidence should compare recent-only fitting, bounded-memory retention, +and full replay using new-region error, return-to-prior-region error, +multistep prediction, estimator error, downstream control performance, memory, +and update time. + +## 6. Application domains + +- **Continual calibration of vehicle, electric-drive, energy-management, and + thermal-system models** across operating regions and explicitly represented + load, temperature, or aging conditions. +- **Control-oriented model learning for state estimation, MPC, online + lookahead, and critic targets**, where the model must remain useful beyond + the latest trajectory. +- **Model learning in complex real environments** where complete offline data + and repeated full retraining are impractical. + +
+
+
\ No newline at end of file diff --git a/research/research_themes/06_constrained_pinn.md b/research/research_themes/06_constrained_pinn.md new file mode 100644 index 000000000..37dab3cad --- /dev/null +++ b/research/research_themes/06_constrained_pinn.md @@ -0,0 +1,360 @@ +--- +title: Constrained PINN +layout: default +group: research +math: true +--- + +
+
+
+ +# Constrained PINN + +> **Core idea.** Constrained PINN learns a complex physics-consistent field or +> model while keeping the governing-equation residual in the objective and +> representing boundary, initial, conservation, and admissibility +> requirements as explicit constraints. Primal–dual training updates the +> learned field or model together with its Lagrange multipliers instead of +> relying only on fixed penalty weights. + +
+ Physics and data define an explicit constrained learning problem whose primal and dual variables produce a physics-consistent field or model. +
+ Physics and data define an explicit constrained learning problem + whose primal and dual variables produce a physics-consistent field or model +
+
+ +*The formulation distinguishes sampled residual minimization from +continuous-domain constraint satisfaction. Writing an explicit constraint +does not by itself establish feasibility or convergence.* + +## 1. Problem definition + +This Theme concerns the model and state-information side of +[Online Learning-Based Optimal Control](/research/research_themes/00_online_learning_based_optimal_control), +but PINNs also apply well beyond control. It therefore begins from the general +problem of learning a physical field or model rather than from a Bellman +decision problem. + +Within the MIC Lab program, Constrained PINN is primarily a supporting +physics-learning capability: it can prepare or update a control-usable model +or state representation for an online estimator, predictor, critic target, or +controller. + +Let $q$ denote the physical field or model and let +$\widehat q_\psi(\zeta)$ be its primal neural approximation, parameterized by +$\psi$. The coordinate $\zeta$ may contain space, time, geometry, material +parameters, and operating conditions. For a governing operator $\mathcal N$, +define the interior physics residual + +$$ +r_\Omega(\zeta;\psi) += +\mathcal N +\!\left( +\widehat q_\psi +\right)(\zeta) +- +b(\zeta), +$$ + +where $b(\zeta)$ is the declared source or forcing term. Let +$\mathcal C_\Omega$ be the interior collocation set; the corresponding loss is + +$$ +\mathcal L_{\mathrm{physics}}(\psi) += +\frac{1}{|\mathcal C_\Omega|} +\sum_{\zeta\in\mathcal C_\Omega} +\left\| +r_\Omega(\zeta;\psi) +\right\|^2. +$$ + +For a forward problem, set +$\mathcal L_{\mathrm{obj}}=\mathcal L_{\mathrm{physics}}$. For an inverse +problem or sparse-data reconstruction, an observation term may be included: + +$$ +\mathcal L_{\mathrm{obj}}(\psi) += +\mathcal L_{\mathrm{physics}}(\psi) ++ +w_{\mathrm{obs}}\mathcal L_{\mathrm{obs}}(\psi), +\qquad +w_{\mathrm{obs}}\ge0. +$$ + +The observation term remains part of the objective rather than one of the +declared physical constraints. Here $\mathcal L_{\mathrm{obs}}$ denotes the +observation-data misfit. + +Boundary, initial, conservation, interface, and admissibility conditions are +represented as equality and inequality residuals: + +$$ +c_j(\zeta;\psi)=0 +\quad \forall\zeta\in\mathcal C_j^{=}, +\qquad +d_m(\zeta;\psi)\le0 +\quad \forall\zeta\in\mathcal C_m^{\le}. +$$ + +Here $j=1,\ldots,J$ and $m=1,\ldots,M$. Scalar residuals are shown for +clarity; components of vector-valued residuals are stacked in the same way. +The sets $\mathcal C_j^{=}$ and $\mathcal C_m^{\le}$ specify where sampled +equality and inequality constraints are enforced. + +For example, + +$$ +\begin{aligned} +c_{\mathrm{BC}}(\zeta;\psi) +&= +\mathcal B\!\left(\widehat q_\psi\right)(\zeta) +- +b_{\partial\Omega}(\zeta),\\ +c_{\mathrm{IC}}(s;\psi) +&= +\widehat q_\psi(s,0)-q_0(s). +\end{aligned} +$$ + +$\mathcal B$ is the boundary operator, $b_{\partial\Omega}$ is the prescribed +boundary data, and $q_0$ is the initial field over the spatial coordinate $s$. + +The defining optimization problem places the physics residual and any +observation loss in the objective, and the remaining requirements after an +explicit $\mathrm{s.t.}$ line: + +$$ +\begin{aligned} +\min_{\psi}\quad +&\mathcal L_{\mathrm{obj}}(\psi)\\ +\mathrm{s.t.}\quad +&c_j(\zeta;\psi)=0 +\quad \forall\zeta\in\mathcal C_j^{=},\\ +&d_m(\zeta;\psi)\le0 +\quad \forall\zeta\in\mathcal C_m^{\le}. +\end{aligned} +$$ + +This equation defines the target learning problem, not a training algorithm. +For the sampled problem, each pair $(j,\zeta)$ in an equality collocation set +has an unrestricted multiplier, and each pair $(m,\zeta)$ in an inequality +collocation set has a nonnegative multiplier. Stacking them gives the dual +vectors $\lambda$ and $\nu$, respectively. Section 5 first gives direct vector +updates and then an optional shared **multiplier-network** parameterization +that generates these entries from $\zeta$. + +## 2. Why this problem matters + +PINNs can learn complex fields or models by combining sparse observations +with governing physics. A conventional PINN usually places PDE residuals, +boundary conditions, initial conditions, data, and conservation terms in one +weighted sum. Large differences in scale and conditioning can cause training +to reduce one term while leaving another physically important condition +poorly satisfied. + +An explicit constrained formulation separates what should be minimized from +what should be satisfied. With appropriate residual scaling, convergence, and +constraint qualifications, the associated Lagrange multipliers can provide +local sensitivity information, and nonzero inequality multipliers can help +identify active constraints. Raw magnitudes are not directly comparable +across differently scaled residuals. This can make the learning problem more +transparent and reduce manual penalty-weight tuning, but it does not +automatically produce an exact physical solution or a convergent algorithm. + +## 3. Key challenges + +- **Feasibility and representation:** the network class, collocation set, and + coupled physical conditions may not admit a feasible solution. +- **Nonconvex primal–dual learning:** simultaneous primal-network and dual + updates can oscillate, diverge, or become poorly scaled. +- **Sampled versus continuous satisfaction:** small residuals at collocation + points do not prove physical consistency throughout the domain or under new + conditions. +- **Multiplier representation and computation:** independently optimized + pointwise multipliers scale with the collocation set; a shared multiplier + network adds approximation choices, regularization, gradient-computation + cost, and another coupled learning problem. + +## 4. Baselines — weighted penalties and hard parameterization + +The standard soft-penalty PINN minimizes + +$$ +\mathcal L_{\mathrm{pen}}(\psi) += +w_\Omega +\mathcal L_{\mathrm{physics}}(\psi) ++ +w_{\mathrm{BC}} +\mathcal L_{\mathrm{BC}}(\psi) ++ +w_{\mathrm{IC}} +\mathcal L_{\mathrm{IC}}(\psi) ++ +\sum_r +w_r\mathcal L_r(\psi), +\qquad +w_\Omega,w_{\mathrm{BC}},w_{\mathrm{IC}},w_r\ge0. +$$ + +$\mathcal L_{\mathrm{BC}}$ and $\mathcal L_{\mathrm{IC}}$ are the sampled +boundary- and initial-residual losses, while $r$ indexes any additional +residual penalties. + +Fixed weights are simple, and adaptive loss balancing can reduce some scale +mismatch. Nevertheless, a finite penalty permits constraint violation, +whereas a very large penalty can make optimization ill-conditioned. + +For a selected simple equality condition, a separate hard baseline is + +$$ +\widehat q_\psi(\zeta) += +q_{\mathrm{known}}(\zeta) ++ +A(\zeta)h_\psi(\zeta) +$$ + +This parameterization is chosen **before training**, rather than applied as a +correction afterward. Here $q_{\mathrm{known}}$ satisfies the prescribed +condition, $h_\psi$ is an unconstrained neural correction, and $A(\zeta)$ is +zero wherever that value condition must hold. For example, the initial +condition $q(s,0)=q_0(s)$ can be encoded as + +$$ +\widehat q_\psi(s,\tau) += +q_0(s)+\tau h_\psi(s,\tau). +$$ + +At physical time $\tau=0$, the neural correction vanishes for every $\psi$, +so the initial condition remains exactly satisfied throughout training. For a +derivative or operator constraint, $A=0$ alone is insufficient: the +parameterization must make the relevant operator applied to the correction +vanish. Hard encoding is therefore useful for selected simple equalities but +can be difficult to construct for complex geometries, coupled interfaces, +inequalities, or changing conditions. + +## 5. Research direction — explicit constraints and primal–dual learning + +At the sampled level, stack equality and inequality residuals over their +collocation sets as $c(\psi)$ and $d(\psi)$. The Lagrangian is + +$$ +\mathscr L(\psi,\lambda,\nu) += +\mathcal L_{\mathrm{obj}}(\psi) ++ +\lambda^\top c(\psi) ++ +\nu^\top d(\psi), +\qquad +\nu\ge0. +$$ + +At primal–dual training iteration $t$, distinct from the physical coordinate +$\tau$ used above, conceptual updates are + +$$ +\begin{aligned} +\psi_{t+1} +&= +\psi_t +- +\rho_{\psi,t} +\nabla_\psi +\mathscr L(\psi_t,\lambda_t,\nu_t),\\ +\lambda_{t+1} +&= +\lambda_t ++ +\rho_{\lambda,t}c(\psi_t),\\ +\nu_{t+1} +&= +\left[ +\nu_t ++ +\rho_{\nu,t}d(\psi_t) +\right]_+ . +\end{aligned} +$$ + +Here $\rho_{\psi,t},\rho_{\lambda,t},\rho_{\nu,t}>0$, and $[\cdot]_+$ denotes +componentwise projection onto the nonnegative orthant. Equality multipliers +$\lambda$ are unrestricted, so their update is signed. The projected +inequality update increases a multiplier for positive violation and can +decrease it when the corresponding constraint is strictly satisfied. The +primal step updates the learned field or model. +Augmented-Lagrangian terms, residual normalization, alternating updates, and +dual regularization are possible stabilization mechanisms rather than +guarantees. + +The updates above optimize one multiplier entry per sampled residual. Shared +**multiplier networks** can instead generate domain-indexed entries: + +$$ +\lambda_j(\zeta) += +\Lambda_{\omega_\lambda,j}(\zeta), +\qquad +\nu_m(\zeta) += +\operatorname{softplus} +\!\left( +R_{\omega_\nu,m}(\zeta) +\right) +\ge0. +$$ + +$\Lambda_{\omega_\lambda,j}$ is the equality-network output, and +$R_{\omega_\nu,m}$ is the raw inequality-network output, with +$\omega_\lambda$ and $\omega_\nu$ denoting their parameters. The equality +multiplier needs no sign transformation, while softplus enforces +$\nu_m\ge0$. At each collocation batch, these outputs supply $\lambda$ and +$\nu$ in the sampled Lagrangian; $\psi$ is updated by descent and the +multiplier-network parameters by dual ascent. This shares a parameterization +across points but restricts the dual class, so sampled feasibility still does +not establish continuous-domain feasibility. + +Future evidence should compare a baseline PINN, fixed and adaptive penalties, +hard encoding where available, directly optimized multiplier vectors with +standard or augmented-Lagrangian updates, and a multiplier-network variant +under matched data and computation budgets. +Metrics should include reference-solution error, physics residual, mean and +maximum constraint violation, conservation defect, training stability, +multiplier behavior, computation, and generalization to new boundary +conditions, parameters, and geometries. + +## 6. Application domains + +- **Structural, contact, and multiphysics systems**, where interface, + conservation, positivity, or material conditions must be distinguished + from the governing residual objective. +- **Sparse-observation inverse problems and state reconstruction**, where + physical constraints supplement incomplete measurements. +- **Synchronous-machine flux-linkage learning and estimation**, where + governing electrical dynamics and physically admissible inductance bounds + guide online learning of a nonlinear magnetic model. A related MIC Lab study + reports simulation validation on a 35-kW IPMSM drive + ([Jang, Ryu, and Choi, 2025](https://ieeexplore.ieee.org/document/11221587/)). +- **EV battery and thermal-management analysis**, including temperature-field + reconstruction under changing cooling conditions and sparse sensing. + +## References + +- S. Jang, M. Ryu, and K. Choi, “Physics-Informed Online Learning of Flux + Linkage Model for Synchronous Machines,” *IECON 2025 — 51st Annual + Conference of the IEEE Industrial Electronics Society*, pp. 1–6, 2025. + [IEEE Xplore](https://ieeexplore.ieee.org/document/11221587/) + · [Full text](https://kaist-mic-lab.github.io/static/pub/2025-physics-informed-IECON.pdf) + + +
+
+
\ No newline at end of file diff --git a/research/research_themes/07_neuro_adaptive_control.md b/research/research_themes/07_neuro_adaptive_control.md new file mode 100644 index 000000000..167ade56d --- /dev/null +++ b/research/research_themes/07_neuro_adaptive_control.md @@ -0,0 +1,292 @@ +--- +title: Neuro-Adaptive Control +layout: default +group: research +math: true +--- + +
+
+
+ +# Neuro-Adaptive Control + +> **Core idea.** Neuro-Adaptive Control directly approximates the unknown +> ideal feedback law for an uncertain plant with a neural policy and adapts +> that policy from +> closed-loop signals. The effect of uncertain dynamics can therefore be +> represented inside the controller without first producing a separately +> reusable dynamics model. + +
+ An uncertain ideal control law is represented by a neural policy whose parameters are adapted online under closed-loop and constraint conditions. +
+ An uncertain ideal control law is represented by a neural policy + whose parameters are adapted online under closed-loop and constraint conditions +
+
+ +*Physical interaction is indexed by $k$, whereas neural-policy updates are +indexed by $t$; the two clocks need not advance at the same rate. Here $t(k)$ +is the latest neural-policy update available at physical step $k$, and $k(t)$ +is the latest physical sample available at learner update $t$. $\phi_t$ +denotes the policy parameter available before update $t$, and $\phi_{t+1}$ +denotes the updated parameter.* + +## 1. Problem definition + +[Online Learning-Based Optimal Control](/research/research_themes/00_online_learning_based_optimal_control) +defines the desired policy through Bellman optimality when the decision +problem and required information are available. Neuro-Adaptive Control +addresses a related but distinct situation: system uncertainty prevents direct +implementation of the ideal feedback law required for a declared tracking, +regulation, or stabilization objective. + +Consider + +$$ +x_{k+1}=f(x_k,u_k,w_k), +\qquad +y_k=h(x_k)+v_k, +$$ + +where $x_k$, $u_k$, and $y_k$ are the state, control input, and measurement; +$f$ and $h$ are the state-transition and measurement maps; $w_k$ represents +process uncertainty or a disturbance; and $v_k$ is measurement noise. When +$x_k$ is not measured, $\widehat x_k$ denotes its estimate. The dynamics may +contain unknown nonlinearities or uncertain parameters. The unavailable target +is first written as the ideal feedback law + +$$ +u_k^{\mathrm{ideal}}=\mu_{\mathrm{ideal}}(x_k). +$$ + +Neuro-Adaptive Control then **parameterizes and approximates** this law with a +selected neural function class: + +$$ +\mu_{\mathrm{ideal}}(x) += +\mu_{\phi^\star}(x) ++ +\varepsilon_{\mathrm{app}}(x). +$$ + +The implemented policy and its online adaptation are + +$$ +\begin{aligned} +u_k +&= +\mu_{\phi_{t(k)}}(\widehat x_k),\\ +\phi_t +&\xrightarrow{\text{online adaptation}} +\phi_{t+1}. +\end{aligned} +$$ + +Here $\phi^\star$ is an ideal parameter in the chosen neural class, +$\varepsilon_{\mathrm{app}}$ is its approximation error, and $\phi_t$ is the +deployed parameter at learner update $t$. The law +$\mu_{\mathrm{ideal}}$ is Bellman-optimal only when that connection is +explicitly established. + +If the controller uses a recurrent or other sequence-processing network, +replace $\widehat x_k$ by a declared history-dependent information vector +$\xi_k$. The network may approximate the complete feedback law or a residual +correction +$u_k=u_{\mathrm{nom},k}+\mu_{\phi_{t(k)}}(\xi_k)$. + +Direct approximation of $\mu_{\mathrm{ideal}}$ folds the control effect of +uncertain dynamics into the policy. It does not automatically identify a +reusable model $\widehat f$. + +Within the overview taxonomy, this Theme is placed in the joint +control–identification lane because closed-loop information about uncertainty +directly changes the controller. This placement does not mean that a reusable +dynamics model is identified. + +## 2. Why this problem matters + +The exact stabilizing or performance-improving feedback often depends on +nonlinearities, disturbances, and parameters that are not available to a +nominal controller. A neural function class can represent richer compensation +than a fixed low-dimensional adaptive law and can update during operation. + +Direct-law approximation may also avoid the full sequence of identifying a +physical model and then redesigning the controller. Its primary benefit, +however, is closed-loop uncertainty compensation. Without explicit mechanisms +for coverage, memory, or retention, it remains closer to adaptation along the +visited trajectory than to task-wide policy learning. + +## 3. Key challenges + +- **Ideal-law approximation and information:** the approximation domain, + $\varepsilon_{\mathrm{app}}$, available state or history information, and + estimator error must be declared. +- **Closed-loop adaptation versus learning:** changing $\phi_t$ changes both + the applied input and future data, while a trajectory-local update need not + recover $\phi^\star$ or produce a reusable task-wide policy. +- **Stability and constraints:** bounded tracking error, bounded neural + weights, admissible control inputs, and state safety are different claims. +- **Excitation and online implementation:** weak excitation can prevent + parameter convergence, and sensitivity calculation, memory, and update time + must remain compatible with the physical control rate. + +## 4. A conventional approach — Lyapunov-derived neuro-adaptation + +A conventional neuro-adaptive controller derives its parameter update from +declared closed-loop error dynamics and a Lyapunov argument. Schematically, + +$$ +\begin{aligned} +u_k +&= +\mu_{\phi_{t(k)}}(\widehat x_k),\\ +\phi_{t+1} +&= +\Pi_{\Phi_{\mathrm{adm}}} +\left[ +\phi_t+ +\Delta\phi_t +\!\left(e_{k(t)},\widehat x_{k(t)}\right) +\right], +\end{aligned} +$$ + +where $e_k$ is the selected tracking or regulation error and +$\Delta\phi_t$ is the update derived for the specific plant and controller. +The projection $\Pi_{\Phi_{\mathrm{adm}}}$ is included only when the +implemented method uses an admissible parameter set. + +Under its stated approximation, gain, excitation, and initialization +conditions, such an update may establish boundedness of selected closed-loop +signals. It need not recover $\phi^\star$, learn a globally reusable policy, +or enforce actuator and state constraints. + +## 5. A constrained direction — optimization-based neuro-adaptation + +A constrained formulation makes the online performance objective and +admissibility conditions explicit. Let $k_t:=k(t)$ denote the newest physical +sample available at learner update $t$: + +$$ +\begin{aligned} +\min_{\phi}\quad +&\mathcal L_{\mathrm{cl},t}(\phi)\\ +\text{s.t.}\quad +&c_{\phi,j}(\phi)\le 0, +\qquad j=1,\ldots,n_\phi,\\ +&c_{u,i} +\!\left( +\mu_\phi(\widehat x_{k_t}), +\widehat x_{k_t} +\right) +\le 0, +\qquad i=1,\ldots,n_u . +\end{aligned} +$$ + +$\mathcal L_{\mathrm{cl},t}$ is a closed-loop error or performance objective; +it is not a supervised loss against the unavailable +$\mu_{\mathrm{ideal}}$. Its +dependence on a candidate $\phi$ must be supplied by a declared sliding-window +or rollout prediction, forward sensitivity, or differentiable instantaneous +surrogate. Previously observed errors alone are constant with respect to a new +candidate $\phi$ and therefore do not define the displayed optimization. +Example constraints include + +$$ +c_{\phi,\ell}(\phi) += +\|\phi_\ell\|_2^2-\bar\phi_\ell^2 +\le 0, +\qquad +c_u(\phi;\widehat x_{k_t}) += +\|\mu_\phi(\widehat x_{k_t})\|_2^2-\bar u^2 +\le 0. +$$ + +$n_\phi$ and $n_u$ count the declared parameter and input constraints, +$\phi_\ell$ is a selected parameter group, and +$\bar\phi_\ell,\bar u>0$ are prescribed bounds. + +The input constraint $c_u$ is evaluated on the policy output at the current +operating condition and can directly encode input amplitude or norm. An input +rate constraint additionally requires the previous applied input, and an +energy constraint requires an accumulated or horizon state. If the implemented +actuator includes saturation or a separate safety filter, the constraint and +analysis must refer to the input actually applied to the system, not only the +raw neural-policy output. + +Let $c_j$ enumerate the declared parameter constraints $c_{\phi,j}$ and +applied-input constraints $c_{u,i}$ after any required state augmentation. A +conceptual primal–dual realization takes online steps toward the constrained +target: + +$$ +\begin{aligned} +\mathscr L_t(\phi,\lambda) +&= +\mathcal L_{\mathrm{cl},t}(\phi) ++ +\sum_j\lambda_jc_j(\phi;\widehat x_{k_t}),\\ +\phi_{t+1} +&= +\phi_t-\eta_{\phi,t} +\nabla_\phi\mathscr L_t(\phi_t,\lambda_t),\\ +\lambda_{j,t+1} +&= +\left[ +\lambda_{j,t} ++ +\eta_{\lambda_j,t} +c_j(\phi_t;\widehat x_{k_t}) +\right]_+ . +\end{aligned} +$$ + +Here $\lambda_{j,0}\ge0$, +$\eta_{\phi,t},\eta_{\lambda_j,t}>0$, and $[\cdot]_+$ denotes projection onto +the nonnegative reals. Transient primal–dual iterates need not be feasible; +when every applied action must satisfy its constraints, an implemented +projection, safety filter, saturation, or fallback and a corresponding +closed-loop analysis remain necessary. + +The **Constrained Optimization-Based Neuro-Adaptive Control +(CoNAC)** [preprint](https://www.techrxiv.org/doi/full/10.36227/techrxiv.172954216.68720680/v1) +is a concrete related instance for uncertain Euler–Lagrange systems. It uses a +DNN to approximate an ideal stabilizing law +and formulates online weight adaptation with weight-norm and input +constraints. The preprint reports bounded tracking errors and DNN weights +under its stated assumptions, together with comparative numerical +simulations. The scope of these guarantees is determined by the plant, +network, feasibility, and adaptation assumptions stated in that work; +extensions to other systems or architectures require corresponding analysis. + +Evidence should distinguish tracking-error boundedness, weight boundedness, +constraint satisfaction, parameter or KKT convergence, control effort, +computation time, and performance outside the theorem conditions. + +## 6. Application domains + +- **Uncertain nonlinear vehicle and mobility subsystems** requiring online + compensation under actuator limits. +- **Robotic and Euler–Lagrange systems**, including manipulators and motion + platforms with tracking objectives. +- **Electric drives and mechatronic systems** subject to torque, current, + voltage, or other input constraints. +- **Full-law or residual neural control** when a reusable physical model is + unavailable or impractical to maintain online. + +## References + +- M. Ryu, D. Hong, and K. Choi, “Constrained Optimization-Based + Neuro-Adaptive Control (CoNAC) for Uncertain Euler–Lagrange Systems Under + Weight and Input Constraints,” TechRxiv preprint, v1, 2024. + [Full text](https://www.techrxiv.org/doi/full/10.36227/techrxiv.172954216.68720680/v1) + +
+
+
\ No newline at end of file diff --git a/research/research_themes/08_structured_critic_adaptation.md b/research/research_themes/08_structured_critic_adaptation.md new file mode 100644 index 000000000..da684972b --- /dev/null +++ b/research/research_themes/08_structured_critic_adaptation.md @@ -0,0 +1,317 @@ +--- +title: Structured Critic Adaptation +layout: default +group: research +math: true +--- + +
+
+
+ +# Structured Critic Adaptation + +> **Core idea.** Structured Critic Adaptation estimates the current +> environment parameter online and uses it to reconfigure a stored base +> critic. The reconfigured critic then supports policy improvement, so an +> environmental change need not trigger unconstrained relearning of the full +> critic. + +
+ Selected transitions identify the environment, a learned structure reconfigures a stored critic, and the resulting critic improves the policy. +
+ Selected transitions identify the environment, + a learned structure reconfigures a stored critic, and the resulting critic improves the policy. +
+
+ +*Physical interaction is indexed by $k$, whereas identification and policy +updates are indexed by $t$. The environment parameter $\eta_k$ may vary with +physical conditions; $\widehat\eta_t$ is the estimate available at learner +update $t$, and $k(t)$ is the latest physical sample associated with that +update. Identification from a window assumes that $\eta_k$ is constant or +slowly varying over that window, or explicitly models its evolution.* + +## 1. Problem definition + +[Online Learning-Based Optimal Control](/research/research_themes/00_online_learning_based_optimal_control) +requires a critic appropriate to the current decision problem. Consider a +family of environments + +$$ +x_{k+1} += +f(x_k,u_k;\eta_k)+w_k, +\qquad +y_k=h(x_k)+v_k, +$$ + +where $x_k$, $u_k$, and $y_k$ are the state, control input, and measurement; +$f$ and $h$ are the state-transition and measurement maps; and $w_k$ and +$v_k$ represent process and measurement uncertainty. The compact parameter +$\eta_k$ represents control-relevant changes in dynamics and, when declared, +in the stage cost $g$ or constraints. For notational simplicity, assume the +state is available; otherwise, an estimator supplies $\widehat x_k$. + +Unless the evolution of $\eta_k$ is included in a Markov decision state, a +condition-specific critic $J_\eta$ below refers to the problem with $\eta$ +fixed over its value horizon. Material variation over that horizon instead +requires a nonstationary formulation such as +[Nonstationary Infinite-Horizon OCP](/research/research_themes/04_nonstationary_infinite_horizon_ocp). + +Let the transition record and the set selected at learner update $t$ be + +$$ +d_k += +(x_k,u_k,g_k,x_{k+1}), +\qquad +g_k +:= +g(x_k,u_k;\eta_k), +\qquad +\mathcal S_t += +\{d_k:k\in\mathcal I_t\}, +$$ + +where $\mathcal I_t$ is the selected set of physical transition indices. +These recent transitions support online identification: + +$$ +\widehat\eta_t +\in +\arg\min_{\eta} +\mathcal L_{\mathrm{id}}(\eta;\mathcal S_t). +$$ + +The loss $\mathcal L_{\mathrm{id}}$ must declare which effects identify +$\eta$: transition residuals when $\eta$ changes the dynamics, observed costs +when it changes the objective, or constraint/activity information when it +changes feasibility. A parameter that affects only an unobserved cost or +constraint cannot be identified from state transitions alone. The single +$\widehat\eta_t$ formulation assumes local constancy or sufficiently slow +variation over $\mathcal I_t$; otherwise, a trajectory-valued condition model +is required. + +Let $J_0$ be a stored approximate base critic for a declared reference +condition $\eta_0$, and let $\beta_\omega(x,\eta)$ be a structural map learned +over representative conditions. The current critic is reconfigured as + +$$ +\widehat J_t(x) +:= +\beta_\omega(x,\widehat\eta_t)J_0(x), +$$ + +and then used for policy improvement: + +$$ +\widehat\mu_{t+1} +\leftarrow +\operatorname{Improve} +\!\left[ +\widehat J_t;\widehat\eta_t +\right]. +$$ + +The generic $\operatorname{Improve}$ operator may denote a declared +model-based Bellman improvement or a policy-network update. One-step lookahead +is therefore a possible implementation, not part of the Theme definition. +Here adaptation means reconfiguration through the identified condition, not +unrestricted online retraining of all critic parameters. The multiplicative +$\beta_\omega J_0$ form is the current schematic hypothesis; additive +corrections or critic-bank interpolation require separate definitions. + +## 2. Why this problem matters + +Direct adaptive-DP updates can repeatedly modify the full critic whenever the +operating condition changes. Such updates may fit the current condition while +overwriting behavior useful under earlier conditions, so a recurring +condition can require relearning. + +If the environment family is organized by an identifiable parameter $\eta$, +a prelearned relation from $(x,\eta)$ to critic shape can reuse structure +across conditions. Online computation is then concentrated on estimating +$\eta$ and improving the policy from the reconfigured critic. This is a +conditional opportunity, not a guaranteed speedup: it depends on +identifiability, offline coverage, and the expressiveness of the critic +family. + +## 3. Key challenges + +- **Environment representation and identification:** $\eta$ must capture + control-relevant variation and be identifiable from the limited, + policy-dependent data in $\mathcal S_t$. +- **Structured-family expressiveness:** the family + $\beta_\omega(x,\eta)J_0(x)$ may be too restrictive, especially near zeros + or sign changes of $J_0$. +- **Error propagation and changing conditions:** identification delay, + structural approximation error, stale parameters, and model error perturb + both the critic and the improved policy. +- **Bellman and closed-loop consistency:** a plausibly shaped critic is not + automatically Bellman-consistent, stabilizing, constraint-satisfying, or + reliable outside the representative offline conditions. + +## 4. A conventional approach — direct critic adaptation + +A conventional adaptive-DP realization updates critic parameters directly +from recent data and then improves the policy: + +$$ +\begin{aligned} +\theta_{t+1} +&\approx +\arg\min_{\theta} +\mathcal L_{\mathrm{critic}}(\theta;\mathcal S_t),\\ +\widehat\mu_{t+1} +&\leftarrow +\operatorname{Improve} +\!\left[ +\widehat J_{\theta_{t+1}} +\right]. +\end{aligned} +$$ + +One representative choice for $\mathcal L_{\mathrm{critic}}$ is a sampled +one-step TD loss. Holding the target critic +$\widehat J_{\bar\theta_t}$ fixed during learner update $t$ and taking +$0<\alpha<1$, define + +$$ +\begin{aligned} +y_{k,t}^{\mathrm{TD}} +&= +g_k ++ +\alpha +\widehat J_{\bar\theta_t}(x_{k+1}),\\ +\mathcal L_{\mathrm{critic}}(\theta;\mathcal S_t) +&= +\frac{1}{|\mathcal S_t|} +\sum_{k\in\mathcal I_t} +\left( +\widehat J_\theta(x_k) +- +y_{k,t}^{\mathrm{TD}} +\right)^2. +\end{aligned} +$$ + +This form evaluates the policy that generated the transitions, subject to the +usual on-policy or correction assumptions. If $\mathcal S_t$ mixes behavior +policies or materially different environment conditions, the window must be +restricted or an explicit off-policy/condition correction must be supplied. +If a model and action minimization are available, an empirical +Bellman-optimality target can be used instead. The selected critic loss and +its policy-evaluation or policy-improvement role must be declared. + +This can track the currently visited condition, but it does not explicitly +preserve a reusable relation between the environment parameter and critic +shape. Under limited data, it may repeatedly overwrite and relearn critic +behavior as conditions change. This is a baseline mechanism, not a claim that +every direct critic update necessarily forgets. + +## 5. Candidate direction — identifier-conditioned critic reconfiguration + +Offline, suppose approximate critics +$\{\widehat J^{(m)}\}$ are available for representative conditions +$\{\eta^{(m)}\}$ under a common state representation, objective, discount, +and solution concept. In the optimal-control case, +$\widehat J^{(m)}\approx J_{\eta^{(m)}}^\star$, where $J_\eta^\star$ denotes +the optimal critic for the fixed-condition problem. Hold the reference critic +$J_0$ fixed and fit the structural map through + +$$ +\omega^\dagger +\in +\arg\min_{\omega} +\sum_m +\left\| +\widehat J^{(m)}(\cdot) +- +\beta_\omega(\cdot,\eta^{(m)})J_0(\cdot) +\right\|_{\nu_m}^{2}, +$$ + +where $\nu_m$ is a declared weighting distribution or measure over +control-relevant states and +$\|e\|_{\nu_m}^2:=\int |e(x)|^2\,d\nu_m(x)$. If $J_0$ is to be learned jointly +instead, it must appear as an optimization variable and a normalization is +needed to remove the scaling ambiguity between $J_0$ and $\beta_\omega$. +Online operation then follows + +$$ +\mathcal S_t +\xrightarrow{\text{online identification}} +\widehat\eta_t, +\qquad +\widehat J_t(x) += +\beta_{\omega^\dagger}(x,\widehat\eta_t)J_0(x), +\qquad +\widehat\mu_{t+1} +\leftarrow +\operatorname{Improve} +\!\left[ +\widehat J_t;\widehat\eta_t +\right]. +$$ + +A useful diagnostic, rather than a guarantee, separates two critic errors. +For compactness, let $\eta_t:=\eta_{k(t)}$ denote the physical environment +condition associated with learner update $t$. The norm below must use the same +declared control-relevant domain or measure for all terms: + +$$ +\begin{aligned} +\left\| +\widehat J_t-J_{\eta_t}^{\star} +\right\| +\le {}& +\left\| +\beta_{\omega^\dagger}(\cdot,\widehat\eta_t)J_0 +- +\beta_{\omega^\dagger}(\cdot,\eta_t)J_0 +\right\|\\ +&+ +\left\| +\beta_{\omega^\dagger}(\cdot,\eta_t)J_0 +- +J_{\eta_t}^{\star} +\right\|. +\end{aligned} +$$ + +The first term audits sensitivity to identification error; the second audits +the structured critic family. Policy-improvement and model errors remain +additional. Research questions include how to select $\eta$, how to learn +$J_0$ and $\beta_\omega$, when an additive or critic-bank structure is more +appropriate, how to represent uncertainty in $\widehat\eta_t$, and when to +reject reconfiguration in favor of a safe fallback. + +Future evidence should compare a fixed $J_0$, direct critic adaptation, the +structured method with oracle $\eta$, and the method with estimated $\eta$. +Relevant metrics include identification error and delay, critic and Bellman +error, closed-loop performance, violations, update latency, memory, and +return-to-prior-condition performance under abrupt, gradual, recurring, and +held-out conditions. + +## 6. Application domains + +- **Mobility controllers under changing mass, load, road–tire condition, + temperature, or component aging:** estimate a compact condition parameter + and reconfigure a stored critic so recurring operating regimes can reuse + previously learned long-run control structure. +- **Electric-drive and mechatronic control** with compact identifiable plant + parameters, such as resistance, inertia, friction, or load, that can + condition a reusable critic family and its subsequent policy improvement. +- **Systems that revisit a finite or smoothly parameterized environment + family:** recover or interpolate an appropriate critic when a known + condition returns, rather than relearning the full value function from the + latest trajectory. The benefit depends on identification quality and + offline coverage of that environment family. + +
+
+
\ No newline at end of file diff --git a/research/research_themes/assets/00_online_learning_based_optimal_control.png b/research/research_themes/assets/00_online_learning_based_optimal_control.png new file mode 100644 index 000000000..4c84b4cd9 Binary files /dev/null and b/research/research_themes/assets/00_online_learning_based_optimal_control.png differ diff --git a/research/research_themes/assets/00_online_learning_based_optimal_control_framework.png b/research/research_themes/assets/00_online_learning_based_optimal_control_framework.png new file mode 100644 index 000000000..7b53da4c8 Binary files /dev/null and b/research/research_themes/assets/00_online_learning_based_optimal_control_framework.png differ diff --git a/research/research_themes/assets/01_real_world_rl.png b/research/research_themes/assets/01_real_world_rl.png new file mode 100644 index 000000000..99923ea97 Binary files /dev/null and b/research/research_themes/assets/01_real_world_rl.png differ diff --git a/research/research_themes/assets/02_online_multistep_lookahead.png b/research/research_themes/assets/02_online_multistep_lookahead.png new file mode 100644 index 000000000..a98995104 Binary files /dev/null and b/research/research_themes/assets/02_online_multistep_lookahead.png differ diff --git a/research/research_themes/assets/03_semantic_critic_learning.png b/research/research_themes/assets/03_semantic_critic_learning.png new file mode 100644 index 000000000..921242699 Binary files /dev/null and b/research/research_themes/assets/03_semantic_critic_learning.png differ diff --git a/research/research_themes/assets/04_nonstationary_infinite_horizon_ocp.png b/research/research_themes/assets/04_nonstationary_infinite_horizon_ocp.png new file mode 100644 index 000000000..ca9bd616b Binary files /dev/null and b/research/research_themes/assets/04_nonstationary_infinite_horizon_ocp.png differ diff --git a/research/research_themes/assets/05_continual_model_learning.svg b/research/research_themes/assets/05_continual_model_learning.svg new file mode 100644 index 000000000..3eb3c0414 --- /dev/null +++ b/research/research_themes/assets/05_continual_model_learning.svg @@ -0,0 +1,63 @@ + + + Continual model learning concept + Selected new transitions and retained prior function behavior enter a continual model update, producing an updated dynamics model for estimation, prediction, and control. + + + + + + + + + + + + + + CONTINUAL MODEL LEARNING + LEARN NEW → RETAIN PRIOR → SUPPORT CONTROL + + + NEW TRANSITIONS + 𝒮ₜ + current operating information + + + PRIOR BEHAVIOR + ψ̂ₜ(x,u) + memory / anchors 𝓜ₜ + bounded representation of the past + + + CONTINUAL MODEL UPDATE + + L_new + λ_retain L_retain + + LEARN NEW + + RETAIN PRIOR + + CONTROL USE + + + UPDATED MODEL + ψ̂ₜ₊₁(x,u) + + + CONTROL USES + estimation · MPC / lookahead + prediction · critic targets + + + fit + + + retain + + + update + + + use online + diff --git a/research/research_themes/assets/06_constrained_pinn.png b/research/research_themes/assets/06_constrained_pinn.png new file mode 100644 index 000000000..d89f4ef34 Binary files /dev/null and b/research/research_themes/assets/06_constrained_pinn.png differ diff --git a/research/research_themes/assets/07_neuro_adaptive_control.svg b/research/research_themes/assets/07_neuro_adaptive_control.svg new file mode 100644 index 000000000..f4e3b40c9 --- /dev/null +++ b/research/research_themes/assets/07_neuro_adaptive_control.svg @@ -0,0 +1,78 @@ + + + Neuro-adaptive control concept + A neural policy produces a raw control for an uncertain plant, an input-constraint layer determines the applied input, and closed-loop error drives a projected parameter-adaptation path. + + + + + + + + + + + + + + + + + NEURO-ADAPTIVE CONTROL + APPROXIMATE IDEAL LAW → ADAPT ONLINE → CONSTRAIN PARAMETERS & INPUTS + + + REFERENCE + + MEASUREMENT + closed-loop information + + + REGRESSOR + ξₖ + + + NEURAL POLICY + μφₜ(ξₖ) + full law or residual correction + + + INPUT-CONSTRAINT + LAYER + raw → admissible uₖ + + + UNCERTAIN PLANT + apply uₖ + observe state / output + + + CLOSED-LOOP ERROR + tracking · regulation · residual + + + ADAPTATION LAW + Δφₜ + closed-loop-derived update + + + PARAMETER + PROJECTION + project to 𝓡adm + + + + + uₖ (raw) + + uₖ + + + measurement feedback + + + + + + projected parameters at t+1 + + diff --git a/research/research_themes/assets/08_structured_critic_adaptation.png b/research/research_themes/assets/08_structured_critic_adaptation.png new file mode 100644 index 000000000..54179e043 Binary files /dev/null and b/research/research_themes/assets/08_structured_critic_adaptation.png differ diff --git a/static/css/custom.css b/static/css/custom.css index 0fcfa2cda..891bdb2bf 100644 --- a/static/css/custom.css +++ b/static/css/custom.css @@ -138,4 +138,31 @@ body { .dropdown-item { padding: 3px; /* font-size: 15px; */ -} \ No newline at end of file +} + +.dropdown-submenu { + position: relative; +} + +.dropdown-submenu > .dropdown-menu { + top: 0; + left: 100%; + margin-top: -0.25rem; +} + +@media (min-width: 992px) { + .dropdown-submenu:hover > .dropdown-menu, + .dropdown-submenu:focus-within > .dropdown-menu { + display: block; + } +} + +@media (max-width: 991.98px) { + .dropdown-submenu > .dropdown-menu { + position: static; + float: none; + margin: 0 0 0 1rem; + border: 0; + box-shadow: none; + } +} diff --git a/static/img/research/Home/09ed3e2ac3dd0d756845166520a84a14.jpg b/static/projects/Home/09ed3e2ac3dd0d756845166520a84a14.jpg similarity index 100% rename from static/img/research/Home/09ed3e2ac3dd0d756845166520a84a14.jpg rename to static/projects/Home/09ed3e2ac3dd0d756845166520a84a14.jpg diff --git a/static/img/research/Home/f161b77526f558ed93e2359accd90dc3.jpg b/static/projects/Home/f161b77526f558ed93e2359accd90dc3.jpg similarity index 100% rename from static/img/research/Home/f161b77526f558ed93e2359accd90dc3.jpg rename to static/projects/Home/f161b77526f558ed93e2359accd90dc3.jpg diff --git a/static/img/research/Problem_statement_CAEV1.jpg b/static/projects/Problem_statement_CAEV1.jpg similarity index 100% rename from static/img/research/Problem_statement_CAEV1.jpg rename to static/projects/Problem_statement_CAEV1.jpg diff --git a/static/img/research/Problem_statement_CAEV2.jpg b/static/projects/Problem_statement_CAEV2.jpg similarity index 100% rename from static/img/research/Problem_statement_CAEV2.jpg rename to static/projects/Problem_statement_CAEV2.jpg diff --git a/static/img/research/Problem_statement_EM.jpg b/static/projects/Problem_statement_EM.jpg similarity index 100% rename from static/img/research/Problem_statement_EM.jpg rename to static/projects/Problem_statement_EM.jpg diff --git a/static/img/research/Problem_statement_HP.jpg b/static/projects/Problem_statement_HP.jpg similarity index 100% rename from static/img/research/Problem_statement_HP.jpg rename to static/projects/Problem_statement_HP.jpg diff --git a/static/img/research/Research overview_a.jpg b/static/projects/Research overview_a.jpg similarity index 100% rename from static/img/research/Research overview_a.jpg rename to static/projects/Research overview_a.jpg diff --git a/static/img/research/Research_Program.jpg b/static/projects/Research_Program.jpg similarity index 100% rename from static/img/research/Research_Program.jpg rename to static/projects/Research_Program.jpg diff --git a/static/img/research/control_learning/control_learning1.jpg b/static/projects/control_learning/control_learning1.jpg similarity index 100% rename from static/img/research/control_learning/control_learning1.jpg rename to static/projects/control_learning/control_learning1.jpg diff --git a/static/img/research/control_learning/control_learning2.jpg b/static/projects/control_learning/control_learning2.jpg similarity index 100% rename from static/img/research/control_learning/control_learning2.jpg rename to static/projects/control_learning/control_learning2.jpg diff --git "a/static/img/research/projects/2020-\354\206\214\355\230\225.jpg" "b/static/projects/img/2020-\354\206\214\355\230\225.jpg" similarity index 100% rename from "static/img/research/projects/2020-\354\206\214\355\230\225.jpg" rename to "static/projects/img/2020-\354\206\214\355\230\225.jpg" diff --git "a/static/img/research/projects/2021-\354\210\230\354\206\214\353\262\204\354\212\244.jpg" "b/static/projects/img/2021-\354\210\230\354\206\214\353\262\204\354\212\244.jpg" similarity index 100% rename from "static/img/research/projects/2021-\354\210\230\354\206\214\353\262\204\354\212\244.jpg" rename to "static/projects/img/2021-\354\210\230\354\206\214\353\262\204\354\212\244.jpg" diff --git "a/static/img/research/projects/2022-\353\217\231\352\270\260\354\240\204\353\217\231\352\270\260\353\245\274.jpg" "b/static/projects/img/2022-\353\217\231\352\270\260\354\240\204\353\217\231\352\270\260\353\245\274.jpg" similarity index 100% rename from "static/img/research/projects/2022-\353\217\231\352\270\260\354\240\204\353\217\231\352\270\260\353\245\274.jpg" rename to "static/projects/img/2022-\353\217\231\352\270\260\354\240\204\353\217\231\352\270\260\353\245\274.jpg" diff --git "a/static/img/research/projects/2022-\353\257\270\353\236\230\355\230\225\354\236\220\353\217\231\354\260\250.jpg" "b/static/projects/img/2022-\353\257\270\353\236\230\355\230\225\354\236\220\353\217\231\354\260\250.jpg" similarity index 100% rename from "static/img/research/projects/2022-\353\257\270\353\236\230\355\230\225\354\236\220\353\217\231\354\260\250.jpg" rename to "static/projects/img/2022-\353\257\270\353\236\230\355\230\225\354\236\220\353\217\231\354\260\250.jpg" diff --git "a/static/img/research/projects/2022-\354\236\220\354\234\250\354\243\274\355\226\211.jpg" "b/static/projects/img/2022-\354\236\220\354\234\250\354\243\274\355\226\211.jpg" similarity index 100% rename from "static/img/research/projects/2022-\354\236\220\354\234\250\354\243\274\355\226\211.jpg" rename to "static/projects/img/2022-\354\236\220\354\234\250\354\243\274\355\226\211.jpg" diff --git "a/static/img/research/projects/2022-\354\240\204\352\270\260\354\236\220\353\217\231\354\260\250\354\235\230.jpg" "b/static/projects/img/2022-\354\240\204\352\270\260\354\236\220\353\217\231\354\260\250\354\235\230.jpg" similarity index 100% rename from "static/img/research/projects/2022-\354\240\204\352\270\260\354\236\220\353\217\231\354\260\250\354\235\230.jpg" rename to "static/projects/img/2022-\354\240\204\352\270\260\354\236\220\353\217\231\354\260\250\354\235\230.jpg" diff --git "a/static/img/research/projects/2022-\354\264\210\354\213\244\352\260\220.jpg" "b/static/projects/img/2022-\354\264\210\354\213\244\352\260\220.jpg" similarity index 100% rename from "static/img/research/projects/2022-\354\264\210\354\213\244\352\260\220.jpg" rename to "static/projects/img/2022-\354\264\210\354\213\244\352\260\220.jpg" diff --git "a/static/img/research/projects/2023-\354\236\220\354\234\250\354\243\274\355\226\211\354\235\204.jpg" "b/static/projects/img/2023-\354\236\220\354\234\250\354\243\274\355\226\211\354\235\204.jpg" similarity index 100% rename from "static/img/research/projects/2023-\354\236\220\354\234\250\354\243\274\355\226\211\354\235\204.jpg" rename to "static/projects/img/2023-\354\236\220\354\234\250\354\243\274\355\226\211\354\235\204.jpg" diff --git "a/static/img/research/projects/2023-\354\240\204\352\270\260\354\260\250.jpg" "b/static/projects/img/2023-\354\240\204\352\270\260\354\260\250.jpg" similarity index 100% rename from "static/img/research/projects/2023-\354\240\204\352\270\260\354\260\250.jpg" rename to "static/projects/img/2023-\354\240\204\352\270\260\354\260\250.jpg" diff --git "a/static/img/research/projects/2024-\352\270\211\353\263\200\353\266\200\355\225\230.jpg" "b/static/projects/img/2024-\352\270\211\353\263\200\353\266\200\355\225\230.jpg" similarity index 100% rename from "static/img/research/projects/2024-\352\270\211\353\263\200\353\266\200\355\225\230.jpg" rename to "static/projects/img/2024-\352\270\211\353\263\200\353\266\200\355\225\230.jpg" diff --git a/static/img/research/projects/2025-RWS.jpg b/static/projects/img/2025-RWS.jpg similarity index 100% rename from static/img/research/projects/2025-RWS.jpg rename to static/projects/img/2025-RWS.jpg diff --git "a/static/img/research/projects/2025-\354\213\254\354\270\265\355\225\231\354\212\265\354\235\230.jpg" "b/static/projects/img/2025-\354\213\254\354\270\265\355\225\231\354\212\265\354\235\230.jpg" similarity index 100% rename from "static/img/research/projects/2025-\354\213\254\354\270\265\355\225\231\354\212\265\354\235\230.jpg" rename to "static/projects/img/2025-\354\213\254\354\270\265\355\225\231\354\212\265\354\235\230.jpg" diff --git "a/static/img/research/projects/2025-\354\225\210\354\240\225\354\204\261.jpg" "b/static/projects/img/2025-\354\225\210\354\240\225\354\204\261.jpg" similarity index 100% rename from "static/img/research/projects/2025-\354\225\210\354\240\225\354\204\261.jpg" rename to "static/projects/img/2025-\354\225\210\354\240\225\354\204\261.jpg"