mirror of
https://github.com/gatieme/LDD-LinuxDeviceDrivers.git
synced 2026-09-22 20:53:34 +08:00
CFS调度器之虚拟时钟...
This commit is contained in:
@@ -3367,7 +3367,8 @@ static void
|
||||
set_next_entity(struct cfs_rq *cfs_rq, struct sched_entity *se)
|
||||
{
|
||||
/* 'current' is not kept within the tree. */
|
||||
if (se->on_rq) {
|
||||
if (se->on_rq) /* 如果se尚在rq队列上 */
|
||||
{
|
||||
/*
|
||||
* Any task has to be enqueued before it get to execute on
|
||||
* a CPU. So account for the time it spent waiting on the
|
||||
@@ -3375,10 +3376,11 @@ set_next_entity(struct cfs_rq *cfs_rq, struct sched_entity *se)
|
||||
*/
|
||||
if (schedstat_enabled())
|
||||
update_stats_wait_end(cfs_rq, se);
|
||||
/* 将se从rq队列中删除 */
|
||||
__dequeue_entity(cfs_rq, se);
|
||||
update_load_avg(se, 1);
|
||||
}
|
||||
|
||||
/* 新sched_entity中的exec_start字段为当前clock_task */
|
||||
update_stats_curr_start(cfs_rq, se);
|
||||
cfs_rq->curr = se;
|
||||
#ifdef CONFIG_SCHEDSTATS
|
||||
@@ -3392,6 +3394,7 @@ set_next_entity(struct cfs_rq *cfs_rq, struct sched_entity *se)
|
||||
se->sum_exec_runtime - se->prev_sum_exec_runtime);
|
||||
}
|
||||
#endif
|
||||
/* //更新task上一次投入运行的从时间 */
|
||||
se->prev_sum_exec_runtime = se->sum_exec_runtime;
|
||||
}
|
||||
|
||||
@@ -3404,20 +3407,41 @@ wakeup_preempt_entity(struct sched_entity *curr, struct sched_entity *se);
|
||||
* 2) pick the "next" process, since someone really wants that to run
|
||||
* 3) pick the "last" process, for cache locality
|
||||
* 4) do not run the "skip" process, if something else is available
|
||||
*
|
||||
* 1. 首先要确保任务组之间的公平, 这也是设置组的原因之一
|
||||
* 2. 其次, 挑选下一个合适的(优先级比较高的)进程
|
||||
* 因为它确实需要马上运行
|
||||
* 3. 如果没有找到条件2中的进程
|
||||
* 那么为了保持良好的局部性
|
||||
* 则选中上一次执行的进程
|
||||
* 4. 只要有任务存在, 就不要让CPU空转, 也就是让CPU运行idle进程
|
||||
*/
|
||||
static struct sched_entity *
|
||||
pick_next_entity(struct cfs_rq *cfs_rq, struct sched_entity *curr)
|
||||
{
|
||||
/* //摘取红黑树最左边的进程 */
|
||||
struct sched_entity *left = __pick_first_entity(cfs_rq);
|
||||
struct sched_entity *se;
|
||||
|
||||
/*
|
||||
* If curr is set we have to see if its left of the leftmost entity
|
||||
* still in the tree, provided there was anything in the tree at all.
|
||||
*
|
||||
* 如果
|
||||
* left == NULL 或者
|
||||
* curr != NULL curr进程比left进程更优(即curr的虚拟运行时间更小)
|
||||
* 说明curr进程是自动放弃运行权利, 且其比最左进程更优
|
||||
* 因此将left指向了curr, 即curr是最优的进程
|
||||
*/
|
||||
if (!left || (curr && entity_before(curr, left)))
|
||||
{
|
||||
left = curr;
|
||||
}
|
||||
|
||||
/* se = left存储了cfs_rq队列中最优的那个进程
|
||||
* 如果进程curr是一个自愿放弃CPU的进程(其比最左进程更优), 则取se = curr
|
||||
* 否则进程se就取红黑树中最左的进程left, 它必然是当前就绪队列上最优的
|
||||
*/
|
||||
se = left; /* ideally we run the leftmost entity */
|
||||
|
||||
/*
|
||||
@@ -3427,9 +3451,13 @@ pick_next_entity(struct cfs_rq *cfs_rq, struct sched_entity *curr)
|
||||
if (cfs_rq->skip == se) {
|
||||
struct sched_entity *second;
|
||||
|
||||
if (se == curr) {
|
||||
if (se == curr)
|
||||
{
|
||||
second = __pick_first_entity(cfs_rq);
|
||||
} else {
|
||||
}
|
||||
else
|
||||
{
|
||||
/* 摘取红黑树上第二左的进程节点 */
|
||||
second = __pick_next_entity(se);
|
||||
if (!second || (curr && entity_before(curr, second)))
|
||||
second = curr;
|
||||
@@ -4394,6 +4422,7 @@ static void dequeue_task_fair(struct rq *rq, struct task_struct *p, int flags)
|
||||
|
||||
for_each_sched_entity(se) {
|
||||
cfs_rq = cfs_rq_of(se);
|
||||
/* */
|
||||
dequeue_entity(cfs_rq, se, flags);
|
||||
|
||||
/*
|
||||
@@ -5466,9 +5495,11 @@ pick_next_task_fair(struct rq *rq, struct task_struct *prev)
|
||||
|
||||
again:
|
||||
#ifdef CONFIG_FAIR_GROUP_SCHED
|
||||
/* 如果nr_running计数器为0, 即当前队列上没有可运行进程,
|
||||
* 则需要调度idle进程 */
|
||||
if (!cfs_rq->nr_running)
|
||||
goto idle;
|
||||
|
||||
/* 如果当前运行进程prev不是被fair调度的普通非实时进程 */
|
||||
if (prev->sched_class != &fair_sched_class)
|
||||
goto simple;
|
||||
|
||||
@@ -5489,7 +5520,11 @@ again:
|
||||
* entity, update_curr() will update its vruntime, otherwise
|
||||
* forget we've ever seen it.
|
||||
*/
|
||||
if (curr) {
|
||||
if (curr)
|
||||
{
|
||||
/* 如果当前进程curr在队列上,
|
||||
* 则需要更新起统计量和虚拟运行时间
|
||||
* 否则设置curr为空 */
|
||||
if (curr->on_rq)
|
||||
update_curr(cfs_rq);
|
||||
else
|
||||
@@ -5504,11 +5539,11 @@ again:
|
||||
if (unlikely(check_cfs_rq_runtime(cfs_rq)))
|
||||
goto simple;
|
||||
}
|
||||
|
||||
/* 选择一个最优的调度实体 */
|
||||
se = pick_next_entity(cfs_rq, curr);
|
||||
cfs_rq = group_cfs_rq(se);
|
||||
} while (cfs_rq);
|
||||
|
||||
} while (cfs_rq); /* 如果被调度的进程仍属于当前组,那么选取下一个可能被调度的任务,以保证组间调度的公平性 */
|
||||
/* 获取调度实体se的进程实体信息 */
|
||||
p = task_of(se);
|
||||
|
||||
/*
|
||||
@@ -5516,18 +5551,22 @@ again:
|
||||
* is a different task than we started out with, try and touch the
|
||||
* least amount of cfs_rqs.
|
||||
*/
|
||||
if (prev != p) {
|
||||
if (prev != p)
|
||||
{
|
||||
struct sched_entity *pse = &prev->se;
|
||||
|
||||
while (!(cfs_rq = is_same_group(se, pse))) {
|
||||
while (!(cfs_rq = is_same_group(se, pse)))
|
||||
{
|
||||
int se_depth = se->depth;
|
||||
int pse_depth = pse->depth;
|
||||
|
||||
if (se_depth <= pse_depth) {
|
||||
if (se_depth <= pse_depth)
|
||||
{
|
||||
put_prev_entity(cfs_rq_of(pse), pse);
|
||||
pse = parent_entity(pse);
|
||||
}
|
||||
if (se_depth >= pse_depth) {
|
||||
if (se_depth >= pse_depth)
|
||||
{
|
||||
set_next_entity(cfs_rq_of(se), se);
|
||||
se = parent_entity(se);
|
||||
}
|
||||
@@ -5550,7 +5589,8 @@ simple:
|
||||
|
||||
put_prev_task(rq, prev);
|
||||
|
||||
do {
|
||||
do
|
||||
{
|
||||
se = pick_next_entity(cfs_rq, NULL);
|
||||
set_next_entity(cfs_rq, se);
|
||||
cfs_rq = group_cfs_rq(se);
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
Linux CFS调度器之队列操作
|
||||
=======
|
||||
|
||||
|
||||
| 日期 | 内核版本 | 架构| 作者 | GitHub| CSDN |
|
||||
| ------- |:-------:|:-------:|:-------:|:-------:|:-------:|
|
||||
| 2016-06-29 | [Linux-4.6](http://lxr.free-electrons.com/source/?v=4.6) | X86 & arm | [gatieme](http://blog.csdn.net/gatieme) | [LinuxDeviceDrivers](https://github.com/gatieme/LDD-LinuxDeviceDrivers) | [Linux进程管理与调度](http://blog.csdn.net/gatieme/article/details/51456569) |
|
||||
|
||||
|
||||
|
||||
CFS负责处理普通非实时进程, 这类进程是我们linux中最普遍的进程
|
||||
|
||||
|
||||
#1 前景回顾
|
||||
-------
|
||||
|
||||
##1.1 CFS调度算法
|
||||
-------
|
||||
|
||||
**CFS调度算法的思想**
|
||||
|
||||
理想状态下每个进程都能获得相同的时间片,并且同时运行在CPU上,但实际上一个CPU同一时刻运行的进程只能有一个。也就是说,当一个进程占用CPU时,其他进程就必须等待。CFS为了实现公平,必须惩罚当前正在运行的进程,以使那些正在等待的进程下次被调度.
|
||||
|
||||
## 负荷权重和虚拟时钟
|
||||
|
||||
**虚拟时钟是红黑树排序的依据**
|
||||
|
||||
具体实现时,CFS通过每个进程的**虚拟运行时间(vruntime)**来衡量哪个进程最值得被调度. CFS中的就绪队列是一棵以vruntime为键值的红黑树,虚拟时间越小的进程越靠近整个红黑树的最左端。因此,调度器每次选择位于红黑树最左端的那个进程,该进程的vruntime最小.
|
||||
|
||||
**优先级计算负荷权重, 负荷权重和当前时间计算出虚拟运行时间**
|
||||
|
||||
虚拟运行时间是通过进程的实际运行时间和进程的权重(weight)计算出来的。在CFS调度器中,将进程优先级这个概念弱化,而是强调进程的权重。一个进程的权重越大,则说明这个进程更需要运行,因此它的虚拟运行时间就越小,这样被调度的机会就越大。而,CFS调度器中的权重在内核是对用户态进程的优先级nice值, 通过prio_to_weight数组进行nice值和权重的转换而计算出来的
|
||||
|
||||
|
||||
**虚拟时钟相关公式**
|
||||
|
||||
linux内核采用了计算公式:
|
||||
|
||||
| 属性 | 公式 | 描述 |
|
||||
|:-------:|:-------:|
|
||||
| ideal_time | sum_runtime *se.weight/cfs_rq.weight | 每个进程应该运行的时间 |
|
||||
| sum_exec_runtime | | 运行队列中所有任务运行完一遍的时间 |
|
||||
| se.weight | | 当前进程的权重 |
|
||||
| cfs.weight | | 整个cfs_rq的总权重 |
|
||||
|
||||
这里se.weight和cfs.weight根据上面讲解我们可以算出, sum_runtime是怎们计算的呢,linux内核中这是个经验值,其经验公式是
|
||||
|
||||
| 条件 | 公式 |
|
||||
|:-------:|:-------:|
|
||||
| 进程数 > sched_nr_latency | sum_runtime=sysctl_sched_min_granularity *nr_running |
|
||||
| 进程数 <=sched_nr_latency | sum_runtime=sysctl_sched_latency = 20ms |
|
||||
|
||||
>注:sysctl_sched_min_granularity =4ms
|
||||
>
|
||||
>sched_nr_latency是内核在一个延迟周期中处理的最大活动进程数目
|
||||
|
||||
linux内核代码中是通过一个叫vruntime的变量来实现上面的原理的,即:
|
||||
|
||||
每一个进程拥有一个vruntime,每次需要调度的时候就选运行队列中拥有最小vruntime的那个进程来运行,vruntime在时钟中断里面被维护,每次时钟中断都要更新当前进程的vruntime,即vruntime以如下公式逐渐增长:
|
||||
|
||||
|
||||
| 条件 | 公式 |
|
||||
|:-------:|:-------:|
|
||||
| curr.nice!=NICE_0_LOAD | vruntime += delta* NICE_0_LOAD/se.weight; |
|
||||
| curr.nice=NICE_0_LOAD | vruntime += delta; |
|
||||
|
||||
|
||||
##1.2 CFS进程入队和出队
|
||||
-------
|
||||
|
||||
enqueue_task_fair和dequeue_task_fair分别用来向CFS就绪队列中添加或者删除进程
|
||||
|
||||
完全公平调度器CFS中有两个函数可用来增删队列的成员:enqueue_task_fair和dequeue_task_fair分别用来向CFS就绪队列中添加或者删除进程
|
||||
|
||||
|
||||
**enqueue_task_fair进程入队**
|
||||
|
||||
向就绪队列中放置新进程的工作由函数enqueue_task_fair函数完成, 该函数定义在[kernel/sched/fair.c, line 5442](http://lxr.free-electrons.com/source/kernel/sched/fair.c?v=4.6#L5442), 其函数原型如下
|
||||
|
||||
该函数将task_struct *p所指向的进程插入到rq所在的就绪队列中, 除了指向所述的就绪队列rq和task_struct的指针外, 该函数还有另外一个参数wakeup. 这使得可以指定入队的进程是否最近才被唤醒并转换为运行状态(此时需指定wakeup = 1), 还是此前就是可运行的(那么wakeup = 0).
|
||||
|
||||
```c
|
||||
static void
|
||||
enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
|
||||
```
|
||||
enqueue_task_fair的执行流程如下
|
||||
|
||||
* 如果通过struct sched_entity的on_rq成员判断进程已经在就绪队列上, 则无事可做.
|
||||
|
||||
* 否则, 具体的工作委托给enqueue_entity完成, 其中内核会借机用update_curr更新统计量
|
||||
在enqueue_entity内部如果需要会调用__enqueue_entity将进程插入到CFS红黑树中合适的结点
|
||||
|
||||
|
||||
**dequeue_task_fair进程出队操作**
|
||||
|
||||
dequeue_task_fair函数在完成睡眠等情况下调度, 将任务从就绪队列中移除
|
||||
|
||||
其执行的过程正好跟enqueue_task_fair的思路相同, 只是操作刚好相反
|
||||
|
||||
|
||||
enqueue_task_fair的执行流程如下
|
||||
|
||||
* 如果通过struct sched_entity的on_rq成员判断进程已经在就绪队列上, 则无事可做.
|
||||
|
||||
* 否则, 具体的工作委托给dequeue_entity完成, 其中内核会借机用update_curr更新统计量
|
||||
在enqueue_entity内部如果需要会调用__dequeue_entity将进程插入到CFS红黑树中合适的结点
|
||||
|
||||
|
||||
#2 今日看点(CFS如何选择最合适的进程)
|
||||
-------
|
||||
|
||||
每个调度器类sched_class都必须提供一个pick_next_task函数用以在就绪队列中选择一个最优的进程来等待调度, 而我们的CFS调度器类中, 选择下一个将要运行的进程由pick_next_task_fair函数来完成
|
||||
|
||||
|
||||
|
||||
# CFS如何选择一个进程
|
||||
-------
|
||||
|
||||
## pick_next_task_fair
|
||||
-------
|
||||
|
||||
选择下一个将要运行的进程pick_next_task_fair执行. 其代码执行流程如下
|
||||
|
||||
如果nr_running计数器为0, 即当前队列上没有可运行进程, 则无事可做, 函数可以立即返回. 否则将具体工作委托给pick
|
||||
|
||||
## put_prev_task
|
||||
-------
|
||||
|
||||
在选中了下一个将被调度执行的进程之后,回到pick_next_task_fair中,执行set_next_entity
|
||||
|
||||
|
||||
|
||||
## pick_next_entity
|
||||
-------
|
||||
|
||||
|
||||
## set_next_entity
|
||||
-------
|
||||
|
||||
|
||||
|
||||
http://blog.csdn.net/sunnybeike/article/details/6918586
|
||||
Reference in New Issue
Block a user