mirror of
https://gitlab.com/redhat/centos-stream/src/kernel/centos-stream-10.git
synced 2026-09-09 00:07:04 +08:00
Merge: [RHEL-10] Update MM-core codebase and its dependencies to upstream v6.17
MR: https://gitlab.com/redhat/centos-stream/src/kernel/centos-stream-10/-/merge_requests/2316 JIRA: https://issues.redhat.com/browse/RHEL-145695 Depends: https://gitlab.com/redhat/centos-stream/src/kernel/centos-stream-10/-/merge_requests/2267 List of CVEs addressed by this update: CVE: CVE-2026-43388 CVE: CVE-2026-23012 CVE: CVE-2026-31653 CVE: CVE-2025-40008 CVE: CVE-2025-40006 CVE: CVE-2025-39877 CVE: CVE-2025-39910 CVE: CVE-2025-39916 CVE: CVE-2025-39909 CVE: CVE-2025-39845 CVE: CVE-2025-39844 CVE: CVE-2025-39899 CVE: CVE-2025-39902 CVE: CVE-2025-39775 CVE: CVE-2025-38686 CVE: CVE-2025-38554 CVE: CVE-2025-39700 CVE: CVE-2025-38681 Set of changes to level the RHEL-10 MM-core and dependencies codebase up to upstream's v6.17 (stable). The following set of "omitted-fixes" is not critical, and the commits in the list below will eventually be picked up later on, without causing any further conflicts, when we wrap up the update work with v6.18 LTS. Omitted-fix: 1442bb87b878 ("s390/boot: Use entire page for PTEs") Omitted-fix: b4a96ab50f36 ("powerpc/kdump: Add support for crashkernel CMA reservation") Omitted-fix: 4ba5a8a7faa6 ("vmw_balloon: indicate success when effectively deflating during migration") Omitted-fix: 51e38e7d40d6 ("mm: add remap_pfn_range_prepare(), remap_pfn_range_complete()") Omitted-fix: dd3b304b9410 ("mm/page_alloc: use xxx_pageblock_isolate() for better reading") Omitted-fix: f04aad36a07c ("mm/ksm: fix flag-dropping behavior in ksm_madvise") Omitted-fix: c373f7f98e6a ("mm/sparse-vmemmap: fix vmemmap accounting underflow") Omitted-fix: 7e89979f6695 ("include/linux/pgtable.h: convert arch_enter_lazy_mmu_mode() and friends to static inlines") Omitted-fix: 84f4928446e6 ("tools/testing/selftests: add merge test for partial msealed range") Omitted-fix: bce1dabd310e ("selftests/mm: fix usage of FORCE_READ() in cow tests") Omitted-fix: d7484f6edd31 ("Docs/mm/damon/design: fix wrong link to intervals goal section") Omitted-fix: 1736047a4e96 ("mm/damon/core: cleanup targets and regions at once on kdamond termination") Omitted-fix: 0199390a6b92 ("mm/damon/sysfs: dealloc repeat_call_control if damon_call() fails") Omitted-fix: 4c04c6b47c36 ("mm/damon/stat: deallocate damon_call() failure leaking damon_ctx") Omitted-fix: 7e6cc35f5283 ("mm/damon/core: trace esz at first setup") Omitted-fix: 2f6ce7e714ef ("mm/damon/stat: change last_refresh_jiffies to a global variable") Omitted-fix: 84481e705ab0 ("mm/damon/stat: monitor all System RAM resources") Omitted-fix: e04ed278d25b ("mm/damon/stat: fix memory leak on damon_start() failure in damon_stat_start()") Omitted-fix: f98590bc08d4 ("mm/damon/stat: detect and use fresh enabled value") Omitted-fix: 7746d72c6405 ("samples/damon/mtier: fail early if address range parameters are invalid") Omitted-fix: c62cff40481c ("samples/damon/mtier: avoid starting DAMON before initialization") Omitted-fix: e6b733ca2f99 ("samples/damon/prcl: avoid starting DAMON before initialization") Omitted-fix: f826edeb888c ("samples/damon/wsse: avoid starting DAMON before initialization") Signed-off-by: Rafael Aquini <raquini@redhat.com> Approved-by: Luiz Capitulino <luizcap@redhat.com> Approved-by: Ricardo Robaina <rrobaina@redhat.com> Approved-by: Eder Zulian <ezulian@redhat.com> Approved-by: David Arcari <darcari@redhat.com> Approved-by: ashelat <ashelat@redhat.com> Approved-by: José Expósito <jexposit@redhat.com> Approved-by: Michael Petlan <mpetlan@redhat.com> Approved-by: CKI KWF Bot <cki-ci-bot+kwf-gitlab-com@redhat.com> Merged-by: CKI GitLab Kmaint Pipeline Bot <26919896-cki-kmaint-pipeline-bot@users.noreply.gitlab.com>
This commit is contained in:
@@ -227,3 +227,12 @@ Contact: Jiaqi Yan <jiaqiyan@google.com>
|
||||
Description:
|
||||
Of the raw poisoned pages on a NUMA node, how many pages are
|
||||
recovered by memory error recovery attempt.
|
||||
|
||||
What: /sys/devices/system/node/nodeX/reclaim
|
||||
Date: June 2025
|
||||
Contact: Linux Memory Management list <linux-mm@kvack.org>
|
||||
Description:
|
||||
Perform user-triggered proactive reclaim on a NUMA node.
|
||||
This interface is equivalent to the memcg variant.
|
||||
|
||||
See Documentation/admin-guide/cgroup-v2.rst
|
||||
|
||||
@@ -44,6 +44,13 @@ Contact: SeongJae Park <sj@kernel.org>
|
||||
Description: Reading this file returns the pid of the kdamond if it is
|
||||
running.
|
||||
|
||||
What: /sys/kernel/mm/damon/admin/kdamonds/<K>/refresh_ms
|
||||
Date: Jul 2025
|
||||
Contact: SeongJae Park <sj@kernel.org>
|
||||
Description: Writing a value to this file sets the time interval for
|
||||
automatic DAMON status file contents update. Writing '0'
|
||||
disables the update. Reading this file returns the value.
|
||||
|
||||
What: /sys/kernel/mm/damon/admin/kdamonds/<K>/contexts/nr_contexts
|
||||
Date: Mar 2022
|
||||
Contact: SeongJae Park <sj@kernel.org>
|
||||
@@ -283,6 +290,12 @@ Contact: SeongJae Park <sj@kernel.org>
|
||||
Description: Writing to and reading from this file sets and gets the current
|
||||
value of the goal metric.
|
||||
|
||||
What: /sys/kernel/mm/damon/admin/kdamonds/<K>/contexts/<C>/schemes/<S>/quotas/goals/<G>/nid
|
||||
Date: Apr 2025
|
||||
Contact: SeongJae Park <sj@kernel.org>
|
||||
Description: Writing to and reading from this file sets and gets the nid
|
||||
parameter of the goal.
|
||||
|
||||
What: /sys/kernel/mm/damon/admin/kdamonds/<K>/contexts/<C>/schemes/<S>/quotas/weights/sz_permil
|
||||
Date: Mar 2022
|
||||
Contact: SeongJae Park <sj@kernel.org>
|
||||
@@ -425,6 +438,28 @@ Description: Directory for DAMON operations set layer-handled DAMOS filters.
|
||||
/sys/kernel/mm/damon/admin/kdamonds/<K>/contexts/<C>/schemes/<S>/filters
|
||||
directory.
|
||||
|
||||
What: /sys/kernel/mm/damon/admin/kdamonds/<K>/contexts/<C>/schemes/<S>/dests/nr_dests
|
||||
Date: Jul 2025
|
||||
Contact: SeongJae Park <sj@kernel.org>
|
||||
Description: Writing a number 'N' to this file creates the number of
|
||||
directories for setting action destinations of the scheme named
|
||||
'0' to 'N-1' under the dests/ directory.
|
||||
|
||||
What: /sys/kernel/mm/damon/admin/kdamonds/<K>/contexts/<C>/schemes/<S>/dests/<D>/id
|
||||
Date: Jul 2025
|
||||
Contact: SeongJae Park <sj@kernel.org>
|
||||
Description: Writing to and reading from this file sets and gets the id of
|
||||
the DAMOS action destination. For DAMOS_MIGRATE_{HOT,COLD}
|
||||
actions, the destination node's node id can be written and
|
||||
read.
|
||||
|
||||
What: /sys/kernel/mm/damon/admin/kdamonds/<K>/contexts/<C>/schemes/<S>/dests/<D>/weight
|
||||
Date: Jul 2025
|
||||
Contact: SeongJae Park <sj@kernel.org>
|
||||
Description: Writing to and reading from this file sets and gets the weight
|
||||
of the DAMOS action destination to select as the destination of
|
||||
each action among the destinations.
|
||||
|
||||
What: /sys/kernel/mm/damon/admin/kdamonds/<K>/contexts/<C>/schemes/<S>/stats/nr_tried
|
||||
Date: Mar 2022
|
||||
Contact: SeongJae Park <sj@kernel.org>
|
||||
|
||||
@@ -37,7 +37,8 @@ Description:
|
||||
The alloc_calls file is read-only and lists the kernel code
|
||||
locations from which allocations for this cache were performed.
|
||||
The alloc_calls file only contains information if debugging is
|
||||
enabled for that cache (see Documentation/mm/slub.rst).
|
||||
enabled for that cache (see
|
||||
Documentation/admin-guide/mm/slab.rst).
|
||||
|
||||
What: /sys/kernel/slab/<cache>/alloc_fastpath
|
||||
Date: February 2008
|
||||
@@ -219,7 +220,7 @@ Contact: Pekka Enberg <penberg@cs.helsinki.fi>,
|
||||
Description:
|
||||
The free_calls file is read-only and lists the locations of
|
||||
object frees if slab debugging is enabled (see
|
||||
Documentation/mm/slub.rst).
|
||||
Documentation/admin-guide/mm/slab.rst).
|
||||
|
||||
What: /sys/kernel/slab/<cache>/free_fastpath
|
||||
Date: February 2008
|
||||
|
||||
@@ -311,6 +311,27 @@ crashkernel syntax
|
||||
|
||||
crashkernel=0,low
|
||||
|
||||
4) crashkernel=size,cma
|
||||
|
||||
Reserve additional crash kernel memory from CMA. This reservation is
|
||||
usable by the first system's userspace memory and kernel movable
|
||||
allocations (memory balloon, zswap). Pages allocated from this memory
|
||||
range will not be included in the vmcore so this should not be used if
|
||||
dumping of userspace memory is intended and it has to be expected that
|
||||
some movable kernel pages may be missing from the dump.
|
||||
|
||||
A standard crashkernel reservation, as described above, is still needed
|
||||
to hold the crash kernel and initrd.
|
||||
|
||||
This option increases the risk of a kdump failure: DMA transfers
|
||||
configured by the first kernel may end up corrupting the second
|
||||
kernel's memory.
|
||||
|
||||
This reservation method is intended for systems that can't afford to
|
||||
sacrifice enough memory for standard crashkernel reservation and where
|
||||
less reliable and possibly incomplete kdump is preferable to no kdump at
|
||||
all.
|
||||
|
||||
Boot into System Kernel
|
||||
-----------------------
|
||||
1) Update the boot loader (such as grub, yaboot, or lilo) configuration
|
||||
|
||||
@@ -982,6 +982,28 @@
|
||||
0: to disable low allocation.
|
||||
It will be ignored when crashkernel=X,high is not used
|
||||
or memory reserved is below 4G.
|
||||
crashkernel=size[KMG],cma
|
||||
[KNL, X86] Reserve additional crash kernel memory from
|
||||
CMA. This reservation is usable by the first system's
|
||||
userspace memory and kernel movable allocations (memory
|
||||
balloon, zswap). Pages allocated from this memory range
|
||||
will not be included in the vmcore so this should not
|
||||
be used if dumping of userspace memory is intended and
|
||||
it has to be expected that some movable kernel pages
|
||||
may be missing from the dump.
|
||||
|
||||
A standard crashkernel reservation, as described above,
|
||||
is still needed to hold the crash kernel and initrd.
|
||||
|
||||
This option increases the risk of a kdump failure: DMA
|
||||
transfers configured by the first kernel may end up
|
||||
corrupting the second kernel's memory.
|
||||
|
||||
This reservation method is intended for systems that
|
||||
can't afford to sacrifice enough memory for standard
|
||||
crashkernel reservation and where less reliable and
|
||||
possibly incomplete kdump is preferable to no kdump at
|
||||
all.
|
||||
|
||||
cryptomgr.notests
|
||||
[KNL] Disable crypto self-tests
|
||||
@@ -1787,6 +1809,27 @@
|
||||
backtraces on all cpus.
|
||||
Format: 0 | 1
|
||||
|
||||
hash_pointers=
|
||||
[KNL,EARLY]
|
||||
By default, when pointers are printed to the console
|
||||
or buffers via the %p format string, that pointer is
|
||||
"hashed", i.e. obscured by hashing the pointer value.
|
||||
This is a security feature that hides actual kernel
|
||||
addresses from unprivileged users, but it also makes
|
||||
debugging the kernel more difficult since unequal
|
||||
pointers can no longer be compared. The choices are:
|
||||
Format: { auto | always | never }
|
||||
Default: auto
|
||||
|
||||
auto - Hash pointers unless slab_debug is enabled.
|
||||
always - Always hash pointers (even if slab_debug is
|
||||
enabled).
|
||||
never - Never hash pointers. This option should only
|
||||
be specified when debugging the kernel. Do
|
||||
not use on production kernels. The boot
|
||||
param "no_hash_pointers" is an alias for
|
||||
this mode.
|
||||
|
||||
hashdist= [KNL,NUMA] Large hashes allocated during boot
|
||||
are distributed across NUMA nodes. Defaults on
|
||||
for 64-bit NUMA, off otherwise.
|
||||
@@ -4097,18 +4140,7 @@
|
||||
|
||||
no_hash_pointers
|
||||
[KNL,EARLY]
|
||||
Force pointers printed to the console or buffers to be
|
||||
unhashed. By default, when a pointer is printed via %p
|
||||
format string, that pointer is "hashed", i.e. obscured
|
||||
by hashing the pointer value. This is a security feature
|
||||
that hides actual kernel addresses from unprivileged
|
||||
users, but it also makes debugging the kernel more
|
||||
difficult since unequal pointers can no longer be
|
||||
compared. However, if this command-line option is
|
||||
specified, then all normal pointers will have their true
|
||||
value printed. This option should only be specified when
|
||||
debugging the kernel. Please do not use on production
|
||||
kernels.
|
||||
Alias for "hash_pointers=never".
|
||||
|
||||
nohibernate [HIBERNATION] Disable hibernation and resume.
|
||||
|
||||
@@ -6469,14 +6501,18 @@
|
||||
slab_debug can create guard zones around objects and
|
||||
may poison objects when not in use. Also tracks the
|
||||
last alloc / free. For more information see
|
||||
Documentation/mm/slub.rst.
|
||||
Documentation/admin-guide/mm/slab.rst.
|
||||
(slub_debug legacy name also accepted for now)
|
||||
|
||||
Using this option implies the "no_hash_pointers"
|
||||
option which can be undone by adding the
|
||||
"hash_pointers=always" option.
|
||||
|
||||
slab_max_order= [MM]
|
||||
Determines the maximum allowed order for slabs.
|
||||
A high setting may cause OOMs due to memory
|
||||
fragmentation. For more information see
|
||||
Documentation/mm/slub.rst.
|
||||
Documentation/admin-guide/mm/slab.rst.
|
||||
(slub_max_order legacy name also accepted for now)
|
||||
|
||||
slab_merge [MM]
|
||||
@@ -6491,13 +6527,14 @@
|
||||
the number of objects indicated. The higher the number
|
||||
of objects the smaller the overhead of tracking slabs
|
||||
and the less frequently locks need to be acquired.
|
||||
For more information see Documentation/mm/slub.rst.
|
||||
For more information see
|
||||
Documentation/admin-guide/mm/slab.rst.
|
||||
(slub_min_objects legacy name also accepted for now)
|
||||
|
||||
slab_min_order= [MM]
|
||||
Determines the minimum page order for slabs. Must be
|
||||
lower or equal to slab_max_order. For more information see
|
||||
Documentation/mm/slub.rst.
|
||||
Documentation/admin-guide/mm/slab.rst.
|
||||
(slub_min_order legacy name also accepted for now)
|
||||
|
||||
slab_nomerge [MM]
|
||||
@@ -6511,7 +6548,8 @@
|
||||
cache (risks via metadata attacks are mostly
|
||||
unchanged). Debug options disable merging on their
|
||||
own.
|
||||
For more information see Documentation/mm/slub.rst.
|
||||
For more information see
|
||||
Documentation/admin-guide/mm/slab.rst.
|
||||
(slub_nomerge legacy name also accepted for now)
|
||||
|
||||
slab_strict_numa [MM]
|
||||
|
||||
@@ -14,3 +14,4 @@ access monitoring and access-aware system operations.
|
||||
usage
|
||||
reclaim
|
||||
lru_sort
|
||||
stat
|
||||
|
||||
@@ -42,32 +42,45 @@ the execution. ::
|
||||
|
||||
$ git clone https://github.com/sjp38/masim; cd masim; make
|
||||
$ sudo damo start "./masim ./configs/stairs.cfg --quiet"
|
||||
$ sudo ./damo show
|
||||
0 addr [85.541 TiB , 85.541 TiB ) (57.707 MiB ) access 0 % age 10.400 s
|
||||
1 addr [85.541 TiB , 85.542 TiB ) (413.285 MiB) access 0 % age 11.400 s
|
||||
2 addr [127.649 TiB , 127.649 TiB) (57.500 MiB ) access 0 % age 1.600 s
|
||||
3 addr [127.649 TiB , 127.649 TiB) (32.500 MiB ) access 0 % age 500 ms
|
||||
4 addr [127.649 TiB , 127.649 TiB) (9.535 MiB ) access 100 % age 300 ms
|
||||
5 addr [127.649 TiB , 127.649 TiB) (8.000 KiB ) access 60 % age 0 ns
|
||||
6 addr [127.649 TiB , 127.649 TiB) (6.926 MiB ) access 0 % age 1 s
|
||||
7 addr [127.998 TiB , 127.998 TiB) (120.000 KiB) access 0 % age 11.100 s
|
||||
8 addr [127.998 TiB , 127.998 TiB) (8.000 KiB ) access 40 % age 100 ms
|
||||
9 addr [127.998 TiB , 127.998 TiB) (4.000 KiB ) access 0 % age 11 s
|
||||
total size: 577.590 MiB
|
||||
$ sudo ./damo stop
|
||||
$ sudo damo report access
|
||||
heatmap: 641111111000000000000000000000000000000000000000000000[...]33333333333333335557984444[...]7
|
||||
# min/max temperatures: -1,840,000,000, 370,010,000, column size: 3.925 MiB
|
||||
0 addr 86.182 TiB size 8.000 KiB access 0 % age 14.900 s
|
||||
1 addr 86.182 TiB size 8.000 KiB access 60 % age 0 ns
|
||||
2 addr 86.182 TiB size 3.422 MiB access 0 % age 4.100 s
|
||||
3 addr 86.182 TiB size 2.004 MiB access 95 % age 2.200 s
|
||||
4 addr 86.182 TiB size 29.688 MiB access 0 % age 14.100 s
|
||||
5 addr 86.182 TiB size 29.516 MiB access 0 % age 16.700 s
|
||||
6 addr 86.182 TiB size 29.633 MiB access 0 % age 17.900 s
|
||||
7 addr 86.182 TiB size 117.652 MiB access 0 % age 18.400 s
|
||||
8 addr 126.990 TiB size 62.332 MiB access 0 % age 9.500 s
|
||||
9 addr 126.990 TiB size 13.980 MiB access 0 % age 5.200 s
|
||||
10 addr 126.990 TiB size 9.539 MiB access 100 % age 3.700 s
|
||||
11 addr 126.990 TiB size 16.098 MiB access 0 % age 6.400 s
|
||||
12 addr 127.987 TiB size 132.000 KiB access 0 % age 2.900 s
|
||||
total size: 314.008 MiB
|
||||
$ sudo damo stop
|
||||
|
||||
The first command of the above example downloads and builds an artificial
|
||||
memory access generator program called ``masim``. The second command asks DAMO
|
||||
to execute the artificial generator process start via the given command and
|
||||
make DAMON monitors the generator process. The third command retrieves the
|
||||
current snapshot of the monitored access pattern of the process from DAMON and
|
||||
shows the pattern in a human readable format.
|
||||
to start the program via the given command and make DAMON monitors the newly
|
||||
started process. The third command retrieves the current snapshot of the
|
||||
monitored access pattern of the process from DAMON and shows the pattern in a
|
||||
human readable format.
|
||||
|
||||
Each line of the output shows which virtual address range (``addr [XX, XX)``)
|
||||
of the process is how frequently (``access XX %``) accessed for how long time
|
||||
(``age XX``). For example, the fifth region of ~9 MiB size is being most
|
||||
frequently accessed for last 300 milliseconds. Finally, the fourth command
|
||||
stops DAMON.
|
||||
The first line of the output shows the relative access temperature (hotness) of
|
||||
the regions in a single row hetmap format. Each column on the heatmap
|
||||
represents regions of same size on the monitored virtual address space. The
|
||||
position of the colun on the row and the number on the column represents the
|
||||
relative location and access temperature of the region. ``[...]`` means
|
||||
unmapped huge regions on the virtual address spaces. The second line shows
|
||||
additional information for better understanding the heatmap.
|
||||
|
||||
Each line of the output from the third line shows which virtual address range
|
||||
(``addr XX size XX``) of the process is how frequently (``access XX %``)
|
||||
accessed for how long time (``age XX``). For example, the evelenth region of
|
||||
~9.5 MiB size is being most frequently accessed for last 3.7 seconds. Finally,
|
||||
the fourth command stops DAMON.
|
||||
|
||||
Note that DAMON can monitor not only virtual address spaces but multiple types
|
||||
of address spaces including the physical address space.
|
||||
@@ -95,7 +108,7 @@ Visualizing Recorded Patterns
|
||||
You can visualize the pattern in a heatmap, showing which memory region
|
||||
(x-axis) got accessed when (y-axis) and how frequently (number).::
|
||||
|
||||
$ sudo damo report heats --heatmap stdout
|
||||
$ sudo damo report heatmap
|
||||
22222222222222222222222222222222222222211111111111111111111111111111111111111100
|
||||
44444444444444444444444444444444444444434444444444444444444444444444444444443200
|
||||
44444444444444444444444444444444444444433444444444444444444444444444444444444200
|
||||
@@ -160,6 +173,6 @@ Data Access Pattern Aware Memory Management
|
||||
Below command makes every memory region of size >=4K that has not accessed for
|
||||
>=60 seconds in your workload to be swapped out. ::
|
||||
|
||||
$ sudo damo schemes --damos_access_rate 0 0 --damos_sz_region 4K max \
|
||||
--damos_age 60s max --damos_action pageout \
|
||||
<pid of your workload>
|
||||
$ sudo damo start --damos_access_rate 0 0 --damos_sz_region 4K max \
|
||||
--damos_age 60s max --damos_action pageout \
|
||||
<pid of your workload>
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
.. SPDX-License-Identifier: GPL-2.0
|
||||
|
||||
===================================
|
||||
Data Access Monitoring Results Stat
|
||||
===================================
|
||||
|
||||
Data Access Monitoring Results Stat (DAMON_STAT) is a static kernel module that
|
||||
is aimed to be used for simple access pattern monitoring. It monitors accesses
|
||||
on the system's entire physical memory using DAMON, and provides simplified
|
||||
access monitoring results statistics, namely idle time percentiles and
|
||||
estimated memory bandwidth.
|
||||
|
||||
Monitoring Accuracy and Overhead
|
||||
================================
|
||||
|
||||
DAMON_STAT uses monitoring intervals :ref:`auto-tuning
|
||||
<damon_design_monitoring_intervals_autotuning>` to make its accuracy high and
|
||||
overhead minimum. It auto-tunes the intervals aiming 4 % of observable access
|
||||
events to be captured in each snapshot, while limiting the resulting sampling
|
||||
events to be 5 milliseconds in minimum and 10 seconds in maximum. On a few
|
||||
production server systems, it resulted in consuming only 0.x % single CPU time,
|
||||
while capturing reasonable quality of access patterns.
|
||||
|
||||
Interface: Module Parameters
|
||||
============================
|
||||
|
||||
To use this feature, you should first ensure your system is running on a kernel
|
||||
that is built with ``CONFIG_DAMON_STAT=y``. The feature can be enabled by
|
||||
default at build time, by setting ``CONFIG_DAMON_STAT_ENABLED_DEFAULT`` true.
|
||||
|
||||
To let sysadmins enable or disable it at boot and/or runtime, and read the
|
||||
monitoring results, DAMON_STAT provides module parameters. Following
|
||||
sections are descriptions of the parameters.
|
||||
|
||||
enabled
|
||||
-------
|
||||
|
||||
Enable or disable DAMON_STAT.
|
||||
|
||||
You can enable DAMON_STAT by setting the value of this parameter as ``Y``.
|
||||
Setting it as ``N`` disables DAMON_STAT. The default value is set by
|
||||
``CONFIG_DAMON_STAT_ENABLED_DEFAULT`` build config option.
|
||||
|
||||
estimated_memory_bandwidth
|
||||
--------------------------
|
||||
|
||||
Estimated memory bandwidth consumption (bytes per second) of the system.
|
||||
|
||||
DAMON_STAT reads observed access events on the current DAMON results snapshot
|
||||
and converts it to memory bandwidth consumption estimation in bytes per second.
|
||||
The resulting metric is exposed to user via this read-only parameter. Because
|
||||
DAMON uses sampling, this is only an estimation of the access intensity rather
|
||||
than accurate memory bandwidth.
|
||||
|
||||
memory_idle_ms_percentiles
|
||||
--------------------------
|
||||
|
||||
Per-byte idle time (milliseconds) percentiles of the system.
|
||||
|
||||
DAMON_STAT calculates how long each byte of the memory was not accessed until
|
||||
now (idle time), based on the current DAMON results snapshot. If DAMON found a
|
||||
region of access frequency (nr_accesses) larger than zero, every byte of the
|
||||
region gets zero idle time. If a region has zero access frequency
|
||||
(nr_accesses), how long the region was keeping the zero access frequency (age)
|
||||
becomes the idle time of every byte of the region. Then, DAMON_STAT exposes
|
||||
the percentiles of the idle time values via this read-only parameter. Reading
|
||||
the parameter returns 101 idle time values in milliseconds, separated by comma.
|
||||
Each value represents 0-th, 1st, 2nd, 3rd, ..., 99th and 100th percentile idle
|
||||
times.
|
||||
@@ -59,11 +59,12 @@ comma (",").
|
||||
|
||||
:ref:`/sys/kernel/mm/damon <sysfs_root>`/admin
|
||||
│ :ref:`kdamonds <sysfs_kdamonds>`/nr_kdamonds
|
||||
│ │ :ref:`0 <sysfs_kdamond>`/state,pid
|
||||
│ │ :ref:`0 <sysfs_kdamond>`/state,pid,refresh_ms
|
||||
│ │ │ :ref:`contexts <sysfs_contexts>`/nr_contexts
|
||||
│ │ │ │ :ref:`0 <sysfs_context>`/avail_operations,operations
|
||||
│ │ │ │ │ :ref:`monitoring_attrs <sysfs_monitoring_attrs>`/
|
||||
│ │ │ │ │ │ intervals/sample_us,aggr_us,update_us
|
||||
│ │ │ │ │ │ │ intervals_goal/access_bp,aggrs,min_sample_us,max_sample_us
|
||||
│ │ │ │ │ │ nr_regions/min,max
|
||||
│ │ │ │ │ :ref:`targets <sysfs_targets>`/nr_targets
|
||||
│ │ │ │ │ │ :ref:`0 <sysfs_target>`/pid_target
|
||||
@@ -80,10 +81,12 @@ comma (",").
|
||||
│ │ │ │ │ │ │ :ref:`quotas <sysfs_quotas>`/ms,bytes,reset_interval_ms,effective_bytes
|
||||
│ │ │ │ │ │ │ │ weights/sz_permil,nr_accesses_permil,age_permil
|
||||
│ │ │ │ │ │ │ │ :ref:`goals <sysfs_schemes_quota_goals>`/nr_goals
|
||||
│ │ │ │ │ │ │ │ │ 0/target_metric,target_value,current_value
|
||||
│ │ │ │ │ │ │ │ │ 0/target_metric,target_value,current_value,nid
|
||||
│ │ │ │ │ │ │ :ref:`watermarks <sysfs_watermarks>`/metric,interval_us,high,mid,low
|
||||
│ │ │ │ │ │ │ :ref:`filters <sysfs_filters>`/nr_filters
|
||||
│ │ │ │ │ │ │ │ 0/type,matching,memcg_id,allow
|
||||
│ │ │ │ │ │ │ :ref:`{core_,ops_,}filters <sysfs_filters>`/nr_filters
|
||||
│ │ │ │ │ │ │ │ 0/type,matching,allow,memcg_path,addr_start,addr_end,target_idx,min,max
|
||||
│ │ │ │ │ │ │ :ref:`dests <damon_sysfs_dests>`/nr_dests
|
||||
│ │ │ │ │ │ │ │ 0/id,weight
|
||||
│ │ │ │ │ │ │ :ref:`stats <sysfs_schemes_stats>`/nr_tried,sz_tried,nr_applied,sz_applied,sz_ops_filter_passed,qt_exceeds
|
||||
│ │ │ │ │ │ │ :ref:`tried_regions <sysfs_schemes_tried_regions>`/total_bytes
|
||||
│ │ │ │ │ │ │ │ 0/start,end,nr_accesses,age,sz_filter_passed
|
||||
@@ -120,8 +123,8 @@ kdamond.
|
||||
kdamonds/<N>/
|
||||
-------------
|
||||
|
||||
In each kdamond directory, two files (``state`` and ``pid``) and one directory
|
||||
(``contexts``) exist.
|
||||
In each kdamond directory, three files (``state``, ``pid`` and ``refresh_ms``)
|
||||
and one directory (``contexts``) exist.
|
||||
|
||||
Reading ``state`` returns ``on`` if the kdamond is currently running, or
|
||||
``off`` if it is not running.
|
||||
@@ -132,6 +135,11 @@ Users can write below commands for the kdamond to the ``state`` file.
|
||||
- ``off``: Stop running.
|
||||
- ``commit``: Read the user inputs in the sysfs files except ``state`` file
|
||||
again.
|
||||
- ``update_tuned_intervals``: Update the contents of ``sample_us`` and
|
||||
``aggr_us`` files of the kdamond with the auto-tuning applied ``sampling
|
||||
interval`` and ``aggregation interval`` for the files. Please refer to
|
||||
:ref:`intervals_goal section <damon_usage_sysfs_monitoring_intervals_goal>`
|
||||
for more details.
|
||||
- ``commit_schemes_quota_goals``: Read the DAMON-based operation schemes'
|
||||
:ref:`quota goals <sysfs_schemes_quota_goals>`.
|
||||
- ``update_schemes_stats``: Update the contents of stats files for each
|
||||
@@ -153,6 +161,13 @@ Users can write below commands for the kdamond to the ``state`` file.
|
||||
|
||||
If the state is ``on``, reading ``pid`` shows the pid of the kdamond thread.
|
||||
|
||||
Users can ask the kernel to periodically update files showing auto-tuned
|
||||
parameters and DAMOS stats instead of manually writing
|
||||
``update_tuned_intervals`` like keywords to ``state`` file. For this, users
|
||||
should write the desired update time interval in milliseconds to ``refresh_ms``
|
||||
file. If the interval is zero, the periodic update is disabled. Reading the
|
||||
file shows currently set time interval.
|
||||
|
||||
``contexts`` directory contains files for controlling the monitoring contexts
|
||||
that this kdamond will execute.
|
||||
|
||||
@@ -213,6 +228,25 @@ writing to and rading from the files.
|
||||
For more details about the intervals and monitoring regions range, please refer
|
||||
to the Design document (:doc:`/mm/damon/design`).
|
||||
|
||||
.. _damon_usage_sysfs_monitoring_intervals_goal:
|
||||
|
||||
contexts/<N>/monitoring_attrs/intervals/intervals_goal/
|
||||
-------------------------------------------------------
|
||||
|
||||
Under the ``intervals`` directory, one directory for automated tuning of
|
||||
``sample_us`` and ``aggr_us``, namely ``intervals_goal`` directory also exists.
|
||||
Under the directory, four files for the auto-tuning control, namely
|
||||
``access_bp``, ``aggrs``, ``min_sample_us`` and ``max_sample_us`` exist.
|
||||
Please refer to the :ref:`design document of the feature
|
||||
<damon_design_monitoring_intervals_autotuning>` for the internal of the tuning
|
||||
mechanism. Reading and writing the four files under ``intervals_goal``
|
||||
directory shows and updates the tuning parameters that described in the
|
||||
:ref:design doc <damon_design_monitoring_intervals_autotuning>` with the same
|
||||
names. The tuning starts with the user-set ``sample_us`` and ``aggr_us``. The
|
||||
tuning-applied current values of the two intervals can be read from the
|
||||
``sample_us`` and ``aggr_us`` files after writing ``update_tuned_intervals`` to
|
||||
the ``state`` file.
|
||||
|
||||
.. _sysfs_targets:
|
||||
|
||||
contexts/<N>/targets/
|
||||
@@ -282,9 +316,10 @@ to ``N-1``. Each directory represents each DAMON-based operation scheme.
|
||||
schemes/<N>/
|
||||
------------
|
||||
|
||||
In each scheme directory, five directories (``access_pattern``, ``quotas``,
|
||||
``watermarks``, ``filters``, ``stats``, and ``tried_regions``) and three files
|
||||
(``action``, ``target_nid`` and ``apply_interval``) exist.
|
||||
In each scheme directory, eight directories (``access_pattern``, ``quotas``,
|
||||
``watermarks``, ``core_filters``, ``ops_filters``, ``filters``, ``dests``,
|
||||
``stats``, and ``tried_regions``) and three files (``action``, ``target_nid``
|
||||
and ``apply_interval``) exist.
|
||||
|
||||
The ``action`` file is for setting and getting the scheme's :ref:`action
|
||||
<damon_design_damos_action>`. The keywords that can be written to and read
|
||||
@@ -364,11 +399,11 @@ number (``N``) to the file creates the number of child directories named ``0``
|
||||
to ``N-1``. Each directory represents each goal and current achievement.
|
||||
Among the multiple feedback, the best one is used.
|
||||
|
||||
Each goal directory contains three files, namely ``target_metric``,
|
||||
``target_value`` and ``current_value``. Users can set and get the three
|
||||
parameters for the quota auto-tuning goals that specified on the :ref:`design
|
||||
doc <damon_design_damos_quotas_auto_tuning>` by writing to and reading from each
|
||||
of the files. Note that users should further write
|
||||
Each goal directory contains four files, namely ``target_metric``,
|
||||
``target_value``, ``current_value`` and ``nid``. Users can set and get the
|
||||
four parameters for the quota auto-tuning goals that specified on the
|
||||
:ref:`design doc <damon_design_damos_quotas_auto_tuning>` by writing to and
|
||||
reading from each of the files. Note that users should further write
|
||||
``commit_schemes_quota_goals`` to the ``state`` file of the :ref:`kdamond
|
||||
directory <sysfs_kdamond>` to pass the feedback to DAMON.
|
||||
|
||||
@@ -395,33 +430,43 @@ The ``interval`` should written in microseconds unit.
|
||||
|
||||
.. _sysfs_filters:
|
||||
|
||||
schemes/<N>/filters/
|
||||
--------------------
|
||||
schemes/<N>/{core\_,ops\_,}filters/
|
||||
-----------------------------------
|
||||
|
||||
The directory for the :ref:`filters <damon_design_damos_filters>` of the given
|
||||
Directories for :ref:`filters <damon_design_damos_filters>` of the given
|
||||
DAMON-based operation scheme.
|
||||
|
||||
In the beginning, this directory has only one file, ``nr_filters``. Writing a
|
||||
``core_filters`` and ``ops_filters`` directories are for the filters handled by
|
||||
the DAMON core layer and operations set layer, respectively. ``filters``
|
||||
directory can be used for installing filters regardless of their handled
|
||||
layers. Filters that requested by ``core_filters`` and ``ops_filters`` will be
|
||||
installed before those of ``filters``. All three directories have same files.
|
||||
|
||||
Use of ``filters`` directory can make expecting evaluation orders of given
|
||||
filters with the files under directory bit confusing. Users are hence
|
||||
recommended to use ``core_filters`` and ``ops_filters`` directories. The
|
||||
``filters`` directory could be deprecated in future.
|
||||
|
||||
In the beginning, the directory has only one file, ``nr_filters``. Writing a
|
||||
number (``N``) to the file creates the number of child directories named ``0``
|
||||
to ``N-1``. Each directory represents each filter. The filters are evaluated
|
||||
in the numeric order.
|
||||
|
||||
Each filter directory contains seven files, namely ``type``, ``matching``,
|
||||
``allow``, ``memcg_path``, ``addr_start``, ``addr_end``, and ``target_idx``.
|
||||
To ``type`` file, you can write one of five special keywords: ``anon`` for
|
||||
anonymous pages, ``memcg`` for specific memory cgroup, ``young`` for young
|
||||
pages, ``addr`` for specific address range (an open-ended interval), or
|
||||
``target`` for specific DAMON monitoring target filtering. Meaning of the
|
||||
types are same to the description on the :ref:`design doc
|
||||
<damon_design_damos_filters>`.
|
||||
Each filter directory contains nine files, namely ``type``, ``matching``,
|
||||
``allow``, ``memcg_path``, ``addr_start``, ``addr_end``, ``min``, ``max``
|
||||
and ``target_idx``. To ``type`` file, you can write the type of the filter.
|
||||
Refer to :ref:`the design doc <damon_design_damos_filters>` for available type
|
||||
names, their meaning and on what layer those are handled.
|
||||
|
||||
In case of the memory cgroup filtering, you can specify the memory cgroup of
|
||||
the interest by writing the path of the memory cgroup from the cgroups mount
|
||||
point to ``memcg_path`` file. In case of the address range filtering, you can
|
||||
specify the start and end address of the range to ``addr_start`` and
|
||||
``addr_end`` files, respectively. For the DAMON monitoring target filtering,
|
||||
you can specify the index of the target between the list of the DAMON context's
|
||||
monitoring targets list to ``target_idx`` file.
|
||||
For ``memcg`` type, you can specify the memory cgroup of the interest by
|
||||
writing the path of the memory cgroup from the cgroups mount point to
|
||||
``memcg_path`` file. For ``addr`` type, you can specify the start and end
|
||||
address of the range (open-ended interval) to ``addr_start`` and ``addr_end``
|
||||
files, respectively. For ``hugepage_size`` type, you can specify the minimum
|
||||
and maximum size of the range (closed interval) to ``min`` and ``max`` files,
|
||||
respectively. For ``target`` type, you can specify the index of the target
|
||||
between the list of the DAMON context's monitoring targets list to
|
||||
``target_idx`` file.
|
||||
|
||||
You can write ``Y`` or ``N`` to ``matching`` file to specify whether the filter
|
||||
is for memory that matches the ``type``. You can write ``Y`` or ``N`` to
|
||||
@@ -431,6 +476,7 @@ the ``type`` and ``matching`` should be allowed or not.
|
||||
For example, below restricts a DAMOS action to be applied to only non-anonymous
|
||||
pages of all memory cgroups except ``/having_care_already``.::
|
||||
|
||||
# cd ops_filters/0/
|
||||
# echo 2 > nr_filters
|
||||
# # disallow anonymous pages
|
||||
echo anon > 0/type
|
||||
@@ -447,6 +493,29 @@ Refer to the :ref:`DAMOS filters design documentation
|
||||
of different ``allow`` works, when each of the filters are supported, and
|
||||
differences on stats.
|
||||
|
||||
.. _damon_sysfs_dests:
|
||||
|
||||
schemes/<N>/dests/
|
||||
------------------
|
||||
|
||||
Directory for specifying the destinations of given DAMON-based operation
|
||||
scheme's action. This directory is ignored if the action of the given scheme
|
||||
is not supporting multiple destinations. Only ``DAMOS_MIGRATE_{HOT,COLD}``
|
||||
actions are supporting multiple destinations.
|
||||
|
||||
In the beginning, the directory has only one file, ``nr_dests``. Writing a
|
||||
number (``N``) to the file creates the number of child directories named ``0``
|
||||
to ``N-1``. Each directory represents each action destination.
|
||||
|
||||
Each destination directory contains two files, namely ``id`` and ``weight``.
|
||||
Users can write and read the identifier of the destination to ``id`` file.
|
||||
For ``DAMOS_MIGRATE_{HOT,COLD}`` actions, the migrate destination node's node
|
||||
id should be written to ``id`` file. Users can write and read the weight of
|
||||
the destination among the given destinations to the ``weight`` file. The
|
||||
weight can be an arbitrary integer. When DAMOS apply the action to each entity
|
||||
of the memory region, it will select the destination of the action based on the
|
||||
relative weights of the destinations.
|
||||
|
||||
.. _sysfs_schemes_stats:
|
||||
|
||||
schemes/<N>/stats/
|
||||
|
||||
@@ -37,6 +37,7 @@ the Linux memory management.
|
||||
numaperf
|
||||
pagemap
|
||||
shrinker_debugfs
|
||||
slab
|
||||
soft-dirty
|
||||
swap_numa
|
||||
transhuge
|
||||
|
||||
@@ -1,13 +1,12 @@
|
||||
==========================
|
||||
Short users guide for SLUB
|
||||
==========================
|
||||
========================================
|
||||
Short users guide for the slab allocator
|
||||
========================================
|
||||
|
||||
The basic philosophy of SLUB is very different from SLAB. SLAB
|
||||
requires rebuilding the kernel to activate debug options for all
|
||||
slab caches. SLUB always includes full debugging but it is off by default.
|
||||
SLUB can enable debugging only for selected slabs in order to avoid
|
||||
an impact on overall system performance which may make a bug more
|
||||
difficult to find.
|
||||
The slab allocator includes full debugging support (when built with
|
||||
CONFIG_SLUB_DEBUG=y) but it is off by default (unless built with
|
||||
CONFIG_SLUB_DEBUG_ON=y). You can enable debugging only for selected
|
||||
slabs in order to avoid an impact on overall system performance which
|
||||
may make a bug more difficult to find.
|
||||
|
||||
In order to switch debugging on one can add an option ``slab_debug``
|
||||
to the kernel command line. That will enable full debugging for
|
||||
@@ -207,7 +207,7 @@ a device with limitations, it needs to be decreased.
|
||||
|
||||
Special note about PCI: PCI-X specification requires PCI-X devices to support
|
||||
64-bit addressing (DAC) for all transactions. And at least one platform (SGI
|
||||
SN2) requires 64-bit consistent allocations to operate correctly when the IO
|
||||
SN2) requires 64-bit coherent allocations to operate correctly when the IO
|
||||
bus is in PCI-X mode.
|
||||
|
||||
For correct operation, you must set the DMA mask to inform the kernel about
|
||||
@@ -226,7 +226,7 @@ used instead:
|
||||
|
||||
int dma_set_mask(struct device *dev, u64 mask);
|
||||
|
||||
The setup for consistent allocations is performed via a call
|
||||
The setup for coherent allocations is performed via a call
|
||||
to dma_set_coherent_mask()::
|
||||
|
||||
int dma_set_coherent_mask(struct device *dev, u64 mask);
|
||||
@@ -293,7 +293,7 @@ it would look like this::
|
||||
|
||||
The coherent mask will always be able to set the same or a smaller mask as
|
||||
the streaming mask. However for the rare case that a device driver only
|
||||
uses consistent allocations, one would have to check the return value from
|
||||
uses coherent allocations, one would have to check the return value from
|
||||
dma_set_coherent_mask().
|
||||
|
||||
Finally, if your device can only drive the low 24-bits of
|
||||
@@ -350,20 +350,20 @@ Types of DMA mappings
|
||||
|
||||
There are two types of DMA mappings:
|
||||
|
||||
- Consistent DMA mappings which are usually mapped at driver
|
||||
- Coherent DMA mappings which are usually mapped at driver
|
||||
initialization, unmapped at the end and for which the hardware should
|
||||
guarantee that the device and the CPU can access the data
|
||||
in parallel and will see updates made by each other without any
|
||||
explicit software flushing.
|
||||
|
||||
Think of "consistent" as "synchronous" or "coherent".
|
||||
Think of "coherent" as "synchronous".
|
||||
|
||||
The current default is to return consistent memory in the low 32
|
||||
The current default is to return coherent memory in the low 32
|
||||
bits of the DMA space. However, for future compatibility you should
|
||||
set the consistent mask even if this default is fine for your
|
||||
set the coherent mask even if this default is fine for your
|
||||
driver.
|
||||
|
||||
Good examples of what to use consistent mappings for are:
|
||||
Good examples of what to use coherent mappings for are:
|
||||
|
||||
- Network card DMA ring descriptors.
|
||||
- SCSI adapter mailbox command data structures.
|
||||
@@ -372,13 +372,13 @@ There are two types of DMA mappings:
|
||||
|
||||
The invariant these examples all require is that any CPU store
|
||||
to memory is immediately visible to the device, and vice
|
||||
versa. Consistent mappings guarantee this.
|
||||
versa. Coherent mappings guarantee this.
|
||||
|
||||
.. important::
|
||||
|
||||
Consistent DMA memory does not preclude the usage of
|
||||
Coherent DMA memory does not preclude the usage of
|
||||
proper memory barriers. The CPU may reorder stores to
|
||||
consistent memory just as it may normal memory. Example:
|
||||
coherent memory just as it may normal memory. Example:
|
||||
if it is important for the device to see the first word
|
||||
of a descriptor updated before the second, you must do
|
||||
something like::
|
||||
@@ -417,10 +417,10 @@ Also, systems with caches that aren't DMA-coherent will work better
|
||||
when the underlying buffers don't share cache lines with other data.
|
||||
|
||||
|
||||
Using Consistent DMA mappings
|
||||
=============================
|
||||
Using Coherent DMA mappings
|
||||
===========================
|
||||
|
||||
To allocate and map large (PAGE_SIZE or so) consistent DMA regions,
|
||||
To allocate and map large (PAGE_SIZE or so) coherent DMA regions,
|
||||
you should do::
|
||||
|
||||
dma_addr_t dma_handle;
|
||||
@@ -437,10 +437,10 @@ __get_free_pages() (but takes size instead of a page order). If your
|
||||
driver needs regions sized smaller than a page, you may prefer using
|
||||
the dma_pool interface, described below.
|
||||
|
||||
The consistent DMA mapping interfaces, will by default return a DMA address
|
||||
The coherent DMA mapping interfaces, will by default return a DMA address
|
||||
which is 32-bit addressable. Even if the device indicates (via the DMA mask)
|
||||
that it may address the upper 32-bits, consistent allocation will only
|
||||
return > 32-bit addresses for DMA if the consistent DMA mask has been
|
||||
that it may address the upper 32-bits, coherent allocation will only
|
||||
return > 32-bit addresses for DMA if the coherent DMA mask has been
|
||||
explicitly changed via dma_set_coherent_mask(). This is true of the
|
||||
dma_pool interface as well.
|
||||
|
||||
@@ -549,7 +549,7 @@ program address space. Such platforms can and do report errors in the
|
||||
kernel logs when the DMA controller hardware detects violation of the
|
||||
permission setting.
|
||||
|
||||
Only streaming mappings specify a direction, consistent mappings
|
||||
Only streaming mappings specify a direction, coherent mappings
|
||||
implicitly have a direction attribute setting of
|
||||
DMA_BIDIRECTIONAL.
|
||||
|
||||
|
||||
@@ -8,15 +8,15 @@ This document describes the DMA API. For a more gentle introduction
|
||||
of the API (and actual examples), see Documentation/core-api/dma-api-howto.rst.
|
||||
|
||||
This API is split into two pieces. Part I describes the basic API.
|
||||
Part II describes extensions for supporting non-consistent memory
|
||||
Part II describes extensions for supporting non-coherent memory
|
||||
machines. Unless you know that your driver absolutely has to support
|
||||
non-consistent platforms (this is usually only legacy platforms) you
|
||||
non-coherent platforms (this is usually only legacy platforms) you
|
||||
should only use the API described in part I.
|
||||
|
||||
Part I - dma_API
|
||||
Part I - DMA API
|
||||
----------------
|
||||
|
||||
To get the dma_API, you must #include <linux/dma-mapping.h>. This
|
||||
To get the DMA API, you must #include <linux/dma-mapping.h>. This
|
||||
provides dma_addr_t and the interfaces described below.
|
||||
|
||||
A dma_addr_t can hold any valid DMA address for the platform. It can be
|
||||
@@ -33,13 +33,13 @@ Part Ia - Using large DMA-coherent buffers
|
||||
dma_alloc_coherent(struct device *dev, size_t size,
|
||||
dma_addr_t *dma_handle, gfp_t flag)
|
||||
|
||||
Consistent memory is memory for which a write by either the device or
|
||||
Coherent memory is memory for which a write by either the device or
|
||||
the processor can immediately be read by the processor or device
|
||||
without having to worry about caching effects. (You may however need
|
||||
to make sure to flush the processor's write buffers before telling
|
||||
devices to read that memory.)
|
||||
|
||||
This routine allocates a region of <size> bytes of consistent memory.
|
||||
This routine allocates a region of <size> bytes of coherent memory.
|
||||
|
||||
It returns a pointer to the allocated region (in the processor's virtual
|
||||
address space) or NULL if the allocation failed.
|
||||
@@ -48,15 +48,14 @@ It also returns a <dma_handle> which may be cast to an unsigned integer the
|
||||
same width as the bus and given to the device as the DMA address base of
|
||||
the region.
|
||||
|
||||
Note: consistent memory can be expensive on some platforms, and the
|
||||
Note: coherent memory can be expensive on some platforms, and the
|
||||
minimum allocation length may be as big as a page, so you should
|
||||
consolidate your requests for consistent memory as much as possible.
|
||||
consolidate your requests for coherent memory as much as possible.
|
||||
The simplest way to do that is to use the dma_pool calls (see below).
|
||||
|
||||
The flag parameter (dma_alloc_coherent() only) allows the caller to
|
||||
specify the ``GFP_`` flags (see kmalloc()) for the allocation (the
|
||||
implementation may choose to ignore flags that affect the location of
|
||||
the returned memory, like GFP_DMA).
|
||||
The flag parameter allows the caller to specify the ``GFP_`` flags (see
|
||||
kmalloc()) for the allocation (the implementation may ignore flags that affect
|
||||
the location of the returned memory, like GFP_DMA).
|
||||
|
||||
::
|
||||
|
||||
@@ -64,19 +63,18 @@ the returned memory, like GFP_DMA).
|
||||
dma_free_coherent(struct device *dev, size_t size, void *cpu_addr,
|
||||
dma_addr_t dma_handle)
|
||||
|
||||
Free a region of consistent memory you previously allocated. dev,
|
||||
size and dma_handle must all be the same as those passed into
|
||||
dma_alloc_coherent(). cpu_addr must be the virtual address returned by
|
||||
the dma_alloc_coherent().
|
||||
Free a previously allocated region of coherent memory. dev, size and dma_handle
|
||||
must all be the same as those passed into dma_alloc_coherent(). cpu_addr must
|
||||
be the virtual address returned by dma_alloc_coherent().
|
||||
|
||||
Note that unlike their sibling allocation calls, these routines
|
||||
may only be called with IRQs enabled.
|
||||
Note that unlike the sibling allocation call, this routine may only be called
|
||||
with IRQs enabled.
|
||||
|
||||
|
||||
Part Ib - Using small DMA-coherent buffers
|
||||
------------------------------------------
|
||||
|
||||
To get this part of the dma_API, you must #include <linux/dmapool.h>
|
||||
To get this part of the DMA API, you must #include <linux/dmapool.h>
|
||||
|
||||
Many drivers need lots of small DMA-coherent memory regions for DMA
|
||||
descriptors or I/O buffers. Rather than allocating in units of a page
|
||||
@@ -246,9 +244,7 @@ Part Id - Streaming DMA mappings
|
||||
Maps a piece of processor virtual memory so it can be accessed by the
|
||||
device and returns the DMA address of the memory.
|
||||
|
||||
The direction for both APIs may be converted freely by casting.
|
||||
However the dma_API uses a strongly typed enumerator for its
|
||||
direction:
|
||||
The DMA API uses a strongly typed enumerator for its direction:
|
||||
|
||||
======================= =============================================
|
||||
DMA_NONE no direction (used for debugging)
|
||||
@@ -325,8 +321,7 @@ DMA_BIDIRECTIONAL direction isn't known
|
||||
enum dma_data_direction direction)
|
||||
|
||||
Unmaps the region previously mapped. All the parameters passed in
|
||||
must be identical to those passed in (and returned) by the mapping
|
||||
API.
|
||||
must be identical to those passed to (and returned by) dma_map_single().
|
||||
|
||||
::
|
||||
|
||||
@@ -775,19 +770,19 @@ memory or doing partial flushes.
|
||||
of two for easy alignment.
|
||||
|
||||
|
||||
Part III - Debug drivers use of the DMA-API
|
||||
Part III - Debug drivers use of the DMA API
|
||||
-------------------------------------------
|
||||
|
||||
The DMA-API as described above has some constraints. DMA addresses must be
|
||||
The DMA API as described above has some constraints. DMA addresses must be
|
||||
released with the corresponding function with the same size for example. With
|
||||
the advent of hardware IOMMUs it becomes more and more important that drivers
|
||||
do not violate those constraints. In the worst case such a violation can
|
||||
result in data corruption up to destroyed filesystems.
|
||||
|
||||
To debug drivers and find bugs in the usage of the DMA-API checking code can
|
||||
To debug drivers and find bugs in the usage of the DMA API checking code can
|
||||
be compiled into the kernel which will tell the developer about those
|
||||
violations. If your architecture supports it you can select the "Enable
|
||||
debugging of DMA-API usage" option in your kernel configuration. Enabling this
|
||||
debugging of DMA API usage" option in your kernel configuration. Enabling this
|
||||
option has a performance impact. Do not enable it in production kernels.
|
||||
|
||||
If you boot the resulting kernel will contain code which does some bookkeeping
|
||||
@@ -826,7 +821,7 @@ example warning message may look like this::
|
||||
<EOI> <4>---[ end trace f6435a98e2a38c0e ]---
|
||||
|
||||
The driver developer can find the driver and the device including a stacktrace
|
||||
of the DMA-API call which caused this warning.
|
||||
of the DMA API call which caused this warning.
|
||||
|
||||
Per default only the first error will result in a warning message. All other
|
||||
errors will only silently counted. This limitation exist to prevent the code
|
||||
@@ -834,7 +829,7 @@ from flooding your kernel log. To support debugging a device driver this can
|
||||
be disabled via debugfs. See the debugfs interface documentation below for
|
||||
details.
|
||||
|
||||
The debugfs directory for the DMA-API debugging code is called dma-api/. In
|
||||
The debugfs directory for the DMA API debugging code is called dma-api/. In
|
||||
this directory the following files can currently be found:
|
||||
|
||||
=============================== ===============================================
|
||||
@@ -882,7 +877,7 @@ dma-api/driver_filter You can write a name of a driver into this file
|
||||
|
||||
If you have this code compiled into your kernel it will be enabled by default.
|
||||
If you want to boot without the bookkeeping anyway you can provide
|
||||
'dma_debug=off' as a boot parameter. This will disable DMA-API debugging.
|
||||
'dma_debug=off' as a boot parameter. This will disable DMA API debugging.
|
||||
Notice that you can not enable it again at runtime. You have to reboot to do
|
||||
so.
|
||||
|
||||
|
||||
@@ -9,6 +9,9 @@ Memory hotplug event notifier
|
||||
|
||||
Hotplugging events are sent to a notification queue.
|
||||
|
||||
Memory notifier
|
||||
----------------
|
||||
|
||||
There are six types of notification defined in ``include/linux/memory.h``:
|
||||
|
||||
MEM_GOING_ONLINE
|
||||
@@ -56,20 +59,18 @@ The third argument (arg) passes a pointer of struct memory_notify::
|
||||
struct memory_notify {
|
||||
unsigned long start_pfn;
|
||||
unsigned long nr_pages;
|
||||
int status_change_nid_normal;
|
||||
int status_change_nid;
|
||||
}
|
||||
|
||||
- start_pfn is start_pfn of online/offline memory.
|
||||
- nr_pages is # of pages of online/offline memory.
|
||||
- status_change_nid_normal is set node id when N_NORMAL_MEMORY of nodemask
|
||||
is (will be) set/clear, if this is -1, then nodemask status is not changed.
|
||||
- status_change_nid is set node id when N_MEMORY of nodemask is (will be)
|
||||
set/clear. It means a new(memoryless) node gets new memory by online and a
|
||||
node loses all memory. If this is -1, then nodemask status is not changed.
|
||||
|
||||
If status_changed_nid* >= 0, callback should create/discard structures for the
|
||||
node if necessary.
|
||||
It is possible to get notified for MEM_CANCEL_ONLINE without having been notified
|
||||
for MEM_GOING_ONLINE, and the same applies to MEM_CANCEL_OFFLINE and
|
||||
MEM_GOING_OFFLINE.
|
||||
This can happen when a consumer fails, meaning we break the callchain and we
|
||||
stop calling the remaining consumers of the notifier.
|
||||
It is then important that users of memory_notify make no assumptions and get
|
||||
prepared to handle such cases.
|
||||
|
||||
The callback routine shall return one of the values
|
||||
NOTIFY_DONE, NOTIFY_OK, NOTIFY_BAD, NOTIFY_STOP
|
||||
@@ -83,6 +84,78 @@ further processing of the notification queue.
|
||||
|
||||
NOTIFY_STOP stops further processing of the notification queue.
|
||||
|
||||
Numa node notifier
|
||||
------------------
|
||||
|
||||
There are six types of notification defined in ``include/linux/node.h``:
|
||||
|
||||
NODE_ADDING_FIRST_MEMORY
|
||||
Generated before memory becomes available to this node for the first time.
|
||||
|
||||
NODE_CANCEL_ADDING_FIRST_MEMORY
|
||||
Generated if NODE_ADDING_FIRST_MEMORY fails.
|
||||
|
||||
NODE_ADDED_FIRST_MEMORY
|
||||
Generated when memory has become available fo this node for the first time.
|
||||
|
||||
NODE_REMOVING_LAST_MEMORY
|
||||
Generated when the last memory available to this node is about to be offlined.
|
||||
|
||||
NODE_CANCEL_REMOVING_LAST_MEMORY
|
||||
Generated when NODE_CANCEL_REMOVING_LAST_MEMORY fails.
|
||||
|
||||
NODE_REMOVED_LAST_MEMORY
|
||||
Generated when the last memory available to this node has been offlined.
|
||||
|
||||
A callback routine can be registered by calling::
|
||||
|
||||
hotplug_node_notifier(callback_func, priority)
|
||||
|
||||
Callback functions with higher values of priority are called before callback
|
||||
functions with lower values.
|
||||
|
||||
A callback function must have the following prototype::
|
||||
|
||||
int callback_func(
|
||||
|
||||
struct notifier_block *self, unsigned long action, void *arg);
|
||||
|
||||
The first argument of the callback function (self) is a pointer to the block
|
||||
of the notifier chain that points to the callback function itself.
|
||||
The second argument (action) is one of the event types described above.
|
||||
The third argument (arg) passes a pointer of struct node_notify::
|
||||
|
||||
struct node_notify {
|
||||
int nid;
|
||||
}
|
||||
|
||||
- nid is the node we are adding or removing memory to.
|
||||
|
||||
It is possible to get notified for NODE_CANCEL_ADDING_FIRST_MEMORY without
|
||||
having been notified for NODE_ADDING_FIRST_MEMORY, and the same applies to
|
||||
NODE_CANCEL_REMOVING_LAST_MEMORY and NODE_REMOVING_LAST_MEMORY.
|
||||
This can happen when a consumer fails, meaning we break the callchain and we
|
||||
stop calling the remaining consumers of the notifier.
|
||||
It is then important that users of node_notify make no assumptions and get
|
||||
prepared to handle such cases.
|
||||
|
||||
The callback routine shall return one of the values
|
||||
NOTIFY_DONE, NOTIFY_OK, NOTIFY_BAD, NOTIFY_STOP
|
||||
defined in ``include/linux/notifier.h``
|
||||
|
||||
NOTIFY_DONE and NOTIFY_OK have no effect on the further processing.
|
||||
|
||||
NOTIFY_BAD is used as response to the NODE_ADDING_FIRST_MEMORY,
|
||||
NODE_REMOVING_LAST_MEMORY, NODE_ADDED_FIRST_MEMORY or
|
||||
NODE_REMOVED_LAST_MEMORY action to cancel hotplugging.
|
||||
It stops further processing of the notification queue.
|
||||
|
||||
NOTIFY_STOP stops further processing of the notification queue.
|
||||
|
||||
Please note that we should not fail for NODE_ADDED_FIRST_MEMORY /
|
||||
NODE_REMOVED_FIRST_MEMORY, as memory_hotplug code cannot rollback at that
|
||||
point anymore.
|
||||
|
||||
Locking Internals
|
||||
=================
|
||||
|
||||
|
||||
@@ -139,4 +139,3 @@ More Memory Management Functions
|
||||
.. kernel-doc:: mm/mmu_notifier.c
|
||||
.. kernel-doc:: mm/balloon_compaction.c
|
||||
.. kernel-doc:: mm/huge_memory.c
|
||||
.. kernel-doc:: mm/io-mapping.c
|
||||
|
||||
@@ -248,10 +248,10 @@ prototypes::
|
||||
int (*writepages)(struct address_space *, struct writeback_control *);
|
||||
bool (*dirty_folio)(struct address_space *, struct folio *folio);
|
||||
void (*readahead)(struct readahead_control *);
|
||||
int (*write_begin)(struct file *, struct address_space *mapping,
|
||||
int (*write_begin)(const struct kiocb *, struct address_space *mapping,
|
||||
loff_t pos, unsigned len,
|
||||
struct folio **foliop, void **fsdata);
|
||||
int (*write_end)(struct file *, struct address_space *mapping,
|
||||
int (*write_end)(const struct kiocb *, struct address_space *mapping,
|
||||
loff_t pos, unsigned len, unsigned copied,
|
||||
struct folio *folio, void *fsdata);
|
||||
sector_t (*bmap)(struct address_space *, sector_t);
|
||||
|
||||
@@ -1182,12 +1182,14 @@ SecPageTables
|
||||
Memory consumed by secondary page tables, this currently includes
|
||||
KVM mmu and IOMMU allocations on x86 and arm64.
|
||||
NFS_Unstable
|
||||
Always zero. Previous counted pages which had been written to
|
||||
Always zero. Previously counted pages which had been written to
|
||||
the server, but has not been committed to stable storage.
|
||||
Bounce
|
||||
Memory used for block device "bounce buffers"
|
||||
Always zero. Previously memory used for block device
|
||||
"bounce buffers".
|
||||
WritebackTmp
|
||||
Memory used by FUSE for temporary writeback buffers
|
||||
Always zero. Previously memory used by FUSE for temporary
|
||||
writeback buffers.
|
||||
CommitLimit
|
||||
Based on the overcommit ratio ('vm.overcommit_ratio'),
|
||||
this is the total amount of memory currently available to
|
||||
|
||||
@@ -822,10 +822,10 @@ cache in your filesystem. The following members are defined:
|
||||
int (*writepages)(struct address_space *, struct writeback_control *);
|
||||
bool (*dirty_folio)(struct address_space *, struct folio *);
|
||||
void (*readahead)(struct readahead_control *);
|
||||
int (*write_begin)(struct file *, struct address_space *mapping,
|
||||
int (*write_begin)(const struct kiocb *, struct address_space *mapping,
|
||||
loff_t pos, unsigned len,
|
||||
struct page **pagep, void **fsdata);
|
||||
int (*write_end)(struct file *, struct address_space *mapping,
|
||||
struct page **pagep, void **fsdata);
|
||||
int (*write_end)(const struct kiocb *, struct address_space *mapping,
|
||||
loff_t pos, unsigned len, unsigned copied,
|
||||
struct folio *folio, void *fsdata);
|
||||
sector_t (*bmap)(struct address_space *, sector_t);
|
||||
|
||||
@@ -236,13 +236,13 @@ SWAP Page Table Helpers
|
||||
========================
|
||||
|
||||
+---------------------------+--------------------------------------------------+
|
||||
| __pte_to_swp_entry | Creates a swapped entry (arch) from a mapped PTE |
|
||||
| __pte_to_swp_entry | Creates a swp_entry_t (arch) from a swap PTE |
|
||||
+---------------------------+--------------------------------------------------+
|
||||
| __swp_to_pte_entry | Creates a mapped PTE from a swapped entry (arch) |
|
||||
| __swp_entry_to_pte | Creates a swap PTE from a swp_entry_t (arch) |
|
||||
+---------------------------+--------------------------------------------------+
|
||||
| __pmd_to_swp_entry | Creates a swapped entry (arch) from a mapped PMD |
|
||||
| __pmd_to_swp_entry | Creates a swp_entry_t (arch) from a swap PMD |
|
||||
+---------------------------+--------------------------------------------------+
|
||||
| __swp_to_pmd_entry | Creates a mapped PMD from a swapped entry (arch) |
|
||||
| __swp_entry_to_pmd | Creates a swap PMD from a swp_entry_t (arch) |
|
||||
+---------------------------+--------------------------------------------------+
|
||||
| is_migration_entry | Tests a migration (read or write) swapped entry |
|
||||
+-------------------------------+----------------------------------------------+
|
||||
|
||||
@@ -203,6 +203,8 @@ This scheme, however, cannot preserve the quality of the output if the
|
||||
assumption is not guaranteed.
|
||||
|
||||
|
||||
.. _damon_design_adaptive_regions_adjustment:
|
||||
|
||||
Adaptive Regions Adjustment
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
@@ -264,6 +266,111 @@ tracepoints. For more details, please refer to the documentations for
|
||||
respectively.
|
||||
|
||||
|
||||
.. _damon_design_monitoring_params_tuning_guide:
|
||||
|
||||
Monitoring Parameters Tuning Guide
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
In short, set ``aggregation interval`` to capture meaningful amount of accesses
|
||||
for the purpose. The amount of accesses can be measured using ``nr_accesses``
|
||||
and ``age`` of regions in the aggregated monitoring results snapshot. The
|
||||
default value of the interval, ``100ms``, turns out to be too short in many
|
||||
cases. Set ``sampling interval`` proportional to ``aggregation interval``. By
|
||||
default, ``1/20`` is recommended as the ratio.
|
||||
|
||||
``Aggregation interval`` should be set as the time interval that the workload
|
||||
can make an amount of accesses for the monitoring purpose, within the interval.
|
||||
If the interval is too short, only small number of accesses are captured. As a
|
||||
result, the monitoring results look everything is samely accessed only rarely.
|
||||
For many purposes, that would be useless. If it is too long, however, the time
|
||||
to converge regions with the :ref:`regions adjustment mechanism
|
||||
<damon_design_adaptive_regions_adjustment>` can be too long, depending on the
|
||||
time scale of the given purpose. This could happen if the workload is actually
|
||||
making only rare accesses but the user thinks the amount of accesses for the
|
||||
monitoring purpose too high. For such cases, the target amount of access to
|
||||
capture per ``aggregation interval`` should carefully reconsidered. Also, note
|
||||
that the captured amount of accesses is represented with not only
|
||||
``nr_accesses``, but also ``age``. For example, even if every region on the
|
||||
monitoring results show zero ``nr_accesses``, regions could still be
|
||||
distinguished using ``age`` values as the recency information.
|
||||
|
||||
Hence the optimum value of ``aggregation interval`` depends on the access
|
||||
intensiveness of the workload. The user should tune the interval based on the
|
||||
amount of access that captured on each aggregated snapshot of the monitoring
|
||||
results.
|
||||
|
||||
Note that the default value of the interval is 100 milliseconds, which is too
|
||||
short in many cases, especially on large systems.
|
||||
|
||||
``Sampling interval`` defines the resolution of each aggregation. If it is set
|
||||
too large, monitoring results will look like every region was samely rarely
|
||||
accessed, or samely frequently accessed. That is, regions become
|
||||
undistinguishable based on access pattern, and therefore the results will be
|
||||
useless in many use cases. If ``sampling interval`` is too small, it will not
|
||||
degrade the resolution, but will increase the monitoring overhead. If it is
|
||||
appropriate enough to provide a resolution of the monitoring results that
|
||||
sufficient for the given purpose, it shouldn't be unnecessarily further
|
||||
lowered. It is recommended to be set proportional to ``aggregation interval``.
|
||||
By default, the ratio is set as ``1/20``, and it is still recommended.
|
||||
|
||||
Based on the manual tuning guide, DAMON provides more intuitive knob-based
|
||||
intervals auto tuning mechanism. Please refer to :ref:`the design document of
|
||||
the feature <damon_design_monitoring_intervals_autotuning>` for detail.
|
||||
|
||||
Refer to below documents for an example tuning based on the above guide.
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 1
|
||||
|
||||
monitoring_intervals_tuning_example
|
||||
|
||||
|
||||
.. _damon_design_monitoring_intervals_autotuning:
|
||||
|
||||
Monitoring Intervals Auto-tuning
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
DAMON provides automatic tuning of the ``sampling interval`` and ``aggregation
|
||||
interval`` based on the :ref:`the tuning guide idea
|
||||
<damon_design_monitoring_params_tuning_guide>`. The tuning mechanism allows
|
||||
users to set the aimed amount of access events to observe via DAMON within
|
||||
given time interval. The target can be specified by the user as a ratio of
|
||||
DAMON-observed access events to the theoretical maximum amount of the events
|
||||
(``access_bp``) that measured within a given number of aggregations
|
||||
(``aggrs``).
|
||||
|
||||
The DAMON-observed access events are calculated in byte granularity based on
|
||||
DAMON :ref:`region assumption <damon_design_region_based_sampling>`. For
|
||||
example, if a region of size ``X`` bytes of ``Y`` ``nr_accesses`` is found, it
|
||||
means ``X * Y`` access events are observed by DAMON. Theoretical maximum
|
||||
access events for the region is calculated in same way, but replacing ``Y``
|
||||
with theoretical maximum ``nr_accesses``, which can be calculated as
|
||||
``aggregation interval / sampling interval``.
|
||||
|
||||
The mechanism calculates the ratio of access events for ``aggrs`` aggregations,
|
||||
and increases or decrease the ``sampleing interval`` and ``aggregation
|
||||
interval`` in same ratio, if the observed access ratio is lower or higher than
|
||||
the target, respectively. The ratio of the intervals change is decided in
|
||||
proportion to the distance between current samples ratio and the target ratio.
|
||||
|
||||
The user can further set the minimum and maximum ``sampling interval`` that can
|
||||
be set by the tuning mechanism using two parameters (``min_sample_us`` and
|
||||
``max_sample_us``). Because the tuning mechanism changes ``sampling interval``
|
||||
and ``aggregation interval`` in same ratio always, the minimum and maximum
|
||||
``aggregation interval`` after each of the tuning changes can automatically set
|
||||
together.
|
||||
|
||||
The tuning is turned off by default, and need to be set explicitly by the user.
|
||||
As a rule of thumbs and the Parreto principle, 4% access samples ratio target
|
||||
is recommended. Note that Parreto principle (80/20 rule) has applied twice.
|
||||
That is, assumes 4% (20% of 20%) DAMON-observed access events ratio (source)
|
||||
to capture 64% (80% multipled by 80%) real access events (outcomes).
|
||||
|
||||
To know how user-space can use this feature via :ref:`DAMON sysfs interface
|
||||
<sysfs_interface>`, refer to :ref:`intervals_goal <sysfs_scheme>` part of
|
||||
the documentation.
|
||||
|
||||
|
||||
.. _damon_design_damos:
|
||||
|
||||
Operation Schemes
|
||||
@@ -345,9 +452,9 @@ that supports each action are as below.
|
||||
- ``lru_deprio``: Deprioritize the region on its LRU lists.
|
||||
Supported by ``paddr`` operations set.
|
||||
- ``migrate_hot``: Migrate the regions prioritizing warmer regions.
|
||||
Supported by ``paddr`` operations set.
|
||||
Supported by ``vaddr``, ``fvaddr`` and ``paddr`` operations set.
|
||||
- ``migrate_cold``: Migrate the regions prioritizing colder regions.
|
||||
Supported by ``paddr`` operations set.
|
||||
Supported by ``vaddr``, ``fvaddr`` and ``paddr`` operations set.
|
||||
- ``stat``: Do nothing but count the statistics.
|
||||
Supported by all operations sets.
|
||||
|
||||
@@ -517,11 +624,21 @@ number of filters for each scheme. Each filter specifies
|
||||
- whether it is to allow (include) or reject (exclude) applying
|
||||
the scheme's action to the memory (``allow``).
|
||||
|
||||
When multiple filters are installed, each filter is evaluated in the installed
|
||||
order. If a part of memory is matched to one of the filter, next filters are
|
||||
ignored. If the memory passes through the filters evaluation stage because it
|
||||
is not matched to any of the filters, applying the scheme's action to it is
|
||||
allowed, same to the behavior when no filter exists.
|
||||
For efficient handling of filters, some types of filters are handled by the
|
||||
core layer, while others are handled by operations set. In the latter case,
|
||||
hence, support of the filter types depends on the DAMON operations set. In
|
||||
case of the core layer-handled filters, the memory regions that excluded by the
|
||||
filter are not counted as the scheme has tried to the region. In contrast, if
|
||||
a memory regions is filtered by an operations set layer-handled filter, it is
|
||||
counted as the scheme has tried. This difference affects the statistics.
|
||||
|
||||
When multiple filters are installed, the group of filters that handled by the
|
||||
core layer are evaluated first. After that, the group of filters that handled
|
||||
by the operations layer are evaluated. Filters in each of the groups are
|
||||
evaluated in the installed order. If a part of memory is matched to one of the
|
||||
filter, next filters are ignored. If the memory passes through the filters
|
||||
evaluation stage because it is not matched to any of the filters, applying the
|
||||
scheme's action to it is allowed, same to the behavior when no filter exists.
|
||||
|
||||
For example, let's assume 1) a filter for allowing anonymous pages and 2)
|
||||
another filter for rejecting young pages are installed in the order. If a page
|
||||
@@ -538,34 +655,25 @@ filter-allowed or filters evaluation stage passed. It means that installing
|
||||
allow-filters at the end of the list makes no practical change but only
|
||||
filters-checking overhead.
|
||||
|
||||
For efficient handling of filters, some types of filters are handled by the
|
||||
core layer, while others are handled by operations set. In the latter case,
|
||||
hence, support of the filter types depends on the DAMON operations set. In
|
||||
case of the core layer-handled filters, the memory regions that excluded by the
|
||||
filter are not counted as the scheme has tried to the region. In contrast, if
|
||||
a memory regions is filtered by an operations set layer-handled filter, it is
|
||||
counted as the scheme has tried. This difference affects the statistics.
|
||||
|
||||
Below ``type`` of filters are currently supported.
|
||||
|
||||
- anonymous page
|
||||
- Applied to pages that containing data that not stored in files.
|
||||
- Handled by operations set layer. Supported by only ``paddr`` set.
|
||||
- memory cgroup
|
||||
- Applied to pages that belonging to a given cgroup.
|
||||
- Handled by operations set layer. Supported by only ``paddr`` set.
|
||||
- young page
|
||||
- Applied to pages that are accessed after the last access check from the
|
||||
scheme.
|
||||
- Handled by operations set layer. Supported by only ``paddr`` set.
|
||||
- address range
|
||||
- Applied to pages that belonging to a given address range.
|
||||
- Handled by the core logic.
|
||||
- DAMON monitoring target
|
||||
- Applied to pages that belonging to a given DAMON monitoring target.
|
||||
- Handled by the core logic.
|
||||
- Core layer handled
|
||||
- addr
|
||||
- Applied to pages that belonging to a given address range.
|
||||
- target
|
||||
- Applied to pages that belonging to a given DAMON monitoring target.
|
||||
- Operations layer handled, supported by only ``paddr`` operations set.
|
||||
- anon
|
||||
- Applied to pages that containing data that not stored in files.
|
||||
- memcg
|
||||
- Applied to pages that belonging to a given cgroup.
|
||||
- young
|
||||
- Applied to pages that are accessed after the last access check from the
|
||||
scheme.
|
||||
- hugepage_size
|
||||
- Applied to pages that managed in a given size range.
|
||||
|
||||
To know how user-space can set the watermarks via :ref:`DAMON sysfs interface
|
||||
To know how user-space can set the filters via :ref:`DAMON sysfs interface
|
||||
<sysfs_interface>`, refer to :ref:`filters <sysfs_filters>` part of the
|
||||
documentation.
|
||||
|
||||
|
||||
@@ -7,9 +7,9 @@ The DAMON subsystem covers the files that are listed in 'DATA ACCESS MONITOR'
|
||||
section of 'MAINTAINERS' file.
|
||||
|
||||
The mailing lists for the subsystem are damon@lists.linux.dev and
|
||||
linux-mm@kvack.org. Patches should be made against the `mm-unstable tree
|
||||
<https://git.kernel.org/akpm/mm/h/mm-unstable>`_ whenever possible and posted
|
||||
to the mailing lists.
|
||||
linux-mm@kvack.org. Patches should be made against the `mm-new tree
|
||||
<https://git.kernel.org/akpm/mm/h/mm-new>`_ whenever possible and posted to the
|
||||
mailing lists.
|
||||
|
||||
SCM Trees
|
||||
---------
|
||||
@@ -17,17 +17,19 @@ SCM Trees
|
||||
There are multiple Linux trees for DAMON development. Patches under
|
||||
development or testing are queued in `damon/next
|
||||
<https://git.kernel.org/sj/h/damon/next>`_ by the DAMON maintainer.
|
||||
Sufficiently reviewed patches will be queued in `mm-unstable
|
||||
<https://git.kernel.org/akpm/mm/h/mm-unstable>`_ by the memory management
|
||||
subsystem maintainer. After more sufficient tests, the patches will be queued
|
||||
in `mm-stable <https://git.kernel.org/akpm/mm/h/mm-stable>`_, and finally
|
||||
pull-requested to the mainline by the memory management subsystem maintainer.
|
||||
Sufficiently reviewed patches will be queued in `mm-new
|
||||
<https://git.kernel.org/akpm/mm/h/mm-new>`_ by the memory management subsystem
|
||||
maintainer. As more sufficient tests are done, the patches will move to
|
||||
`mm-unstable <https://git.kernel.org/akpm/mm/h/mm-unstable>`_ and then to
|
||||
`mm-stable <https://git.kernel.org/akpm/mm/h/mm-stable>`_. And finally those
|
||||
will be pull-requested to the mainline by the memory management subsystem
|
||||
maintainer.
|
||||
|
||||
Note again the patches for `mm-unstable tree
|
||||
<https://git.kernel.org/akpm/mm/h/mm-unstable>`_ are queued by the memory
|
||||
management subsystem maintainer. If the patches requires some patches in
|
||||
`damon/next tree <https://git.kernel.org/sj/h/damon/next>`_ which not yet merged
|
||||
in mm-unstable, please make sure the requirement is clearly specified.
|
||||
Note again the patches for `mm-new tree
|
||||
<https://git.kernel.org/akpm/mm/h/mm-new>`_ are queued by the memory management
|
||||
subsystem maintainer. If the patches requires some patches in `damon/next tree
|
||||
<https://git.kernel.org/sj/h/damon/next>`_ which not yet merged in mm-new,
|
||||
please make sure the requirement is clearly specified.
|
||||
|
||||
Submit checklist addendum
|
||||
-------------------------
|
||||
@@ -53,7 +55,8 @@ Further doing below and putting the results will be helpful.
|
||||
Key cycle dates
|
||||
---------------
|
||||
|
||||
Patches can be sent anytime. Key cycle dates of the `mm-unstable
|
||||
Patches can be sent anytime. Key cycle dates of the `mm-new
|
||||
<https://git.kernel.org/akpm/mm/h/mm-new>`_, `mm-unstable
|
||||
<https://git.kernel.org/akpm/mm/h/mm-unstable>`_ and `mm-stable
|
||||
<https://git.kernel.org/akpm/mm/h/mm-stable>`_ trees depend on the memory
|
||||
management subsystem maintainer.
|
||||
|
||||
@@ -0,0 +1,247 @@
|
||||
.. SPDX-License-Identifier: GPL-2.0
|
||||
|
||||
=================================================
|
||||
DAMON Moniting Interval Parameters Tuning Example
|
||||
=================================================
|
||||
|
||||
DAMON's monitoring parameters need tuning based on given workload and the
|
||||
monitoring purpose. There is a :ref:`tuning guide
|
||||
<damon_design_monitoring_params_tuning_guide>` for that. This document
|
||||
provides an example tuning based on the guide.
|
||||
|
||||
Setup
|
||||
=====
|
||||
|
||||
For below example, DAMON of Linux kernel v6.11 and `damo
|
||||
<https://github.com/damonitor/damo>`_ (DAMON user-space tool) v2.5.9 was used to
|
||||
monitor and visualize access patterns on the physical address space of a system
|
||||
running a real-world server workload.
|
||||
|
||||
5ms/100ms intervals: Too Short Interval
|
||||
=======================================
|
||||
|
||||
Let's start by capturing the access pattern snapshot on the physical address
|
||||
space of the system using DAMON, with the default interval parameters (5
|
||||
milliseconds and 100 milliseconds for the sampling and the aggregation
|
||||
intervals, respectively). Wait ten minutes between the start of DAMON and
|
||||
the capturing of the snapshot, to show a meaningful time-wise access patterns.
|
||||
::
|
||||
|
||||
# damo start
|
||||
# sleep 600
|
||||
# damo record --snapshot 0 1
|
||||
# damo stop
|
||||
|
||||
Then, list the DAMON-found regions of different access patterns, sorted by the
|
||||
"access temperature". "Access temperature" is a metric representing the
|
||||
access-hotness of a region. It is calculated as a weighted sum of the access
|
||||
frequency and the age of the region. If the access frequency is 0 %, the
|
||||
temperature is multipled by minus one. That is, if a region is not accessed,
|
||||
it gets minus temperature and it gets lower as not accessed for longer time.
|
||||
The sorting is in temperature-ascendint order, so the region at the top of the
|
||||
list is the coldest, and the one at the bottom is the hottest one. ::
|
||||
|
||||
# damo report access --sort_regions_by temperature
|
||||
0 addr 16.052 GiB size 5.985 GiB access 0 % age 5.900 s # coldest
|
||||
1 addr 22.037 GiB size 6.029 GiB access 0 % age 5.300 s
|
||||
2 addr 28.065 GiB size 6.045 GiB access 0 % age 5.200 s
|
||||
3 addr 10.069 GiB size 5.983 GiB access 0 % age 4.500 s
|
||||
4 addr 4.000 GiB size 6.069 GiB access 0 % age 4.400 s
|
||||
5 addr 62.008 GiB size 3.992 GiB access 0 % age 3.700 s
|
||||
6 addr 56.795 GiB size 5.213 GiB access 0 % age 3.300 s
|
||||
7 addr 39.393 GiB size 6.096 GiB access 0 % age 2.800 s
|
||||
8 addr 50.782 GiB size 6.012 GiB access 0 % age 2.800 s
|
||||
9 addr 34.111 GiB size 5.282 GiB access 0 % age 2.300 s
|
||||
10 addr 45.489 GiB size 5.293 GiB access 0 % age 1.800 s # hottest
|
||||
total size: 62.000 GiB
|
||||
|
||||
The list shows not seemingly hot regions, and only minimum access pattern
|
||||
diversity. Every region has zero access frequency. The number of region is
|
||||
10, which is the default ``min_nr_regions value``. Size of each region is also
|
||||
nearly idential. We can suspect this is because “adaptive regions adjustment”
|
||||
mechanism was not well working. As the guide suggested, we can get relative
|
||||
hotness of regions using ``age`` as the recency information. That would be
|
||||
better than nothing, but given the fact that the longest age is only about 6
|
||||
seconds while we waited about ten minuts, it is unclear how useful this will
|
||||
be.
|
||||
|
||||
The temperature ranges to total size of regions of each range histogram
|
||||
visualization of the results also shows no interesting distribution pattern. ::
|
||||
|
||||
# damo report access --style temperature-sz-hist
|
||||
<temperature> <total size>
|
||||
[-,590,000,000, -,549,000,000) 5.985 GiB |********** |
|
||||
[-,549,000,000, -,508,000,000) 12.074 GiB |********************|
|
||||
[-,508,000,000, -,467,000,000) 0 B | |
|
||||
[-,467,000,000, -,426,000,000) 12.052 GiB |********************|
|
||||
[-,426,000,000, -,385,000,000) 0 B | |
|
||||
[-,385,000,000, -,344,000,000) 3.992 GiB |******* |
|
||||
[-,344,000,000, -,303,000,000) 5.213 GiB |********* |
|
||||
[-,303,000,000, -,262,000,000) 12.109 GiB |********************|
|
||||
[-,262,000,000, -,221,000,000) 5.282 GiB |********* |
|
||||
[-,221,000,000, -,180,000,000) 0 B | |
|
||||
[-,180,000,000, -,139,000,000) 5.293 GiB |********* |
|
||||
total size: 62.000 GiB
|
||||
|
||||
In short, the parameters provide poor quality monitoring results for hot
|
||||
regions detection. According to the :ref:`guide
|
||||
<damon_design_monitoring_params_tuning_guide>`, this is due to the too short
|
||||
aggregation interval.
|
||||
|
||||
100ms/2s intervals: Starts Showing Small Hot Regions
|
||||
====================================================
|
||||
|
||||
Following the guide, increase the interval 20 times (100 milliseocnds and 2
|
||||
seconds for sampling and aggregation intervals, respectively). ::
|
||||
|
||||
# damo start -s 100ms -a 2s
|
||||
# sleep 600
|
||||
# damo record --snapshot 0 1
|
||||
# damo stop
|
||||
# damo report access --sort_regions_by temperature
|
||||
0 addr 10.180 GiB size 6.117 GiB access 0 % age 7 m 8 s # coldest
|
||||
1 addr 49.275 GiB size 6.195 GiB access 0 % age 6 m 14 s
|
||||
2 addr 62.421 GiB size 3.579 GiB access 0 % age 6 m 4 s
|
||||
3 addr 40.154 GiB size 6.127 GiB access 0 % age 5 m 40 s
|
||||
4 addr 16.296 GiB size 6.182 GiB access 0 % age 5 m 32 s
|
||||
5 addr 34.254 GiB size 5.899 GiB access 0 % age 5 m 24 s
|
||||
6 addr 46.281 GiB size 2.995 GiB access 0 % age 5 m 20 s
|
||||
7 addr 28.420 GiB size 5.835 GiB access 0 % age 5 m 6 s
|
||||
8 addr 4.000 GiB size 6.180 GiB access 0 % age 4 m 16 s
|
||||
9 addr 22.478 GiB size 5.942 GiB access 0 % age 3 m 58 s
|
||||
10 addr 55.470 GiB size 915.645 MiB access 0 % age 3 m 6 s
|
||||
11 addr 56.364 GiB size 6.056 GiB access 0 % age 2 m 8 s
|
||||
12 addr 56.364 GiB size 4.000 KiB access 95 % age 16 s
|
||||
13 addr 49.275 GiB size 4.000 KiB access 100 % age 8 m 24 s # hottest
|
||||
total size: 62.000 GiB
|
||||
# damo report access --style temperature-sz-hist
|
||||
<temperature> <total size>
|
||||
[-42,800,000,000, -33,479,999,000) 22.018 GiB |***************** |
|
||||
[-33,479,999,000, -24,159,998,000) 27.090 GiB |********************|
|
||||
[-24,159,998,000, -14,839,997,000) 6.836 GiB |****** |
|
||||
[-14,839,997,000, -5,519,996,000) 6.056 GiB |***** |
|
||||
[-5,519,996,000, 3,800,005,000) 4.000 KiB |* |
|
||||
[3,800,005,000, 13,120,006,000) 0 B | |
|
||||
[13,120,006,000, 22,440,007,000) 0 B | |
|
||||
[22,440,007,000, 31,760,008,000) 0 B | |
|
||||
[31,760,008,000, 41,080,009,000) 0 B | |
|
||||
[41,080,009,000, 50,400,010,000) 0 B | |
|
||||
[50,400,010,000, 59,720,011,000) 4.000 KiB |* |
|
||||
total size: 62.000 GiB
|
||||
|
||||
DAMON found two distinct 4 KiB regions that pretty hot. The regions are also
|
||||
well aged. The hottest 4 KiB region was keeping the access frequency for about
|
||||
8 minutes, and the coldest region was keeping no access for about 7 minutes.
|
||||
The distribution on the histogram also looks like having a pattern.
|
||||
|
||||
Especially, the finding of the 4 KiB regions among the 62 GiB total memory
|
||||
shows DAMON’s adaptive regions adjustment is working as designed.
|
||||
|
||||
Still the number of regions is close to the ``min_nr_regions``, and sizes of
|
||||
cold regions are similar, though. Apparently it is improved, but it still has
|
||||
rooms to improve.
|
||||
|
||||
400ms/8s intervals: Pretty Improved Results
|
||||
===========================================
|
||||
|
||||
Increase the intervals four times (400 milliseconds and 8 seconds
|
||||
for sampling and aggregation intervals, respectively). ::
|
||||
|
||||
# damo start -s 400ms -a 8s
|
||||
# sleep 600
|
||||
# damo record --snapshot 0 1
|
||||
# damo stop
|
||||
# damo report access --sort_regions_by temperature
|
||||
0 addr 64.492 GiB size 1.508 GiB access 0 % age 6 m 48 s # coldest
|
||||
1 addr 21.749 GiB size 5.674 GiB access 0 % age 6 m 8 s
|
||||
2 addr 27.422 GiB size 5.801 GiB access 0 % age 6 m
|
||||
3 addr 49.431 GiB size 8.675 GiB access 0 % age 5 m 28 s
|
||||
4 addr 33.223 GiB size 5.645 GiB access 0 % age 5 m 12 s
|
||||
5 addr 58.321 GiB size 6.170 GiB access 0 % age 5 m 4 s
|
||||
[...]
|
||||
25 addr 6.615 GiB size 297.531 MiB access 15 % age 0 ns
|
||||
26 addr 9.513 GiB size 12.000 KiB access 20 % age 0 ns
|
||||
27 addr 9.511 GiB size 108.000 KiB access 25 % age 0 ns
|
||||
28 addr 9.513 GiB size 20.000 KiB access 25 % age 0 ns
|
||||
29 addr 9.511 GiB size 12.000 KiB access 30 % age 0 ns
|
||||
30 addr 9.520 GiB size 4.000 KiB access 40 % age 0 ns
|
||||
[...]
|
||||
41 addr 9.520 GiB size 4.000 KiB access 80 % age 56 s
|
||||
42 addr 9.511 GiB size 12.000 KiB access 100 % age 6 m 16 s
|
||||
43 addr 58.321 GiB size 4.000 KiB access 100 % age 6 m 24 s
|
||||
44 addr 9.512 GiB size 4.000 KiB access 100 % age 6 m 48 s
|
||||
45 addr 58.106 GiB size 4.000 KiB access 100 % age 6 m 48 s # hottest
|
||||
total size: 62.000 GiB
|
||||
# damo report access --style temperature-sz-hist
|
||||
<temperature> <total size>
|
||||
[-40,800,000,000, -32,639,999,000) 21.657 GiB |********************|
|
||||
[-32,639,999,000, -24,479,998,000) 17.938 GiB |***************** |
|
||||
[-24,479,998,000, -16,319,997,000) 16.885 GiB |**************** |
|
||||
[-16,319,997,000, -8,159,996,000) 586.879 MiB |* |
|
||||
[-8,159,996,000, 5,000) 4.946 GiB |***** |
|
||||
[5,000, 8,160,006,000) 260.000 KiB |* |
|
||||
[8,160,006,000, 16,320,007,000) 0 B | |
|
||||
[16,320,007,000, 24,480,008,000) 0 B | |
|
||||
[24,480,008,000, 32,640,009,000) 0 B | |
|
||||
[32,640,009,000, 40,800,010,000) 16.000 KiB |* |
|
||||
[40,800,010,000, 48,960,011,000) 8.000 KiB |* |
|
||||
total size: 62.000 GiB
|
||||
|
||||
The number of regions having different access patterns has significantly
|
||||
increased. Size of each region is also more varied. Total size of non-zero
|
||||
access frequency regions is also significantly increased. Maybe this is already
|
||||
good enough to make some meaningful memory management efficieny changes.
|
||||
|
||||
800ms/16s intervals: Another bias
|
||||
=================================
|
||||
|
||||
Further double the intervals (800 milliseconds and 16 seconds for sampling
|
||||
and aggregation intervals, respectively). The results is more improved for the
|
||||
hot regions detection, but starts looking degrading cold regions detection. ::
|
||||
|
||||
# damo start -s 800ms -a 16s
|
||||
# sleep 600
|
||||
# damo record --snapshot 0 1
|
||||
# damo stop
|
||||
# damo report access --sort_regions_by temperature
|
||||
0 addr 64.781 GiB size 1.219 GiB access 0 % age 4 m 48 s
|
||||
1 addr 24.505 GiB size 2.475 GiB access 0 % age 4 m 16 s
|
||||
2 addr 26.980 GiB size 504.273 MiB access 0 % age 4 m
|
||||
3 addr 29.443 GiB size 2.462 GiB access 0 % age 4 m
|
||||
4 addr 37.264 GiB size 5.645 GiB access 0 % age 4 m
|
||||
5 addr 31.905 GiB size 5.359 GiB access 0 % age 3 m 44 s
|
||||
[...]
|
||||
20 addr 8.711 GiB size 40.000 KiB access 5 % age 2 m 40 s
|
||||
21 addr 27.473 GiB size 1.970 GiB access 5 % age 4 m
|
||||
22 addr 48.185 GiB size 4.625 GiB access 5 % age 4 m
|
||||
23 addr 47.304 GiB size 902.117 MiB access 10 % age 4 m
|
||||
24 addr 8.711 GiB size 4.000 KiB access 100 % age 4 m
|
||||
25 addr 20.793 GiB size 3.713 GiB access 5 % age 4 m 16 s
|
||||
26 addr 8.773 GiB size 4.000 KiB access 100 % age 4 m 16 s
|
||||
total size: 62.000 GiB
|
||||
# damo report access --style temperature-sz-hist
|
||||
<temperature> <total size>
|
||||
[-28,800,000,000, -23,359,999,000) 12.294 GiB |***************** |
|
||||
[-23,359,999,000, -17,919,998,000) 9.753 GiB |************* |
|
||||
[-17,919,998,000, -12,479,997,000) 15.131 GiB |********************|
|
||||
[-12,479,997,000, -7,039,996,000) 0 B | |
|
||||
[-7,039,996,000, -1,599,995,000) 7.506 GiB |********** |
|
||||
[-1,599,995,000, 3,840,006,000) 6.127 GiB |********* |
|
||||
[3,840,006,000, 9,280,007,000) 0 B | |
|
||||
[9,280,007,000, 14,720,008,000) 136.000 KiB |* |
|
||||
[14,720,008,000, 20,160,009,000) 40.000 KiB |* |
|
||||
[20,160,009,000, 25,600,010,000) 11.188 GiB |*************** |
|
||||
[25,600,010,000, 31,040,011,000) 4.000 KiB |* |
|
||||
total size: 62.000 GiB
|
||||
|
||||
It found more non-zero access frequency regions. The number of regions is still
|
||||
much higher than the ``min_nr_regions``, but it is reduced from that of the
|
||||
previous setup. And apparently the distribution seems bit biased to hot
|
||||
regions.
|
||||
|
||||
Conclusion
|
||||
==========
|
||||
|
||||
With the above experimental tuning results, we can conclude the theory and the
|
||||
guide makes sense to at least this workload, and could be applied to similar
|
||||
cases.
|
||||
@@ -56,7 +56,6 @@ documentation, or deleted if it has served its purpose.
|
||||
page_owner
|
||||
page_table_check
|
||||
remap_file_pages
|
||||
slub
|
||||
split_page_table_lock
|
||||
transhuge
|
||||
unevictable-lru
|
||||
|
||||
@@ -146,18 +146,33 @@ Steps:
|
||||
18. The new page is moved to the LRU and can be scanned by the swapper,
|
||||
etc. again.
|
||||
|
||||
Non-LRU page migration
|
||||
======================
|
||||
movable_ops page migration
|
||||
==========================
|
||||
|
||||
Although migration originally aimed for reducing the latency of memory
|
||||
accesses for NUMA, compaction also uses migration to create high-order
|
||||
pages. For compaction purposes, it is also useful to be able to move
|
||||
non-LRU pages, such as zsmalloc and virtio-balloon pages.
|
||||
Selected typed, non-folio pages (e.g., pages inflated in a memory balloon,
|
||||
zsmalloc pages) can be migrated using the movable_ops migration framework.
|
||||
|
||||
If a driver wants to make its pages movable, it should define a struct
|
||||
movable_operations. It then needs to call __SetPageMovable() on each
|
||||
page that it may be able to move. This uses the ``page->mapping`` field,
|
||||
so this field is not available for the driver to use for other purposes.
|
||||
The "struct movable_operations" provide callbacks specific to a page type
|
||||
for isolating, migrating and un-isolating (putback) these pages.
|
||||
|
||||
Once a page is indicated as having movable_ops, that condition must not
|
||||
change until the page was freed back to the buddy. This includes not
|
||||
changing/clearing the page type and not changing/clearing the
|
||||
PG_movable_ops page flag.
|
||||
|
||||
Arbitrary drivers cannot currently make use of this framework, as it
|
||||
requires:
|
||||
|
||||
(a) a page type
|
||||
(b) indicating them as possibly having movable_ops in page_has_movable_ops()
|
||||
based on the page type
|
||||
(c) returning the movable_ops from page_movable_ops() based on the page
|
||||
type
|
||||
(d) not reusing the PG_movable_ops and PG_movable_ops_isolated page flags
|
||||
for other purposes
|
||||
|
||||
For example, balloon drivers can make use of this framework through the
|
||||
balloon-compaction infrastructure residing in the core kernel.
|
||||
|
||||
Monitoring Migration
|
||||
=====================
|
||||
|
||||
@@ -338,10 +338,272 @@ Statistics
|
||||
|
||||
Zones
|
||||
=====
|
||||
As we have mentioned, each zone in memory is described by a ``struct zone``
|
||||
which is an element of the ``node_zones`` array of the node it belongs to.
|
||||
``struct zone`` is the core data structure of the page allocator. A zone
|
||||
represents a range of physical memory and may have holes.
|
||||
|
||||
.. admonition:: Stub
|
||||
The page allocator uses the GFP flags, see :ref:`mm-api-gfp-flags`, specified by
|
||||
a memory allocation to determine the highest zone in a node from which the
|
||||
memory allocation can allocate memory. The page allocator first allocates memory
|
||||
from that zone, if the page allocator can't allocate the requested amount of
|
||||
memory from the zone, it will allocate memory from the next lower zone in the
|
||||
node, the process continues up to and including the lowest zone. For example, if
|
||||
a node contains ``ZONE_DMA32``, ``ZONE_NORMAL`` and ``ZONE_MOVABLE`` and the
|
||||
highest zone of a memory allocation is ``ZONE_MOVABLE``, the order of the zones
|
||||
from which the page allocator allocates memory is ``ZONE_MOVABLE`` >
|
||||
``ZONE_NORMAL`` > ``ZONE_DMA32``.
|
||||
|
||||
This section is incomplete. Please list and describe the appropriate fields.
|
||||
At runtime, free pages in a zone are in the Per-CPU Pagesets (PCP) or free areas
|
||||
of the zone. The Per-CPU Pagesets are a vital mechanism in the kernel's memory
|
||||
management system. By handling most frequent allocations and frees locally on
|
||||
each CPU, the Per-CPU Pagesets improve performance and scalability, especially
|
||||
on systems with many cores. The page allocator in the kernel employs a two-step
|
||||
strategy for memory allocation, starting with the Per-CPU Pagesets before
|
||||
falling back to the buddy allocator. Pages are transferred between the Per-CPU
|
||||
Pagesets and the global free areas (managed by the buddy allocator) in batches.
|
||||
This minimizes the overhead of frequent interactions with the global buddy
|
||||
allocator.
|
||||
|
||||
Architecture specific code calls free_area_init() to initializes zones.
|
||||
|
||||
Zone structure
|
||||
--------------
|
||||
The zones structure ``struct zone`` is defined in ``include/linux/mmzone.h``.
|
||||
Here we briefly describe fields of this structure:
|
||||
|
||||
General
|
||||
~~~~~~~
|
||||
|
||||
``_watermark``
|
||||
The watermarks for this zone. When the amount of free pages in a zone is below
|
||||
the min watermark, boosting is ignored, an allocation may trigger direct
|
||||
reclaim and direct compaction, it is also used to throttle direct reclaim.
|
||||
When the amount of free pages in a zone is below the low watermark, kswapd is
|
||||
woken up. When the amount of free pages in a zone is above the high watermark,
|
||||
kswapd stops reclaiming (a zone is balanced) when the
|
||||
``NUMA_BALANCING_MEMORY_TIERING`` bit of ``sysctl_numa_balancing_mode`` is not
|
||||
set. The promo watermark is used for memory tiering and NUMA balancing. When
|
||||
the amount of free pages in a zone is above the promo watermark, kswapd stops
|
||||
reclaiming when the ``NUMA_BALANCING_MEMORY_TIERING`` bit of
|
||||
``sysctl_numa_balancing_mode`` is set. The watermarks are set by
|
||||
``__setup_per_zone_wmarks()``. The min watermark is calculated according to
|
||||
``vm.min_free_kbytes`` sysctl. The other three watermarks are set according
|
||||
to the distance between two watermarks. The distance itself is calculated
|
||||
taking ``vm.watermark_scale_factor`` sysctl into account.
|
||||
|
||||
``watermark_boost``
|
||||
The number of pages which are used to boost watermarks to increase reclaim
|
||||
pressure to reduce the likelihood of future fallbacks and wake kswapd now
|
||||
as the node may be balanced overall and kswapd will not wake naturally.
|
||||
|
||||
``nr_reserved_highatomic``
|
||||
The number of pages which are reserved for high-order atomic allocations.
|
||||
|
||||
``nr_free_highatomic``
|
||||
The number of free pages in reserved highatomic pageblocks
|
||||
|
||||
``lowmem_reserve``
|
||||
The array of the amounts of the memory reserved in this zone for memory
|
||||
allocations. For example, if the highest zone a memory allocation can
|
||||
allocate memory from is ``ZONE_MOVABLE``, the amount of memory reserved in
|
||||
this zone for this allocation is ``lowmem_reserve[ZONE_MOVABLE]`` when
|
||||
attempting to allocate memory from this zone. This is a mechanism the page
|
||||
allocator uses to prevent allocations which could use ``highmem`` from using
|
||||
too much ``lowmem``. For some specialised workloads on ``highmem`` machines,
|
||||
it is dangerous for the kernel to allow process memory to be allocated from
|
||||
the ``lowmem`` zone. This is because that memory could then be pinned via the
|
||||
``mlock()`` system call, or by unavailability of swapspace.
|
||||
``vm.lowmem_reserve_ratio`` sysctl determines how aggressive the kernel is in
|
||||
defending these lower zones. This array is recalculated by
|
||||
``setup_per_zone_lowmem_reserve()`` at runtime if ``vm.lowmem_reserve_ratio``
|
||||
sysctl changes.
|
||||
|
||||
``node``
|
||||
The index of the node this zone belongs to. Available only when
|
||||
``CONFIG_NUMA`` is enabled because there is only one zone in a UMA system.
|
||||
|
||||
``zone_pgdat``
|
||||
Pointer to the ``struct pglist_data`` of the node this zone belongs to.
|
||||
|
||||
``per_cpu_pageset``
|
||||
Pointer to the Per-CPU Pagesets (PCP) allocated and initialized by
|
||||
``setup_zone_pageset()``. By handling most frequent allocations and frees
|
||||
locally on each CPU, PCP improves performance and scalability on systems with
|
||||
many cores.
|
||||
|
||||
``pageset_high_min``
|
||||
Copied to the ``high_min`` of the Per-CPU Pagesets for faster access.
|
||||
|
||||
``pageset_high_max``
|
||||
Copied to the ``high_max`` of the Per-CPU Pagesets for faster access.
|
||||
|
||||
``pageset_batch``
|
||||
Copied to the ``batch`` of the Per-CPU Pagesets for faster access. The
|
||||
``batch``, ``high_min`` and ``high_max`` of the Per-CPU Pagesets are used to
|
||||
calculate the number of elements the Per-CPU Pagesets obtain from the buddy
|
||||
allocator under a single hold of the lock for efficiency. They are also used
|
||||
to decide if the Per-CPU Pagesets return pages to the buddy allocator in page
|
||||
free process.
|
||||
|
||||
``pageblock_flags``
|
||||
The pointer to the flags for the pageblocks in the zone (see
|
||||
``include/linux/pageblock-flags.h`` for flags list). The memory is allocated
|
||||
in ``setup_usemap()``. Each pageblock occupies ``NR_PAGEBLOCK_BITS`` bits.
|
||||
Defined only when ``CONFIG_FLATMEM`` is enabled. The flags is stored in
|
||||
``mem_section`` when ``CONFIG_SPARSEMEM`` is enabled.
|
||||
|
||||
``zone_start_pfn``
|
||||
The start pfn of the zone. It is initialized by
|
||||
``calculate_node_totalpages()``.
|
||||
|
||||
``managed_pages``
|
||||
The present pages managed by the buddy system, which is calculated as:
|
||||
``managed_pages`` = ``present_pages`` - ``reserved_pages``, ``reserved_pages``
|
||||
includes pages allocated by the memblock allocator. It should be used by page
|
||||
allocator and vm scanner to calculate all kinds of watermarks and thresholds.
|
||||
It is accessed using ``atomic_long_xxx()`` functions. It is initialized in
|
||||
``free_area_init_core()`` and then is reinitialized when memblock allocator
|
||||
frees pages into buddy system.
|
||||
|
||||
``spanned_pages``
|
||||
The total pages spanned by the zone, including holes, which is calculated as:
|
||||
``spanned_pages`` = ``zone_end_pfn`` - ``zone_start_pfn``. It is initialized
|
||||
by ``calculate_node_totalpages()``.
|
||||
|
||||
``present_pages``
|
||||
The physical pages existing within the zone, which is calculated as:
|
||||
``present_pages`` = ``spanned_pages`` - ``absent_pages`` (pages in holes). It
|
||||
may be used by memory hotplug or memory power management logic to figure out
|
||||
unmanaged pages by checking (``present_pages`` - ``managed_pages``). Write
|
||||
access to ``present_pages`` at runtime should be protected by
|
||||
``mem_hotplug_begin/done()``. Any reader who can't tolerant drift of
|
||||
``present_pages`` should use ``get_online_mems()`` to get a stable value. It
|
||||
is initialized by ``calculate_node_totalpages()``.
|
||||
|
||||
``present_early_pages``
|
||||
The present pages existing within the zone located on memory available since
|
||||
early boot, excluding hotplugged memory. Defined only when
|
||||
``CONFIG_MEMORY_HOTPLUG`` is enabled and initialized by
|
||||
``calculate_node_totalpages()``.
|
||||
|
||||
``cma_pages``
|
||||
The pages reserved for CMA use. These pages behave like ``ZONE_MOVABLE`` when
|
||||
they are not used for CMA. Defined only when ``CONFIG_CMA`` is enabled.
|
||||
|
||||
``name``
|
||||
The name of the zone. It is a pointer to the corresponding element of
|
||||
the ``zone_names`` array.
|
||||
|
||||
``nr_isolate_pageblock``
|
||||
Number of isolated pageblocks. It is used to solve incorrect freepage counting
|
||||
problem due to racy retrieving migratetype of pageblock. Protected by
|
||||
``zone->lock``. Defined only when ``CONFIG_MEMORY_ISOLATION`` is enabled.
|
||||
|
||||
``span_seqlock``
|
||||
The seqlock to protect ``zone_start_pfn`` and ``spanned_pages``. It is a
|
||||
seqlock because it has to be read outside of ``zone->lock``, and it is done in
|
||||
the main allocator path. However, the seqlock is written quite infrequently.
|
||||
Defined only when ``CONFIG_MEMORY_HOTPLUG`` is enabled.
|
||||
|
||||
``initialized``
|
||||
The flag indicating if the zone is initialized. Set by
|
||||
``init_currently_empty_zone()`` during boot.
|
||||
|
||||
``free_area``
|
||||
The array of free areas, where each element corresponds to a specific order
|
||||
which is a power of two. The buddy allocator uses this structure to manage
|
||||
free memory efficiently. When allocating, it tries to find the smallest
|
||||
sufficient block, if the smallest sufficient block is larger than the
|
||||
requested size, it will be recursively split into the next smaller blocks
|
||||
until the required size is reached. When a page is freed, it may be merged
|
||||
with its buddy to form a larger block. It is initialized by
|
||||
``zone_init_free_lists()``.
|
||||
|
||||
``unaccepted_pages``
|
||||
The list of pages to be accepted. All pages on the list are ``MAX_PAGE_ORDER``.
|
||||
Defined only when ``CONFIG_UNACCEPTED_MEMORY`` is enabled.
|
||||
|
||||
``flags``
|
||||
The zone flags. The least three bits are used and defined by
|
||||
``enum zone_flags``. ``ZONE_BOOSTED_WATERMARK`` (bit 0): zone recently boosted
|
||||
watermarks. Cleared when kswapd is woken. ``ZONE_RECLAIM_ACTIVE`` (bit 1):
|
||||
kswapd may be scanning the zone. ``ZONE_BELOW_HIGH`` (bit 2): zone is below
|
||||
high watermark.
|
||||
|
||||
``lock``
|
||||
The main lock that protects the internal data structures of the page allocator
|
||||
specific to the zone, especially protects ``free_area``.
|
||||
|
||||
``percpu_drift_mark``
|
||||
When free pages are below this point, additional steps are taken when reading
|
||||
the number of free pages to avoid per-cpu counter drift allowing watermarks
|
||||
to be breached. It is updated in ``refresh_zone_stat_thresholds()``.
|
||||
|
||||
Compaction control
|
||||
~~~~~~~~~~~~~~~~~~
|
||||
|
||||
``compact_cached_free_pfn``
|
||||
The PFN where compaction free scanner should start in the next scan.
|
||||
|
||||
``compact_cached_migrate_pfn``
|
||||
The PFNs where compaction migration scanner should start in the next scan.
|
||||
This array has two elements: the first one is used in ``MIGRATE_ASYNC`` mode,
|
||||
and the other one is used in ``MIGRATE_SYNC`` mode.
|
||||
|
||||
``compact_init_migrate_pfn``
|
||||
The initial migration PFN which is initialized to 0 at boot time, and to the
|
||||
first pageblock with migratable pages in the zone after a full compaction
|
||||
finishes. It is used to check if a scan is a whole zone scan or not.
|
||||
|
||||
``compact_init_free_pfn``
|
||||
The initial free PFN which is initialized to 0 at boot time and to the last
|
||||
pageblock with free ``MIGRATE_MOVABLE`` pages in the zone. It is used to check
|
||||
if it is the start of a scan.
|
||||
|
||||
``compact_considered``
|
||||
The number of compactions attempted since last failure. It is reset in
|
||||
``defer_compaction()`` when a compaction fails to result in a page allocation
|
||||
success. It is increased by 1 in ``compaction_deferred()`` when a compaction
|
||||
should be skipped. ``compaction_deferred()`` is called before
|
||||
``compact_zone()`` is called, ``compaction_defer_reset()`` is called when
|
||||
``compact_zone()`` returns ``COMPACT_SUCCESS``, ``defer_compaction()`` is
|
||||
called when ``compact_zone()`` returns ``COMPACT_PARTIAL_SKIPPED`` or
|
||||
``COMPACT_COMPLETE``.
|
||||
|
||||
``compact_defer_shift``
|
||||
The number of compactions skipped before trying again is
|
||||
``1<<compact_defer_shift``. It is increased by 1 in ``defer_compaction()``.
|
||||
It is reset in ``compaction_defer_reset()`` when a direct compaction results
|
||||
in a page allocation success. Its maximum value is ``COMPACT_MAX_DEFER_SHIFT``.
|
||||
|
||||
``compact_order_failed``
|
||||
The minimum compaction failed order. It is set in ``compaction_defer_reset()``
|
||||
when a compaction succeeds and in ``defer_compaction()`` when a compaction
|
||||
fails to result in a page allocation success.
|
||||
|
||||
``compact_blockskip_flush``
|
||||
Set to true when compaction migration scanner and free scanner meet, which
|
||||
means the ``PB_compact_skip`` bits should be cleared.
|
||||
|
||||
``contiguous``
|
||||
Set to true when the zone is contiguous (in other words, no hole).
|
||||
|
||||
Statistics
|
||||
~~~~~~~~~~
|
||||
|
||||
``vm_stat``
|
||||
VM statistics for the zone. The items tracked are defined by
|
||||
``enum zone_stat_item``.
|
||||
|
||||
``vm_numa_event``
|
||||
VM NUMA event statistics for the zone. The items tracked are defined by
|
||||
``enum numa_stat_item``.
|
||||
|
||||
``per_cpu_zonestats``
|
||||
Per-CPU VM statistics for the zone. It records VM statistics and VM NUMA event
|
||||
statistics on a per-CPU basis. It reduces updates to the global ``vm_stat``
|
||||
and ``vm_numa_event`` fields of the zone to improve performance.
|
||||
|
||||
.. _pages:
|
||||
|
||||
|
||||
@@ -62,7 +62,6 @@ memory_notify结构体的指针::
|
||||
struct memory_notify {
|
||||
unsigned long start_pfn;
|
||||
unsigned long nr_pages;
|
||||
int status_change_nid_normal;
|
||||
int status_change_nid;
|
||||
}
|
||||
|
||||
@@ -70,8 +69,6 @@ memory_notify结构体的指针::
|
||||
|
||||
- nr_pages是在线/离线内存的页数。
|
||||
|
||||
- status_change_nid_normal是当nodemask的N_NORMAL_MEMORY被设置/清除时设置节
|
||||
点id,如果是-1,则nodemask状态不改变。
|
||||
|
||||
- status_change_nid是当nodemask的N_MEMORY被(将)设置/清除时设置的节点id。这
|
||||
意味着一个新的(没上线的)节点通过联机获得新的内存,而一个节点失去了所有的内
|
||||
|
||||
@@ -6354,6 +6354,7 @@ F: Documentation/mm/damon/
|
||||
F: include/linux/damon.h
|
||||
F: include/trace/events/damon.h
|
||||
F: mm/damon/
|
||||
F: samples/damon/
|
||||
F: tools/testing/selftests/damon/
|
||||
|
||||
DAVICOM FAST ETHERNET (DMFE) NETWORK DRIVER
|
||||
|
||||
@@ -7,6 +7,7 @@ config ALPHA
|
||||
select ARCH_HAS_DMA_OPS if PCI
|
||||
select ARCH_MIGHT_HAVE_PC_PARPORT
|
||||
select ARCH_MIGHT_HAVE_PC_SERIO
|
||||
select ARCH_MODULE_NEEDS_WEAK_PER_CPU if SMP
|
||||
select ARCH_NO_PREEMPT
|
||||
select ARCH_NO_SG_CHAIN
|
||||
select ARCH_USE_CMPXCHG_LOCKREF
|
||||
|
||||
@@ -9,10 +9,9 @@
|
||||
* way above 4G.
|
||||
*
|
||||
* Always use weak definitions for percpu variables in modules.
|
||||
* Therefore, we have enabled CONFIG_ARCH_MODULE_NEEDS_WEAK_PER_CPU
|
||||
* in the Kconfig.
|
||||
*/
|
||||
#if defined(MODULE) && defined(CONFIG_SMP)
|
||||
#define ARCH_NEEDS_WEAK_PER_CPU
|
||||
#endif
|
||||
|
||||
#include <asm-generic/percpu.h>
|
||||
|
||||
|
||||
@@ -1004,7 +1004,7 @@ static void __init reserve_crashkernel(void)
|
||||
total_mem = get_total_mem();
|
||||
ret = parse_crashkernel(boot_command_line, total_mem,
|
||||
&crash_size, &crash_base,
|
||||
NULL, NULL);
|
||||
NULL, NULL, NULL);
|
||||
/* invalid value specified or crashkernel=0 */
|
||||
if (ret || !crash_size)
|
||||
return;
|
||||
|
||||
+1
-1
@@ -268,7 +268,7 @@ do_page_fault(unsigned long addr, unsigned int fsr, struct pt_regs *regs)
|
||||
int sig, code;
|
||||
vm_fault_t fault;
|
||||
unsigned int flags = FAULT_FLAG_DEFAULT;
|
||||
unsigned long vm_flags = VM_ACCESS_FLAGS;
|
||||
vm_flags_t vm_flags = VM_ACCESS_FLAGS;
|
||||
|
||||
if (kprobe_page_fault(regs, fsr))
|
||||
return 0;
|
||||
|
||||
@@ -11,10 +11,10 @@
|
||||
#include <linux/shmem_fs.h>
|
||||
#include <linux/types.h>
|
||||
|
||||
static inline unsigned long arch_calc_vm_prot_bits(unsigned long prot,
|
||||
static inline vm_flags_t arch_calc_vm_prot_bits(unsigned long prot,
|
||||
unsigned long pkey)
|
||||
{
|
||||
unsigned long ret = 0;
|
||||
vm_flags_t ret = 0;
|
||||
|
||||
if (system_supports_bti() && (prot & PROT_BTI))
|
||||
ret |= VM_ARM64_BTI;
|
||||
@@ -34,8 +34,8 @@ static inline unsigned long arch_calc_vm_prot_bits(unsigned long prot,
|
||||
}
|
||||
#define arch_calc_vm_prot_bits(prot, pkey) arch_calc_vm_prot_bits(prot, pkey)
|
||||
|
||||
static inline unsigned long arch_calc_vm_flag_bits(struct file *file,
|
||||
unsigned long flags)
|
||||
static inline vm_flags_t arch_calc_vm_flag_bits(struct file *file,
|
||||
unsigned long flags)
|
||||
{
|
||||
/*
|
||||
* Only allow MTE on anonymous mappings as these are guaranteed to be
|
||||
@@ -68,7 +68,7 @@ static inline bool arch_validate_prot(unsigned long prot,
|
||||
}
|
||||
#define arch_validate_prot(prot, addr) arch_validate_prot(prot, addr)
|
||||
|
||||
static inline bool arch_validate_flags(unsigned long vm_flags)
|
||||
static inline bool arch_validate_flags(vm_flags_t vm_flags)
|
||||
{
|
||||
if (system_supports_mte()) {
|
||||
/*
|
||||
|
||||
@@ -1620,6 +1620,14 @@ static inline void update_mmu_cache_range(struct vm_fault *vmf,
|
||||
*/
|
||||
#define arch_wants_old_prefaulted_pte cpu_has_hw_af
|
||||
|
||||
/*
|
||||
* Request exec memory is read into pagecache in at least 64K folios. This size
|
||||
* can be contpte-mapped when 4K base pages are in use (16 pages into 1 iTLB
|
||||
* entry), and HPA can coalesce it (4 pages into 1 TLB entry) when 16K base
|
||||
* pages are in use.
|
||||
*/
|
||||
#define exec_folio_order() ilog2(SZ_64K >> PAGE_SHIFT)
|
||||
|
||||
static inline bool pud_sect_supported(void)
|
||||
{
|
||||
return PAGE_SIZE == SZ_4K;
|
||||
@@ -1636,6 +1644,16 @@ extern void ptep_modify_prot_commit(struct vm_area_struct *vma,
|
||||
unsigned long addr, pte_t *ptep,
|
||||
pte_t old_pte, pte_t new_pte);
|
||||
|
||||
#define modify_prot_start_ptes modify_prot_start_ptes
|
||||
extern pte_t modify_prot_start_ptes(struct vm_area_struct *vma,
|
||||
unsigned long addr, pte_t *ptep,
|
||||
unsigned int nr);
|
||||
|
||||
#define modify_prot_commit_ptes modify_prot_commit_ptes
|
||||
extern void modify_prot_commit_ptes(struct vm_area_struct *vma, unsigned long addr,
|
||||
pte_t *ptep, pte_t old_pte, pte_t pte,
|
||||
unsigned int nr);
|
||||
|
||||
#ifdef CONFIG_ARM64_CONTPTE
|
||||
|
||||
/*
|
||||
|
||||
@@ -322,16 +322,6 @@ static inline bool arch_tlbbatch_should_defer(struct mm_struct *mm)
|
||||
return true;
|
||||
}
|
||||
|
||||
/*
|
||||
* If mprotect/munmap/etc occurs during TLB batched flushing, we need to
|
||||
* synchronise all the TLBI issued with a DSB to avoid the race mentioned in
|
||||
* flush_tlb_batched_pending().
|
||||
*/
|
||||
static inline void arch_flush_tlb_batched_pending(struct mm_struct *mm)
|
||||
{
|
||||
dsb(ish);
|
||||
}
|
||||
|
||||
/*
|
||||
* To support TLB batched flush for multiple pages unmapping, we only send
|
||||
* the TLBI for each page in arch_tlbbatch_add_pending() and wait for the
|
||||
|
||||
@@ -555,7 +555,7 @@ static int __kprobes do_page_fault(unsigned long far, unsigned long esr,
|
||||
const struct fault_info *inf;
|
||||
struct mm_struct *mm = current->mm;
|
||||
vm_fault_t fault;
|
||||
unsigned long vm_flags;
|
||||
vm_flags_t vm_flags;
|
||||
unsigned int mm_flags = FAULT_FLAG_DEFAULT;
|
||||
unsigned long addr = untagged_addr(far);
|
||||
struct vm_area_struct *vma;
|
||||
|
||||
@@ -106,7 +106,7 @@ static void __init arch_reserve_crashkernel(void)
|
||||
|
||||
ret = parse_crashkernel(boot_command_line, memblock_phys_mem_size(),
|
||||
&crash_size, &crash_base,
|
||||
&low_size, &high);
|
||||
&low_size, NULL, &high);
|
||||
if (ret)
|
||||
return;
|
||||
|
||||
|
||||
@@ -81,7 +81,7 @@ static int __init adjust_protection_map(void)
|
||||
}
|
||||
arch_initcall(adjust_protection_map);
|
||||
|
||||
pgprot_t vm_get_page_prot(unsigned long vm_flags)
|
||||
pgprot_t vm_get_page_prot(vm_flags_t vm_flags)
|
||||
{
|
||||
ptdesc_t prot;
|
||||
|
||||
|
||||
+23
-5
@@ -26,6 +26,7 @@
|
||||
#include <linux/set_memory.h>
|
||||
#include <linux/kfence.h>
|
||||
#include <linux/pkeys.h>
|
||||
#include <linux/mm_inline.h>
|
||||
|
||||
#include <asm/barrier.h>
|
||||
#include <asm/cputype.h>
|
||||
@@ -1519,24 +1520,41 @@ static int __init prevent_bootmem_remove_init(void)
|
||||
early_initcall(prevent_bootmem_remove_init);
|
||||
#endif
|
||||
|
||||
pte_t ptep_modify_prot_start(struct vm_area_struct *vma, unsigned long addr, pte_t *ptep)
|
||||
pte_t modify_prot_start_ptes(struct vm_area_struct *vma, unsigned long addr,
|
||||
pte_t *ptep, unsigned int nr)
|
||||
{
|
||||
pte_t pte = get_and_clear_ptes(vma->vm_mm, addr, ptep, nr);
|
||||
|
||||
if (alternative_has_cap_unlikely(ARM64_WORKAROUND_2645198)) {
|
||||
/*
|
||||
* Break-before-make (BBM) is required for all user space mappings
|
||||
* when the permission changes from executable to non-executable
|
||||
* in cases where cpu is affected with errata #2645198.
|
||||
*/
|
||||
if (pte_user_exec(ptep_get(ptep)))
|
||||
return ptep_clear_flush(vma, addr, ptep);
|
||||
if (pte_accessible(vma->vm_mm, pte) && pte_user_exec(pte))
|
||||
__flush_tlb_range(vma, addr, nr * PAGE_SIZE,
|
||||
PAGE_SIZE, true, 3);
|
||||
}
|
||||
return ptep_get_and_clear(vma->vm_mm, addr, ptep);
|
||||
|
||||
return pte;
|
||||
}
|
||||
|
||||
pte_t ptep_modify_prot_start(struct vm_area_struct *vma, unsigned long addr, pte_t *ptep)
|
||||
{
|
||||
return modify_prot_start_ptes(vma, addr, ptep, 1);
|
||||
}
|
||||
|
||||
void modify_prot_commit_ptes(struct vm_area_struct *vma, unsigned long addr,
|
||||
pte_t *ptep, pte_t old_pte, pte_t pte,
|
||||
unsigned int nr)
|
||||
{
|
||||
set_ptes(vma->vm_mm, addr, ptep, pte, nr);
|
||||
}
|
||||
|
||||
void ptep_modify_prot_commit(struct vm_area_struct *vma, unsigned long addr, pte_t *ptep,
|
||||
pte_t old_pte, pte_t pte)
|
||||
{
|
||||
set_pte_at(vma->vm_mm, addr, ptep, pte);
|
||||
modify_prot_commit_ptes(vma, addr, ptep, old_pte, pte, 1);
|
||||
}
|
||||
|
||||
/*
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
// SPDX-License-Identifier: GPL-2.0
|
||||
#include <linux/debugfs.h>
|
||||
#include <linux/memory_hotplug.h>
|
||||
#include <linux/seq_file.h>
|
||||
|
||||
#include <asm/ptdump.h>
|
||||
@@ -9,9 +8,7 @@ static int ptdump_show(struct seq_file *m, void *v)
|
||||
{
|
||||
struct ptdump_info *info = m->private;
|
||||
|
||||
get_online_mems();
|
||||
ptdump_walk(m, info);
|
||||
put_online_mems();
|
||||
return 0;
|
||||
}
|
||||
DEFINE_SHOW_ATTRIBUTE(ptdump);
|
||||
|
||||
@@ -10,20 +10,6 @@
|
||||
|
||||
uint64_t pmd_to_entrylo(unsigned long pmd_val);
|
||||
|
||||
#define __HAVE_ARCH_PREPARE_HUGEPAGE_RANGE
|
||||
static inline int prepare_hugepage_range(struct file *file,
|
||||
unsigned long addr,
|
||||
unsigned long len)
|
||||
{
|
||||
unsigned long task_size = STACK_TOP;
|
||||
|
||||
if (len > task_size)
|
||||
return -ENOMEM;
|
||||
if (task_size - len < addr)
|
||||
return -EINVAL;
|
||||
return 0;
|
||||
}
|
||||
|
||||
#define __HAVE_ARCH_HUGE_PTEP_GET_AND_CLEAR
|
||||
static inline pte_t huge_ptep_get_and_clear(struct mm_struct *mm,
|
||||
unsigned long addr, pte_t *ptep,
|
||||
|
||||
@@ -265,7 +265,7 @@ static void __init arch_reserve_crashkernel(void)
|
||||
return;
|
||||
|
||||
ret = parse_crashkernel(boot_command_line, memblock_phys_mem_size(),
|
||||
&crash_size, &crash_base, &low_size, &high);
|
||||
&crash_size, &crash_base, &low_size, NULL, &high);
|
||||
if (ret)
|
||||
return;
|
||||
|
||||
|
||||
@@ -117,7 +117,7 @@ static int __set_memory(unsigned long addr, int numpages, pgprot_t set_mask, pgp
|
||||
return 0;
|
||||
|
||||
mmap_write_lock(&init_mm);
|
||||
ret = walk_page_range_novma(&init_mm, start, end, &pageattr_ops, NULL, &masks);
|
||||
ret = walk_kernel_page_table_range(start, end, &pageattr_ops, NULL, &masks);
|
||||
mmap_write_unlock(&init_mm);
|
||||
|
||||
flush_tlb_kernel_range(start, end);
|
||||
|
||||
@@ -11,20 +11,6 @@
|
||||
|
||||
#include <asm/page.h>
|
||||
|
||||
#define __HAVE_ARCH_PREPARE_HUGEPAGE_RANGE
|
||||
static inline int prepare_hugepage_range(struct file *file,
|
||||
unsigned long addr,
|
||||
unsigned long len)
|
||||
{
|
||||
unsigned long task_size = STACK_TOP;
|
||||
|
||||
if (len > task_size)
|
||||
return -ENOMEM;
|
||||
if (task_size - len < addr)
|
||||
return -EINVAL;
|
||||
return 0;
|
||||
}
|
||||
|
||||
#define __HAVE_ARCH_HUGE_PTEP_GET_AND_CLEAR
|
||||
static inline pte_t huge_ptep_get_and_clear(struct mm_struct *mm,
|
||||
unsigned long addr, pte_t *ptep,
|
||||
|
||||
@@ -458,7 +458,7 @@ static void __init mips_parse_crashkernel(void)
|
||||
total_mem = memblock_phys_mem_size();
|
||||
ret = parse_crashkernel(boot_command_line, total_mem,
|
||||
&crash_size, &crash_base,
|
||||
NULL, NULL);
|
||||
NULL, NULL, NULL);
|
||||
if (ret != 0 || crash_size <= 0)
|
||||
return;
|
||||
|
||||
|
||||
@@ -75,7 +75,7 @@ void *arch_dma_set_uncached(void *cpu_addr, size_t size)
|
||||
* them and setting the cache-inhibit bit.
|
||||
*/
|
||||
mmap_write_lock(&init_mm);
|
||||
error = walk_page_range_novma(&init_mm, va, va + size,
|
||||
error = walk_kernel_page_table_range(va, va + size,
|
||||
&set_nocache_walk_ops, NULL, NULL);
|
||||
mmap_write_unlock(&init_mm);
|
||||
|
||||
@@ -90,7 +90,7 @@ void arch_dma_clear_uncached(void *cpu_addr, size_t size)
|
||||
|
||||
mmap_write_lock(&init_mm);
|
||||
/* walk_page_range shouldn't be able to fail here */
|
||||
WARN_ON(walk_page_range_novma(&init_mm, va, va + size,
|
||||
WARN_ON(walk_kernel_page_table_range(va, va + size,
|
||||
&clear_nocache_walk_ops, NULL, NULL));
|
||||
mmap_write_unlock(&init_mm);
|
||||
}
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
|
||||
#include <asm/book3s/64/hash-pkey.h>
|
||||
|
||||
static inline u64 vmflag_to_pte_pkey_bits(u64 vm_flags)
|
||||
static inline u64 vmflag_to_pte_pkey_bits(vm_flags_t vm_flags)
|
||||
{
|
||||
if (!mmu_has_feature(MMU_FTR_PKEY))
|
||||
return 0x0UL;
|
||||
|
||||
@@ -14,7 +14,7 @@
|
||||
#include <asm/cpu_has_feature.h>
|
||||
#include <asm/firmware.h>
|
||||
|
||||
static inline unsigned long arch_calc_vm_prot_bits(unsigned long prot,
|
||||
static inline vm_flags_t arch_calc_vm_prot_bits(unsigned long prot,
|
||||
unsigned long pkey)
|
||||
{
|
||||
#ifdef CONFIG_PPC_MEM_KEYS
|
||||
|
||||
@@ -30,9 +30,9 @@ extern u32 reserved_allocation_mask; /* bits set for reserved keys */
|
||||
#endif
|
||||
|
||||
|
||||
static inline u64 pkey_to_vmflag_bits(u16 pkey)
|
||||
static inline vm_flags_t pkey_to_vmflag_bits(u16 pkey)
|
||||
{
|
||||
return (((u64)pkey << VM_PKEY_SHIFT) & ARCH_VM_PKEY_FLAGS);
|
||||
return (((vm_flags_t)pkey << VM_PKEY_SHIFT) & ARCH_VM_PKEY_FLAGS);
|
||||
}
|
||||
|
||||
static inline int vma_pkey(struct vm_area_struct *vma)
|
||||
|
||||
@@ -335,7 +335,7 @@ static __init u64 fadump_calculate_reserve_size(void)
|
||||
* memory at a predefined offset.
|
||||
*/
|
||||
ret = parse_crashkernel(boot_command_line, memblock_phys_mem_size(),
|
||||
&size, &base, NULL, NULL);
|
||||
&size, &base, NULL, NULL, NULL);
|
||||
if (ret == 0 && size > 0) {
|
||||
unsigned long max_size;
|
||||
|
||||
|
||||
@@ -132,7 +132,7 @@ void __init arch_reserve_crashkernel(void)
|
||||
|
||||
/* use common parsing */
|
||||
ret = parse_crashkernel(boot_command_line, total_mem_sz, &crash_size,
|
||||
&crash_base, NULL, NULL);
|
||||
&crash_base, NULL, NULL, NULL);
|
||||
|
||||
if (ret)
|
||||
return;
|
||||
|
||||
@@ -393,7 +393,7 @@ static int kvmppc_memslot_page_merge(struct kvm *kvm,
|
||||
{
|
||||
unsigned long gfn = memslot->base_gfn;
|
||||
unsigned long end, start = gfn_to_hva(kvm, gfn);
|
||||
unsigned long vm_flags;
|
||||
vm_flags_t vm_flags;
|
||||
int ret = 0;
|
||||
struct vm_area_struct *vma;
|
||||
int merge_flag = (merge) ? MADV_MERGEABLE : MADV_UNMERGEABLE;
|
||||
|
||||
@@ -643,7 +643,7 @@ unsigned long memremap_compat_align(void)
|
||||
EXPORT_SYMBOL_GPL(memremap_compat_align);
|
||||
#endif
|
||||
|
||||
pgprot_t vm_get_page_prot(unsigned long vm_flags)
|
||||
pgprot_t vm_get_page_prot(vm_flags_t vm_flags)
|
||||
{
|
||||
unsigned long prot;
|
||||
|
||||
|
||||
@@ -1122,18 +1122,25 @@ int __meminit radix__vmemmap_populate(unsigned long start, unsigned long end, in
|
||||
pte_t *pte;
|
||||
|
||||
/*
|
||||
* Make sure we align the start vmemmap addr so that we calculate
|
||||
* the correct start_pfn in altmap boundary check to decided whether
|
||||
* we should use altmap or RAM based backing memory allocation. Also
|
||||
* the address need to be aligned for set_pte operation.
|
||||
|
||||
* If the start addr is already PMD_SIZE aligned we will try to use
|
||||
* a pmd mapping. We don't want to be too aggressive here beacause
|
||||
* that will cause more allocations in RAM. So only if the namespace
|
||||
* vmemmap start addr is PMD_SIZE aligned we will use PMD mapping.
|
||||
* If altmap is present, Make sure we align the start vmemmap addr
|
||||
* to PAGE_SIZE so that we calculate the correct start_pfn in
|
||||
* altmap boundary check to decide whether we should use altmap or
|
||||
* RAM based backing memory allocation. Also the address need to be
|
||||
* aligned for set_pte operation. If the start addr is already
|
||||
* PMD_SIZE aligned and with in the altmap boundary then we will
|
||||
* try to use a pmd size altmap mapping else we go for page size
|
||||
* mapping.
|
||||
*
|
||||
* If altmap is not present, align the vmemmap addr to PMD_SIZE and
|
||||
* always allocate a PMD size page for vmemmap backing.
|
||||
*
|
||||
*/
|
||||
|
||||
start = ALIGN_DOWN(start, PAGE_SIZE);
|
||||
if (altmap)
|
||||
start = ALIGN_DOWN(start, PAGE_SIZE);
|
||||
else
|
||||
start = ALIGN_DOWN(start, PMD_SIZE);
|
||||
|
||||
for (addr = start; addr < end; addr = next) {
|
||||
next = pmd_addr_end(addr, end);
|
||||
|
||||
@@ -1159,7 +1166,7 @@ int __meminit radix__vmemmap_populate(unsigned long start, unsigned long end, in
|
||||
* in altmap block allocation failures, in which case
|
||||
* we fallback to RAM for vmemmap allocation.
|
||||
*/
|
||||
if (!IS_ALIGNED(addr, PMD_SIZE) || (altmap &&
|
||||
if (altmap && (!IS_ALIGNED(addr, PMD_SIZE) ||
|
||||
altmap_cross_boundary(altmap, addr, PMD_SIZE))) {
|
||||
/*
|
||||
* make sure we don't create altmap mappings
|
||||
@@ -1173,7 +1180,7 @@ int __meminit radix__vmemmap_populate(unsigned long start, unsigned long end, in
|
||||
vmemmap_set_pmd(pmd, p, node, addr, next);
|
||||
pr_debug("PMD_SIZE vmemmap mapping\n");
|
||||
continue;
|
||||
} else if (altmap) {
|
||||
} else {
|
||||
/*
|
||||
* A vmemmap block allocation can fail due to
|
||||
* alignment requirements and we trying to align
|
||||
|
||||
@@ -178,7 +178,7 @@ static void __init get_crash_kernel(void *fdt, unsigned long size)
|
||||
int ret;
|
||||
|
||||
ret = parse_crashkernel(boot_command_line, size, &crash_size,
|
||||
&crash_base, NULL, NULL);
|
||||
&crash_base, NULL, NULL, NULL);
|
||||
if (ret != 0 || crash_size == 0)
|
||||
return;
|
||||
if (crash_base == 0)
|
||||
|
||||
@@ -532,7 +532,6 @@ static int cmm_migratepage(struct balloon_dev_info *b_dev_info,
|
||||
|
||||
spin_lock_irqsave(&b_dev_info->pages_lock, flags);
|
||||
balloon_page_insert(b_dev_info, newpage);
|
||||
balloon_page_delete(page);
|
||||
b_dev_info->isolated_pages--;
|
||||
spin_unlock_irqrestore(&b_dev_info->pages_lock, flags);
|
||||
|
||||
@@ -542,6 +541,7 @@ static int cmm_migratepage(struct balloon_dev_info *b_dev_info,
|
||||
*/
|
||||
plpar_page_set_active(page);
|
||||
|
||||
balloon_page_finalize(page);
|
||||
/* balloon page list reference */
|
||||
put_page(page);
|
||||
|
||||
|
||||
@@ -29,7 +29,7 @@ struct pci_controller *init_phb_dynamic(struct device_node *dn)
|
||||
nid = of_node_to_nid(dn);
|
||||
if (likely((nid) >= 0)) {
|
||||
if (!node_online(nid)) {
|
||||
if (__register_one_node(nid)) {
|
||||
if (register_one_node(nid)) {
|
||||
pr_err("PCI: Failed to register node %d\n", nid);
|
||||
} else {
|
||||
update_numa_distance(dn);
|
||||
|
||||
@@ -63,7 +63,6 @@ void flush_pud_tlb_range(struct vm_area_struct *vma, unsigned long start,
|
||||
bool arch_tlbbatch_should_defer(struct mm_struct *mm);
|
||||
void arch_tlbbatch_add_pending(struct arch_tlbflush_unmap_batch *batch,
|
||||
struct mm_struct *mm, unsigned long start, unsigned long end);
|
||||
void arch_flush_tlb_batched_pending(struct mm_struct *mm);
|
||||
void arch_tlbbatch_flush(struct arch_tlbflush_unmap_batch *batch);
|
||||
|
||||
extern unsigned long tlb_flush_all_threshold;
|
||||
|
||||
@@ -1404,7 +1404,7 @@ static void __init arch_reserve_crashkernel(void)
|
||||
|
||||
ret = parse_crashkernel(boot_command_line, memblock_phys_mem_size(),
|
||||
&crash_size, &crash_base,
|
||||
&low_size, &high);
|
||||
&low_size, NULL, &high);
|
||||
if (ret)
|
||||
return;
|
||||
|
||||
|
||||
@@ -299,7 +299,7 @@ static int __set_memory(unsigned long addr, int numpages, pgprot_t set_mask,
|
||||
if (ret)
|
||||
goto unlock;
|
||||
|
||||
ret = walk_page_range_novma(&init_mm, lm_start, lm_end,
|
||||
ret = walk_kernel_page_table_range(lm_start, lm_end,
|
||||
&pageattr_ops, NULL, &masks);
|
||||
if (ret)
|
||||
goto unlock;
|
||||
@@ -317,13 +317,13 @@ static int __set_memory(unsigned long addr, int numpages, pgprot_t set_mask,
|
||||
if (ret)
|
||||
goto unlock;
|
||||
|
||||
ret = walk_page_range_novma(&init_mm, lm_start, lm_end,
|
||||
ret = walk_kernel_page_table_range(lm_start, lm_end,
|
||||
&pageattr_ops, NULL, &masks);
|
||||
if (ret)
|
||||
goto unlock;
|
||||
}
|
||||
|
||||
ret = walk_page_range_novma(&init_mm, start, end, &pageattr_ops, NULL,
|
||||
ret = walk_kernel_page_table_range(start, end, &pageattr_ops, NULL,
|
||||
&masks);
|
||||
|
||||
unlock:
|
||||
@@ -335,7 +335,7 @@ unlock:
|
||||
*/
|
||||
flush_tlb_all();
|
||||
#else
|
||||
ret = walk_page_range_novma(&init_mm, start, end, &pageattr_ops, NULL,
|
||||
ret = walk_kernel_page_table_range(start, end, &pageattr_ops, NULL,
|
||||
&masks);
|
||||
|
||||
mmap_write_unlock(&init_mm);
|
||||
|
||||
@@ -6,7 +6,6 @@
|
||||
#include <linux/efi.h>
|
||||
#include <linux/init.h>
|
||||
#include <linux/debugfs.h>
|
||||
#include <linux/memory_hotplug.h>
|
||||
#include <linux/seq_file.h>
|
||||
#include <linux/ptdump.h>
|
||||
|
||||
@@ -413,9 +412,7 @@ bool ptdump_check_wx(void)
|
||||
|
||||
static int ptdump_show(struct seq_file *m, void *v)
|
||||
{
|
||||
get_online_mems();
|
||||
ptdump_walk(m, m->private);
|
||||
put_online_mems();
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -234,11 +234,6 @@ void arch_tlbbatch_add_pending(struct arch_tlbflush_unmap_batch *batch,
|
||||
mmu_notifier_arch_invalidate_secondary_tlbs(mm, start, end);
|
||||
}
|
||||
|
||||
void arch_flush_tlb_batched_pending(struct mm_struct *mm)
|
||||
{
|
||||
flush_tlb_mm(mm);
|
||||
}
|
||||
|
||||
void arch_tlbbatch_flush(struct arch_tlbflush_unmap_batch *batch)
|
||||
{
|
||||
__flush_tlb_range(NULL, &batch->cpumask,
|
||||
|
||||
@@ -130,6 +130,7 @@ config S390
|
||||
select ARCH_INLINE_WRITE_UNLOCK_IRQ
|
||||
select ARCH_INLINE_WRITE_UNLOCK_IRQRESTORE
|
||||
select ARCH_MHP_MEMMAP_ON_MEMORY_ENABLE
|
||||
select ARCH_MODULE_NEEDS_WEAK_PER_CPU
|
||||
select ARCH_STACKWALK
|
||||
select ARCH_SUPPORTS_ATOMIC_RMW
|
||||
select ARCH_SUPPORTS_DEBUG_PAGEALLOC
|
||||
|
||||
@@ -16,10 +16,9 @@
|
||||
* For 64 bit module code, the module may be more than 4G above the
|
||||
* per cpu area, use weak definitions to force the compiler to
|
||||
* generate external references.
|
||||
* Therefore, we have enabled CONFIG_ARCH_MODULE_NEEDS_WEAK_PER_CPU
|
||||
* in the Kconfig.
|
||||
*/
|
||||
#if defined(MODULE)
|
||||
#define ARCH_NEEDS_WEAK_PER_CPU
|
||||
#endif
|
||||
|
||||
/*
|
||||
* We use a compare-and-swap loop since that uses less cpu cycles than
|
||||
|
||||
@@ -602,7 +602,7 @@ static void __init reserve_crashkernel(void)
|
||||
int rc;
|
||||
|
||||
rc = parse_crashkernel(boot_command_line, ident_map_size,
|
||||
&crash_size, &crash_base, NULL, NULL);
|
||||
&crash_size, &crash_base, NULL, NULL, NULL);
|
||||
|
||||
crash_base = ALIGN(crash_base, KEXEC_CRASH_MEM_ALIGN);
|
||||
crash_size = ALIGN(crash_size, KEXEC_CRASH_MEM_ALIGN);
|
||||
|
||||
@@ -245,11 +245,9 @@ static int ptdump_show(struct seq_file *m, void *v)
|
||||
.marker = markers,
|
||||
};
|
||||
|
||||
get_online_mems();
|
||||
mutex_lock(&cpa_mutex);
|
||||
ptdump_walk_pgd(&st.ptdump, &init_mm, NULL);
|
||||
mutex_unlock(&cpa_mutex);
|
||||
put_online_mems();
|
||||
return 0;
|
||||
}
|
||||
DEFINE_SHOW_ATTRIBUTE(ptdump);
|
||||
|
||||
@@ -170,11 +170,6 @@ void pte_free_defer(struct mm_struct *mm, pgtable_t pgtable)
|
||||
struct ptdesc *ptdesc = virt_to_ptdesc(pgtable);
|
||||
|
||||
call_rcu(&ptdesc->pt_rcu_head, pte_free_now);
|
||||
/*
|
||||
* THPs are not allowed for KVM guests. Warn if pgste ever reaches here.
|
||||
* Turn to the generic pte_free_defer() version once gmap is removed.
|
||||
*/
|
||||
WARN_ON_ONCE(mm_has_pgste(mm));
|
||||
}
|
||||
#endif /* CONFIG_TRANSPARENT_HUGEPAGE */
|
||||
|
||||
|
||||
@@ -321,7 +321,6 @@ pte_t ptep_modify_prot_start(struct vm_area_struct *vma, unsigned long addr,
|
||||
int nodat;
|
||||
struct mm_struct *mm = vma->vm_mm;
|
||||
|
||||
preempt_disable();
|
||||
pgste = ptep_xchg_start(mm, addr, ptep);
|
||||
nodat = !!(pgste_val(pgste) & _PGSTE_GPS_NODAT);
|
||||
old = ptep_flush_lazy(mm, addr, ptep, nodat);
|
||||
@@ -346,7 +345,6 @@ void ptep_modify_prot_commit(struct vm_area_struct *vma, unsigned long addr,
|
||||
} else {
|
||||
set_pte(ptep, pte);
|
||||
}
|
||||
preempt_enable();
|
||||
}
|
||||
|
||||
static inline void pmdp_idte_local(struct mm_struct *mm,
|
||||
|
||||
+2
-3
@@ -63,13 +63,12 @@ void *vmem_crst_alloc(unsigned long val)
|
||||
|
||||
pte_t __ref *vmem_pte_alloc(void)
|
||||
{
|
||||
unsigned long size = PTRS_PER_PTE * sizeof(pte_t);
|
||||
pte_t *pte;
|
||||
|
||||
if (slab_is_available())
|
||||
pte = (pte_t *) page_table_alloc(&init_mm);
|
||||
pte = (pte_t *)page_table_alloc(&init_mm);
|
||||
else
|
||||
pte = (pte_t *) memblock_alloc(size, size);
|
||||
pte = (pte_t *)memblock_alloc(PAGE_SIZE, PAGE_SIZE);
|
||||
if (!pte)
|
||||
return NULL;
|
||||
memset64((u64 *)pte, _PAGE_INVALID, PTRS_PER_PTE);
|
||||
|
||||
@@ -146,7 +146,7 @@ void __init reserve_crashkernel(void)
|
||||
return;
|
||||
|
||||
ret = parse_crashkernel(boot_command_line, memblock_phys_mem_size(),
|
||||
&crash_size, &crash_base, NULL, NULL);
|
||||
&crash_size, &crash_base, NULL, NULL, NULL);
|
||||
if (ret == 0 && crash_size > 0) {
|
||||
crashk_res.start = crash_base;
|
||||
crashk_res.end = crash_base + crash_size - 1;
|
||||
|
||||
@@ -50,11 +50,6 @@ static inline int huge_ptep_set_access_flags(struct vm_area_struct *vma,
|
||||
return changed;
|
||||
}
|
||||
|
||||
#define __HAVE_ARCH_HUGETLB_FREE_PGD_RANGE
|
||||
void hugetlb_free_pgd_range(struct mmu_gather *tlb, unsigned long addr,
|
||||
unsigned long end, unsigned long floor,
|
||||
unsigned long ceiling);
|
||||
|
||||
#include <asm-generic/hugetlb.h>
|
||||
|
||||
#endif /* _ASM_SPARC64_HUGETLB_H */
|
||||
|
||||
@@ -28,7 +28,7 @@ static inline void ipi_set_tstate_mcde(void *arg)
|
||||
}
|
||||
|
||||
#define arch_calc_vm_prot_bits(prot, pkey) sparc_calc_vm_prot_bits(prot)
|
||||
static inline unsigned long sparc_calc_vm_prot_bits(unsigned long prot)
|
||||
static inline vm_flags_t sparc_calc_vm_prot_bits(unsigned long prot)
|
||||
{
|
||||
if (adi_capable() && (prot & PROT_ADI)) {
|
||||
struct pt_regs *regs;
|
||||
@@ -58,7 +58,7 @@ static inline int sparc_validate_prot(unsigned long prot, unsigned long addr)
|
||||
/* arch_validate_flags() - Ensure combination of flags is valid for a
|
||||
* VMA.
|
||||
*/
|
||||
static inline bool arch_validate_flags(unsigned long vm_flags)
|
||||
static inline bool arch_validate_flags(vm_flags_t vm_flags)
|
||||
{
|
||||
/* If ADI is being enabled on this VMA, check for ADI
|
||||
* capability on the platform and ensure VMA is suitable
|
||||
|
||||
@@ -295,122 +295,3 @@ pte_t huge_ptep_get_and_clear(struct mm_struct *mm, unsigned long addr,
|
||||
|
||||
return entry;
|
||||
}
|
||||
|
||||
static void hugetlb_free_pte_range(struct mmu_gather *tlb, pmd_t *pmd,
|
||||
unsigned long addr)
|
||||
{
|
||||
pgtable_t token = pmd_pgtable(*pmd);
|
||||
|
||||
pmd_clear(pmd);
|
||||
pte_free_tlb(tlb, token, addr);
|
||||
mm_dec_nr_ptes(tlb->mm);
|
||||
}
|
||||
|
||||
static void hugetlb_free_pmd_range(struct mmu_gather *tlb, pud_t *pud,
|
||||
unsigned long addr, unsigned long end,
|
||||
unsigned long floor, unsigned long ceiling)
|
||||
{
|
||||
pmd_t *pmd;
|
||||
unsigned long next;
|
||||
unsigned long start;
|
||||
|
||||
start = addr;
|
||||
pmd = pmd_offset(pud, addr);
|
||||
do {
|
||||
next = pmd_addr_end(addr, end);
|
||||
if (pmd_none(*pmd))
|
||||
continue;
|
||||
if (is_hugetlb_pmd(*pmd))
|
||||
pmd_clear(pmd);
|
||||
else
|
||||
hugetlb_free_pte_range(tlb, pmd, addr);
|
||||
} while (pmd++, addr = next, addr != end);
|
||||
|
||||
start &= PUD_MASK;
|
||||
if (start < floor)
|
||||
return;
|
||||
if (ceiling) {
|
||||
ceiling &= PUD_MASK;
|
||||
if (!ceiling)
|
||||
return;
|
||||
}
|
||||
if (end - 1 > ceiling - 1)
|
||||
return;
|
||||
|
||||
pmd = pmd_offset(pud, start);
|
||||
pud_clear(pud);
|
||||
pmd_free_tlb(tlb, pmd, start);
|
||||
mm_dec_nr_pmds(tlb->mm);
|
||||
}
|
||||
|
||||
static void hugetlb_free_pud_range(struct mmu_gather *tlb, p4d_t *p4d,
|
||||
unsigned long addr, unsigned long end,
|
||||
unsigned long floor, unsigned long ceiling)
|
||||
{
|
||||
pud_t *pud;
|
||||
unsigned long next;
|
||||
unsigned long start;
|
||||
|
||||
start = addr;
|
||||
pud = pud_offset(p4d, addr);
|
||||
do {
|
||||
next = pud_addr_end(addr, end);
|
||||
if (pud_none_or_clear_bad(pud))
|
||||
continue;
|
||||
if (is_hugetlb_pud(*pud))
|
||||
pud_clear(pud);
|
||||
else
|
||||
hugetlb_free_pmd_range(tlb, pud, addr, next, floor,
|
||||
ceiling);
|
||||
} while (pud++, addr = next, addr != end);
|
||||
|
||||
start &= PGDIR_MASK;
|
||||
if (start < floor)
|
||||
return;
|
||||
if (ceiling) {
|
||||
ceiling &= PGDIR_MASK;
|
||||
if (!ceiling)
|
||||
return;
|
||||
}
|
||||
if (end - 1 > ceiling - 1)
|
||||
return;
|
||||
|
||||
pud = pud_offset(p4d, start);
|
||||
p4d_clear(p4d);
|
||||
pud_free_tlb(tlb, pud, start);
|
||||
mm_dec_nr_puds(tlb->mm);
|
||||
}
|
||||
|
||||
void hugetlb_free_pgd_range(struct mmu_gather *tlb,
|
||||
unsigned long addr, unsigned long end,
|
||||
unsigned long floor, unsigned long ceiling)
|
||||
{
|
||||
pgd_t *pgd;
|
||||
p4d_t *p4d;
|
||||
unsigned long next;
|
||||
|
||||
addr &= PMD_MASK;
|
||||
if (addr < floor) {
|
||||
addr += PMD_SIZE;
|
||||
if (!addr)
|
||||
return;
|
||||
}
|
||||
if (ceiling) {
|
||||
ceiling &= PMD_MASK;
|
||||
if (!ceiling)
|
||||
return;
|
||||
}
|
||||
if (end - 1 > ceiling - 1)
|
||||
end -= PMD_SIZE;
|
||||
if (addr > end - 1)
|
||||
return;
|
||||
|
||||
pgd = pgd_offset(tlb->mm, addr);
|
||||
p4d = p4d_offset(pgd, addr);
|
||||
do {
|
||||
next = p4d_addr_end(addr, end);
|
||||
if (p4d_none_or_clear_bad(p4d))
|
||||
continue;
|
||||
hugetlb_free_pud_range(tlb, p4d, addr, next, floor, ceiling);
|
||||
} while (p4d++, addr = next, addr != end);
|
||||
}
|
||||
|
||||
@@ -3202,7 +3202,7 @@ void copy_highpage(struct page *to, struct page *from)
|
||||
}
|
||||
EXPORT_SYMBOL(copy_highpage);
|
||||
|
||||
pgprot_t vm_get_page_prot(unsigned long vm_flags)
|
||||
pgprot_t vm_get_page_prot(vm_flags_t vm_flags)
|
||||
{
|
||||
unsigned long prot = pgprot_val(protection_map[vm_flags &
|
||||
(VM_READ|VM_WRITE|VM_EXEC|VM_SHARED)]);
|
||||
|
||||
@@ -36,6 +36,9 @@ static inline bool pgtable_l5_enabled(void)
|
||||
#define pgtable_l5_enabled() cpu_feature_enabled(X86_FEATURE_LA57)
|
||||
#endif /* USE_EARLY_PGTABLE_L5 */
|
||||
|
||||
#define ARCH_PAGE_TABLE_SYNC_MASK \
|
||||
(pgtable_l5_enabled() ? PGTBL_PGD_MODIFIED : PGTBL_P4D_MODIFIED)
|
||||
|
||||
extern unsigned int pgdir_shift;
|
||||
extern unsigned int ptrs_per_p4d;
|
||||
|
||||
|
||||
@@ -356,11 +356,6 @@ static inline void arch_tlbbatch_add_pending(struct arch_tlbflush_unmap_batch *b
|
||||
mmu_notifier_arch_invalidate_secondary_tlbs(mm, 0, -1UL);
|
||||
}
|
||||
|
||||
static inline void arch_flush_tlb_batched_pending(struct mm_struct *mm)
|
||||
{
|
||||
flush_tlb_mm(mm);
|
||||
}
|
||||
|
||||
extern void arch_tlbbatch_flush(struct arch_tlbflush_unmap_batch *batch);
|
||||
|
||||
static inline bool pte_flags_need_flush(unsigned long oldflags,
|
||||
|
||||
@@ -138,7 +138,7 @@ struct its_array its_pages;
|
||||
|
||||
static void *__its_alloc(struct its_array *pages)
|
||||
{
|
||||
void *page __free(execmem) = execmem_alloc(EXECMEM_MODULE_TEXT, PAGE_SIZE);
|
||||
void *page __free(execmem) = execmem_alloc_rw(EXECMEM_MODULE_TEXT, PAGE_SIZE);
|
||||
if (!page)
|
||||
return NULL;
|
||||
|
||||
@@ -238,7 +238,6 @@ static void *its_alloc(void)
|
||||
if (!page)
|
||||
return NULL;
|
||||
|
||||
execmem_make_temp_rw(page, PAGE_SIZE);
|
||||
if (pages == &its_pages)
|
||||
set_memory_x((unsigned long)page, 1);
|
||||
|
||||
|
||||
@@ -279,7 +279,7 @@ static struct sgx_encl_page *__sgx_encl_load_page(struct sgx_encl *encl,
|
||||
|
||||
static struct sgx_encl_page *sgx_encl_load_page_in_vma(struct sgx_encl *encl,
|
||||
unsigned long addr,
|
||||
unsigned long vm_flags)
|
||||
vm_flags_t vm_flags)
|
||||
{
|
||||
unsigned long vm_prot_bits = vm_flags & VM_ACCESS_FLAGS;
|
||||
struct sgx_encl_page *entry;
|
||||
@@ -520,9 +520,9 @@ static void sgx_vma_open(struct vm_area_struct *vma)
|
||||
* Return: 0 on success, -EACCES otherwise
|
||||
*/
|
||||
int sgx_encl_may_map(struct sgx_encl *encl, unsigned long start,
|
||||
unsigned long end, unsigned long vm_flags)
|
||||
unsigned long end, vm_flags_t vm_flags)
|
||||
{
|
||||
unsigned long vm_prot_bits = vm_flags & VM_ACCESS_FLAGS;
|
||||
vm_flags_t vm_prot_bits = vm_flags & VM_ACCESS_FLAGS;
|
||||
struct sgx_encl_page *page;
|
||||
unsigned long count = 0;
|
||||
int ret = 0;
|
||||
@@ -605,7 +605,7 @@ static int sgx_encl_debug_write(struct sgx_encl *encl, struct sgx_encl_page *pag
|
||||
*/
|
||||
static struct sgx_encl_page *sgx_encl_reserve_page(struct sgx_encl *encl,
|
||||
unsigned long addr,
|
||||
unsigned long vm_flags)
|
||||
vm_flags_t vm_flags)
|
||||
{
|
||||
struct sgx_encl_page *entry;
|
||||
|
||||
|
||||
@@ -101,7 +101,7 @@ static inline int sgx_encl_find(struct mm_struct *mm, unsigned long addr,
|
||||
}
|
||||
|
||||
int sgx_encl_may_map(struct sgx_encl *encl, unsigned long start,
|
||||
unsigned long end, unsigned long vm_flags);
|
||||
unsigned long end, vm_flags_t vm_flags);
|
||||
|
||||
bool current_is_ksgxd(void);
|
||||
void sgx_encl_release(struct kref *ref);
|
||||
|
||||
+22
-4
@@ -163,10 +163,10 @@ static struct crash_mem *fill_up_crash_elf_data(void)
|
||||
return NULL;
|
||||
|
||||
/*
|
||||
* Exclusion of crash region and/or crashk_low_res may cause
|
||||
* another range split. So add extra two slots here.
|
||||
* Exclusion of crash region, crashk_low_res and/or crashk_cma_ranges
|
||||
* may cause range splits. So add extra slots here.
|
||||
*/
|
||||
nr_ranges += 2;
|
||||
nr_ranges += 2 + crashk_cma_cnt;
|
||||
cmem = vzalloc(struct_size(cmem, ranges, nr_ranges));
|
||||
if (!cmem)
|
||||
return NULL;
|
||||
@@ -184,6 +184,7 @@ static struct crash_mem *fill_up_crash_elf_data(void)
|
||||
static int elf_header_exclude_ranges(struct crash_mem *cmem)
|
||||
{
|
||||
int ret = 0;
|
||||
int i;
|
||||
|
||||
/* Exclude the low 1M because it is always reserved */
|
||||
ret = crash_exclude_mem_range(cmem, 0, SZ_1M - 1);
|
||||
@@ -198,8 +199,17 @@ static int elf_header_exclude_ranges(struct crash_mem *cmem)
|
||||
if (crashk_low_res.end)
|
||||
ret = crash_exclude_mem_range(cmem, crashk_low_res.start,
|
||||
crashk_low_res.end);
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
return ret;
|
||||
for (i = 0; i < crashk_cma_cnt; ++i) {
|
||||
ret = crash_exclude_mem_range(cmem, crashk_cma_ranges[i].start,
|
||||
crashk_cma_ranges[i].end);
|
||||
if (ret)
|
||||
return ret;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int prepare_elf64_ram_headers_callback(struct resource *res, void *arg)
|
||||
@@ -374,6 +384,14 @@ int crash_setup_memmap_entries(struct kimage *image, struct boot_params *params)
|
||||
add_e820_entry(params, &ei);
|
||||
}
|
||||
|
||||
for (i = 0; i < crashk_cma_cnt; ++i) {
|
||||
ei.addr = crashk_cma_ranges[i].start;
|
||||
ei.size = crashk_cma_ranges[i].end -
|
||||
crashk_cma_ranges[i].start + 1;
|
||||
ei.type = E820_TYPE_RAM;
|
||||
add_e820_entry(params, &ei);
|
||||
}
|
||||
|
||||
out:
|
||||
vfree(cmem);
|
||||
return ret;
|
||||
|
||||
@@ -268,7 +268,7 @@ void arch_ftrace_update_code(int command)
|
||||
|
||||
static inline void *alloc_tramp(unsigned long size)
|
||||
{
|
||||
return execmem_alloc(EXECMEM_FTRACE, size);
|
||||
return execmem_alloc_rw(EXECMEM_FTRACE, size);
|
||||
}
|
||||
static inline void tramp_free(void *tramp)
|
||||
{
|
||||
|
||||
@@ -490,24 +490,6 @@ static int prepare_singlestep(kprobe_opcode_t *buf, struct kprobe *p,
|
||||
return len;
|
||||
}
|
||||
|
||||
/* Make page to RO mode when allocate it */
|
||||
void *alloc_insn_page(void)
|
||||
{
|
||||
void *page;
|
||||
|
||||
page = execmem_alloc(EXECMEM_KPROBES, PAGE_SIZE);
|
||||
if (!page)
|
||||
return NULL;
|
||||
|
||||
/*
|
||||
* TODO: Once additional kernel code protection mechanisms are set, ensure
|
||||
* that the page was not maliciously altered and it is still zeroed.
|
||||
*/
|
||||
set_memory_rox((unsigned long)page, 1);
|
||||
|
||||
return page;
|
||||
}
|
||||
|
||||
/* Kprobe x86 instruction emulation - only regs->ip or IF flag modifiers */
|
||||
|
||||
static void kprobe_emulate_ifmodifiers(struct kprobe *p, struct pt_regs *regs)
|
||||
|
||||
@@ -578,7 +578,7 @@ static void __init memblock_x86_reserve_range_setup_data(void)
|
||||
|
||||
static void __init arch_reserve_crashkernel(void)
|
||||
{
|
||||
unsigned long long crash_base, crash_size, low_size = 0;
|
||||
unsigned long long crash_base, crash_size, low_size = 0, cma_size = 0;
|
||||
bool high = false;
|
||||
int ret;
|
||||
|
||||
@@ -587,7 +587,7 @@ static void __init arch_reserve_crashkernel(void)
|
||||
|
||||
ret = parse_crashkernel(boot_command_line, memblock_phys_mem_size(),
|
||||
&crash_size, &crash_base,
|
||||
&low_size, &high);
|
||||
&low_size, &cma_size, &high);
|
||||
if (ret)
|
||||
return;
|
||||
|
||||
@@ -597,6 +597,7 @@ static void __init arch_reserve_crashkernel(void)
|
||||
}
|
||||
|
||||
reserve_crashkernel_generic(crash_size, crash_base, low_size, high);
|
||||
reserve_crashkernel_cma(cma_size);
|
||||
}
|
||||
|
||||
static struct resource standard_io_resources[] = {
|
||||
|
||||
+17
-7
@@ -1063,13 +1063,9 @@ unsigned long arch_max_swapfile_size(void)
|
||||
static struct execmem_info execmem_info __ro_after_init;
|
||||
|
||||
#ifdef CONFIG_ARCH_HAS_EXECMEM_ROX
|
||||
void execmem_fill_trapping_insns(void *ptr, size_t size, bool writeable)
|
||||
void execmem_fill_trapping_insns(void *ptr, size_t size)
|
||||
{
|
||||
/* fill memory with INT3 instructions */
|
||||
if (writeable)
|
||||
memset(ptr, INT3_INSN_OPCODE, size);
|
||||
else
|
||||
text_poke_set(ptr, INT3_INSN_OPCODE, size);
|
||||
memset(ptr, INT3_INSN_OPCODE, size);
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1102,7 +1098,21 @@ struct execmem_info __init *execmem_arch_setup(void)
|
||||
.pgprot = pgprot,
|
||||
.alignment = MODULE_ALIGN,
|
||||
},
|
||||
[EXECMEM_KPROBES ... EXECMEM_BPF] = {
|
||||
[EXECMEM_KPROBES] = {
|
||||
.flags = flags,
|
||||
.start = start,
|
||||
.end = MODULES_END,
|
||||
.pgprot = PAGE_KERNEL_ROX,
|
||||
.alignment = MODULE_ALIGN,
|
||||
},
|
||||
[EXECMEM_FTRACE] = {
|
||||
.flags = flags,
|
||||
.start = start,
|
||||
.end = MODULES_END,
|
||||
.pgprot = pgprot,
|
||||
.alignment = MODULE_ALIGN,
|
||||
},
|
||||
[EXECMEM_BPF] = {
|
||||
.flags = EXECMEM_KASAN_SHADOW,
|
||||
.start = start,
|
||||
.end = MODULES_END,
|
||||
|
||||
@@ -223,6 +223,24 @@ static void sync_global_pgds(unsigned long start, unsigned long end)
|
||||
sync_global_pgds_l4(start, end);
|
||||
}
|
||||
|
||||
/*
|
||||
* Make kernel mappings visible in all page tables in the system.
|
||||
* This is necessary except when the init task populates kernel mappings
|
||||
* during the boot process. In that case, all processes originating from
|
||||
* the init task copies the kernel mappings, so there is no issue.
|
||||
* Otherwise, missing synchronization could lead to kernel crashes due
|
||||
* to missing page table entries for certain kernel mappings.
|
||||
*
|
||||
* Synchronization is performed at the top level, which is the PGD in
|
||||
* 5-level paging systems. But in 4-level paging systems, however,
|
||||
* pgd_populate() is a no-op, so synchronization is done at the P4D level.
|
||||
* sync_global_pgds() handles this difference between paging levels.
|
||||
*/
|
||||
void arch_sync_kernel_mappings(unsigned long start, unsigned long end)
|
||||
{
|
||||
sync_global_pgds(start, end);
|
||||
}
|
||||
|
||||
/*
|
||||
* NOTE: This function is marked __ref because it calls __init function
|
||||
* (alloc_bootmem_pages). It's safe to do it ONLY when after_bootmem == 0.
|
||||
|
||||
@@ -32,7 +32,7 @@ void add_encrypt_protection_map(void)
|
||||
protection_map[i] = pgprot_encrypted(protection_map[i]);
|
||||
}
|
||||
|
||||
pgprot_t vm_get_page_prot(unsigned long vm_flags)
|
||||
pgprot_t vm_get_page_prot(vm_flags_t vm_flags)
|
||||
{
|
||||
unsigned long val = pgprot_val(protection_map[vm_flags &
|
||||
(VM_READ|VM_WRITE|VM_EXEC|VM_SHARED)]);
|
||||
|
||||
+9
-6
@@ -500,18 +500,21 @@ static void blkdev_readahead(struct readahead_control *rac)
|
||||
mpage_readahead(rac, blkdev_get_block);
|
||||
}
|
||||
|
||||
static int blkdev_write_begin(struct file *file, struct address_space *mapping,
|
||||
loff_t pos, unsigned len, struct folio **foliop, void **fsdata)
|
||||
static int blkdev_write_begin(const struct kiocb *iocb,
|
||||
struct address_space *mapping, loff_t pos,
|
||||
unsigned len, struct folio **foliop,
|
||||
void **fsdata)
|
||||
{
|
||||
return block_write_begin(mapping, pos, len, foliop, blkdev_get_block);
|
||||
}
|
||||
|
||||
static int blkdev_write_end(struct file *file, struct address_space *mapping,
|
||||
loff_t pos, unsigned len, unsigned copied, struct folio *folio,
|
||||
void *fsdata)
|
||||
static int blkdev_write_end(const struct kiocb *iocb,
|
||||
struct address_space *mapping,
|
||||
loff_t pos, unsigned len, unsigned copied,
|
||||
struct folio *folio, void *fsdata)
|
||||
{
|
||||
int ret;
|
||||
ret = block_write_end(file, mapping, pos, len, copied, folio, fsdata);
|
||||
ret = block_write_end(pos, len, copied, folio);
|
||||
|
||||
folio_unlock(folio);
|
||||
folio_put(folio);
|
||||
|
||||
@@ -967,10 +967,10 @@ static int hmat_callback(struct notifier_block *self,
|
||||
unsigned long action, void *arg)
|
||||
{
|
||||
struct memory_target *target;
|
||||
struct memory_notify *mnb = arg;
|
||||
int pxm, nid = mnb->status_change_nid;
|
||||
struct node_notify *nn = arg;
|
||||
int pxm, nid = nn->nid;
|
||||
|
||||
if (nid == NUMA_NO_NODE || action != MEM_ONLINE)
|
||||
if (action != NODE_ADDED_FIRST_MEMORY)
|
||||
return NOTIFY_OK;
|
||||
|
||||
pxm = node_to_pxm(nid);
|
||||
@@ -1123,7 +1123,7 @@ static __init int hmat_init(void)
|
||||
hmat_register_targets();
|
||||
|
||||
/* Keep the table and structures if the notifier may use them */
|
||||
if (hotplug_memory_notifier(hmat_callback, HMAT_CALLBACK_PRI))
|
||||
if (hotplug_node_notifier(hmat_callback, HMAT_CALLBACK_PRI))
|
||||
goto out_put;
|
||||
|
||||
if (!hmat_set_default_dram_perf())
|
||||
|
||||
+29
-63
@@ -112,6 +112,27 @@ static const struct attribute_group *node_access_node_groups[] = {
|
||||
NULL,
|
||||
};
|
||||
|
||||
#ifdef CONFIG_MEMORY_HOTPLUG
|
||||
static BLOCKING_NOTIFIER_HEAD(node_chain);
|
||||
|
||||
int register_node_notifier(struct notifier_block *nb)
|
||||
{
|
||||
return blocking_notifier_chain_register(&node_chain, nb);
|
||||
}
|
||||
EXPORT_SYMBOL(register_node_notifier);
|
||||
|
||||
void unregister_node_notifier(struct notifier_block *nb)
|
||||
{
|
||||
blocking_notifier_chain_unregister(&node_chain, nb);
|
||||
}
|
||||
EXPORT_SYMBOL(unregister_node_notifier);
|
||||
|
||||
int node_notify(unsigned long val, void *v)
|
||||
{
|
||||
return blocking_notifier_call_chain(&node_chain, val, v);
|
||||
}
|
||||
#endif
|
||||
|
||||
static void node_remove_accesses(struct node *node)
|
||||
{
|
||||
struct node_access_nodes *c, *cnext;
|
||||
@@ -479,7 +500,7 @@ static ssize_t node_read_meminfo(struct device *dev,
|
||||
nid, K(node_page_state(pgdat, NR_SECONDARY_PAGETABLE)),
|
||||
nid, 0UL,
|
||||
nid, 0UL,
|
||||
nid, K(node_page_state(pgdat, NR_WRITEBACK_TEMP)),
|
||||
nid, 0UL,
|
||||
nid, K(sreclaimable +
|
||||
node_page_state(pgdat, NR_KERNEL_MISC_RECLAIMABLE)),
|
||||
nid, K(sreclaimable + sunreclaimable),
|
||||
@@ -638,6 +659,7 @@ static int register_node(struct node *node, int num)
|
||||
} else {
|
||||
hugetlb_register_node(node);
|
||||
compaction_register_node(node);
|
||||
reclaim_register_node(node);
|
||||
}
|
||||
|
||||
return error;
|
||||
@@ -654,6 +676,7 @@ void unregister_node(struct node *node)
|
||||
{
|
||||
hugetlb_unregister_node(node);
|
||||
compaction_unregister_node(node);
|
||||
reclaim_unregister_node(node);
|
||||
node_remove_accesses(node);
|
||||
node_remove_caches(node);
|
||||
device_unregister(&node->dev);
|
||||
@@ -757,15 +780,6 @@ int unregister_cpu_under_node(unsigned int cpu, unsigned int nid)
|
||||
}
|
||||
|
||||
#ifdef CONFIG_MEMORY_HOTPLUG
|
||||
static int __ref get_nid_for_pfn(unsigned long pfn)
|
||||
{
|
||||
#ifdef CONFIG_DEFERRED_STRUCT_PAGE_INIT
|
||||
if (system_state < SYSTEM_RUNNING)
|
||||
return early_pfn_to_nid(pfn);
|
||||
#endif
|
||||
return pfn_to_nid(pfn);
|
||||
}
|
||||
|
||||
static void do_register_memory_block_under_node(int nid,
|
||||
struct memory_block *mem_blk,
|
||||
enum meminit_context context)
|
||||
@@ -792,46 +806,6 @@ static void do_register_memory_block_under_node(int nid,
|
||||
ret);
|
||||
}
|
||||
|
||||
/* register memory section under specified node if it spans that node */
|
||||
static int register_mem_block_under_node_early(struct memory_block *mem_blk,
|
||||
void *arg)
|
||||
{
|
||||
unsigned long memory_block_pfns = memory_block_size_bytes() / PAGE_SIZE;
|
||||
unsigned long start_pfn = section_nr_to_pfn(mem_blk->start_section_nr);
|
||||
unsigned long end_pfn = start_pfn + memory_block_pfns - 1;
|
||||
int nid = *(int *)arg;
|
||||
unsigned long pfn;
|
||||
|
||||
for (pfn = start_pfn; pfn <= end_pfn; pfn++) {
|
||||
int page_nid;
|
||||
|
||||
/*
|
||||
* memory block could have several absent sections from start.
|
||||
* skip pfn range from absent section
|
||||
*/
|
||||
if (!pfn_in_present_section(pfn)) {
|
||||
pfn = round_down(pfn + PAGES_PER_SECTION,
|
||||
PAGES_PER_SECTION) - 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
/*
|
||||
* We need to check if page belongs to nid only at the boot
|
||||
* case because node's ranges can be interleaved.
|
||||
*/
|
||||
page_nid = get_nid_for_pfn(pfn);
|
||||
if (page_nid < 0)
|
||||
continue;
|
||||
if (page_nid != nid)
|
||||
continue;
|
||||
|
||||
do_register_memory_block_under_node(nid, mem_blk, MEMINIT_EARLY);
|
||||
return 0;
|
||||
}
|
||||
/* mem section does not span the specified node */
|
||||
return 0;
|
||||
}
|
||||
|
||||
/*
|
||||
* During hotplug we know that all pages in the memory block belong to the same
|
||||
* node.
|
||||
@@ -888,24 +862,16 @@ static void register_memory_blocks_under_nodes(void)
|
||||
}
|
||||
}
|
||||
|
||||
void register_memory_blocks_under_node(int nid, unsigned long start_pfn,
|
||||
unsigned long end_pfn,
|
||||
enum meminit_context context)
|
||||
void register_memory_blocks_under_node_hotplug(int nid, unsigned long start_pfn,
|
||||
unsigned long end_pfn)
|
||||
{
|
||||
walk_memory_blocks_func_t func;
|
||||
|
||||
if (context == MEMINIT_HOTPLUG)
|
||||
func = register_mem_block_under_node_hotplug;
|
||||
else
|
||||
func = register_mem_block_under_node_early;
|
||||
|
||||
walk_memory_blocks(PFN_PHYS(start_pfn), PFN_PHYS(end_pfn - start_pfn),
|
||||
(void *)&nid, func);
|
||||
(void *)&nid, register_mem_block_under_node_hotplug);
|
||||
return;
|
||||
}
|
||||
#endif /* CONFIG_MEMORY_HOTPLUG */
|
||||
|
||||
int __register_one_node(int nid)
|
||||
int register_one_node(int nid)
|
||||
{
|
||||
int error;
|
||||
int cpu;
|
||||
@@ -1016,7 +982,7 @@ void __init node_dev_init(void)
|
||||
* to already created cpu devices.
|
||||
*/
|
||||
for_each_online_node(i) {
|
||||
ret = __register_one_node(i);
|
||||
ret = register_one_node(i);
|
||||
if (ret)
|
||||
panic("%s() failed to add node: %d\n", __func__, ret);
|
||||
}
|
||||
|
||||
@@ -2507,12 +2507,12 @@ static int cxl_region_perf_attrs_callback(struct notifier_block *nb,
|
||||
unsigned long action, void *arg)
|
||||
{
|
||||
struct cxl_region *cxlr = container_of(nb, struct cxl_region,
|
||||
memory_notifier);
|
||||
struct memory_notify *mnb = arg;
|
||||
int nid = mnb->status_change_nid;
|
||||
node_notifier);
|
||||
struct node_notify *nn = arg;
|
||||
int nid = nn->nid;
|
||||
int region_nid;
|
||||
|
||||
if (nid == NUMA_NO_NODE || action != MEM_ONLINE)
|
||||
if (action != NODE_ADDED_FIRST_MEMORY)
|
||||
return NOTIFY_DONE;
|
||||
|
||||
/*
|
||||
@@ -3583,7 +3583,7 @@ static void shutdown_notifiers(void *_cxlr)
|
||||
{
|
||||
struct cxl_region *cxlr = _cxlr;
|
||||
|
||||
unregister_memory_notifier(&cxlr->memory_notifier);
|
||||
unregister_node_notifier(&cxlr->node_notifier);
|
||||
unregister_mt_adistance_algorithm(&cxlr->adist_notifier);
|
||||
}
|
||||
|
||||
@@ -3634,9 +3634,9 @@ static int cxl_region_probe(struct device *dev)
|
||||
* CXL_CONFIG_COMMIT is also responsible for releasing the driver.
|
||||
*/
|
||||
|
||||
cxlr->memory_notifier.notifier_call = cxl_region_perf_attrs_callback;
|
||||
cxlr->memory_notifier.priority = CXL_CALLBACK_PRI;
|
||||
register_memory_notifier(&cxlr->memory_notifier);
|
||||
cxlr->node_notifier.notifier_call = cxl_region_perf_attrs_callback;
|
||||
cxlr->node_notifier.priority = CXL_CALLBACK_PRI;
|
||||
register_node_notifier(&cxlr->node_notifier);
|
||||
|
||||
cxlr->adist_notifier.notifier_call = cxl_region_calculate_adistance;
|
||||
cxlr->adist_notifier.priority = 100;
|
||||
|
||||
+2
-2
@@ -514,7 +514,7 @@ enum cxl_partition_mode {
|
||||
* @flags: Region state flags
|
||||
* @params: active + config params for the region
|
||||
* @coord: QoS access coordinates for the region
|
||||
* @memory_notifier: notifier for setting the access coordinates to node
|
||||
* @node_notifier: notifier for setting the access coordinates to node
|
||||
* @adist_notifier: notifier for calculating the abstract distance of node
|
||||
*/
|
||||
struct cxl_region {
|
||||
@@ -527,7 +527,7 @@ struct cxl_region {
|
||||
unsigned long flags;
|
||||
struct cxl_region_params params;
|
||||
struct access_coordinate coord[ACCESS_COORDINATE_MAX];
|
||||
struct notifier_block memory_notifier;
|
||||
struct notifier_block node_notifier;
|
||||
struct notifier_block adist_notifier;
|
||||
};
|
||||
|
||||
|
||||
@@ -325,7 +325,7 @@ void __shmem_writeback(size_t size, struct address_space *mapping)
|
||||
if (folio_mapped(folio))
|
||||
folio_redirty_for_writepage(&wbc, folio);
|
||||
else
|
||||
error = shmem_writeout(folio, &wbc);
|
||||
error = shmem_writeout(folio, NULL, NULL);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -114,15 +114,8 @@ ttm_backup_backup_page(struct file *backup, struct page *page,
|
||||
|
||||
if (writeback && !folio_mapped(to_folio) &&
|
||||
folio_clear_dirty_for_io(to_folio)) {
|
||||
struct writeback_control wbc = {
|
||||
.sync_mode = WB_SYNC_NONE,
|
||||
.nr_to_write = SWAP_CLUSTER_MAX,
|
||||
.range_start = 0,
|
||||
.range_end = LLONG_MAX,
|
||||
.for_reclaim = 1,
|
||||
};
|
||||
folio_set_reclaim(to_folio);
|
||||
ret = shmem_writeout(to_folio, &wbc);
|
||||
ret = shmem_writeout(to_folio, NULL, NULL);
|
||||
if (!folio_test_writeback(to_folio))
|
||||
folio_clear_reclaim(to_folio);
|
||||
/*
|
||||
|
||||
@@ -1778,8 +1778,7 @@ static int vmballoon_migratepage(struct balloon_dev_info *b_dev_info,
|
||||
* @pages_lock . We keep holding @comm_lock since we will need it in a
|
||||
* second.
|
||||
*/
|
||||
balloon_page_delete(page);
|
||||
|
||||
balloon_page_finalize(page);
|
||||
put_page(page);
|
||||
|
||||
/* Inflate */
|
||||
|
||||
@@ -866,15 +866,13 @@ static int virtballoon_migratepage(struct balloon_dev_info *vb_dev_info,
|
||||
tell_host(vb, vb->inflate_vq);
|
||||
|
||||
/* balloon's page migration 2nd step -- deflate "page" */
|
||||
spin_lock_irqsave(&vb_dev_info->pages_lock, flags);
|
||||
balloon_page_delete(page);
|
||||
spin_unlock_irqrestore(&vb_dev_info->pages_lock, flags);
|
||||
vb->num_pfns = VIRTIO_BALLOON_PAGES_PER_PAGE;
|
||||
set_page_pfns(vb, vb->pfns, page);
|
||||
tell_host(vb, vb->deflate_vq);
|
||||
|
||||
mutex_unlock(&vb->balloon_lock);
|
||||
|
||||
balloon_page_finalize(page);
|
||||
put_page(page); /* balloon reference */
|
||||
|
||||
return MIGRATEPAGE_SUCCESS;
|
||||
|
||||
@@ -1243,7 +1243,7 @@ static int virtio_mem_fake_offline(struct virtio_mem *vm, unsigned long pfn,
|
||||
if (atomic_read(&vm->config_changed))
|
||||
return -EAGAIN;
|
||||
|
||||
rc = alloc_contig_range(pfn, pfn + nr_pages, MIGRATE_MOVABLE,
|
||||
rc = alloc_contig_range(pfn, pfn + nr_pages, ACR_FLAGS_NONE,
|
||||
GFP_KERNEL);
|
||||
if (rc == -ENOMEM)
|
||||
/* whoops, out of memory */
|
||||
|
||||
+1
-1
@@ -516,7 +516,7 @@ const struct file_operations v9fs_file_operations = {
|
||||
.open = v9fs_file_open,
|
||||
.release = v9fs_dir_release,
|
||||
.lock = v9fs_file_lock,
|
||||
.mmap = generic_file_readonly_mmap,
|
||||
.mmap_prepare = generic_file_readonly_mmap_prepare,
|
||||
.splice_read = v9fs_file_splice_read,
|
||||
.splice_write = iter_file_splice_write,
|
||||
.fsync = v9fs_file_fsync,
|
||||
|
||||
+1
-1
@@ -25,7 +25,7 @@
|
||||
const struct file_operations adfs_file_operations = {
|
||||
.llseek = generic_file_llseek,
|
||||
.read_iter = generic_file_read_iter,
|
||||
.mmap = generic_file_mmap,
|
||||
.mmap_prepare = generic_file_mmap_prepare,
|
||||
.fsync = generic_file_fsync,
|
||||
.write_iter = generic_file_write_iter,
|
||||
.splice_read = filemap_splice_read,
|
||||
|
||||
+5
-4
@@ -53,13 +53,14 @@ static void adfs_write_failed(struct address_space *mapping, loff_t to)
|
||||
truncate_pagecache(inode, inode->i_size);
|
||||
}
|
||||
|
||||
static int adfs_write_begin(struct file *file, struct address_space *mapping,
|
||||
loff_t pos, unsigned len,
|
||||
struct folio **foliop, void **fsdata)
|
||||
static int adfs_write_begin(const struct kiocb *iocb,
|
||||
struct address_space *mapping,
|
||||
loff_t pos, unsigned len,
|
||||
struct folio **foliop, void **fsdata)
|
||||
{
|
||||
int ret;
|
||||
|
||||
ret = cont_write_begin(file, mapping, pos, len, foliop, fsdata,
|
||||
ret = cont_write_begin(iocb, mapping, pos, len, foliop, fsdata,
|
||||
adfs_get_block,
|
||||
&ADFS_I(mapping->host)->mmu_private);
|
||||
if (unlikely(ret))
|
||||
|
||||
+16
-12
@@ -415,13 +415,14 @@ affs_direct_IO(struct kiocb *iocb, struct iov_iter *iter)
|
||||
return ret;
|
||||
}
|
||||
|
||||
static int affs_write_begin(struct file *file, struct address_space *mapping,
|
||||
loff_t pos, unsigned len,
|
||||
struct folio **foliop, void **fsdata)
|
||||
static int affs_write_begin(const struct kiocb *iocb,
|
||||
struct address_space *mapping,
|
||||
loff_t pos, unsigned len,
|
||||
struct folio **foliop, void **fsdata)
|
||||
{
|
||||
int ret;
|
||||
|
||||
ret = cont_write_begin(file, mapping, pos, len, foliop, fsdata,
|
||||
ret = cont_write_begin(iocb, mapping, pos, len, foliop, fsdata,
|
||||
affs_get_block,
|
||||
&AFFS_I(mapping->host)->mmu_private);
|
||||
if (unlikely(ret))
|
||||
@@ -430,14 +431,15 @@ static int affs_write_begin(struct file *file, struct address_space *mapping,
|
||||
return ret;
|
||||
}
|
||||
|
||||
static int affs_write_end(struct file *file, struct address_space *mapping,
|
||||
loff_t pos, unsigned int len, unsigned int copied,
|
||||
static int affs_write_end(const struct kiocb *iocb,
|
||||
struct address_space *mapping, loff_t pos,
|
||||
unsigned int len, unsigned int copied,
|
||||
struct folio *folio, void *fsdata)
|
||||
{
|
||||
struct inode *inode = mapping->host;
|
||||
int ret;
|
||||
|
||||
ret = generic_write_end(file, mapping, pos, len, copied, folio, fsdata);
|
||||
ret = generic_write_end(iocb, mapping, pos, len, copied, folio, fsdata);
|
||||
|
||||
/* Clear Archived bit on file writes, as AmigaOS would do */
|
||||
if (AFFS_I(inode)->i_protect & FIBF_ARCHIVED) {
|
||||
@@ -645,7 +647,8 @@ static int affs_read_folio_ofs(struct file *file, struct folio *folio)
|
||||
return err;
|
||||
}
|
||||
|
||||
static int affs_write_begin_ofs(struct file *file, struct address_space *mapping,
|
||||
static int affs_write_begin_ofs(const struct kiocb *iocb,
|
||||
struct address_space *mapping,
|
||||
loff_t pos, unsigned len,
|
||||
struct folio **foliop, void **fsdata)
|
||||
{
|
||||
@@ -684,9 +687,10 @@ static int affs_write_begin_ofs(struct file *file, struct address_space *mapping
|
||||
return err;
|
||||
}
|
||||
|
||||
static int affs_write_end_ofs(struct file *file, struct address_space *mapping,
|
||||
loff_t pos, unsigned len, unsigned copied,
|
||||
struct folio *folio, void *fsdata)
|
||||
static int affs_write_end_ofs(const struct kiocb *iocb,
|
||||
struct address_space *mapping,
|
||||
loff_t pos, unsigned len, unsigned copied,
|
||||
struct folio *folio, void *fsdata)
|
||||
{
|
||||
struct inode *inode = mapping->host;
|
||||
struct super_block *sb = inode->i_sb;
|
||||
@@ -998,7 +1002,7 @@ const struct file_operations affs_file_operations = {
|
||||
.llseek = generic_file_llseek,
|
||||
.read_iter = generic_file_read_iter,
|
||||
.write_iter = generic_file_write_iter,
|
||||
.mmap = generic_file_mmap,
|
||||
.mmap_prepare = generic_file_mmap_prepare,
|
||||
.open = affs_file_open,
|
||||
.release = affs_file_release,
|
||||
.fsync = affs_file_fsync,
|
||||
|
||||
+1
-1
@@ -336,7 +336,7 @@ int backing_file_mmap(struct file *file, struct vm_area_struct *vma,
|
||||
WARN_ON_ONCE(ctx->user_file != vma->vm_file))
|
||||
return -EIO;
|
||||
|
||||
if (!file->f_op->mmap)
|
||||
if (!can_mmap_file(file))
|
||||
return -ENODEV;
|
||||
|
||||
vma_set_file(vma, file);
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user