This revision moves the `cache_key` to the first level
dict of the cache, instead of the last one.
Doing so, we reduce the number of times the reference
to the cache key is stored in the dict.
For instance,
for 100.000 records, 20 fields and 2 env (e.g. with and without sudo)
formerly, there were 100.000 * 20 * 2 occurences of cache key references
now, there is only 2 references.
Storing references to an object consumes memory.
Therefore, by reducing the number of object references
in the cache, we reduce the memory consumed by the cache.
Also, we reduce the time to access a value in the cache
as the cache size is smaller.
The time and memory consumption are therefore improved,
while keeping the advantages of revision
d7190a3fd0
which was about sharing the cache of fields
which do not depends on the context, but
only on the cursor and user id.
This revision relies on the fact there are less different references
to the cache key then references to fields/records.
Indeed, this is more likely to have 100.000 different records stored
in the cache rather than 100.000 different environments.
Here is the Python proof of concept that was used
to make the conclusion that setting the cache_key
in the first level dict of the cache is more efficient.
```Python
import os
import psutil
import time
from collections import defaultdict
cr = object()
uid = 1
fields = [object() for i in range(20)]
number_items = 500000
p = psutil.Process(os.getpid())
m = p.memory_info().rss
s = time.time()
cache_key = (cr, uid)
cache = defaultdict(lambda: defaultdict(dict))
for field in fields:
for i in range(number_items):
cache[field][i][cache_key] = 5.0
# cache[cache_key][field][i] = 5.0
print('Memory: %s' % (p.memory_info().rss - m,))
print('Time: %s' % (time.time() - s,))
```
- Using `cache[field][i][cache_key]`:
- Time: 3.17s
- Memory: 3138MB
- Using `cache[cache_key][field][i]`:
- Time: 1.43s
- Memory: 756MB
Even worse, when the cache key tuple is instantiated inside the loop,
for the former cache structure (e.g. `cache[field][i][(cr, uid)]`),
the time goes from 3.17s to 25.63s and the memory from 3138MB to 3773MB
Here is the same proof of concept, but using the Odoo API and Cache:
```Python
import os
import psutil
import time
from odoo.api import Cache
model = env['res.users']
records = [model.new() for i in range(100000)]
p = psutil.Process(os.getpid())
m = p.memory_info().rss
s = time.time()
cache = Cache()
char_fields = [field for field in model._fields.values() if field.type == 'char']
for field in char_fields:
for record in records:
cache.set(record, field, 'test')
print('Memory: %s' % (p.memory_info().rss - m,))
print('Time: %s' % (time.time() - s,))
```
- Before (`cache[field][record_id][cache_key]` and cache_key tuple instantiated in the loop):
- Time: 4.12s
- Memory: 810MB
- After (`cache[cache_key][field][record_id]` and cache_key tuple stored in the env and re-used):
- Time: 1.63s
- Memory: 125MB
This can be played in an Odoo shell, for instance
by storing it in `/tmp/test.py`, and then
piping it to the Odoo shell:
`cat /tmp/test.py | ./odoo-bin shell -d 12.0`
closes odoo/odoo#29676
closes odoo/odoo#30554
146 lines
5.2 KiB
Python
146 lines
5.2 KiB
Python
# -*- coding: utf-8 -*-
|
|
# Part of Odoo. See LICENSE file for full copyright and licensing details.
|
|
|
|
import os
|
|
import psutil
|
|
|
|
from odoo.tests.common import TransactionCase
|
|
|
|
|
|
class TestRecordCache(TransactionCase):
|
|
|
|
def test_cache(self):
|
|
""" Check the record cache object. """
|
|
Model = self.env['res.partner']
|
|
name = type(Model).name
|
|
ref = type(Model).ref
|
|
|
|
cache = self.env.cache
|
|
|
|
def check1(record, field, value):
|
|
# value is None means no value in cache
|
|
self.assertEqual(cache.contains(record, field), value is not None)
|
|
self.assertEqual(cache.contains_value(record, field), value is not None)
|
|
self.assertEqual(cache.get_value(record, field), value)
|
|
try:
|
|
self.assertEqual(cache.get(record, field), value)
|
|
self.assertIsNotNone(value)
|
|
except KeyError:
|
|
self.assertIsNone(value)
|
|
self.assertIsNone(cache.get_special(record, field))
|
|
self.assertEqual(field in cache.get_fields(record), value is not None)
|
|
self.assertEqual(record in cache.get_records(record, field), value is not None)
|
|
|
|
def check(record, name_val, ref_val):
|
|
""" check the values of fields 'name' and 'ref' on record. """
|
|
check1(record, name, name_val)
|
|
check1(record, ref, ref_val)
|
|
|
|
foo1, bar1 = Model.browse([1, 2])
|
|
foo2, bar2 = Model.sudo(self.env.ref('base.user_demo')).browse([1, 2])
|
|
self.assertNotEqual(foo1.env.uid, foo2.env.uid)
|
|
|
|
# cache is empty
|
|
cache.invalidate()
|
|
check(foo1, None, None)
|
|
check(foo2, None, None)
|
|
check(bar1, None, None)
|
|
check(bar2, None, None)
|
|
self.assertCountEqual(cache.get_missing_ids(foo1 + bar1, name), [1, 2])
|
|
self.assertCountEqual(cache.get_missing_ids(foo2 + bar2, name), [1, 2])
|
|
|
|
# set values in one environment only
|
|
for rec in [foo1, bar1]:
|
|
cache.set(rec, name, 'NAME1')
|
|
cache.set(rec, ref, 'REF1')
|
|
check(foo1, 'NAME1', 'REF1')
|
|
check(foo2, None, None)
|
|
check(bar1, 'NAME1', 'REF1')
|
|
check(bar2, None, None)
|
|
self.assertCountEqual(cache.get_missing_ids(foo1 + bar1, name), [])
|
|
self.assertCountEqual(cache.get_missing_ids(foo2 + bar2, name), [1, 2])
|
|
|
|
# set values in both environments
|
|
for rec in [foo2, bar2]:
|
|
cache.set(rec, name, 'NAME2')
|
|
cache.set(rec, ref, 'REF2')
|
|
check(foo1, 'NAME1', 'REF1')
|
|
check(foo2, 'NAME2', 'REF2')
|
|
check(bar1, 'NAME1', 'REF1')
|
|
check(bar2, 'NAME2', 'REF2')
|
|
self.assertCountEqual(cache.get_missing_ids(foo1 + bar1, name), [])
|
|
self.assertCountEqual(cache.get_missing_ids(foo2 + bar2, name), [])
|
|
|
|
# remove value in one environment
|
|
cache.remove(foo1, name)
|
|
check(foo1, None, 'REF1')
|
|
check(foo2, 'NAME2', 'REF2')
|
|
check(bar1, 'NAME1', 'REF1')
|
|
check(bar2, 'NAME2', 'REF2')
|
|
self.assertCountEqual(cache.get_missing_ids(foo1 + bar1, name), [1])
|
|
self.assertCountEqual(cache.get_missing_ids(foo2 + bar2, name), [])
|
|
|
|
# partial invalidation
|
|
cache.invalidate([(name, None), (ref, foo1.ids)])
|
|
check(foo1, None, None)
|
|
check(foo2, None, None)
|
|
check(bar1, None, 'REF1')
|
|
check(bar2, None, 'REF2')
|
|
|
|
# total invalidation
|
|
cache.invalidate()
|
|
check(foo1, None, None)
|
|
check(foo2, None, None)
|
|
check(bar1, None, None)
|
|
check(bar2, None, None)
|
|
|
|
# set a special value
|
|
cache.set_special(foo1, name, lambda: '42')
|
|
self.assertTrue(cache.contains(foo1, name))
|
|
self.assertFalse(cache.contains_value(foo1, name))
|
|
self.assertEqual(cache.get(foo1, name), '42')
|
|
self.assertIsNone(cache.get_value(foo1, name))
|
|
self.assertIsNotNone(cache.get_special(foo1, name))
|
|
|
|
# copy cache
|
|
for rec in [foo1, bar1]:
|
|
cache.set(rec, name, 'NAME1')
|
|
cache.set(rec, ref, 'REF1')
|
|
check(foo1, 'NAME1', 'REF1')
|
|
check(foo2, None, None)
|
|
check(bar1, 'NAME1', 'REF1')
|
|
check(bar2, None, None)
|
|
|
|
cache.copy(foo1 + bar1, foo2.env)
|
|
check(foo1, 'NAME1', 'REF1')
|
|
check(foo2, 'NAME1', 'REF1')
|
|
check(bar1, 'NAME1', 'REF1')
|
|
check(bar2, 'NAME1', 'REF1')
|
|
|
|
def test_memory(self):
|
|
""" Check memory consumption of the cache. """
|
|
NB_RECORDS = 100000
|
|
MAX_MEMORY = 100
|
|
|
|
cache = self.env.cache
|
|
model = self.env['res.partner']
|
|
records = [model.new() for index in range(NB_RECORDS)]
|
|
|
|
process = psutil.Process(os.getpid())
|
|
rss0 = process.memory_info().rss
|
|
|
|
char_names = [
|
|
'name', 'display_name', 'email', 'website', 'phone', 'mobile',
|
|
'street', 'street2', 'city', 'zip', 'vat', 'ref',
|
|
]
|
|
for name in char_names:
|
|
field = model._fields[name]
|
|
for record in records:
|
|
cache.set(record, field, 'test')
|
|
|
|
mem_usage = process.memory_info().rss - rss0
|
|
self.assertLess(
|
|
mem_usage, MAX_MEMORY * 1024 * 1024,
|
|
"Caching %s records must take less than %sMB of memory" % (NB_RECORDS, MAX_MEMORY),
|
|
)
|