Files
odoo_source/odoo/addons/base/tests/test_cache.py
T
Denis Ledoux 8fb76c425d [IMP] api: improve storing performance of the api cache
This revision moves the `cache_key` to the first level
dict of the cache, instead of the last one.

Doing so, we reduce the number of times the reference
to the cache key is stored in the dict.

For instance,
for 100.000 records, 20 fields and 2 env (e.g. with and without sudo)
formerly, there were 100.000 * 20 * 2 occurences of cache key references
now, there is only 2 references.

Storing references to an object consumes memory.
Therefore, by reducing the number of object references
in the cache, we reduce the memory consumed by the cache.
Also, we reduce the time to access a value in the cache
as the cache size is smaller.

The time and memory consumption are therefore improved,
while keeping the advantages of revision
d7190a3fd0
which was about sharing the cache of fields
which do not depends on the context, but
only on the cursor and user id.

This revision relies on the fact there are less different references
to the cache key then references to fields/records.
Indeed, this is more likely to have 100.000 different records stored
in the cache rather than 100.000 different environments.

Here is the Python proof of concept that was used
to make the conclusion that setting the cache_key
in the first level dict of the cache is more efficient.
```Python
import os
import psutil
import time

from collections import defaultdict

cr = object()
uid = 1
fields = [object() for i in range(20)]
number_items = 500000

p = psutil.Process(os.getpid())
m = p.memory_info().rss
s = time.time()

cache_key = (cr, uid)
cache = defaultdict(lambda: defaultdict(dict))
for field in fields:
    for i in range(number_items):
        cache[field][i][cache_key] = 5.0
        # cache[cache_key][field][i] = 5.0

print('Memory: %s' % (p.memory_info().rss - m,))
print('Time: %s' % (time.time() - s,))

```
- Using `cache[field][i][cache_key]`:
   - Time: 3.17s
   - Memory: 3138MB
- Using `cache[cache_key][field][i]`:
   - Time: 1.43s
   - Memory: 756MB

Even worse, when the cache key tuple is instantiated inside the loop,
for the former cache structure (e.g. `cache[field][i][(cr, uid)]`),
the time goes from 3.17s to 25.63s and the memory from 3138MB to 3773MB

Here is the same proof of concept, but using the Odoo API and Cache:
```Python
import os
import psutil
import time

from odoo.api import Cache

model = env['res.users']
records = [model.new() for i in range(100000)]

p = psutil.Process(os.getpid())
m = p.memory_info().rss
s = time.time()

cache = Cache()
char_fields = [field for field in model._fields.values() if field.type == 'char']
for field in char_fields:
    for record in records:
        cache.set(record, field, 'test')

print('Memory: %s' % (p.memory_info().rss - m,))
print('Time: %s' % (time.time() - s,))
```
- Before (`cache[field][record_id][cache_key]` and cache_key tuple instantiated in the loop):
   - Time: 4.12s
   - Memory: 810MB
- After (`cache[cache_key][field][record_id]` and cache_key tuple stored in the env and re-used):
   - Time: 1.63s
   - Memory: 125MB

This can be played in an Odoo shell, for instance
by storing it in `/tmp/test.py`, and then
piping it to the Odoo shell:
`cat /tmp/test.py | ./odoo-bin shell -d 12.0`

closes odoo/odoo#29676

closes odoo/odoo#30554
2019-01-25 16:25:27 +00:00

146 lines
5.2 KiB
Python

# -*- coding: utf-8 -*-
# Part of Odoo. See LICENSE file for full copyright and licensing details.
import os
import psutil
from odoo.tests.common import TransactionCase
class TestRecordCache(TransactionCase):
def test_cache(self):
""" Check the record cache object. """
Model = self.env['res.partner']
name = type(Model).name
ref = type(Model).ref
cache = self.env.cache
def check1(record, field, value):
# value is None means no value in cache
self.assertEqual(cache.contains(record, field), value is not None)
self.assertEqual(cache.contains_value(record, field), value is not None)
self.assertEqual(cache.get_value(record, field), value)
try:
self.assertEqual(cache.get(record, field), value)
self.assertIsNotNone(value)
except KeyError:
self.assertIsNone(value)
self.assertIsNone(cache.get_special(record, field))
self.assertEqual(field in cache.get_fields(record), value is not None)
self.assertEqual(record in cache.get_records(record, field), value is not None)
def check(record, name_val, ref_val):
""" check the values of fields 'name' and 'ref' on record. """
check1(record, name, name_val)
check1(record, ref, ref_val)
foo1, bar1 = Model.browse([1, 2])
foo2, bar2 = Model.sudo(self.env.ref('base.user_demo')).browse([1, 2])
self.assertNotEqual(foo1.env.uid, foo2.env.uid)
# cache is empty
cache.invalidate()
check(foo1, None, None)
check(foo2, None, None)
check(bar1, None, None)
check(bar2, None, None)
self.assertCountEqual(cache.get_missing_ids(foo1 + bar1, name), [1, 2])
self.assertCountEqual(cache.get_missing_ids(foo2 + bar2, name), [1, 2])
# set values in one environment only
for rec in [foo1, bar1]:
cache.set(rec, name, 'NAME1')
cache.set(rec, ref, 'REF1')
check(foo1, 'NAME1', 'REF1')
check(foo2, None, None)
check(bar1, 'NAME1', 'REF1')
check(bar2, None, None)
self.assertCountEqual(cache.get_missing_ids(foo1 + bar1, name), [])
self.assertCountEqual(cache.get_missing_ids(foo2 + bar2, name), [1, 2])
# set values in both environments
for rec in [foo2, bar2]:
cache.set(rec, name, 'NAME2')
cache.set(rec, ref, 'REF2')
check(foo1, 'NAME1', 'REF1')
check(foo2, 'NAME2', 'REF2')
check(bar1, 'NAME1', 'REF1')
check(bar2, 'NAME2', 'REF2')
self.assertCountEqual(cache.get_missing_ids(foo1 + bar1, name), [])
self.assertCountEqual(cache.get_missing_ids(foo2 + bar2, name), [])
# remove value in one environment
cache.remove(foo1, name)
check(foo1, None, 'REF1')
check(foo2, 'NAME2', 'REF2')
check(bar1, 'NAME1', 'REF1')
check(bar2, 'NAME2', 'REF2')
self.assertCountEqual(cache.get_missing_ids(foo1 + bar1, name), [1])
self.assertCountEqual(cache.get_missing_ids(foo2 + bar2, name), [])
# partial invalidation
cache.invalidate([(name, None), (ref, foo1.ids)])
check(foo1, None, None)
check(foo2, None, None)
check(bar1, None, 'REF1')
check(bar2, None, 'REF2')
# total invalidation
cache.invalidate()
check(foo1, None, None)
check(foo2, None, None)
check(bar1, None, None)
check(bar2, None, None)
# set a special value
cache.set_special(foo1, name, lambda: '42')
self.assertTrue(cache.contains(foo1, name))
self.assertFalse(cache.contains_value(foo1, name))
self.assertEqual(cache.get(foo1, name), '42')
self.assertIsNone(cache.get_value(foo1, name))
self.assertIsNotNone(cache.get_special(foo1, name))
# copy cache
for rec in [foo1, bar1]:
cache.set(rec, name, 'NAME1')
cache.set(rec, ref, 'REF1')
check(foo1, 'NAME1', 'REF1')
check(foo2, None, None)
check(bar1, 'NAME1', 'REF1')
check(bar2, None, None)
cache.copy(foo1 + bar1, foo2.env)
check(foo1, 'NAME1', 'REF1')
check(foo2, 'NAME1', 'REF1')
check(bar1, 'NAME1', 'REF1')
check(bar2, 'NAME1', 'REF1')
def test_memory(self):
""" Check memory consumption of the cache. """
NB_RECORDS = 100000
MAX_MEMORY = 100
cache = self.env.cache
model = self.env['res.partner']
records = [model.new() for index in range(NB_RECORDS)]
process = psutil.Process(os.getpid())
rss0 = process.memory_info().rss
char_names = [
'name', 'display_name', 'email', 'website', 'phone', 'mobile',
'street', 'street2', 'city', 'zip', 'vat', 'ref',
]
for name in char_names:
field = model._fields[name]
for record in records:
cache.set(record, field, 'test')
mem_usage = process.memory_info().rss - rss0
self.assertLess(
mem_usage, MAX_MEMORY * 1024 * 1024,
"Caching %s records must take less than %sMB of memory" % (NB_RECORDS, MAX_MEMORY),
)