Compare commits
370
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1f0e0070a7 | ||
|
|
688fdc2ef5 | ||
|
|
2c7a5eb06c | ||
|
|
b5c4f90781 | ||
|
|
47efeea70f | ||
|
|
b468a88070 | ||
|
|
4e3996becf | ||
|
|
aeb02c58d4 | ||
|
|
9578c79f29 | ||
|
|
8433e48097 | ||
|
|
37722f350f | ||
|
|
bea5030ece | ||
|
|
de88ffaecd | ||
|
|
4ce9e795c8 | ||
|
|
bbc5638f1d | ||
|
|
68196ca74e | ||
|
|
9854bacfe9 | ||
|
|
60cb09324c | ||
|
|
e0c77a646a | ||
|
|
8275653c67 | ||
|
|
31ba768e27 | ||
|
|
f6f4838a75 | ||
|
|
d531afe765 | ||
|
|
369320b390 | ||
|
|
4688676eaa | ||
|
|
690dc76d53 | ||
|
|
bc066e37a7 | ||
|
|
080d43fd8d | ||
|
|
81cf8d6cde | ||
|
|
6333996d93 | ||
|
|
ef7dd1557c | ||
|
|
9bd77ae875 | ||
|
|
8c028b4487 | ||
|
|
7c90c1e4c3 | ||
|
|
a8ab206849 | ||
|
|
97f776792f | ||
|
|
779f4146e3 | ||
|
|
ca57a223ca | ||
|
|
10a8a3c706 | ||
|
|
8ccf87ad04 | ||
|
|
c4ab81ebe1 | ||
|
|
ebdadb3aee | ||
|
|
db194d733f | ||
|
|
97773d6dc7 | ||
|
|
80b220c1e3 | ||
|
|
7acfc411d8 | ||
|
|
c0bdbb7664 | ||
|
|
bc900c2ef1 | ||
|
|
436c296609 | ||
|
|
40a1e21348 | ||
|
|
633c7ee1e9 | ||
|
|
f35333a7b6 | ||
|
|
3d12bdb556 | ||
|
|
c5a083e2bc | ||
|
|
8fea844d70 | ||
|
|
c45bd07cea | ||
|
|
f98dead1ab | ||
|
|
7fe76c185d | ||
|
|
45552fdb97 | ||
|
|
0dece65b7b | ||
|
|
888f8dc4da | ||
|
|
2a38c5200c | ||
|
|
f30cb87e60 | ||
|
|
bbab55e6d6 | ||
|
|
62ad18d8ff | ||
|
|
ed51ad5be6 | ||
|
|
667cde1685 | ||
|
|
aaa1038a70 | ||
|
|
4be1b0437f | ||
|
|
635197de38 | ||
|
|
22cd908a49 | ||
|
|
ffd9bab0ef | ||
|
|
393209e624 | ||
|
|
4616f8cb0e | ||
|
|
afd9bb5a09 | ||
|
|
4f0624d35c | ||
|
|
3ec54e510c | ||
|
|
b2af10b11c | ||
|
|
0c2d0f0ffd | ||
|
|
5d97b4c84a | ||
|
|
3fb3936a8a | ||
|
|
efebfc67ed | ||
|
|
bde5c75eba | ||
|
|
a8c88a499a | ||
|
|
33ad54d6de | ||
|
|
39eea8490f | ||
|
|
85d0fb66b6 | ||
|
|
f0bef7f75d | ||
|
|
56cf86a246 | ||
|
|
6702cffe72 | ||
|
|
113ac7ad4a | ||
|
|
e97e48a695 | ||
|
|
f084b6e95f | ||
|
|
9e119f33e6 | ||
|
|
9697157d6b | ||
|
|
ce784d7af7 | ||
|
|
0c89989519 | ||
|
|
fcb28b8c9f | ||
|
|
613e594fa9 | ||
|
|
00bf7eeb75 | ||
|
|
6dee101593 | ||
|
|
677e08f723 | ||
|
|
3019647339 | ||
|
|
d1ba26af20 | ||
|
|
7497a3f021 | ||
|
|
bce87c40c1 | ||
|
|
c27c125d61 | ||
|
|
eb97f60cac | ||
|
|
669a40dd5e | ||
|
|
127824c026 | ||
|
|
3c38c8ec40 | ||
|
|
2e586fa61c | ||
|
|
7b3eb943e2 | ||
|
|
19439e45ef | ||
|
|
db3dc70ebf | ||
|
|
27eac13767 | ||
|
|
54c67f2777 | ||
|
|
39a812b134 | ||
|
|
9f36b17fee | ||
|
|
25c4a6710e | ||
|
|
93747f7758 | ||
|
|
4ffc27100d | ||
|
|
ffeed4a6a4 | ||
|
|
0a94ac98a3 | ||
|
|
c23947c28f | ||
|
|
262ce13f4d | ||
|
|
40e26d68b6 | ||
|
|
17b5c629ac | ||
|
|
acccfbcaad | ||
|
|
2d434fe61f | ||
|
|
6d7b6250c5 | ||
|
|
ef30f1d1bd | ||
|
|
4848c7f813 | ||
|
|
39f34b7c85 | ||
|
|
969070f1ae | ||
|
|
1b37c38bbf | ||
|
|
cc0f2c4e91 | ||
|
|
8a21e02862 | ||
|
|
342f06e7b0 | ||
|
|
88510aaa42 | ||
|
|
4749918014 | ||
|
|
7da75e10eb | ||
|
|
8d5e47a957 | ||
|
|
7fcbad44b9 | ||
|
|
38bb4981b8 | ||
|
|
bcc4c29d10 | ||
|
|
3ba44608f3 | ||
|
|
2d3441b127 | ||
|
|
1889ac6372 | ||
|
|
4d2ebd65fb | ||
|
|
40725372fe | ||
|
|
74128e0f26 | ||
|
|
840b29499a | ||
|
|
da87a151df | ||
|
|
9dd1a8f409 | ||
|
|
6a2509b96e | ||
|
|
7a2c8515a2 | ||
|
|
4561159218 | ||
|
|
cadba6f6aa | ||
|
|
a765b5fe5f | ||
|
|
595f2023e3 | ||
|
|
7946a2a272 | ||
|
|
191eac4424 | ||
|
|
7ed62ffd2c | ||
|
|
61107d0d02 | ||
|
|
ba8bb1d5f4 | ||
|
|
ad81a811fc | ||
|
|
64920a96c7 | ||
|
|
6e30ce0130 | ||
|
|
3bbf06a70a | ||
|
|
b8fd39c6a6 | ||
|
|
81e942a4d2 | ||
|
|
bec45b53fa | ||
|
|
af3bdb3ff2 | ||
|
|
fb5577bedd | ||
|
|
cb91971fd9 | ||
|
|
0a2cff5c12 | ||
|
|
752ed0ccb5 | ||
|
|
efe270b502 | ||
|
|
7aef68ab14 | ||
|
|
f96bd22c38 | ||
|
|
e7542dadbd | ||
|
|
434747894c | ||
|
|
ed62cf5da7 | ||
|
|
1388dd9e27 | ||
|
|
d11f8b3b76 | ||
|
|
6902de3b94 | ||
|
|
d912bf56d7 | ||
|
|
d1e61dc420 | ||
|
|
375316fbf3 | ||
|
|
602d0671f1 | ||
|
|
9df82a44ec | ||
|
|
8c22c753a8 | ||
|
|
9884143291 | ||
|
|
64d15cf3b6 | ||
|
|
50f1fed187 | ||
|
|
2f54b3ca5a | ||
|
|
5c78c315ee | ||
|
|
3620b5889d | ||
|
|
3478e59dd3 | ||
|
|
374f2dfe3e | ||
|
|
20ed85c864 | ||
|
|
c56820fb6c | ||
|
|
5f4023f120 | ||
|
|
0c92e2ac74 | ||
|
|
12463c61b8 | ||
|
|
f471da0fed | ||
|
|
76dfdb6012 | ||
|
|
95055996ad | ||
|
|
3bc5f27cda | ||
|
|
7f38f8c654 | ||
|
|
75f0ad9882 | ||
|
|
41cf4c3a79 | ||
|
|
5acb4882a8 | ||
|
|
962a1673f9 | ||
|
|
bfceeb56e5 | ||
|
|
475492a65f | ||
|
|
4f4ce697e4 | ||
|
|
f24749112e | ||
|
|
658c7b42ee | ||
|
|
f2a731711d | ||
|
|
928e41b82c | ||
|
|
298927d631 | ||
|
|
a0e602f0b2 | ||
|
|
cd2bd6b9e1 | ||
|
|
6416b25818 | ||
|
|
84fc270e0e | ||
|
|
44d3dcee34 | ||
|
|
378cae88bc | ||
|
|
acf2f7a189 | ||
|
|
32c77ab190 | ||
|
|
238fd11d6b | ||
|
|
2034b970b9 | ||
|
|
0433ac1087 | ||
|
|
be1cbdee76 | ||
|
|
cec543eee7 | ||
|
|
5059902947 | ||
|
|
3fa18d32d9 | ||
|
|
a836031d36 | ||
|
|
a32453394a | ||
|
|
50550ac6b6 | ||
|
|
dc890dde5e | ||
|
|
0ffafe3f62 | ||
|
|
86f05a0c23 | ||
|
|
658d1da5ce | ||
|
|
d7382e1ce2 | ||
|
|
3cd3e1ec83 | ||
|
|
e0b125f6a6 | ||
|
|
2bca8b4fb8 | ||
|
|
ab0755ad90 | ||
|
|
1bd3e63cd4 | ||
|
|
c20846122f | ||
|
|
793726adb2 | ||
|
|
a5c93dc176 | ||
|
|
2499a3f42a | ||
|
|
eb15dd65ec | ||
|
|
e4f0c5b1dd | ||
|
|
3976366913 | ||
|
|
20e609f952 | ||
|
|
338d686b4c | ||
|
|
a416018b92 | ||
|
|
7cc5c50240 | ||
|
|
e791923493 | ||
|
|
ae3011e88f | ||
|
|
66eda2214c | ||
|
|
b6229125a3 | ||
|
|
6fa6786a9c | ||
|
|
723bd03548 | ||
|
|
f0e0c8035a | ||
|
|
d0aba55b0c | ||
|
|
6673841bf9 | ||
|
|
99de0f6637 | ||
|
|
a5f2fa4d76 | ||
|
|
3268019216 | ||
|
|
57112f9de6 | ||
|
|
92ab21c99a | ||
|
|
0927136ffd | ||
|
|
7fb0589692 | ||
|
|
c023ad407d | ||
|
|
c7d3fa2c1e | ||
|
|
cd8d82c226 | ||
|
|
31ee58967f | ||
|
|
51ba2d9d21 | ||
|
|
c98edc4f77 | ||
|
|
328b924bd2 | ||
|
|
52d63cff60 | ||
|
|
d553aaf9eb | ||
|
|
0342dc863b | ||
|
|
ca8586b213 | ||
|
|
0e4231a900 | ||
|
|
0f9487f589 | ||
|
|
b976899f75 | ||
|
|
49a76b2ef4 | ||
|
|
c6905cbc98 | ||
|
|
7712b3fee0 | ||
|
|
fa3f96e464 | ||
|
|
698e149297 | ||
|
|
b758333f42 | ||
|
|
2cf33d6d03 | ||
|
|
b93c145cde | ||
|
|
f3dc52336f | ||
|
|
395902b9c7 | ||
|
|
314090fa56 | ||
|
|
33a8cf4be4 | ||
|
|
dcbf371077 | ||
|
|
48c08b869e | ||
|
|
50e63fb1f8 | ||
|
|
5ea20a11a9 | ||
|
|
b952f93e78 | ||
|
|
911af41ec1 | ||
|
|
a89fd63f4c | ||
|
|
38472012db | ||
|
|
51a814732b | ||
|
|
b28e11c69d | ||
|
|
322b4a0741 | ||
|
|
0391482ec7 | ||
|
|
444b8a6b39 | ||
|
|
77fe6704d8 | ||
|
|
b54d75e8d8 | ||
|
|
8304ddca5c | ||
|
|
59564f8c70 | ||
|
|
7bbcae574a | ||
|
|
430872b958 | ||
|
|
01f91de54b | ||
|
|
841340acd4 | ||
|
|
b7cf611595 | ||
|
|
de9f64cd3a | ||
|
|
351e698971 | ||
|
|
4f2c4bae74 | ||
|
|
e3d0d69af8 | ||
|
|
0829936940 | ||
|
|
508c959d2c | ||
|
|
bcd3c2ae45 | ||
|
|
6bea4cfc68 | ||
|
|
9891ccffaf | ||
|
|
88b1cea98c | ||
|
|
c4e60ca2a8 | ||
|
|
70d34b097f | ||
|
|
06ed53e4ec | ||
|
|
59a4f1cecc | ||
|
|
67e0442645 | ||
|
|
8f8f1d2176 | ||
|
|
e930e28540 | ||
|
|
3daba3429f | ||
|
|
5db4f113ac | ||
|
|
639867bb32 | ||
|
|
d8fc4b6177 | ||
|
|
a2acb9a025 | ||
|
|
cb6c329d32 | ||
|
|
220616a151 | ||
|
|
33209542d6 | ||
|
|
1966896850 | ||
|
|
53e67fb59f | ||
|
|
2c0ef08ee1 | ||
|
|
7d3b3129c5 | ||
|
|
5b86a66fb2 | ||
|
|
e7b31018eb | ||
|
|
3a56041da6 | ||
|
|
9b7e7a179e | ||
|
|
1b0ea035b8 | ||
|
|
d4baf89de5 | ||
|
|
b72772f1b9 | ||
|
|
f2ac744bc7 | ||
|
|
72becd75be | ||
|
|
d02d6b14eb | ||
|
|
30902d0fe0 | ||
|
|
fcb8196d3d | ||
|
|
c388f543d2 | ||
|
|
e6dba1f89f | ||
|
|
7ebf72705e |
@@ -0,0 +1,9 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" width="1280" height="640" viewBox="0 0 1280 640" role="img" aria-label="infi.clickhouse_orm">
|
||||
<rect width="1280" height="640" fill="#0A0A0A"/>
|
||||
<svg x="96" y="215" width="210" height="210" viewBox="0 0 67 67"><path d="M22.21 67V44.6369H0V67H22.21Z" fill="#fff"/><path d="M66.7038 22.3184H22.2534L0.0878906 44.6367H44.4634L66.7038 22.3184Z" fill="#fff"/><path d="M22.21 0H0V22.3184H22.21V0Z" fill="#fff"/><path d="M66.7198 0H44.5098V22.3184H66.7198V0Z" fill="#fff"/><path d="M66.7198 67V44.6369H44.5098V67H66.7198Z" fill="#fff"/></svg>
|
||||
<text x="378" y="276" font-family="Inter,system-ui,-apple-system,sans-serif" font-size="78" font-weight="800" letter-spacing="-2" fill="#ffffff">infi.clickhouse_orm</text>
|
||||
<text x="378" y="322" font-family="Inter,system-ui,sans-serif" font-size="30" fill="#ffffff" opacity=".66">A Python library for working with the ClickHouse database…</text>
|
||||
<rect x="378" y="338" width="806" height="3" rx="1.5" fill="#ffffff" opacity=".9"/>
|
||||
<text x="378" y="390" font-family="Inter,system-ui,sans-serif" font-size="24" font-weight="600" fill="#ffffff" opacity=".5">github.com/hanzoai</text>
|
||||
<text x="1184" y="390" text-anchor="end" font-family="Inter,system-ui,sans-serif" font-size="24" font-weight="600" fill="#ffffff" opacity=".5">hanzo.ai</text>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 1.3 KiB |
+5
-2
@@ -18,7 +18,6 @@
|
||||
bin/
|
||||
build/
|
||||
dist/
|
||||
setup.py
|
||||
|
||||
# buildout
|
||||
buildout.in.cfg
|
||||
@@ -58,4 +57,8 @@ buildout.in
|
||||
src/infi/clickhouse_orm/__version__.py
|
||||
bootstrap.py
|
||||
|
||||
htmldocs/
|
||||
htmldocs/
|
||||
cover/
|
||||
|
||||
# tox
|
||||
.tox/
|
||||
|
||||
+149
@@ -1,6 +1,155 @@
|
||||
Change Log
|
||||
==========
|
||||
|
||||
v2.1.0
|
||||
------
|
||||
- Support for model constraints
|
||||
- Support for data skipping indexes
|
||||
- Support for mutations: `QuerySet.update` and `QuerySet.delete`
|
||||
- Added functions for working with external dictionaries
|
||||
- Support FINAL for `ReplacingMergeTree` (chripede)
|
||||
- Added `DateTime64Field` (NiyazNz)
|
||||
- Make `DateTimeField` and `DateTime64Field` timezone-aware (NiyazNz)
|
||||
|
||||
**Backwards incompatible changes**
|
||||
|
||||
Previously, `DateTimeField` always converted its value from the database timezone to UTC. This is no longer the case: the field's value now preserves the timezone it was defined with, or if not specified - the database's global timezone. This change has no effect if your database timezone is set to UTC.
|
||||
|
||||
v2.0.1
|
||||
------
|
||||
- Remove unnecessary import of `six`
|
||||
|
||||
v2.0.0
|
||||
------
|
||||
- Dropped support for Python 2.x
|
||||
- New flexible syntax for database expressions and functions
|
||||
- Expressions as default values for model fields
|
||||
- Support for IPv4 and IPv6 fields
|
||||
- Automatic generation of models by inspecting existing tables
|
||||
- Convenient ways to import ORM classes
|
||||
|
||||
See [What's new in version 2](docs/whats_new_in_version_2.md) for details.
|
||||
|
||||
v1.4.0
|
||||
------
|
||||
- Added primary_key parameter to MergeTree engines (M1hacka)
|
||||
- Support negative enum values (Romamo)
|
||||
|
||||
v1.3.0
|
||||
------
|
||||
- Support LowCardinality columns in ad-hoc queries
|
||||
- Support for LIMIT BY in querysets (utapyngo)
|
||||
|
||||
v1.2.0
|
||||
------
|
||||
- Add support for per-field compression codecs (rbelio, Chocorean)
|
||||
- Add support for low cardinality fields (rbelio)
|
||||
|
||||
v1.1.0
|
||||
------
|
||||
- Add PREWHERE support to querysets (M1hacka)
|
||||
- Add WITH TOTALS support to querysets (M1hacka)
|
||||
- Extend date field range (trthhrtz)
|
||||
- Fix parsing of server errors in ClickHouse v19.3.3+
|
||||
- Fix pagination when asking for the last page on a query that matches no records
|
||||
- Use HTTP Basic Authentication instead of passing the credentials in the URL
|
||||
- Support default/alias/materialized for nullable fields
|
||||
- Add UUIDField (kpotehin)
|
||||
- Add `log_statements` parameter to database initializer
|
||||
- Fix test_merge which fails on ClickHouse v19.8.3
|
||||
- Fix querysets using the SystemPart model
|
||||
|
||||
v1.0.4
|
||||
------
|
||||
- Added `timeout` parameter to database initializer (SUHAR1K)
|
||||
- Added `verify_ssl_cert` parameter to database initializer
|
||||
- Added `final()` method to querysets (M1hacka)
|
||||
- Fixed a migrations problem - cannot add a new materialized field after a regular field
|
||||
|
||||
v1.0.3
|
||||
------
|
||||
- Bug fix: `QuerySet.count()` ignores slicing
|
||||
- Bug fix: wrong parentheses when building queries using Q objects
|
||||
- Support Decimal fields
|
||||
- Added `Database.add_setting` method
|
||||
|
||||
v1.0.2
|
||||
----------
|
||||
- Include alias and materialized fields in queryset results
|
||||
- Check for database existence, to allow delayed creation
|
||||
- Added `Database.does_table_exist` method
|
||||
- Support for `IS NULL` and `IS NOT NULL` in querysets (kalombos)
|
||||
|
||||
v1.0.1
|
||||
------
|
||||
- NullableField: take extra_null_values into account in `validate` and `to_python`
|
||||
- Added `Field.isinstance` method
|
||||
- Validate the inner field passed to `ArrayField`
|
||||
|
||||
v1.0.0
|
||||
------
|
||||
- Add support for compound filters with Q objects (desile)
|
||||
- Add support for BETWEEN operator (desile)
|
||||
- Distributed engine support (tsionyx)
|
||||
- `_fields` and `_writable_fields` are OrderedDicts - note that this might break backwards compatibility (tsionyx)
|
||||
- Improve error messages returned from the database with the `ServerError` class (tsionyx)
|
||||
- Added support for custom partitioning (M1hacka)
|
||||
- Added attribute `server_version` to Database class (M1hacka)
|
||||
- Changed `Engine.create_table_sql()`, `Engine.drop_table_sql()`, `Model.create_table_sql()`, `Model.drop_table_sql()` parameter to db from db_name (M1hacka)
|
||||
- Fix parsing of datetime column type when it includes a timezone (M1hacka)
|
||||
- Rename `Model.system` to `Model._system` to prevent collision with a column that has the same name
|
||||
- Rename `Model.readonly` to `Model._readonly` to prevent collision with a column that has the same name
|
||||
- The `field_names` argument to `Model.to_tsv` is now mandatory
|
||||
- Improve creation time of model instances by keeping a dictionary of default values
|
||||
- Fix queryset bug when field name contains double underscores (YouCanKeepSilence)
|
||||
- Prevent exception when determining timezone of old ClickHouse versions (vv-p)
|
||||
|
||||
v0.9.8
|
||||
------
|
||||
- Bug fix: add field names list explicitly to Database.insert method (anci)
|
||||
- Added RunPython and RunSQL migrations (M1hacka)
|
||||
- Allow ISO-formatted datetime values (tsionyx)
|
||||
- Show field name in error message when invalid value assigned (tsionyx)
|
||||
- Bug fix: select query fails when query contains '$' symbol (M1hacka)
|
||||
- Prevent problems with AlterTable migrations related to field order (M1hacka)
|
||||
- Added documentation about custom fields.
|
||||
|
||||
v0.9.7
|
||||
------
|
||||
- Add `distinct` method to querysets
|
||||
- Add `AlterTableWithBuffer` migration operation
|
||||
- Support Merge engine (M1hacka)
|
||||
|
||||
v0.9.6
|
||||
------
|
||||
- Fix python3 compatibility (TvoroG)
|
||||
- Nullable arrays not supported in latest ClickHouse version
|
||||
- system.parts table no longer includes "replicated" column in latest ClickHouse version
|
||||
|
||||
v0.9.5
|
||||
------
|
||||
- Added `QuerySet.paginate()`
|
||||
- Support for basic aggregation in querysets
|
||||
|
||||
v0.9.4
|
||||
------
|
||||
- Migrations: when creating a table for a `BufferModel`, create the underlying table too if necessary
|
||||
|
||||
v0.9.3
|
||||
------
|
||||
- Changed license from PSF to BSD
|
||||
- Nullable fields support (yamiou)
|
||||
- Support for queryset slicing
|
||||
|
||||
v0.9.2
|
||||
------
|
||||
- Added `ne` and `not_in` queryset operators
|
||||
- Querysets no longer have a default order unless `order_by` is called
|
||||
- Added `autocreate` flag to database initializer
|
||||
- Fix some Python 2/3 incompatibilities (TvoroG, tsionyx)
|
||||
- To work around a JOIN bug in ClickHouse, `$table` now inserts only the table name,
|
||||
and the database name is sent in the query params instead
|
||||
|
||||
v0.9.0
|
||||
------
|
||||
- Major new feature: building model queries using QuerySets
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
Copyright (c) 2017 INFINIDAT
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are met:
|
||||
|
||||
1. Redistributions of source code must retain the above copyright notice, this
|
||||
list of conditions and the following disclaimer.
|
||||
|
||||
2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimer in the documentation
|
||||
and/or other materials provided with the distribution.
|
||||
|
||||
3. Neither the name of the copyright holder nor the names of its contributors
|
||||
may be used to endorse or promote products derived from this software
|
||||
without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
||||
ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
@@ -0,0 +1,19 @@
|
||||
# datastore_orm
|
||||
|
||||
**Org:** hanzoai · **Ecosystem:** hanzo · **Path:** `/Users/a/work/hanzo/hanzoai/datastore_orm`
|
||||
**Origin:** https://github.com/hanzoai/datastore_orm.git
|
||||
|
||||
## Discovery
|
||||
|
||||
This file (`CLAUDE.md`) is the canonical agent-facing readme; `LLM.md` is a symlink to it. Update either name and both stay in sync.
|
||||
|
||||
## Where to look first
|
||||
|
||||
- `README.md` — human-facing overview (if present)
|
||||
- `package.json` / `Cargo.toml` / `pyproject.toml` / `go.mod` — language & deps
|
||||
- `.github/workflows/` — CI surface
|
||||
- `docs/` — extended docs (if present)
|
||||
|
||||
## Sibling repos
|
||||
|
||||
See the org-level `LLM.md` at `/Users/a/work/hanzo/hanzoai/LLM.md` for the full inventory of sibling repos and inter-repo dependencies.
|
||||
@@ -1,3 +1,5 @@
|
||||
<p align="center"><img src=".github/hero.svg" alt="infi.clickhouse_orm" width="880"></p>
|
||||
|
||||
Introduction
|
||||
============
|
||||
|
||||
@@ -8,10 +10,7 @@ Let's jump right in with a simple example of monitoring CPU usage. First we need
|
||||
connect to the database and create a table for the model:
|
||||
|
||||
```python
|
||||
from infi.clickhouse_orm.database import Database
|
||||
from infi.clickhouse_orm.models import Model
|
||||
from infi.clickhouse_orm.fields import *
|
||||
from infi.clickhouse_orm.engines import Memory
|
||||
from datastore_orm import Database, Model, DateTimeField, UInt16Field, Float32Field, Memory, F
|
||||
|
||||
class CPUStats(Model):
|
||||
|
||||
@@ -45,13 +44,16 @@ Querying the table is easy, using either the query builder or raw SQL:
|
||||
|
||||
```python
|
||||
# Calculate what percentage of the time CPU 1 was over 95% busy
|
||||
total = CPUStats.objects_in(db).filter(cpu_id=1).count()
|
||||
busy = CPUStats.objects_in(db).filter(cpu_id=1, cpu_percent__gt=95).count()
|
||||
print 'CPU 1 was busy {:.2f}% of the time'.format(busy * 100.0 / total)
|
||||
queryset = CPUStats.objects_in(db)
|
||||
total = queryset.filter(CPUStats.cpu_id == 1).count()
|
||||
busy = queryset.filter(CPUStats.cpu_id == 1, CPUStats.cpu_percent > 95).count()
|
||||
print('CPU 1 was busy {:.2f}% of the time'.format(busy * 100.0 / total))
|
||||
|
||||
# Calculate the average usage per CPU
|
||||
for row in db.select('SELECT cpu_id, avg(cpu_percent) AS average FROM demo.cpustats GROUP BY cpu_id'):
|
||||
print 'CPU {row.cpu_id}: {row.average:.2f}%'.format(row=row)
|
||||
for row in queryset.aggregate(CPUStats.cpu_id, average=F.avg(CPUStats.cpu_percent)):
|
||||
print('CPU {row.cpu_id}: {row.average:.2f}%'.format(row=row))
|
||||
```
|
||||
|
||||
To learn more please visit the [documentation](docs/toc.md).
|
||||
This and other examples can be found in the `examples` folder.
|
||||
|
||||
To learn more please visit the [documentation](docs/toc.md).
|
||||
|
||||
+10
-8
@@ -4,31 +4,33 @@ newest = false
|
||||
download-cache = .cache
|
||||
develop = .
|
||||
parts =
|
||||
relative-paths = true
|
||||
|
||||
[project]
|
||||
name = infi.clickhouse_orm
|
||||
name = datastore_orm
|
||||
company = Infinidat
|
||||
namespace_packages = ['infi']
|
||||
install_requires = [
|
||||
'iso8601 >= 0.1.12',
|
||||
'pytz',
|
||||
'requests',
|
||||
'setuptools',
|
||||
'six'
|
||||
'setuptools'
|
||||
]
|
||||
version_file = src/infi/clickhouse_orm/__version__.py
|
||||
description = A Python library for working with the ClickHouse database
|
||||
long_description = A Python library for working with the ClickHouse database
|
||||
console_scripts = []
|
||||
gui_scripts = []
|
||||
package_data = []
|
||||
upgrade_code = {58530fba-3932-11e6-a20e-7071bc32067f}
|
||||
product_name = infi.clickhouse_orm
|
||||
product_name = datastore_orm
|
||||
post_install_script_name = None
|
||||
pre_uninstall_script_name = None
|
||||
homepage = https://github.com/Infinidat/infi.clickhouse_orm
|
||||
homepage = https://github.com/Infinidat/datastore_orm
|
||||
|
||||
[isolated-python]
|
||||
recipe = infi.recipe.python
|
||||
version = v2.7.9.4
|
||||
version = v3.8.0.2
|
||||
|
||||
[setup.py]
|
||||
recipe = infi.recipe.template.version
|
||||
@@ -43,10 +45,10 @@ output = ${project:version_file}
|
||||
dependent-scripts = true
|
||||
recipe = infi.recipe.console_scripts
|
||||
eggs = ${project:name}
|
||||
ipython
|
||||
ipython<6
|
||||
nose
|
||||
coverage
|
||||
enum34
|
||||
enum-compat
|
||||
infi.unittest
|
||||
infi.traceback
|
||||
memory_profiler
|
||||
|
||||
+2689
-137
File diff suppressed because it is too large
Load Diff
@@ -1,7 +1,7 @@
|
||||
Contributing
|
||||
============
|
||||
|
||||
This project is hosted on GitHub - [https://github.com/Infinidat/infi.clickhouse_orm/](https://github.com/Infinidat/infi.clickhouse_orm/).
|
||||
This project is hosted on GitHub - [https://github.com/Infinidat/datastore_orm/](https://github.com/Infinidat/datastore_orm/).
|
||||
|
||||
Please open an issue there if you encounter a bug or want to request a feature.
|
||||
Pull requests are also welcome.
|
||||
@@ -12,7 +12,7 @@ Building
|
||||
After cloning the project, run the following commands:
|
||||
|
||||
easy_install -U infi.projector
|
||||
cd infi.clickhouse_orm
|
||||
cd datastore_orm
|
||||
projector devenv build
|
||||
|
||||
A `setup.py` file will be generated, which you can use to install the development version of the package:
|
||||
@@ -28,8 +28,12 @@ To run the tests, ensure that the ClickHouse server is running on <http://localh
|
||||
|
||||
To see test coverage information run:
|
||||
|
||||
bin/nosetests --with-coverage --cover-package=infi.clickhouse_orm
|
||||
bin/nosetests --with-coverage --cover-package=datastore_orm
|
||||
|
||||
To test with tox, ensure that the setup.py is present (otherwise run `bin/buildout buildout:develop= setup.py`) and run:
|
||||
|
||||
pip install tox
|
||||
tox
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
|
||||
Expressions
|
||||
===========
|
||||
|
||||
One of the ORM's core concepts is _expressions_, which are composed using functions, operators and model fields. Expressions are used in multiple places in the ORM:
|
||||
|
||||
- When defining [field options](field_options.md) - `default`, `alias` and `materialized`.
|
||||
- In [table engine](table_engines.md) parameters for engines in the `MergeTree` family.
|
||||
- In [queryset](querysets.md) methods such as `filter`, `exclude`, `order_by`, `aggregate` and `limit_by`.
|
||||
|
||||
Using Expressions
|
||||
-----------------
|
||||
|
||||
Expressions usually include ClickHouse database functions, which are made available by the `F` class. Here's a simple function:
|
||||
```python
|
||||
from datastore_orm import F
|
||||
expr = F.today()
|
||||
```
|
||||
|
||||
Functions that accept arguments can be composed, just like when using SQL:
|
||||
```python
|
||||
expr = F.toDayOfWeek(F.today())
|
||||
```
|
||||
|
||||
You can see the SQL expression that is represented by an ORM expression by calling its `to_sql` method or converting it to a string:
|
||||
```python
|
||||
>>> print(expr)
|
||||
toDayOfWeek(today())
|
||||
```
|
||||
|
||||
### Operators
|
||||
|
||||
ORM expressions support Python's standard arithmetic operators, so you can compose expressions using `+`, `-`, `*`, `/`, `//` and `%`. For example:
|
||||
```python
|
||||
# A random integer between 1 and 10
|
||||
F.rand() % 10 + 1
|
||||
```
|
||||
|
||||
There is also support for comparison operators (`<`, `<=`, `==`, `>=`, `>`, `!=`) and logical operators (`&`, `|`, `~`, `^`) which are often used for filtering querysets:
|
||||
```python
|
||||
# Is it Friday the 13th?
|
||||
(F.toDayOfWeek(F.today()) == 6) & (F.toDayOfMonth(F.today()) == 13)
|
||||
```
|
||||
|
||||
Note that Python's bitwise operators (`&`, `|`, `~`, `^`) have higher precedence than comparison operators, so always use parentheses when combining these two types of operators in an expression. Otherwise the resulting SQL might be different than what you would expect.
|
||||
|
||||
### Referring to model fields
|
||||
|
||||
To refer to a model field inside an expression, use `<class>.<field>` syntax, for example:
|
||||
```python
|
||||
# Convert the temperature from Celsius to Fahrenheit
|
||||
Sensor.temperature * 1.8 + 32
|
||||
```
|
||||
|
||||
Inside model class definitions omit the class name:
|
||||
```python
|
||||
class Person(Model):
|
||||
height_cm = Float32Field()
|
||||
height_inch = Float32Field(alias=height_cm/2.54)
|
||||
...
|
||||
```
|
||||
|
||||
### Parametric functions
|
||||
|
||||
Some of ClickHouse's aggregate functions can accept one or more parameters - constants for initialization that affect the way the function works. The syntax is two pairs of brackets instead of one. The first is for parameters, and the second is for arguments. For example:
|
||||
```python
|
||||
# Most common last names
|
||||
F.topK(5)(Person.last_name)
|
||||
# Find 90th, 95th and 99th percentile of heights
|
||||
F.quantiles(0.9, 0.95, 0.99)(Person.height)
|
||||
```
|
||||
|
||||
### Creating new "functions"
|
||||
|
||||
Since expressions are just Python objects until they get converted to SQL, it is possible to invent new "functions" by combining existing ones into useful building blocks. For example, we can create a reusable expression that takes a string and trims whitespace, converts it to uppercase, and changes blanks to underscores:
|
||||
```python
|
||||
def normalize_string(s):
|
||||
return F.replaceAll(F.upper(F.trimBoth(s)), ' ', '_')
|
||||
```
|
||||
|
||||
Then we can use this expression anywhere we need it:
|
||||
```python
|
||||
class Event(Model):
|
||||
code = StringField()
|
||||
normalized_code = StringField(materialized=normalize_string(code))
|
||||
```
|
||||
|
||||
### Which functions are available?
|
||||
|
||||
ClickHouse has many hundreds of functions, and new ones often get added. Many, but not all of them, are already covered by the ORM. If you encounter a function that the database supports but is not available in the `F` class, please report this via a GitHub issue. You can still use the function by providing its name:
|
||||
```python
|
||||
expr = F("someFunctionName", arg1, arg2, ...)
|
||||
```
|
||||
|
||||
Note that higher-order database functions (those that use lambda expressions) are not supported.
|
||||
|
||||
---
|
||||
|
||||
[<< Models and Databases](models_and_databases.md) | [Table of Contents](toc.md) | [Importing ORM Classes >>](importing_orm_classes.md)
|
||||
@@ -0,0 +1,112 @@
|
||||
Field Options
|
||||
=============
|
||||
|
||||
All field types accept the following arguments:
|
||||
|
||||
- default
|
||||
- alias
|
||||
- materialized
|
||||
- readonly
|
||||
- codec
|
||||
|
||||
Note that `default`, `alias` and `materialized` are mutually exclusive - you cannot use more than one of them in a single field.
|
||||
|
||||
## default
|
||||
|
||||
Specifies a default value to use for the field. If not given, the field will have a default value based on its type: empty string for string fields, zero for numeric fields, etc.
|
||||
The default value can be a Python value suitable for the field type, or an expression. For example:
|
||||
```python
|
||||
class Event(Model):
|
||||
|
||||
name = StringField(default="EVENT")
|
||||
repeated = UInt32Field(default=1)
|
||||
created = DateTimeField(default=F.now())
|
||||
|
||||
engine = Memory()
|
||||
...
|
||||
```
|
||||
When creating a model instance, any fields you do not specify get their default value. Fields that use a default expression are assigned a sentinel value of `datastore_orm.utils.NO_VALUE` instead. For example:
|
||||
```python
|
||||
>>> event = Event()
|
||||
>>> print(event.to_dict())
|
||||
{'name': 'EVENT', 'repeated': 1, 'created': <NO_VALUE>}
|
||||
```
|
||||
:warning: Due to a bug in ClickHouse versions prior to 20.1.2.4, insertion of records with expressions for default values may fail.
|
||||
|
||||
## alias / materialized
|
||||
|
||||
The `alias` and `materialized` attributes expect an expression that gets calculated by the database. The difference is that `alias` fields are calculated on the fly, while `materialized` fields are calculated when the record is inserted, and are stored on disk.
|
||||
You can use any expression, and can refer to other model fields. For example:
|
||||
```python
|
||||
class Event(Model):
|
||||
|
||||
created = DateTimeField()
|
||||
created_date = DateTimeField(materialized=F.toDate(created))
|
||||
name = StringField()
|
||||
normalized_name = StringField(alias=F.upper(F.trim(name)))
|
||||
|
||||
engine = Memory()
|
||||
```
|
||||
For backwards compatibility with older versions of the ORM, you can pass the expression as an SQL string:
|
||||
```python
|
||||
created_date = DateTimeField(materialized="toDate(created)")
|
||||
```
|
||||
Both field types can't be inserted into the database directly, so they are ignored when using the `Database.insert()` method. ClickHouse does not return the field values if you use `"SELECT * FROM ..."` - you have to list these field names explicitly in the query.
|
||||
|
||||
Usage:
|
||||
```python
|
||||
obj = Event(created=datetime.now(), name='MyEvent')
|
||||
db = Database('my_test_db')
|
||||
db.insert([obj])
|
||||
# All values will be retrieved from database
|
||||
db.select('SELECT created, created_date, username, name FROM $db.event', model_class=Event)
|
||||
# created_date and username will contain a default value
|
||||
db.select('SELECT * FROM $db.event', model_class=Event)
|
||||
```
|
||||
When creating a model instance, any alias or materialized fields are assigned a sentinel value of `datastore_orm.utils.NO_VALUE` since their real values can only be known after insertion to the database.
|
||||
|
||||
## codec
|
||||
|
||||
This attribute specifies the compression algorithm to use for the field (instead of the default data compression algorithm defined in server settings).
|
||||
|
||||
Supported compression algorithms:
|
||||
|
||||
| Codec | Argument | Comment
|
||||
| -------------------- | -------------------------------------------| ----------------------------------------------------
|
||||
| NONE | None | No compression.
|
||||
| LZ4 | None | LZ4 compression.
|
||||
| LZ4HC(`level`) | Possible `level` range: [3, 12]. | Default value: 9. Greater values stands for better compression and higher CPU usage. Recommended value range: [4,9].
|
||||
| ZSTD(`level`) | Possible `level`range: [1, 22]. | Default value: 1. Greater values stands for better compression and higher CPU usage. Levels >= 20, should be used with caution, as they require more memory.
|
||||
| Delta(`delta_bytes`) | Possible `delta_bytes` range: 1, 2, 4 , 8. | Default value for `delta_bytes` is `sizeof(type)` if it is equal to 1, 2,4 or 8 and equals to 1 otherwise.
|
||||
|
||||
Codecs can be combined by separating their names with commas. The default database codec is not included into pipeline (if it should be applied to a field, you have to specify it explicitly in pipeline).
|
||||
|
||||
Recommended usage for codecs:
|
||||
- When values for particular metric do not differ significantly from point to point, delta-encoding allows to reduce disk space usage significantly.
|
||||
- DateTime works great with pipeline of Delta, ZSTD and the column size can be compressed to 2-3% of its original size (given a smooth datetime data)
|
||||
- Numeric types usually enjoy best compression rates with ZSTD
|
||||
- String types enjoy good compression rates with LZ4HC
|
||||
|
||||
Example:
|
||||
```python
|
||||
class Stats(Model):
|
||||
|
||||
id = UInt64Field(codec='ZSTD(10)')
|
||||
timestamp = DateTimeField(codec='Delta,ZSTD')
|
||||
timestamp_date = DateField(codec='Delta(4),ZSTD(22)')
|
||||
metadata_id = Int64Field(codec='LZ4')
|
||||
status = StringField(codec='LZ4HC(10)')
|
||||
calculation = NullableField(Float32Field(), codec='ZSTD')
|
||||
alerts = ArrayField(FixedStringField(length=15), codec='Delta(2),LZ4HC')
|
||||
|
||||
engine = MergeTree('timestamp_date', ('id', 'timestamp'))
|
||||
```
|
||||
Note: This feature is supported on ClickHouse version 19.1.16 and above. Codec arguments will be ignored by the ORM for older versions of ClickHouse.
|
||||
|
||||
## readonly
|
||||
|
||||
This attribute is set automatically for fields with `alias` or `materialized` attributes, you do not need to pass it yourself.
|
||||
|
||||
---
|
||||
|
||||
[<< Querysets](querysets.md) | [Table of Contents](toc.md) | [Field Types >>](field_types.md)
|
||||
+147
-61
@@ -1,109 +1,195 @@
|
||||
Field Types
|
||||
===========
|
||||
|
||||
Currently the following field types are supported:
|
||||
See: [ClickHouse Documentation](https://clickhouse.tech/docs/en/sql-reference/data-types/)
|
||||
|
||||
The following field types are supported:
|
||||
|
||||
| Class | DB Type | Pythonic Type | Comments
|
||||
| ------------------ | ---------- | --------------------- | -----------------------------------------------------
|
||||
| StringField | String | str | Encoded as UTF-8 when written to ClickHouse
|
||||
| FixedStringField | FixedString| str | Encoded as UTF-8 when written to ClickHouse
|
||||
| DateField | Date | datetime.date | Range 1970-01-01 to 2105-12-31
|
||||
| DateTimeField | DateTime | datetime.datetime | Minimal value is 1970-01-01 00:00:00; Timezone aware
|
||||
| DateTime64Field | DateTime64 | datetime.datetime | Minimal value is 1970-01-01 00:00:00; Timezone aware
|
||||
| Int8Field | Int8 | int | Range -128 to 127
|
||||
| Int16Field | Int16 | int | Range -32768 to 32767
|
||||
| Int32Field | Int32 | int | Range -2147483648 to 2147483647
|
||||
| Int64Field | Int64 | int | Range -9223372036854775808 to 9223372036854775807
|
||||
| UInt8Field | UInt8 | int | Range 0 to 255
|
||||
| UInt16Field | UInt16 | int | Range 0 to 65535
|
||||
| UInt32Field | UInt32 | int | Range 0 to 4294967295
|
||||
| UInt64Field | UInt64 | int | Range 0 to 18446744073709551615
|
||||
| Float32Field | Float32 | float |
|
||||
| Float64Field | Float64 | float |
|
||||
| DecimalField | Decimal | Decimal | Pythonic values are rounded to fit the scale of the database field
|
||||
| Decimal32Field | Decimal32 | Decimal | Ditto
|
||||
| Decimal64Field | Decimal64 | Decimal | Ditto
|
||||
| Decimal128Field | Decimal128 | Decimal | Ditto
|
||||
| UUIDField | UUID | uuid.UUID |
|
||||
| IPv4Field | IPv4 | ipaddress.IPv4Address |
|
||||
| IPv6Field | IPv6 | ipaddress.IPv6Address |
|
||||
| Enum8Field | Enum8 | Enum | See below
|
||||
| Enum16Field | Enum16 | Enum | See below
|
||||
| ArrayField | Array | list | See below
|
||||
| NullableField | Nullable | See below | See below
|
||||
|
||||
| Class | DB Type | Pythonic Type | Comments
|
||||
| ------------------ | ---------- | ------------------- | -----------------------------------------------------
|
||||
| StringField | String | unicode | Encoded as UTF-8 when written to ClickHouse
|
||||
| FixedStringField | String | unicode | Encoded as UTF-8 when written to ClickHouse
|
||||
| DateField | Date | datetime.date | Range 1970-01-01 to 2038-01-19
|
||||
| DateTimeField | DateTime | datetime.datetime | Minimal value is 1970-01-01 00:00:00; Always in UTC
|
||||
| Int8Field | Int8 | int | Range -128 to 127
|
||||
| Int16Field | Int16 | int | Range -32768 to 32767
|
||||
| Int32Field | Int32 | int | Range -2147483648 to 2147483647
|
||||
| Int64Field | Int64 | int/long | Range -9223372036854775808 to 9223372036854775807
|
||||
| UInt8Field | UInt8 | int | Range 0 to 255
|
||||
| UInt16Field | UInt16 | int | Range 0 to 65535
|
||||
| UInt32Field | UInt32 | int | Range 0 to 4294967295
|
||||
| UInt64Field | UInt64 | int/long | Range 0 to 18446744073709551615
|
||||
| Float32Field | Float32 | float |
|
||||
| Float64Field | Float64 | float |
|
||||
| Enum8Field | Enum8 | Enum | See below
|
||||
| Enum16Field | Enum16 | Enum | See below
|
||||
| ArrayField | Array | list | See below
|
||||
|
||||
DateTimeField and Time Zones
|
||||
----------------------------
|
||||
|
||||
A `DateTimeField` can be assigned values from one of the following types:
|
||||
`DateTimeField` and `DateTime64Field` can accept a `timezone` parameter (either the timezone name or a `pytz` timezone instance). This timezone will be used as the column timezone in ClickHouse. If not provided, the fields will use the timezone defined in the database configuration.
|
||||
|
||||
A `DateTimeField` and `DateTime64Field` can be assigned values from one of the following types:
|
||||
|
||||
- datetime
|
||||
- date
|
||||
- integer - number of seconds since the Unix epoch
|
||||
- string in `YYYY-MM-DD HH:MM:SS` format
|
||||
- float (DateTime64Field only) - number of seconds and microseconds since the Unix epoch
|
||||
- string in `YYYY-MM-DD HH:MM:SS` format or [ISO 8601](https://en.wikipedia.org/wiki/ISO_8601)-compatible format
|
||||
|
||||
The assigned value always gets converted to a timezone-aware `datetime` in UTC. If the assigned value is a timezone-aware `datetime` in another timezone, it will be converted to UTC. Otherwise, the assigned value is assumed to already be in UTC.
|
||||
The assigned value always gets converted to a timezone-aware `datetime` in UTC. The only exception is when the assigned value is a timezone-aware `datetime`, in which case it will not be changed.
|
||||
|
||||
DateTime values that are read from the database are kept in the database-defined timezone - either the one defined for the field, or the global timezone defined in the database configuration.
|
||||
|
||||
It is strongly recommended to set the server timezone to UTC and to store all datetime values in that timezone, in order to prevent confusion and subtle bugs. Conversion to a different timezone should only be performed when the value needs to be displayed.
|
||||
|
||||
DateTime values that are read from the database are also converted to UTC. ClickHouse formats them according to the timezone of the server, and the ORM makes the necessary conversions. This requires a ClickHouse
|
||||
version which is new enough to support the `timezone()` function, otherwise it is assumed to be using UTC. In any case, we recommend settings the server timezone to UTC in order to prevent confusion.
|
||||
|
||||
Working with enum fields
|
||||
------------------------
|
||||
|
||||
`Enum8Field` and `Enum16Field` provide support for working with ClickHouse enum columns. They accept strings or integers as values, and convert them to the matching Pythonic Enum member.
|
||||
|
||||
Python 3.4 and higher supports Enums natively. When using previous Python versions you need to install the enum34 library.
|
||||
|
||||
Example of a model with an enum field:
|
||||
|
||||
Gender = Enum('Gender', 'male female unspecified')
|
||||
```python
|
||||
Gender = Enum('Gender', 'male female unspecified')
|
||||
|
||||
class Person(models.Model):
|
||||
class Person(Model):
|
||||
|
||||
first_name = fields.StringField()
|
||||
last_name = fields.StringField()
|
||||
birthday = fields.DateField()
|
||||
gender = fields.Enum32Field(Gender)
|
||||
first_name = StringField()
|
||||
last_name = StringField()
|
||||
birthday = DateField()
|
||||
gender = Enum32Field(Gender)
|
||||
|
||||
engine = engines.MergeTree('birthday', ('first_name', 'last_name', 'birthday'))
|
||||
engine = MergeTree('birthday', ('first_name', 'last_name', 'birthday'))
|
||||
|
||||
suzy = Person(first_name='Suzy', last_name='Jones', gender=Gender.female)
|
||||
suzy = Person(first_name='Suzy', last_name='Jones', gender=Gender.female)
|
||||
```
|
||||
|
||||
Working with array fields
|
||||
-------------------------
|
||||
|
||||
You can create array fields containing any data type, for example:
|
||||
|
||||
class SensorData(models.Model):
|
||||
```python
|
||||
class SensorData(Model):
|
||||
|
||||
date = fields.DateField()
|
||||
temperatures = fields.ArrayField(fields.Float32Field())
|
||||
humidity_levels = fields.ArrayField(fields.UInt8Field())
|
||||
date = DateField()
|
||||
temperatures = ArrayField(Float32Field())
|
||||
humidity_levels = ArrayField(UInt8Field())
|
||||
|
||||
engine = engines.MergeTree('date', ('date',))
|
||||
engine = MergeTree('date', ('date',))
|
||||
|
||||
data = SensorData(date=date.today(), temperatures=[25.5, 31.2, 28.7], humidity_levels=[41, 39, 66])
|
||||
data = SensorData(date=date.today(), temperatures=[25.5, 31.2, 28.7], humidity_levels=[41, 39, 66])
|
||||
```
|
||||
|
||||
Working with materialized and alias fields
|
||||
------------------------------------------
|
||||
Note that multidimensional arrays are not supported yet by the ORM.
|
||||
|
||||
ClickHouse provides an opportunity to create MATERIALIZED and ALIAS Fields.
|
||||
Working with nullable fields
|
||||
----------------------------
|
||||
[ClickHouse provides a NULL value support](https://clickhouse.tech/docs/en/sql-reference/data-types/nullable/).
|
||||
|
||||
See documentation [here](https://clickhouse.yandex/reference_en.html#Default%20values).
|
||||
Wrapping another field in a `NullableField` makes it possible to assign `None` to that field. For example:
|
||||
|
||||
Both field types can't be inserted into the database directly, so they are ignored when using the `Database.insert()` method. ClickHouse does not return the field values if you use `"SELECT * FROM ..."` - you have to list these field names explicitly in the query.
|
||||
```python
|
||||
class EventData(Model):
|
||||
|
||||
Usage:
|
||||
date = DateField()
|
||||
comment = NullableField(StringField(), extra_null_values={''})
|
||||
score = NullableField(UInt8Field())
|
||||
serie = NullableField(ArrayField(UInt8Field()))
|
||||
|
||||
class Event(models.Model):
|
||||
engine = MergeTree('date', ('date',))
|
||||
|
||||
created = fields.DateTimeField()
|
||||
created_date = fields.DateTimeField(materialized='toDate(created)')
|
||||
name = fields.StringField()
|
||||
username = fields.StringField(alias='name')
|
||||
|
||||
engine = engines.MergeTree('created_date', ('created_date', 'created'))
|
||||
score_event = EventData(date=date.today(), comment=None, score=5, serie=None)
|
||||
comment_event = EventData(date=date.today(), comment='Excellent!', score=None, serie=None)
|
||||
another_event = EventData(date=date.today(), comment='', score=None, serie=None)
|
||||
action_event = EventData(date=date.today(), comment='', score=None, serie=[1, 2, 3])
|
||||
```
|
||||
|
||||
obj = Event(created=datetime.now(), name='MyEvent')
|
||||
db = Database('my_test_db')
|
||||
db.insert([obj])
|
||||
# All values will be retrieved from database
|
||||
db.select('SELECT created, created_date, username, name FROM $db.event', model_class=Event)
|
||||
# created_date and username will contain a default value
|
||||
db.select('SELECT * FROM $db.event', model_class=Event)
|
||||
The `extra_null_values` parameter is an iterable of additional values that should be converted
|
||||
to `None`.
|
||||
|
||||
NOTE: `ArrayField` of `NullableField` is not supported. Also `EnumField` cannot be nullable.
|
||||
|
||||
NOTE: Using `Nullable` almost always negatively affects performance, keep this in mind when designing your databases.
|
||||
|
||||
Working with LowCardinality fields
|
||||
----------------------------------
|
||||
Starting with version 19.0 ClickHouse offers a new type of field to improve the performance of queries
|
||||
and compaction of columns for low entropy data.
|
||||
|
||||
[More specifically](https://github.com/tech/ClickHouse/issues/4074) LowCardinality data type builds dictionaries automatically. It can use multiple different dictionaries if necessarily.
|
||||
If the number of distinct values is pretty large, the dictionaries become local, several different dictionaries will be used for different ranges of data. For example, if you have too many distinct values in total, but only less than about a million values each day - then the queries by day will be processed efficiently, and queries for larger ranges will be processed rather efficiently.
|
||||
|
||||
LowCardinality works independently of (generic) fields compression.
|
||||
LowCardinality fields are subsequently compressed as usual.
|
||||
The compression ratios of LowCardinality fields for text data may be significantly better than without LowCardinality.
|
||||
|
||||
LowCardinality will give performance boost, in the form of processing speed, if the number of distinct values is less than a few millions. This is because data is processed in dictionary encoded form.
|
||||
|
||||
You can find further information [here](https://clickhouse.tech/docs/en/sql-reference/data-types/lowcardinality/).
|
||||
|
||||
Usage example:
|
||||
```python
|
||||
class LowCardinalityModel(Model):
|
||||
date = DateField()
|
||||
string = LowCardinalityField(StringField())
|
||||
nullable = LowCardinalityField(NullableField(StringField()))
|
||||
array = ArrayField(LowCardinalityField(DateField()))
|
||||
...
|
||||
```
|
||||
|
||||
Note: `LowCardinality` field with an inner array field is not supported. Use an `ArrayField` with a `LowCardinality` inner field as seen in the example.
|
||||
|
||||
Creating custom field types
|
||||
---------------------------
|
||||
Sometimes it is convenient to use data types that are supported in Python, but have no corresponding column type in ClickHouse. In these cases it is possible to define a custom field class that knows how to convert the Pythonic object to a suitable representation in the database, and vice versa.
|
||||
|
||||
For example, we can create a BooleanField which will hold `True` and `False` values, but write them to the database as 0 and 1 (in a `UInt8` column). For this purpose we'll subclass the `Field` class, and implement two methods:
|
||||
|
||||
- `to_python` which converts any supported value to a `bool`. The method should know how to handle strings (which typically come from the database), booleans, and possibly other valid options. In case the value is not supported, it should raise a `ValueError`.
|
||||
- `to_db_string` which converts a `bool` into a string for writing to the database.
|
||||
|
||||
Here's the full implementation:
|
||||
|
||||
```python
|
||||
from datastore_orm import Field
|
||||
|
||||
class BooleanField(Field):
|
||||
|
||||
# The ClickHouse column type to use
|
||||
db_type = 'UInt8'
|
||||
|
||||
# The default value
|
||||
class_default = False
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
# Convert valid values to bool
|
||||
if value in (1, '1', True):
|
||||
return True
|
||||
elif value in (0, '0', False):
|
||||
return False
|
||||
else:
|
||||
raise ValueError('Invalid value for BooleanField: %r' % value)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
# The value was already converted by to_python, so it's a bool
|
||||
return '1' if value else '0'
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
[<< Querysets](querysets.md) | [Table of Contents](toc.md) | [Table Engines >>](table_engines.md)
|
||||
[<< Field Options](field_options.md) | [Table of Contents](toc.md) | [Table Engines >>](table_engines.md)
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
|
||||
Importing ORM Classes
|
||||
=====================
|
||||
|
||||
The ORM supports different styles of importing and referring to its classes, so choose what works for you from the options below.
|
||||
|
||||
Importing Everything
|
||||
--------------------
|
||||
|
||||
It is safe to use `import *` from `datastore_orm` or its submodules. Only classes that are needed by users of the ORM will get imported, and nothing else:
|
||||
```python
|
||||
from datastore_orm import *
|
||||
```
|
||||
This is exactly equivalent to the following import statements:
|
||||
```python
|
||||
from datastore_orm.database import *
|
||||
from datastore_orm.engines import *
|
||||
from datastore_orm.fields import *
|
||||
from datastore_orm.funcs import *
|
||||
from datastore_orm.migrations import *
|
||||
from datastore_orm.models import *
|
||||
from datastore_orm.query import *
|
||||
from datastore_orm.system_models import *
|
||||
```
|
||||
By importing everything, all of the ORM's public classes can be used directly. For example:
|
||||
```python
|
||||
from datastore_orm import *
|
||||
|
||||
class Event(Model):
|
||||
|
||||
name = StringField(default="EVENT")
|
||||
repeated = UInt32Field(default=1)
|
||||
created = DateTimeField(default=F.now())
|
||||
|
||||
engine = Memory()
|
||||
```
|
||||
|
||||
Importing Everything into a Namespace
|
||||
-------------------------------------
|
||||
|
||||
To prevent potential name clashes and to make the code more readable, you can import the ORM's classes into a namespace of your choosing, e.g. `orm`. For brevity, it is recommended to import the `F` class explicitly:
|
||||
```python
|
||||
import datastore_orm as orm
|
||||
from datastore_orm import F
|
||||
|
||||
class Event(orm.Model):
|
||||
|
||||
name = orm.StringField(default="EVENT")
|
||||
repeated = orm.UInt32Field(default=1)
|
||||
created = orm.DateTimeField(default=F.now())
|
||||
|
||||
engine = orm.Memory()
|
||||
```
|
||||
|
||||
Importing Specific Submodules
|
||||
-----------------------------
|
||||
|
||||
It is possible to import only the submodules you need, and use their names to qualify the ORM's class names. This option is more verbose, but makes it clear where each class comes from. For example:
|
||||
```python
|
||||
from datastore_orm import models, fields, engines, F
|
||||
|
||||
class Event(models.Model):
|
||||
|
||||
name = fields.StringField(default="EVENT")
|
||||
repeated = fields.UInt32Field(default=1)
|
||||
created = fields.DateTimeField(default=F.now())
|
||||
|
||||
engine = engines.Memory()
|
||||
```
|
||||
|
||||
Importing Specific Classes
|
||||
--------------------------
|
||||
|
||||
If you prefer, you can import only the specific ORM classes that you need directly from `datastore_orm`:
|
||||
```python
|
||||
from datastore_orm import Model, StringField, UInt32Field, DateTimeField, F, Memory
|
||||
|
||||
class Event(Model):
|
||||
|
||||
name = StringField(default="EVENT")
|
||||
repeated = UInt32Field(default=1)
|
||||
created = DateTimeField(default=F.now())
|
||||
|
||||
engine = Memory()
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
[<< Expressions](expressions.md) | [Table of Contents](toc.md) | [Querysets >>](querysets.md)
|
||||
+4
-4
@@ -1,16 +1,16 @@
|
||||
Overview
|
||||
========
|
||||
|
||||
This project is simple ORM for working with the [ClickHouse database](https://clickhouse.yandex/). It allows you to define model classes whose instances can be written to the database and read from it.
|
||||
This project is simple ORM for working with the [ClickHouse database](https://clickhouse.tech/). It allows you to define model classes whose instances can be written to the database and read from it.
|
||||
|
||||
It was tested on Python 2.7 and 3.5.
|
||||
Version 1.x supports Python 2.7 and 3.5+. Version 2.x dropped support for Python 2.7, and works only with Python 3.5+.
|
||||
|
||||
Installation
|
||||
------------
|
||||
|
||||
To install infi.clickhouse_orm:
|
||||
To install datastore_orm:
|
||||
|
||||
pip install infi.clickhouse_orm
|
||||
pip install datastore_orm
|
||||
|
||||
---
|
||||
|
||||
|
||||
+106
-32
@@ -8,34 +8,105 @@ Database instances connect to a specific ClickHouse database for running queries
|
||||
Defining Models
|
||||
---------------
|
||||
|
||||
Models are defined in a way reminiscent of Django's ORM:
|
||||
Models are defined in a way reminiscent of Django's ORM, by subclassing `Model`:
|
||||
```python
|
||||
from datastore_orm import Model, StringField, DateField, Float32Field, MergeTree
|
||||
|
||||
from infi.clickhouse_orm import models, fields, engines
|
||||
class Person(Model):
|
||||
|
||||
class Person(models.Model):
|
||||
first_name = StringField()
|
||||
last_name = StringField()
|
||||
birthday = DateField()
|
||||
height = Float32Field()
|
||||
|
||||
first_name = fields.StringField()
|
||||
last_name = fields.StringField()
|
||||
birthday = fields.DateField()
|
||||
height = fields.Float32Field()
|
||||
engine = MergeTree('birthday', ('first_name', 'last_name', 'birthday'))
|
||||
```
|
||||
|
||||
engine = engines.MergeTree('birthday', ('first_name', 'last_name', 'birthday'))
|
||||
The columns in the database table are represented by model fields. Each field has a type, which matches the type of the corresponding database column. All the supported fields types are listed [here](field_types.md).
|
||||
|
||||
It is possible to provide a default value for a field, instead of its "natural" default (empty string for string fields, zero for numeric fields etc.). Alternatively it is possible to pass alias or materialized parameters (see below for usage examples). Only one of `default`, `alias` and `materialized` parameters can be provided.
|
||||
A model must have an `engine`, which determines how its table is stored on disk (if at all), and what capabilities it has. For more details about table engines see [here](table_engines.md).
|
||||
|
||||
For more details see [Field Types](field_types.md) and [Table Engines](table_engines.md).
|
||||
### Default values
|
||||
|
||||
Each field has a "natural" default value - empty string for string fields, zero for numeric fields etc. To specify a different value use the `default` parameter:
|
||||
|
||||
first_name = StringField(default="anonymous")
|
||||
|
||||
For additional details see [here](field_options.md).
|
||||
|
||||
### Null values
|
||||
|
||||
To allow null values in a field, wrap it inside a `NullableField`:
|
||||
|
||||
birthday = NullableField(DateField())
|
||||
|
||||
In this case, the default value for that field becomes `null` unless otherwise specified.
|
||||
|
||||
For more information about `NullableField` see [Field Types](field_types.md).
|
||||
|
||||
### Materialized fields
|
||||
|
||||
The value of a materialized field is calculated from other fields in the model. For example:
|
||||
|
||||
year_born = Int16Field(materialized=F.toYear(birthday))
|
||||
|
||||
Materialized fields are read-only, meaning that their values are not sent to the database when inserting records.
|
||||
|
||||
For additional details see [here](field_options.md).
|
||||
|
||||
### Alias fields
|
||||
|
||||
An alias field is a field whose value is calculated by ClickHouse on the fly, as a function of other fields. It is not physically stored by the database. For example:
|
||||
|
||||
weekday_born = field.UInt8Field(alias=F.toDayOfWeek(birthday))
|
||||
|
||||
Alias fields are read-only, meaning that their values are not sent to the database when inserting records.
|
||||
|
||||
For additional details see [here](field_options.md).
|
||||
|
||||
### Table Names
|
||||
|
||||
The table name used for the model is its class name, converted to lowercase. To override the default name, implement the `table_name` method:
|
||||
```python
|
||||
class Person(Model):
|
||||
|
||||
class Person(models.Model):
|
||||
...
|
||||
|
||||
...
|
||||
@classmethod
|
||||
def table_name(cls):
|
||||
return 'people'
|
||||
```
|
||||
|
||||
### Model Constraints
|
||||
|
||||
It is possible to define constraints which ClickHouse verifies when data is inserted. Trying to insert invalid records will raise a `ServerError`. Each constraint has a name and an expression to validate. For example:
|
||||
```python
|
||||
class Person(Model):
|
||||
|
||||
...
|
||||
|
||||
# Ensure that the birthday is not a future date
|
||||
birthday_is_in_the_past = Constraint(birthday <= F.today())
|
||||
```
|
||||
|
||||
### Data Skipping Indexes
|
||||
|
||||
Models that use an engine from the `MergeTree` family can define additional indexes over one or more columns or expressions. These indexes are used in SELECT queries for reducing the amount of data to read from the disk by skipping big blocks of data that do not satisfy the query's conditions.
|
||||
|
||||
For example:
|
||||
```python
|
||||
class Person(Model):
|
||||
|
||||
...
|
||||
|
||||
# A minmax index that can help find people taller or shorter than some height
|
||||
height_index = Index(height, type=Index.minmax(), granularity=2)
|
||||
|
||||
# A trigram index that can help find substrings inside people names
|
||||
names_index = Index((F.lower(first_name), F.lower(last_name)),
|
||||
type=Index.ngrambf_v1(3, 256, 2, 0), granularity=1)
|
||||
```
|
||||
|
||||
@classmethod
|
||||
def table_name(cls):
|
||||
return 'people'
|
||||
|
||||
Using Models
|
||||
------------
|
||||
@@ -55,14 +126,14 @@ When values are assigned to model fields, they are immediately converted to thei
|
||||
>>> suzy.birthday = 0.5
|
||||
ValueError: Invalid value for DateField - 0.5
|
||||
>>> suzy.birthday = '1922-05-31'
|
||||
ValueError: DateField out of range - 1922-05-31 is not between 1970-01-01 and 2038-01-19
|
||||
ValueError: DateField out of range - 1922-05-31 is not between 1970-01-01 and 2105-12-31
|
||||
|
||||
Inserting to the Database
|
||||
-------------------------
|
||||
|
||||
To write your instances to ClickHouse, you need a `Database` instance:
|
||||
|
||||
from infi.clickhouse_orm.database import Database
|
||||
from datastore_orm import Database
|
||||
|
||||
db = Database('my_test_db')
|
||||
|
||||
@@ -87,19 +158,19 @@ Reading from the Database
|
||||
Loading model instances from the database is simple:
|
||||
|
||||
for person in db.select("SELECT * FROM my_test_db.person", model_class=Person):
|
||||
print person.first_name, person.last_name
|
||||
print(person.first_name, person.last_name)
|
||||
|
||||
Do not include a `FORMAT` clause in the query, since the ORM automatically sets the format to `TabSeparatedWithNamesAndTypes`.
|
||||
|
||||
It is possible to select only a subset of the columns, and the rest will receive their default values:
|
||||
|
||||
for person in db.select("SELECT first_name FROM my_test_db.person WHERE last_name='Smith'", model_class=Person):
|
||||
print person.first_name
|
||||
print(person.first_name)
|
||||
|
||||
The ORM provides a way to build simple queries without writing SQL by hand. The previous snippet can be written like this:
|
||||
|
||||
for person in Person.objects_in(db).filter(last_name='Smith').only('first_name'):
|
||||
print person.first_name
|
||||
for person in Person.objects_in(db).filter(Person.last_name == 'Smith').only('first_name'):
|
||||
print(person.first_name)
|
||||
|
||||
See [Querysets](querysets.md) for more information.
|
||||
|
||||
@@ -110,14 +181,20 @@ Reading without a Model
|
||||
When running a query, specifying a model class is not required. In case you do not provide a model class, an ad-hoc class will be defined based on the column names and types returned by the query:
|
||||
|
||||
for row in db.select("SELECT max(height) as max_height FROM my_test_db.person"):
|
||||
print row.max_height
|
||||
print(row.max_height)
|
||||
|
||||
This is a very convenient feature that saves you the need to define a model for each query, while still letting you work with Pythonic column values and an elegant syntax.
|
||||
|
||||
It is also possible to generate a model class on the fly for an existing table in the database using `get_model_for_table`. This is particularly useful for querying system tables, for example:
|
||||
|
||||
QueryLog = db.get_model_for_table('query_log', system_table=True)
|
||||
for row in QueryLog.objects_in(db).filter(QueryLog.query_duration_ms > 10000):
|
||||
print(row.query)
|
||||
|
||||
SQL Placeholders
|
||||
----------------
|
||||
|
||||
There are a couple of special placeholders that you can use inside the SQL to make it easier to write: `$db` and `$table`. The first one is replaced by the database name, and the second is replaced by the database name plus table name (but is available only when the model is specified).
|
||||
There are a couple of special placeholders that you can use inside the SQL to make it easier to write: `$db` and `$table`. The first one is replaced by the database name, and the second is replaced by the table name (but is available only when the model is specified).
|
||||
|
||||
So instead of this:
|
||||
|
||||
@@ -125,11 +202,9 @@ So instead of this:
|
||||
|
||||
you can use:
|
||||
|
||||
db.select("SELECT * FROM $db.person", model_class=Person)
|
||||
db.select("SELECT * FROM $db.$table", model_class=Person)
|
||||
|
||||
or even:
|
||||
|
||||
db.select("SELECT * FROM $table", model_class=Person)
|
||||
Note: normally it is not necessary to specify the database name, since it's already sent in the query parameters to ClickHouse. It is enough to specify the table name.
|
||||
|
||||
Counting
|
||||
--------
|
||||
@@ -148,9 +223,9 @@ It is possible to paginate through model instances:
|
||||
|
||||
>>> order_by = 'first_name, last_name'
|
||||
>>> page = db.paginate(Person, order_by, page_num=1, page_size=10)
|
||||
>>> print page.number_of_objects
|
||||
>>> print(page.number_of_objects)
|
||||
2507
|
||||
>>> print page.pages_total
|
||||
>>> print(page.pages_total)
|
||||
251
|
||||
>>> for person in page.objects:
|
||||
>>> # do something
|
||||
@@ -160,8 +235,7 @@ The `paginate` method returns a `namedtuple` containing the following fields:
|
||||
- `objects` - the list of objects in this page
|
||||
- `number_of_objects` - total number of objects in all pages
|
||||
- `pages_total` - total number of pages
|
||||
- `number` - the page number, starting from 1; the special value -1
|
||||
may be used to retrieve the last page
|
||||
- `number` - the page number, starting from 1; the special value -1 may be used to retrieve the last page
|
||||
- `page_size` - the number of objects per page
|
||||
|
||||
You can optionally pass conditions to the query:
|
||||
@@ -173,4 +247,4 @@ Note that `order_by` must be chosen so that the ordering is unique, otherwise th
|
||||
|
||||
---
|
||||
|
||||
[<< Overview](index.md) | [Table of Contents](toc.md) | [Querysets >>](querysets.md)
|
||||
[<< Overview](index.md) | [Table of Contents](toc.md) | [Expressions >>](expressions.md)
|
||||
+207
-33
@@ -1,14 +1,15 @@
|
||||
|
||||
Querysets
|
||||
=========
|
||||
|
||||
A queryset is an object that represents a database query using a specific Model. It is lazy, meaning that it does not hit the database until you iterate over its matching rows (model instances). To create a base queryset for a model class, use:
|
||||
|
||||
qs = Person.objects_in(database)
|
||||
|
||||
|
||||
This queryset matches all Person instances in the database. You can get these instances using iteration:
|
||||
|
||||
for person in qs:
|
||||
print person.first_name, person.last_name
|
||||
print(person.first_name, person.last_name)
|
||||
|
||||
Filtering
|
||||
---------
|
||||
@@ -16,26 +17,88 @@ Filtering
|
||||
The `filter` and `exclude` methods are used for filtering the matching instances. Calling these methods returns a new queryset instance, with the added conditions. For example:
|
||||
|
||||
>>> qs = Person.objects_in(database)
|
||||
>>> qs = qs.filter(first_name__startswith='V').exclude(birthday__lt='2000-01-01')
|
||||
>>> qs = qs.filter(F.like(Person.first_name, 'V%')).exclude(Person.birthday < '2000-01-01')
|
||||
>>> qs.conditions_as_sql()
|
||||
u"first_name LIKE 'V%' AND NOT (birthday < '2000-01-01')"
|
||||
|
||||
It is possible to specify several fields to filter or exclude by:
|
||||
"first_name LIKE 'V%' AND NOT (birthday < '2000-01-01')"
|
||||
|
||||
>>> qs = Person.objects_in(database).filter(last_name='Smith', height__gt=1.75)
|
||||
It is possible to specify several expressions to filter or exclude by, and they will be ANDed together:
|
||||
|
||||
>>> qs = Person.objects_in(database).filter(Person.last_name == 'Smith', Person.height > 1.75)
|
||||
>>> qs.conditions_as_sql()
|
||||
u"last_name = 'Smith' AND height > 1.75"
|
||||
"last_name = 'Smith' AND height > 1.75"
|
||||
|
||||
There are different operators that can be used, by passing `<fieldname>__<operator>=<value>` (two underscores separate the field name from the operator). In case no operator is given, `eq` is used by default. Below are all the supported operators.
|
||||
For compound conditions you can use the overloaded operators `&` (AND), `|` (OR) and `~` (NOT):
|
||||
|
||||
>>> qs = Person.objects_in(database)
|
||||
>>> qs = qs.filter(((Person.first_name == 'Ciaran') & (Person.last_name == 'Carver')) | (Person.height <= 1.8) & ~(Person.first_name = 'David'))
|
||||
>>> qs.conditions_as_sql()
|
||||
"((first_name = 'Ciaran' AND last_name = 'Carver') OR height <= 1.8) AND (NOT (first_name = 'David'))"
|
||||
|
||||
Note that Python's bitwise operators (`&`, `|`, `~`, `^`) have higher precedence than comparison operators, so always use parentheses when combining these two types of operators in an expression. Otherwise the resulting SQL might be different than what you would expect.
|
||||
|
||||
### Using `IN` and `NOT IN`
|
||||
|
||||
Filtering queries using ClickHouse's `IN` and `NOT IN` operators requires using the `isIn` and `isNotIn` functions (trying to use Python's `in` keyword will not work!).
|
||||
For example:
|
||||
```python
|
||||
# Is it Monday, Tuesday or Wednesday?
|
||||
F.isIn(F.toDayOfWeek(F.now()), [1, 2, 3])
|
||||
# This will not work:
|
||||
F.toDayOfWeek(F.now()) in [1, 2, 3]
|
||||
```
|
||||
|
||||
In case of model fields, there is a simplified syntax:
|
||||
```python
|
||||
# Filtering using F.isIn:
|
||||
qs.filter(F.isIn(Person.first_name, ['Robert', 'Rob', 'Robbie']))
|
||||
# Simpler syntax using isIn directly on the field:
|
||||
qs.filter(Person.first_name.isIn(['Robert', 'Rob', 'Robbie']))
|
||||
```
|
||||
|
||||
The `isIn` and `isNotIn` functions expect either a list/tuple of values, or another queryset (a subquery). For example if we want to select only people with Irish last names:
|
||||
```python
|
||||
# Last name is in a list of values
|
||||
qs = Person.objects_in(database).filter(Person.last_name.isIn(["Murphy", "O'Sullivan"]))
|
||||
# Last name is in a subquery
|
||||
subquery = IrishLastName.objects_in(database).only("name")
|
||||
qs = Person.objects_in(database).filter(Person.last_name.isIn(subquery))
|
||||
```
|
||||
|
||||
### Specifying PREWHERE conditions
|
||||
|
||||
By default conditions from `filter` and `exclude` methods are add to `WHERE` clause.
|
||||
For better aggregation performance you can add them to `PREWHERE` section by adding a `prewhere=True` parameter:
|
||||
|
||||
>>> qs = Person.objects_in(database)
|
||||
>>> qs = qs.filter(F.like(Person.first_name, 'V%'), prewhere=True)
|
||||
>>> qs.conditions_as_sql(prewhere=True)
|
||||
"first_name LIKE 'V%'"
|
||||
|
||||
### Old-style filter conditions
|
||||
|
||||
Prior to version 2 of the ORM, filtering conditions were limited to a predefined set of operators, and complex expressions were not supported. This old syntax is still available, so you can use it alongside or even intermixed with new-style functions and expressions.
|
||||
|
||||
The old syntax uses keyword arguments to the `filter` and `exclude` methods, that are built as `<fieldname>__<operator>=<value>` (two underscores separate the field name from the operator). In case no operator is given, `eq` is used by default. For example:
|
||||
```python
|
||||
qs = Position.objects.in(database)
|
||||
# New style
|
||||
qs = qs.filter(Position.x > 100, Position.y < 20, Position.terrain == 'water')
|
||||
# Old style
|
||||
qs = qs.filter(x__gt=100, y__lt=20, terrain='water')
|
||||
```
|
||||
Below are all the supported operators.
|
||||
|
||||
| Operator | Equivalent SQL | Comments |
|
||||
| -------- | -------------------------------------------- | ---------------------------------- |
|
||||
| `eq` | `field = value` | |
|
||||
| `ne` | `field != value` | |
|
||||
| `gt` | `field > value` | |
|
||||
| `gte` | `field >= value` | |
|
||||
| `lt` | `field < value` | |
|
||||
| `lte` | `field <= value` | |
|
||||
| `in` | `field IN (values)` | See below |
|
||||
| `between` | `field BETWEEN value1 AND value2` | |
|
||||
| `in` | `field IN (values)` | |
|
||||
| `not_in` | `field NOT IN (values)` | |
|
||||
| `contains` | `field LIKE '%value%'` | For string fields only |
|
||||
| `startswith` | `field LIKE 'value%'` | For string fields only |
|
||||
| `endswith` | `field LIKE '%value'` | For string fields only |
|
||||
@@ -44,33 +107,13 @@ There are different operators that can be used, by passing `<fieldname>__<operat
|
||||
| `iendswith` | `lowerUTF8(field) LIKE lowerUTF8('%value')` | For string fields only |
|
||||
| `iexact` | `lowerUTF8(field) = lowerUTF8(value)` | For string fields only |
|
||||
|
||||
### Using the `in` Operator
|
||||
|
||||
The `in` operator expects one of three types of values:
|
||||
* A list or tuple of simple values
|
||||
* A string, which is used verbatim as the contents of the parentheses
|
||||
* Another queryset (subquery)
|
||||
|
||||
For example if we want to select only people with Irish last names:
|
||||
|
||||
# A list of simple values
|
||||
qs = Person.objects_in(database).filter(last_name__in=["Murphy", "O'Sullivan"])
|
||||
|
||||
# A string
|
||||
subquery = "SELECT name from $db.irishlastname"
|
||||
qs = Person.objects_in(database).filter(last_name__in=subquery)
|
||||
|
||||
# A queryset
|
||||
subquery = IrishLastName.objects_in(database).only("name")
|
||||
qs = Person.objects_in(database).filter(last_name__in=subquery)
|
||||
|
||||
Counting and Checking Existence
|
||||
-------------------------------
|
||||
|
||||
Use the `count` method to get the number of matches:
|
||||
|
||||
Person.objects_in(database).count()
|
||||
|
||||
|
||||
To check if there are any matches at all, you can use any of the following equivalent options:
|
||||
|
||||
if qs.count(): ...
|
||||
@@ -83,11 +126,13 @@ Ordering
|
||||
The sorting order of the results can be controlled using the `order_by` method:
|
||||
|
||||
qs = Person.objects_in(database).order_by('last_name', 'first_name')
|
||||
|
||||
|
||||
The default order is ascending. To use descending order, add a minus sign before the field name:
|
||||
|
||||
qs = Person.objects_in(database).order_by('-height')
|
||||
|
||||
If you do not use `order_by`, the rows are returned in arbitrary order.
|
||||
|
||||
Omitting Fields
|
||||
---------------
|
||||
|
||||
@@ -95,7 +140,136 @@ When some of the model fields aren't needed, it is more efficient to omit them f
|
||||
|
||||
qs = Person.objects_in(database).only('first_name', 'birthday')
|
||||
|
||||
Distinct
|
||||
--------
|
||||
|
||||
Adds a DISTINCT clause to the query, meaning that any duplicate rows in the results will be omitted.
|
||||
|
||||
>>> Person.objects_in(database).only('first_name').count()
|
||||
100
|
||||
>>> Person.objects_in(database).only('first_name').distinct().count()
|
||||
94
|
||||
|
||||
Final
|
||||
-----
|
||||
|
||||
This method can be used only with `CollapsingMergeTree` engine.
|
||||
Adds a FINAL modifier to the query, meaning that the selected data is fully "collapsed" by the engine's sign field.
|
||||
|
||||
>>> Person.objects_in(database).count()
|
||||
100
|
||||
>>> Person.objects_in(database).final().count()
|
||||
94
|
||||
|
||||
Slicing
|
||||
-------
|
||||
|
||||
It is possible to get a specific item from the queryset by index:
|
||||
|
||||
qs = Person.objects_in(database).order_by('last_name', 'first_name')
|
||||
first = qs[0]
|
||||
|
||||
It is also possible to get a range a instances using a slice. This returns a queryset, that you can either iterate over or convert to a list.
|
||||
|
||||
qs = Person.objects_in(database).order_by('last_name', 'first_name')
|
||||
first_ten_people = list(qs[:10])
|
||||
next_ten_people = list(qs[10:20])
|
||||
|
||||
You should use `order_by` to ensure a consistent ordering of the results.
|
||||
|
||||
Trying to use negative indexes or a slice with a step (e.g. [0 : 100 : 2]) is not supported and will raise an `AssertionError`.
|
||||
|
||||
Pagination
|
||||
----------
|
||||
|
||||
Similar to `Database.paginate`, you can go over the queryset results one page at a time:
|
||||
|
||||
>>> qs = Person.objects_in(database).order_by('last_name', 'first_name')
|
||||
>>> page = qs.paginate(page_num=1, page_size=10)
|
||||
>>> print(page.number_of_objects)
|
||||
2507
|
||||
>>> print(page.pages_total)
|
||||
251
|
||||
>>> for person in page.objects:
|
||||
>>> # do something
|
||||
|
||||
The `paginate` method returns a `namedtuple` containing the following fields:
|
||||
|
||||
- `objects` - the list of objects in this page
|
||||
- `number_of_objects` - total number of objects in all pages
|
||||
- `pages_total` - total number of pages
|
||||
- `number` - the page number, starting from 1; the special value -1 may be used to retrieve the last page
|
||||
- `page_size` - the number of objects per page
|
||||
|
||||
Note that you should use `QuerySet.order_by` so that the ordering is unique, otherwise there might be inconsistencies in the pagination (such as an instance that appears on two different pages).
|
||||
|
||||
Mutations
|
||||
---------
|
||||
|
||||
To delete all records that match a queryset's conditions use the `delete` method:
|
||||
|
||||
Person.objects_in(database).filter(first_name='Max').delete()
|
||||
|
||||
To update records that match a queryset's conditions call the `update` method and provide the field names to update and the expressions to use (as keyword arguments):
|
||||
|
||||
Person.objects_in(database).filter(first_name='Max').update(first_name='Maximilian')
|
||||
|
||||
Note a few caveats:
|
||||
|
||||
- ClickHouse cannot update columns that are used in the calculation of the primary or the partition key.
|
||||
- Mutations happen in the background, so they are not immediate.
|
||||
- Only tables in the `MergeTree` family support mutations.
|
||||
|
||||
Aggregation
|
||||
-----------
|
||||
|
||||
It is possible to use aggregation functions over querysets using the `aggregate` method. The simplest form of aggregation works over all rows in the queryset:
|
||||
|
||||
>>> qs = Person.objects_in(database).aggregate(average_height=F.avg(Person.height))
|
||||
>>> print(qs.count())
|
||||
1
|
||||
>>> for row in qs: print(row.average_height)
|
||||
1.71
|
||||
|
||||
The returned row or rows are no longer instances of the base model (`Person` in this example), but rather instances of an ad-hoc model that includes only the fields specified in the call to `aggregate`.
|
||||
|
||||
You can pass fields from the model that will be included in the query. By default, they will be also used in the GROUP BY clause. For example to count the number of people per last name you could do this:
|
||||
|
||||
qs = Person.objects_in(database).aggregate(Person.last_name, num=F.count())
|
||||
|
||||
The underlying SQL query would be something like this:
|
||||
|
||||
SELECT last_name, count() AS num
|
||||
FROM person
|
||||
GROUP BY last_name
|
||||
|
||||
If you would like to control the GROUP BY explicitly, use the `group_by` method. This is useful when you need to group by a calculated field, instead of a field that exists in the model. For example, to count the number of people born on each weekday:
|
||||
|
||||
qs = Person.objects_in(database).aggregate(weekday=F.toDayOfWeek(Person.birthday), num=F.count()).group_by('weekday')
|
||||
|
||||
This queryset is translated to:
|
||||
|
||||
SELECT toDayOfWeek(birthday) AS weekday, count() AS num
|
||||
FROM person
|
||||
GROUP BY weekday
|
||||
|
||||
After calling `aggregate` you can still use most of the regular queryset methods, such as `count`, `order_by` and `paginate`. It is not possible, however, to call `only` or `aggregate`. It is also not possible to filter the aggregated queryset on calculated fields, only on fields that exist in the model.
|
||||
|
||||
### Adding totals
|
||||
|
||||
If you limit aggregation results, it might be useful to get total aggregation values for all rows.
|
||||
To achieve this, you can use `with_totals` method. It will return extra row (last) with
|
||||
values aggregated for all rows suitable for filters.
|
||||
|
||||
qs = Person.objects_in(database).aggregate(Person.first_name, num=F.count()).with_totals().order_by('-count')[:3]
|
||||
>>> print(qs.count())
|
||||
4
|
||||
>>> for row in qs:
|
||||
>>> print("'{}': {}".format(row.first_name, row.count))
|
||||
'Cassandra': 2
|
||||
'Alexandra': 2
|
||||
'': 100
|
||||
|
||||
---
|
||||
|
||||
[<< Models and Databases](models_and_databases.md) | [Table of Contents](toc.md) | [Field Types >>](field_types.md)
|
||||
[<< Importing ORM Classes](importing_orm_classes.md) | [Table of Contents](toc.md) | [Field Options >>](field_options.md)
|
||||
|
||||
+12
-12
@@ -1,7 +1,7 @@
|
||||
Class Reference
|
||||
===============
|
||||
|
||||
infi.clickhouse_orm.database
|
||||
datastore_orm.database
|
||||
----------------------------
|
||||
|
||||
### Database
|
||||
@@ -104,7 +104,7 @@ Extends Exception
|
||||
|
||||
Raised when a database operation fails.
|
||||
|
||||
infi.clickhouse_orm.models
|
||||
datastore_orm.models
|
||||
--------------------------
|
||||
|
||||
### Model
|
||||
@@ -119,12 +119,12 @@ invalid values will cause a `ValueError` to be raised.
|
||||
Unrecognized field names will cause an `AttributeError`.
|
||||
|
||||
|
||||
#### Model.create_table_sql(db_name)
|
||||
#### Model.create_table_sql(db)
|
||||
|
||||
Returns the SQL command for creating a table for this model.
|
||||
|
||||
|
||||
#### Model.drop_table_sql(db_name)
|
||||
#### Model.drop_table_sql(db)
|
||||
|
||||
Returns the SQL command for deleting this model's table.
|
||||
|
||||
@@ -197,12 +197,12 @@ invalid values will cause a `ValueError` to be raised.
|
||||
Unrecognized field names will cause an `AttributeError`.
|
||||
|
||||
|
||||
#### BufferModel.create_table_sql(db_name)
|
||||
#### BufferModel.create_table_sql(db)
|
||||
|
||||
Returns the SQL command for creating a table for this model.
|
||||
|
||||
|
||||
#### BufferModel.drop_table_sql(db_name)
|
||||
#### BufferModel.drop_table_sql(db)
|
||||
|
||||
Returns the SQL command for deleting this model's table.
|
||||
|
||||
@@ -263,7 +263,7 @@ Returns the instance's column values as a tab-separated line. A newline is not i
|
||||
- `include_readonly`: if false, returns only fields that can be inserted into database.
|
||||
|
||||
|
||||
infi.clickhouse_orm.fields
|
||||
datastore_orm.fields
|
||||
--------------------------
|
||||
|
||||
### Field
|
||||
@@ -419,7 +419,7 @@ Extends BaseEnumField
|
||||
#### Enum16Field(enum_cls, default=None, alias=None, materialized=None)
|
||||
|
||||
|
||||
infi.clickhouse_orm.engines
|
||||
datastore_orm.engines
|
||||
---------------------------
|
||||
|
||||
### Engine
|
||||
@@ -448,7 +448,7 @@ Extends Engine
|
||||
Extends Engine
|
||||
|
||||
Here we define Buffer engine
|
||||
Read more here https://clickhouse.yandex/reference_en.html#Buffer
|
||||
Read more here https://clickhouse.tech/reference_en.html#Buffer
|
||||
|
||||
#### Buffer(main_model, num_layers=16, min_time=10, max_time=100, min_rows=10000, max_rows=1000000, min_bytes=10000000, max_bytes=100000000)
|
||||
|
||||
@@ -474,7 +474,7 @@ Extends MergeTree
|
||||
#### ReplacingMergeTree(date_col, key_cols, ver_col=None, sampling_expr=None, index_granularity=8192, replica_table_path=None, replica_name=None)
|
||||
|
||||
|
||||
infi.clickhouse_orm.query
|
||||
datastore_orm.query
|
||||
-------------------------
|
||||
|
||||
### QuerySet
|
||||
@@ -482,9 +482,9 @@ infi.clickhouse_orm.query
|
||||
#### QuerySet(model_cls, database)
|
||||
|
||||
|
||||
#### conditions_as_sql()
|
||||
#### conditions_as_sql(prewhere=True)
|
||||
|
||||
Return the contents of the queryset's WHERE clause.
|
||||
Return the contents of the queryset's WHERE or `PREWHERE` clause.
|
||||
|
||||
|
||||
#### count()
|
||||
|
||||
@@ -22,7 +22,7 @@ To write migrations, create a Python package. Then create a python file for the
|
||||
|
||||
Each migration file is expected to contain a list of `operations`, for example:
|
||||
|
||||
from infi.clickhouse_orm import migrations
|
||||
from datastore_orm import migrations
|
||||
from analytics import models
|
||||
|
||||
operations = [
|
||||
@@ -32,15 +32,20 @@ Each migration file is expected to contain a list of `operations`, for example:
|
||||
|
||||
The following operations are supported:
|
||||
|
||||
**CreateTable**
|
||||
|
||||
A migration operation that creates a table for a given model class.
|
||||
### CreateTable
|
||||
|
||||
**DropTable**
|
||||
A migration operation that creates a table for a given model class. If the table already exists, the operation does nothing.
|
||||
|
||||
A migration operation that drops the table of a given model class.
|
||||
In case the model class is a `BufferModel`, the operation first creates the underlying on-disk table, and then creates the buffer table.
|
||||
|
||||
**AlterTable**
|
||||
|
||||
### DropTable
|
||||
|
||||
A migration operation that drops the table of a given model class. If the table does not exist, the operation does nothing.
|
||||
|
||||
|
||||
### AlterTable
|
||||
|
||||
A migration operation that compares the table of a given model class to the model’s fields, and alters the table to match the model. The operation can:
|
||||
|
||||
@@ -50,6 +55,48 @@ A migration operation that compares the table of a given model class to the mode
|
||||
|
||||
Default values are not altered by this operation.
|
||||
|
||||
|
||||
### AlterTableWithBuffer
|
||||
|
||||
A compound migration operation for altering a buffer table and its underlying on-disk table. The buffer table is dropped, the on-disk table is altered, and then the buffer table is re-created. This is the procedure recommended in the ClickHouse documentation for handling scenarios in which the underlying table needs to be modified.
|
||||
|
||||
Applying this migration operation to a regular table has the same effect as an `AlterTable` operation.
|
||||
|
||||
|
||||
### AlterConstraints
|
||||
|
||||
A migration operation that adds new constraints from the model to the database table, and drops obsolete ones. Constraints are identified by their names, so a change in an existing constraint will not be detected unless its name was changed too. ClickHouse does not check that the constraints hold for existing data in the table.
|
||||
|
||||
|
||||
### RunPython
|
||||
|
||||
A migration operation that runs a Python function. The function receives the `Database` instance to operate on.
|
||||
|
||||
def forward(database):
|
||||
database.insert([
|
||||
TestModel(field=1)
|
||||
])
|
||||
|
||||
operations = [
|
||||
migrations.RunPython(forward),
|
||||
]
|
||||
|
||||
|
||||
### RunSQL
|
||||
|
||||
A migration operation that runs raw SQL statements. It expects a string containing an SQL statements, or a list of statements.
|
||||
|
||||
Example:
|
||||
|
||||
operations = [
|
||||
RunSQL('INSERT INTO `test_table` (field) VALUES (1)'),
|
||||
RunSQL([
|
||||
'INSERT INTO `test_table` (field) VALUES (2)',
|
||||
'INSERT INTO `test_table` (field) VALUES (3)'
|
||||
])
|
||||
]
|
||||
|
||||
|
||||
Running Migrations
|
||||
------------------
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
System Models
|
||||
=============
|
||||
|
||||
[Clickhouse docs](https://clickhouse.yandex/reference_en.html#System%20tables).
|
||||
[Clickhouse docs](https://clickhouse.tech/docs/en/operations/system-tables/).
|
||||
|
||||
System models are read only models for implementing part of the system's functionality, and for providing access to information about how the system is working.
|
||||
|
||||
@@ -14,7 +14,7 @@ Currently the following system models are supported:
|
||||
Partitions and Parts
|
||||
--------------------
|
||||
|
||||
[ClickHouse docs](https://clickhouse.yandex/reference_en.html#Manipulations%20with%20partitions%20and%20parts).
|
||||
[ClickHouse docs](https://clickhouse.tech/docs/en/sql-reference/statements/alter/#alter_manipulations-with-partitions).
|
||||
|
||||
A partition in a table is data for a single calendar month. Table "system.parts" contains information about each part.
|
||||
|
||||
@@ -30,8 +30,7 @@ A partition in a table is data for a single calendar month. Table "system.parts"
|
||||
|
||||
Usage example:
|
||||
|
||||
from infi.clickhouse_orm.database import Database
|
||||
from infi.clickhouse_orm.system_models import SystemPart
|
||||
from datastore_orm import Database, SystemPart
|
||||
db = Database('my_test_db', db_url='http://192.168.1.1:8050', username='scott', password='tiger')
|
||||
partitions = SystemPart.get_active(db, conditions='') # Getting all active partitions of the database
|
||||
if len(partitions) > 0:
|
||||
|
||||
+66
-20
@@ -1,7 +1,7 @@
|
||||
Table Engines
|
||||
=============
|
||||
|
||||
See: [ClickHouse Documentation](https://clickhouse.yandex/reference_en.html#Table+engines)
|
||||
See: [ClickHouse Documentation](https://clickhouse.tech/docs/en/engines/table-engines/)
|
||||
|
||||
Each model must have an engine instance, used when creating the table in ClickHouse.
|
||||
|
||||
@@ -15,6 +15,8 @@ The following engines are supported by the ORM:
|
||||
- SummingMergeTree / ReplicatedSummingMergeTree
|
||||
- ReplacingMergeTree / ReplicatedReplacingMergeTree
|
||||
- Buffer
|
||||
- Merge
|
||||
- Distributed
|
||||
|
||||
|
||||
Simple Engines
|
||||
@@ -22,11 +24,11 @@ Simple Engines
|
||||
|
||||
`TinyLog`, `Log` and `Memory` engines do not require any parameters:
|
||||
|
||||
engine = engines.TinyLog()
|
||||
engine = TinyLog()
|
||||
|
||||
engine = engines.Log()
|
||||
|
||||
engine = engines.Memory()
|
||||
engine = Log()
|
||||
|
||||
engine = Memory()
|
||||
|
||||
|
||||
Engines in the MergeTree Family
|
||||
@@ -34,50 +36,81 @@ Engines in the MergeTree Family
|
||||
|
||||
To define a `MergeTree` engine, supply the date column name and the names (or expressions) for the key columns:
|
||||
|
||||
engine = engines.MergeTree('EventDate', ('CounterID', 'EventDate'))
|
||||
engine = MergeTree('EventDate', ('CounterID', 'EventDate'))
|
||||
|
||||
You may also provide a sampling expression:
|
||||
|
||||
engine = engines.MergeTree('EventDate', ('CounterID', 'EventDate'), sampling_expr='intHash32(UserID)')
|
||||
engine = MergeTree('EventDate', ('CounterID', 'EventDate'), sampling_expr=F.intHash32(UserID))
|
||||
|
||||
A `CollapsingMergeTree` engine is defined in a similar manner, but requires also a sign column:
|
||||
|
||||
engine = engines.CollapsingMergeTree('EventDate', ('CounterID', 'EventDate'), 'Sign')
|
||||
engine = CollapsingMergeTree('EventDate', ('CounterID', 'EventDate'), 'Sign')
|
||||
|
||||
For a `SummingMergeTree` you can optionally specify the summing columns:
|
||||
|
||||
engine = engines.SummingMergeTree('EventDate', ('OrderID', 'EventDate', 'BannerID'),
|
||||
summing_cols=('Shows', 'Clicks', 'Cost'))
|
||||
engine = SummingMergeTree('EventDate', ('OrderID', 'EventDate', 'BannerID'),
|
||||
summing_cols=('Shows', 'Clicks', 'Cost'))
|
||||
|
||||
For a `ReplacingMergeTree` you can optionally specify the version column:
|
||||
|
||||
engine = engines.ReplacingMergeTree('EventDate', ('OrderID', 'EventDate', 'BannerID'), ver_col='Version')
|
||||
engine = ReplacingMergeTree('EventDate', ('OrderID', 'EventDate', 'BannerID'), ver_col='Version')
|
||||
|
||||
### Custom partitioning
|
||||
|
||||
ClickHouse supports [custom partitioning](https://clickhouse.tech/docs/en/engines/table-engines/mergetree-family/custom-partitioning-key/) expressions since version 1.1.54310
|
||||
|
||||
You can use custom partitioning with any `MergeTree` family engine.
|
||||
To set custom partitioning:
|
||||
|
||||
* Instead of specifying the `date_col` (first) constructor parameter, pass a tuple of field names or expressions in the `order_by` (second) constructor parameter.
|
||||
* Add `partition_key` parameter. It should be a tuple of expressions, by which partitions are built.
|
||||
|
||||
Standard monthly partitioning by date column can be specified using the `toYYYYMM(date)` function.
|
||||
|
||||
Example:
|
||||
|
||||
engine = ReplacingMergeTree(order_by=('OrderID', 'EventDate', 'BannerID'), ver_col='Version',
|
||||
partition_key=(F.toYYYYMM(EventDate), 'BannerID'))
|
||||
|
||||
|
||||
### Primary key
|
||||
ClickHouse supports [custom primary key](https://clickhouse.tech/docs/en/engines/table-engines/mergetree-family/mergetree/#primary-keys-and-indexes-in-queries) expressions since version 1.1.54310
|
||||
|
||||
You can use custom primary key with any `MergeTree` family engine.
|
||||
To set custom partitioning add `primary_key` parameter. It should be a tuple of expressions, by which partitions are built.
|
||||
|
||||
By default primary key is equal to order_by expression
|
||||
|
||||
Example:
|
||||
|
||||
engine = ReplacingMergeTree(order_by=('OrderID', 'EventDate', 'BannerID'), ver_col='Version',
|
||||
partition_key=(F.toYYYYMM(EventDate), 'BannerID'), primary_key=('OrderID',))
|
||||
|
||||
### Data Replication
|
||||
|
||||
Any of the above engines can be converted to a replicated engine (e.g. `ReplicatedMergeTree`) by adding two parameters, `replica_table_path` and `replica_name`:
|
||||
|
||||
engine = engines.MergeTree('EventDate', ('CounterID', 'EventDate'),
|
||||
replica_table_path='/clickhouse/tables/{layer}-{shard}/hits',
|
||||
replica_name='{replica}')
|
||||
engine = MergeTree('EventDate', ('CounterID', 'EventDate'),
|
||||
replica_table_path='/clickhouse/tables/{layer}-{shard}/hits',
|
||||
replica_name='{replica}')
|
||||
|
||||
|
||||
Buffer Engine
|
||||
-------------
|
||||
|
||||
A `Buffer` engine is only used in conjunction with a `BufferModel`.
|
||||
The model should be a subclass of both `models.BufferModel` and the main model.
|
||||
The model should be a subclass of both `BufferModel` and the main model.
|
||||
The main model is also passed to the engine:
|
||||
|
||||
class PersonBuffer(models.BufferModel, Person):
|
||||
class PersonBuffer(BufferModel, Person):
|
||||
|
||||
engine = engines.Buffer(Person)
|
||||
engine = Buffer(Person)
|
||||
|
||||
Additional buffer parameters can optionally be specified:
|
||||
|
||||
engine = engines.Buffer(Person, num_layers=16, min_time=10,
|
||||
max_time=100, min_rows=10000, max_rows=1000000,
|
||||
min_bytes=10000000, max_bytes=100000000)
|
||||
engine = Buffer(Person, num_layers=16, min_time=10,
|
||||
max_time=100, min_rows=10000, max_rows=1000000,
|
||||
min_bytes=10000000, max_bytes=100000000)
|
||||
|
||||
Then you can insert objects into Buffer model and they will be handled by ClickHouse properly:
|
||||
|
||||
@@ -87,6 +120,19 @@ Then you can insert objects into Buffer model and they will be handled by ClickH
|
||||
db.insert([dan, suzy])
|
||||
|
||||
|
||||
Merge Engine
|
||||
-------------
|
||||
|
||||
[ClickHouse docs](https://clickhouse.tech/docs/en/operations/table_engines/merge/)
|
||||
|
||||
A `Merge` engine is only used in conjunction with a `MergeModel`.
|
||||
This table does not store data itself, but allows reading from any number of other tables simultaneously. So you can't insert in it.
|
||||
Engine parameter specifies re2 (similar to PCRE) regular expression, from which data is selected.
|
||||
|
||||
class MergeTable(MergeModel):
|
||||
engine = Merge('^table_prefix')
|
||||
|
||||
|
||||
---
|
||||
|
||||
[<< Field Types](field_types.md) | [Table of Contents](toc.md) | [Schema Migrations >>](schema_migrations.md)
|
||||
+76
-23
@@ -5,7 +5,13 @@
|
||||
|
||||
* [Models and Databases](models_and_databases.md#models-and-databases)
|
||||
* [Defining Models](models_and_databases.md#defining-models)
|
||||
* [Default values](models_and_databases.md#default-values)
|
||||
* [Null values](models_and_databases.md#null-values)
|
||||
* [Materialized fields](models_and_databases.md#materialized-fields)
|
||||
* [Alias fields](models_and_databases.md#alias-fields)
|
||||
* [Table Names](models_and_databases.md#table-names)
|
||||
* [Model Constraints](models_and_databases.md#model-constraints)
|
||||
* [Data Skipping Indexes](models_and_databases.md#data-skipping-indexes)
|
||||
* [Using Models](models_and_databases.md#using-models)
|
||||
* [Inserting to the Database](models_and_databases.md#inserting-to-the-database)
|
||||
* [Reading from the Database](models_and_databases.md#reading-from-the-database)
|
||||
@@ -16,25 +22,52 @@
|
||||
|
||||
* [Querysets](querysets.md#querysets)
|
||||
* [Filtering](querysets.md#filtering)
|
||||
* [Using the in Operator](querysets.md#using-the-in-operator)
|
||||
* [Using IN and NOT IN](querysets.md#using-in-and-not-in)
|
||||
* [Specifying PREWHERE conditions](querysets.md#specifying-prewhere-conditions)
|
||||
* [Old-style filter conditions](querysets.md#old-style-filter-conditions)
|
||||
* [Counting and Checking Existence](querysets.md#counting-and-checking-existence)
|
||||
* [Ordering](querysets.md#ordering)
|
||||
* [Omitting Fields](querysets.md#omitting-fields)
|
||||
* [Distinct](querysets.md#distinct)
|
||||
* [Final](querysets.md#final)
|
||||
* [Slicing](querysets.md#slicing)
|
||||
* [Pagination](querysets.md#pagination)
|
||||
* [Mutations](querysets.md#mutations)
|
||||
* [Aggregation](querysets.md#aggregation)
|
||||
* [Adding totals](querysets.md#adding-totals)
|
||||
|
||||
* [Field Options](field_options.md#field-options)
|
||||
* [default](field_options.md#default)
|
||||
* [alias / materialized](field_options.md#alias-/-materialized)
|
||||
* [codec](field_options.md#codec)
|
||||
* [readonly](field_options.md#readonly)
|
||||
|
||||
* [Field Types](field_types.md#field-types)
|
||||
* [DateTimeField and Time Zones](field_types.md#datetimefield-and-time-zones)
|
||||
* [Working with enum fields](field_types.md#working-with-enum-fields)
|
||||
* [Working with array fields](field_types.md#working-with-array-fields)
|
||||
* [Working with materialized and alias fields](field_types.md#working-with-materialized-and-alias-fields)
|
||||
* [Working with nullable fields](field_types.md#working-with-nullable-fields)
|
||||
* [Working with LowCardinality fields](field_types.md#working-with-lowcardinality-fields)
|
||||
* [Creating custom field types](field_types.md#creating-custom-field-types)
|
||||
|
||||
* [Table Engines](table_engines.md#table-engines)
|
||||
* [Simple Engines](table_engines.md#simple-engines)
|
||||
* [Engines in the MergeTree Family](table_engines.md#engines-in-the-mergetree-family)
|
||||
* [Custom partitioning](table_engines.md#custom-partitioning)
|
||||
* [Primary key](table_engines.md#primary-key)
|
||||
* [Data Replication](table_engines.md#data-replication)
|
||||
* [Buffer Engine](table_engines.md#buffer-engine)
|
||||
* [Merge Engine](table_engines.md#merge-engine)
|
||||
|
||||
* [Schema Migrations](schema_migrations.md#schema-migrations)
|
||||
* [Writing Migrations](schema_migrations.md#writing-migrations)
|
||||
* [CreateTable](schema_migrations.md#createtable)
|
||||
* [DropTable](schema_migrations.md#droptable)
|
||||
* [AlterTable](schema_migrations.md#altertable)
|
||||
* [AlterTableWithBuffer](schema_migrations.md#altertablewithbuffer)
|
||||
* [AlterConstraints](schema_migrations.md#alterconstraints)
|
||||
* [RunPython](schema_migrations.md#runpython)
|
||||
* [RunSQL](schema_migrations.md#runsql)
|
||||
* [Running Migrations](schema_migrations.md#running-migrations)
|
||||
|
||||
* [System Models](system_models.md#system-models)
|
||||
@@ -45,44 +78,64 @@
|
||||
* [Tests](contributing.md#tests)
|
||||
|
||||
* [Class Reference](class_reference.md#class-reference)
|
||||
* [infi.clickhouse_orm.database](class_reference.md#infi.clickhouse_orm.database)
|
||||
* [datastore_orm.database](class_reference.md#inficlickhouse_ormdatabase)
|
||||
* [Database](class_reference.md#database)
|
||||
* [DatabaseException](class_reference.md#databaseexception)
|
||||
* [infi.clickhouse_orm.models](class_reference.md#infi.clickhouse_orm.models)
|
||||
* [datastore_orm.models](class_reference.md#inficlickhouse_ormmodels)
|
||||
* [Model](class_reference.md#model)
|
||||
* [BufferModel](class_reference.md#buffermodel)
|
||||
* [infi.clickhouse_orm.fields](class_reference.md#infi.clickhouse_orm.fields)
|
||||
* [Field](class_reference.md#field)
|
||||
* [StringField](class_reference.md#stringfield)
|
||||
* [DateField](class_reference.md#datefield)
|
||||
* [DateTimeField](class_reference.md#datetimefield)
|
||||
* [BaseIntField](class_reference.md#baseintfield)
|
||||
* [BaseFloatField](class_reference.md#basefloatfield)
|
||||
* [BaseEnumField](class_reference.md#baseenumfield)
|
||||
* [MergeModel](class_reference.md#mergemodel)
|
||||
* [DistributedModel](class_reference.md#distributedmodel)
|
||||
* [Constraint](class_reference.md#constraint)
|
||||
* [Index](class_reference.md#index)
|
||||
* [datastore_orm.fields](class_reference.md#inficlickhouse_ormfields)
|
||||
* [ArrayField](class_reference.md#arrayfield)
|
||||
* [BaseEnumField](class_reference.md#baseenumfield)
|
||||
* [BaseFloatField](class_reference.md#basefloatfield)
|
||||
* [BaseIntField](class_reference.md#baseintfield)
|
||||
* [DateField](class_reference.md#datefield)
|
||||
* [DateTime64Field](class_reference.md#datetime64field)
|
||||
* [DateTimeField](class_reference.md#datetimefield)
|
||||
* [Decimal128Field](class_reference.md#decimal128field)
|
||||
* [Decimal32Field](class_reference.md#decimal32field)
|
||||
* [Decimal64Field](class_reference.md#decimal64field)
|
||||
* [DecimalField](class_reference.md#decimalfield)
|
||||
* [Enum16Field](class_reference.md#enum16field)
|
||||
* [Enum8Field](class_reference.md#enum8field)
|
||||
* [Field](class_reference.md#field)
|
||||
* [FixedStringField](class_reference.md#fixedstringfield)
|
||||
* [UInt8Field](class_reference.md#uint8field)
|
||||
* [UInt16Field](class_reference.md#uint16field)
|
||||
* [UInt32Field](class_reference.md#uint32field)
|
||||
* [UInt64Field](class_reference.md#uint64field)
|
||||
* [Int8Field](class_reference.md#int8field)
|
||||
* [Float32Field](class_reference.md#float32field)
|
||||
* [Float64Field](class_reference.md#float64field)
|
||||
* [IPv4Field](class_reference.md#ipv4field)
|
||||
* [IPv6Field](class_reference.md#ipv6field)
|
||||
* [Int16Field](class_reference.md#int16field)
|
||||
* [Int32Field](class_reference.md#int32field)
|
||||
* [Int64Field](class_reference.md#int64field)
|
||||
* [Float32Field](class_reference.md#float32field)
|
||||
* [Float64Field](class_reference.md#float64field)
|
||||
* [Enum8Field](class_reference.md#enum8field)
|
||||
* [Enum16Field](class_reference.md#enum16field)
|
||||
* [infi.clickhouse_orm.engines](class_reference.md#infi.clickhouse_orm.engines)
|
||||
* [Int8Field](class_reference.md#int8field)
|
||||
* [LowCardinalityField](class_reference.md#lowcardinalityfield)
|
||||
* [NullableField](class_reference.md#nullablefield)
|
||||
* [StringField](class_reference.md#stringfield)
|
||||
* [UInt16Field](class_reference.md#uint16field)
|
||||
* [UInt32Field](class_reference.md#uint32field)
|
||||
* [UInt64Field](class_reference.md#uint64field)
|
||||
* [UInt8Field](class_reference.md#uint8field)
|
||||
* [UUIDField](class_reference.md#uuidfield)
|
||||
* [datastore_orm.engines](class_reference.md#inficlickhouse_ormengines)
|
||||
* [Engine](class_reference.md#engine)
|
||||
* [TinyLog](class_reference.md#tinylog)
|
||||
* [Log](class_reference.md#log)
|
||||
* [Memory](class_reference.md#memory)
|
||||
* [MergeTree](class_reference.md#mergetree)
|
||||
* [Buffer](class_reference.md#buffer)
|
||||
* [Merge](class_reference.md#merge)
|
||||
* [Distributed](class_reference.md#distributed)
|
||||
* [CollapsingMergeTree](class_reference.md#collapsingmergetree)
|
||||
* [SummingMergeTree](class_reference.md#summingmergetree)
|
||||
* [ReplacingMergeTree](class_reference.md#replacingmergetree)
|
||||
* [infi.clickhouse_orm.query](class_reference.md#infi.clickhouse_orm.query)
|
||||
* [datastore_orm.query](class_reference.md#inficlickhouse_ormquery)
|
||||
* [QuerySet](class_reference.md#queryset)
|
||||
* [AggregateQuerySet](class_reference.md#aggregatequeryset)
|
||||
* [Q](class_reference.md#q)
|
||||
* [datastore_orm.funcs](class_reference.md#inficlickhouse_ormfuncs)
|
||||
* [F](class_reference.md#f)
|
||||
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
What's New in Version 2
|
||||
=======================
|
||||
|
||||
## Python 3.5+ Only
|
||||
|
||||
This version of the ORM no longer support Python 2.
|
||||
|
||||
## New flexible syntax for database expressions and functions
|
||||
|
||||
Expressions that use model fields, database functions and Python operators are now first-class citizens of the ORM. They provide infinite expressivity and flexibility when defining models and generating queries.
|
||||
|
||||
Example of expressions in model definition:
|
||||
```python
|
||||
class Temperature(Model):
|
||||
|
||||
station_id = UInt16Field()
|
||||
timestamp = DateTimeField(default=F.now()) # function as default value
|
||||
degrees_celsius = Float32Field()
|
||||
degrees_fahrenheit = Float32Field(alias=degrees_celsius * 1.8 + 32) # expression as field alias
|
||||
|
||||
# expressions in engine definition
|
||||
engine = MergeTree(partition_key=[F.toYYYYMM(timestamp)], order_by=[station_id, timestamp])
|
||||
```
|
||||
|
||||
Example of expressions in queries:
|
||||
```python
|
||||
db = Database('default')
|
||||
start = F.toStartOfMonth(F.now())
|
||||
expr = (Temperature.timestamp > start) & (Temperature.station_id == 123) & (Temperature.degrees_celsius > 30)
|
||||
for t in Temperature.objects_in(db).filter(expr):
|
||||
print(t.timestamp, t.degrees_celsius)
|
||||
```
|
||||
|
||||
See [Expressions](expressions.md).
|
||||
|
||||
## Support for IPv4 and IPv6 fields
|
||||
|
||||
Two new fields classes were added: `IPv4Field` and `IPv6Field`. Their values are represented by Python's `ipaddress.IPv4Address` and `ipaddress.IPv6Address`.
|
||||
|
||||
See [Field Types](field_types.md).
|
||||
|
||||
## Automatic generation of models by inspecting existing tables
|
||||
|
||||
It is now easy to generate a model class on the fly for an existing table in the database using `Database.get_model_for_table`. This is particularly useful for querying system tables, for example:
|
||||
```python
|
||||
QueryLog = db.get_model_for_table('query_log', system_table=True)
|
||||
for row in QueryLog.objects_in(db).filter(QueryLog.query_duration_ms > 10000):
|
||||
print(row.query)
|
||||
```
|
||||
|
||||
## Convenient ways to import ORM classes
|
||||
|
||||
You can now import all ORM classes directly from `datastore_orm`, without worrying about sub-modules. For example:
|
||||
```python
|
||||
from datastore_orm import Database, Model, StringField, DateTimeField, MergeTree
|
||||
```
|
||||
See [Importing ORM Classes](importing_orm_classes.md).
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
/env/
|
||||
@@ -0,0 +1,22 @@
|
||||
# CPU Usage
|
||||
|
||||
This basic example uses `psutil` to collect a simple time-series of per-CPU usage percent. It then prints out some aggregate statistics based on the collected data.
|
||||
|
||||
## Running the code
|
||||
|
||||
Create a virtualenv and install the required libraries:
|
||||
```
|
||||
virtualenv -p python3.6 env
|
||||
source env/bin/activate
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
|
||||
Run the `collect` script to populate the database with the CPU statistics. Let it run for a bit before pressing CTRL+C.
|
||||
```
|
||||
python collect.py
|
||||
```
|
||||
|
||||
Run the `results` script to display the CPU statistics:
|
||||
```
|
||||
python results.py
|
||||
```
|
||||
@@ -0,0 +1,20 @@
|
||||
import psutil, time, datetime
|
||||
from datastore_orm import Database
|
||||
from models import CPUStats
|
||||
|
||||
|
||||
db = Database('demo')
|
||||
db.create_table(CPUStats)
|
||||
|
||||
|
||||
psutil.cpu_percent(percpu=True) # first sample should be discarded
|
||||
|
||||
while True:
|
||||
time.sleep(1)
|
||||
stats = psutil.cpu_percent(percpu=True)
|
||||
timestamp = datetime.datetime.now()
|
||||
print(timestamp)
|
||||
db.insert([
|
||||
CPUStats(timestamp=timestamp, cpu_id=cpu_id, cpu_percent=cpu_percent)
|
||||
for cpu_id, cpu_percent in enumerate(stats)
|
||||
])
|
||||
@@ -0,0 +1,11 @@
|
||||
from datastore_orm import Model, DateTimeField, UInt16Field, Float32Field, Memory
|
||||
|
||||
|
||||
class CPUStats(Model):
|
||||
|
||||
timestamp = DateTimeField()
|
||||
cpu_id = UInt16Field()
|
||||
cpu_percent = Float32Field()
|
||||
|
||||
engine = Memory()
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
datastore_orm
|
||||
psutil
|
||||
@@ -0,0 +1,13 @@
|
||||
from datastore_orm import Database, F
|
||||
from models import CPUStats
|
||||
|
||||
|
||||
db = Database('demo')
|
||||
queryset = CPUStats.objects_in(db)
|
||||
total = queryset.filter(CPUStats.cpu_id == 1).count()
|
||||
busy = queryset.filter(CPUStats.cpu_id == 1, CPUStats.cpu_percent > 95).count()
|
||||
print('CPU 1 was busy {:.2f}% of the time'.format(busy * 100.0 / total))
|
||||
|
||||
# Calculate the average usage per CPU
|
||||
for row in queryset.aggregate(CPUStats.cpu_id, average=F.avg(CPUStats.cpu_percent)):
|
||||
print('CPU {row.cpu_id}: {row.average:.2f}%'.format(row=row))
|
||||
@@ -0,0 +1 @@
|
||||
/env/
|
||||
@@ -0,0 +1,36 @@
|
||||
# DB Explorer
|
||||
|
||||
This is a simple Flask web application that connects to ClickHouse and displays the list of existing databases. Clicking on a database name drills down into it, showing its list of tables. Clicking on a table drills down further, showing details about the table and its columns.
|
||||
|
||||
For each table or column, the application displays the compressed size on disk, the uncompressed size, and the ratio between them. Additionally, several pie charts are shown - top tables by size, top tables by rows, and top columns by size (in a table).
|
||||
|
||||
The pie charts are generated using the `pygal` charting library.
|
||||
|
||||
ORM concepts that are demonstrated by this example:
|
||||
|
||||
- Creating ORM models from existing tables using `Database.get_model_for_table`
|
||||
- Queryset filtering
|
||||
- Queryset aggregation
|
||||
|
||||
## Running the code
|
||||
|
||||
Create a virtualenv and install the required libraries:
|
||||
```
|
||||
virtualenv -p python3.6 env
|
||||
source env/bin/activate
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
|
||||
Run the server and open http://127.0.0.1:5000/ in your browser:
|
||||
```
|
||||
python server.py
|
||||
```
|
||||
|
||||
By default the server connects to ClickHouse running on http://localhost:8123/ without a username or password, but you can change this using command line arguments:
|
||||
```
|
||||
python server.py http://myclickhouse:8123/
|
||||
```
|
||||
or:
|
||||
```
|
||||
python server.py http://myclickhouse:8123/ admin secret123
|
||||
```
|
||||
@@ -0,0 +1,62 @@
|
||||
import pygal
|
||||
from pygal.style import RotateStyle
|
||||
from jinja2.filters import do_filesizeformat
|
||||
|
||||
|
||||
# Formatting functions
|
||||
number_formatter = lambda v: '{:,}'.format(v)
|
||||
bytes_formatter = lambda v: do_filesizeformat(v, True)
|
||||
|
||||
|
||||
def tables_piechart(db, by_field, value_formatter):
|
||||
'''
|
||||
Generate a pie chart of the top n tables in the database.
|
||||
`db` - the database instance
|
||||
`by_field` - the field name to sort by
|
||||
`value_formatter` - a function to use for formatting the numeric values
|
||||
'''
|
||||
Tables = db.get_model_for_table('tables', system_table=True)
|
||||
qs = Tables.objects_in(db).filter(database=db.db_name, is_temporary=False).exclude(engine='Buffer')
|
||||
tuples = [(getattr(table, by_field), table.name) for table in qs]
|
||||
return _generate_piechart(tuples, value_formatter)
|
||||
|
||||
|
||||
def columns_piechart(db, tbl_name, by_field, value_formatter):
|
||||
'''
|
||||
Generate a pie chart of the top n columns in the table.
|
||||
`db` - the database instance
|
||||
`tbl_name` - the table name
|
||||
`by_field` - the field name to sort by
|
||||
`value_formatter` - a function to use for formatting the numeric values
|
||||
'''
|
||||
ColumnsTable = db.get_model_for_table('columns', system_table=True)
|
||||
qs = ColumnsTable.objects_in(db).filter(database=db.db_name, table=tbl_name)
|
||||
tuples = [(getattr(col, by_field), col.name) for col in qs]
|
||||
return _generate_piechart(tuples, value_formatter)
|
||||
|
||||
|
||||
def _get_top_tuples(tuples, n=15):
|
||||
'''
|
||||
Given a list of tuples (value, name), this function sorts
|
||||
the list and returns only the top n results. All other tuples
|
||||
are aggregated to a single "others" tuple.
|
||||
'''
|
||||
non_zero_tuples = [t for t in tuples if t[0]]
|
||||
sorted_tuples = sorted(non_zero_tuples, reverse=True)
|
||||
if len(sorted_tuples) > n:
|
||||
others = (sum(t[0] for t in sorted_tuples[n:]), 'others')
|
||||
sorted_tuples = sorted_tuples[:n] + [others]
|
||||
return sorted_tuples
|
||||
|
||||
|
||||
def _generate_piechart(tuples, value_formatter):
|
||||
'''
|
||||
Generates a pie chart.
|
||||
`tuples` - a list of (value, name) tuples to include in the chart
|
||||
`value_formatter` - a function to use for formatting the values
|
||||
'''
|
||||
style = RotateStyle('#9e6ffe', background='white', legend_font_family='Roboto', legend_font_size=18, tooltip_font_family='Roboto', tooltip_font_size=24)
|
||||
chart = pygal.Pie(style=style, margin=0, title=' ', value_formatter=value_formatter, truncate_legend=-1)
|
||||
for t in _get_top_tuples(tuples):
|
||||
chart.add(t[1], t[0])
|
||||
return chart.render(is_unicode=True, disable_xml_declaration=True)
|
||||
@@ -0,0 +1,15 @@
|
||||
certifi==2020.4.5.2
|
||||
chardet==3.0.4
|
||||
click==7.1.2
|
||||
Flask==1.1.2
|
||||
idna==2.9
|
||||
infi.clickhouse-orm==2.0.1
|
||||
iso8601==0.1.12
|
||||
itsdangerous==1.1.0
|
||||
Jinja2==2.11.2
|
||||
MarkupSafe==1.1.1
|
||||
pygal==2.4.0
|
||||
pytz==2020.1
|
||||
requests==2.23.0
|
||||
urllib3==1.25.9
|
||||
Werkzeug==1.0.1
|
||||
@@ -0,0 +1,87 @@
|
||||
from datastore_orm import Database, F
|
||||
from charts import tables_piechart, columns_piechart, number_formatter, bytes_formatter
|
||||
from flask import Flask
|
||||
from flask import render_template
|
||||
import sys
|
||||
|
||||
|
||||
app = Flask(__name__)
|
||||
|
||||
|
||||
@app.route('/')
|
||||
def homepage_view():
|
||||
'''
|
||||
Root view that lists all databases.
|
||||
'''
|
||||
db = _get_db('system')
|
||||
# Get all databases in the system.databases table
|
||||
DatabasesTable = db.get_model_for_table('databases', system_table=True)
|
||||
databases = DatabasesTable.objects_in(db).exclude(name='system')
|
||||
databases = databases.order_by(F.lower(DatabasesTable.name))
|
||||
# Generate the page
|
||||
return render_template('homepage.html', db=db, databases=databases)
|
||||
|
||||
|
||||
@app.route('/<db_name>/')
|
||||
def database_view(db_name):
|
||||
'''
|
||||
A view that displays information about a single database.
|
||||
'''
|
||||
db = _get_db(db_name)
|
||||
# Get all the tables in the database, by aggregating information from system.columns
|
||||
ColumnsTable = db.get_model_for_table('columns', system_table=True)
|
||||
tables = ColumnsTable.objects_in(db).filter(database=db_name).aggregate(
|
||||
ColumnsTable.table,
|
||||
compressed_size=F.sum(ColumnsTable.data_compressed_bytes),
|
||||
uncompressed_size=F.sum(ColumnsTable.data_uncompressed_bytes),
|
||||
ratio=F.sum(ColumnsTable.data_uncompressed_bytes) / F.sum(ColumnsTable.data_compressed_bytes)
|
||||
)
|
||||
tables = tables.order_by(F.lower(ColumnsTable.table))
|
||||
# Generate the page
|
||||
return render_template('database.html',
|
||||
db=db,
|
||||
tables=tables,
|
||||
tables_piechart_by_rows=tables_piechart(db, 'total_rows', value_formatter=number_formatter),
|
||||
tables_piechart_by_size=tables_piechart(db, 'total_bytes', value_formatter=bytes_formatter),
|
||||
)
|
||||
|
||||
|
||||
@app.route('/<db_name>/<tbl_name>/')
|
||||
def table_view(db_name, tbl_name):
|
||||
'''
|
||||
A view that displays information about a single table.
|
||||
'''
|
||||
db = _get_db(db_name)
|
||||
# Get table information from system.tables
|
||||
TablesTable = db.get_model_for_table('tables', system_table=True)
|
||||
tbl_info = TablesTable.objects_in(db).filter(database=db_name, name=tbl_name)[0]
|
||||
# Get the SQL used for creating the table
|
||||
create_table_sql = db.raw('SHOW CREATE TABLE %s FORMAT TabSeparatedRaw' % tbl_name)
|
||||
# Get all columns in the table from system.columns
|
||||
ColumnsTable = db.get_model_for_table('columns', system_table=True)
|
||||
columns = ColumnsTable.objects_in(db).filter(database=db_name, table=tbl_name)
|
||||
# Generate the page
|
||||
return render_template('table.html',
|
||||
db=db,
|
||||
tbl_name=tbl_name,
|
||||
tbl_info=tbl_info,
|
||||
create_table_sql=create_table_sql,
|
||||
columns=columns,
|
||||
piechart=columns_piechart(db, tbl_name, 'data_compressed_bytes', value_formatter=bytes_formatter),
|
||||
)
|
||||
|
||||
|
||||
def _get_db(db_name):
|
||||
'''
|
||||
Returns a Database instance using connection information
|
||||
from the command line arguments (optional).
|
||||
'''
|
||||
db_url = sys.argv[1] if len(sys.argv) > 1 else 'http://localhost:8123/'
|
||||
username = sys.argv[2] if len(sys.argv) > 2 else None
|
||||
password = sys.argv[3] if len(sys.argv) > 3 else None
|
||||
return Database(db_name, db_url, username, password, readonly=True)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
_get_db('system') # fail early on db connection problems
|
||||
app.run(debug=True)
|
||||
@@ -0,0 +1,22 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<title>ClickHouse Explorer</title>
|
||||
<link rel="icon" href="data:,">
|
||||
<link rel="stylesheet" href="https://fonts.googleapis.com/css?family=Roboto:300,300italic,700,700italic">
|
||||
<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/normalize/8.0.1/normalize.css">
|
||||
<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/milligram/1.4.0/milligram.css">
|
||||
<script type="text/javascript" src="http://kozea.github.com/pygal.js/latest/pygal-tooltips.min.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
|
||||
<div class="container">
|
||||
|
||||
{% block contents %}
|
||||
{% endblock %}
|
||||
|
||||
</div>
|
||||
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,54 @@
|
||||
{% extends "base.html" %}
|
||||
|
||||
{% block contents %}
|
||||
|
||||
<h1>{{ db.db_name }}</h1>
|
||||
|
||||
<p>
|
||||
<a href="..">Home</a>
|
||||
»
|
||||
{{ db.db_name }}
|
||||
</p>
|
||||
|
||||
<div class="row">
|
||||
|
||||
<div class="column">
|
||||
<h2>Top Tables by Size</h2>
|
||||
{% autoescape false %}
|
||||
{{ tables_piechart_by_size }}
|
||||
{% endautoescape %}
|
||||
</div>
|
||||
|
||||
<div class="column">
|
||||
<h2>Top Tables by Rows</h2>
|
||||
{% autoescape false %}
|
||||
{{ tables_piechart_by_rows }}
|
||||
{% endautoescape %}
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<h2>Tables ({{ tables.count() }})</h2>
|
||||
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Name</th>
|
||||
<th>Uncompressed Size</th>
|
||||
<th>Compressed Size</th>
|
||||
<th>Compression Ratio</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{% for table in tables %}
|
||||
<tr>
|
||||
<td><a href="{{ table.table|urlencode }}/">{{ table.table }}</a></th>
|
||||
<td>{{ table.uncompressed_size|filesizeformat(true) }}</td>
|
||||
<td>{{ table.compressed_size|filesizeformat(true) }}</td>
|
||||
<td>{% if table.uncompressed_size %} {{ "%.2f" % table.ratio }} {% else %} 1 {% endif %} : 1</td>
|
||||
</tr>
|
||||
{% endfor %}
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,41 @@
|
||||
{% extends "base.html" %}
|
||||
|
||||
{% block contents %}
|
||||
|
||||
|
||||
<div class="row">
|
||||
|
||||
<div class="column-50">
|
||||
|
||||
<h1>ClickHouse Explorer</h1>
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<th>URL</th>
|
||||
<td>{{ db.db_url }}</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<th>Version</th>
|
||||
<td>{{ db.server_version|join('.') }}</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<th>Timezone</th>
|
||||
<td>{{ db.server_timezone }}</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<h2>Databases ({{ databases.count() }})</h2>
|
||||
<ul>
|
||||
{% for d in databases %}
|
||||
<li>
|
||||
<a href="{{ d.name|urlencode }}/">{{ d.name }}</a>
|
||||
</li>
|
||||
{% endfor %}
|
||||
</ul>
|
||||
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,79 @@
|
||||
{% extends "base.html" %}
|
||||
|
||||
{% block contents %}
|
||||
|
||||
|
||||
<p>
|
||||
<a href="../..">Home</a>
|
||||
»
|
||||
<a href="..">{{ db.db_name }}</a>
|
||||
»
|
||||
{{ tbl_name }}
|
||||
</p>
|
||||
|
||||
<h1>{{ tbl_name }}</h1>
|
||||
|
||||
<div class="row">
|
||||
|
||||
<div class="column">
|
||||
<h2>Details</h2>
|
||||
<table>
|
||||
<tr>
|
||||
<th>Total rows</th>
|
||||
<td>{{ "{:,}".format(tbl_info.total_rows) }}</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<th>Total size</th>
|
||||
<td>{{ tbl_info.total_bytes|filesizeformat(true) }}</td>
|
||||
</tr>
|
||||
{% if tbl_info.total_rows %}
|
||||
<tr>
|
||||
<th>Average row size</th>
|
||||
<td>{{ (tbl_info.total_bytes / tbl_info.total_rows)|filesizeformat(true) }}</td>
|
||||
</tr>
|
||||
{% endif %}
|
||||
<tr>
|
||||
<th>Engine</th>
|
||||
<td>{{ tbl_info.engine }}</td>
|
||||
</tr>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<div class="column">
|
||||
<h2>Top Columns by Size</h2>
|
||||
{% autoescape false %}
|
||||
{{ piechart }}
|
||||
{% endautoescape %}
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<h2>Columns ({{ columns.count() }})</h2>
|
||||
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Name</th>
|
||||
<th>Type</th>
|
||||
<th>Uncompressed Size</th>
|
||||
<th>Compressed Size</th>
|
||||
<th>Compression Ratio</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{% for col in columns %}
|
||||
<tr>
|
||||
<td>{{ col.name }}</td>
|
||||
<td>{{ col.type }}</td>
|
||||
<td>{{ col.data_uncompressed_bytes|filesizeformat(true) }}</td>
|
||||
<td>{{ col.data_compressed_bytes|filesizeformat(true) }}</td>
|
||||
<td>{% if col.data_compressed_bytes %} {{ "%.2f" % (col.data_uncompressed_bytes / col.data_compressed_bytes) }} {% else %} 1 {% endif %} : 1</td>
|
||||
</tr>
|
||||
{% endfor %}
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
<h2>Table Definition</h2>
|
||||
<pre><code>{{ create_table_sql }}</code></pre>
|
||||
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,2 @@
|
||||
/ebooks/
|
||||
/env/
|
||||
@@ -0,0 +1,80 @@
|
||||
# Full Text Search
|
||||
|
||||
This example shows how ClickHouse might be used for searching for word sequences in texts. It's a nice proof of concept, but for production use there are probably better solutions, such as Elasticsearch.
|
||||
|
||||
## Running the code
|
||||
|
||||
Create a virtualenv and install the required libraries:
|
||||
```
|
||||
virtualenv -p python3.6 env
|
||||
source env/bin/activate
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
Run the `download_ebooks` script to download a dozen classical books from [The Gutenberg Project](http://www.gutenberg.org/):
|
||||
```
|
||||
python download_ebooks.py
|
||||
```
|
||||
Run the `load` script to populate the database with the downloaded texts:
|
||||
```
|
||||
python load.py
|
||||
```
|
||||
And finally, run the full text search:
|
||||
```
|
||||
python search.py "cheshire cat"
|
||||
```
|
||||
Asterisks can be used as wildcards (each asterisk stands for one word):
|
||||
```
|
||||
python search.py "much * than"
|
||||
```
|
||||
|
||||
## How it works
|
||||
|
||||
The `models.py` file defines an ORM model for storing each word in the indexed texts:
|
||||
```python
|
||||
class Fragment(Model):
|
||||
|
||||
language = LowCardinalityField(StringField(default='EN'))
|
||||
document = LowCardinalityField(StringField())
|
||||
idx = UInt64Field()
|
||||
word = StringField()
|
||||
stem = StringField()
|
||||
|
||||
# An index for faster search by document and fragment idx
|
||||
index = Index((document, idx), type=Index.minmax(), granularity=1)
|
||||
|
||||
# The primary key allows efficient lookup of stems
|
||||
engine = MergeTree(order_by=(stem, document, idx), partition_key=('language',))
|
||||
```
|
||||
The `document` (name) and `idx` (running number of the word inside the document) fields identify the specific word. The `word` field stores the original word as it appears in the text, while the `stem` contains the word after normalization, and that's the field which is used for matching the search terms. Stemming the words makes the matching less strict, so that searching for "swallowed" will also find documents that mention "swallow" or "swallowing".
|
||||
|
||||
Here's what some records in the fragment table might look like:
|
||||
|
||||
| language | document | idx | word | stem |
|
||||
|----------|-------------------------|------|------------------|---------------|
|
||||
| EN | Moby Dick; or The Whale | 4510 | whenever | whenev |
|
||||
| EN | Moby Dick; or The Whale | 4511 | it | it |
|
||||
| EN | Moby Dick; or The Whale | 4512 | is | is |
|
||||
| EN | Moby Dick; or The Whale | 4513 | a | a |
|
||||
| EN | Moby Dick; or The Whale | 4514 | damp, | damp |
|
||||
| EN | Moby Dick; or The Whale | 4515 | drizzly | drizzli |
|
||||
| EN | Moby Dick; or The Whale | 4516 | November | novemb |
|
||||
| EN | Moby Dick; or The Whale | 4517 | in | in |
|
||||
| EN | Moby Dick; or The Whale | 4518 | my | my |
|
||||
| EN | Moby Dick; or The Whale | 4519 | soul; | soul |
|
||||
|
||||
Let's say we're looking for the terms "drizzly November". Finding the first in the sequence (after stemming it) is fast and easy:
|
||||
```python
|
||||
query = Fragment.objects_in(db).filter(stem='drizzli').only(Fragment.document, Fragment.idx)
|
||||
```
|
||||
We're interested only in the `document` and `idx` fields, since they identify a specific word.
|
||||
|
||||
To find the next word in the search terms, we need a subquery similar to the first one, with an additional condition that its index will be one greater than the index of the first word:
|
||||
```python
|
||||
subquery = Fragment.objects_in(db).filter(stem='novemb').only(Fragment.document, Fragment.idx)
|
||||
query = query.filter(F.isIn((Fragment.document, Fragment.idx + 1), subquery))
|
||||
```
|
||||
And so on, by adding another subquery for each additional search term we can construct the whole sequence of words.
|
||||
|
||||
As for wildcard support, when encountering a wildcard in the search terms we simply skip it - it does not need a subquery (since it can match any word). It only increases the index count so that the query conditions will "skip" one word in the sequence.
|
||||
|
||||
The algorithm for building this compound query can be found in the `build_query` function.
|
||||
@@ -0,0 +1,27 @@
|
||||
import requests
|
||||
import os
|
||||
|
||||
|
||||
def download_ebook(id):
|
||||
print(id, end=' ')
|
||||
# Download the ebook's text
|
||||
r = requests.get('https://www.gutenberg.org/files/{id}/{id}-0.txt'.format(id=id))
|
||||
if r.status_code == 404:
|
||||
print('NOT FOUND, SKIPPING')
|
||||
return
|
||||
r.raise_for_status()
|
||||
# Find the ebook's title
|
||||
text = r.content.decode('utf-8')
|
||||
for line in text.splitlines():
|
||||
if line.startswith('Title:'):
|
||||
title = line[6:].strip()
|
||||
print(title)
|
||||
# Save the ebook
|
||||
with open('ebooks/{}.txt'.format(title), 'wb') as f:
|
||||
f.write(r.content)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
os.makedirs('ebooks', exist_ok=True)
|
||||
for i in [1342, 11, 84, 2701, 25525, 1661, 98, 74, 43, 215, 1400, 76]:
|
||||
download_ebook(i)
|
||||
@@ -0,0 +1,61 @@
|
||||
import sys
|
||||
import nltk
|
||||
from nltk.stem.porter import PorterStemmer
|
||||
from glob import glob
|
||||
from datastore_orm import Database
|
||||
from models import Fragment
|
||||
|
||||
|
||||
def trim_punctuation(word):
|
||||
'''
|
||||
Trim punctuation characters from the beginning and end of the word
|
||||
'''
|
||||
start = end = len(word)
|
||||
for i in range(len(word)):
|
||||
if word[i].isalnum():
|
||||
start = min(start, i)
|
||||
end = i + 1
|
||||
return word[start : end]
|
||||
|
||||
|
||||
def parse_file(filename):
|
||||
'''
|
||||
Parses a text file at the give path.
|
||||
Returns a generator of tuples (original_word, stemmed_word)
|
||||
The original_word may include punctuation characters.
|
||||
'''
|
||||
stemmer = PorterStemmer()
|
||||
with open(filename, 'r', encoding='utf-8') as f:
|
||||
for line in f:
|
||||
for word in line.split():
|
||||
yield (word, stemmer.stem(trim_punctuation(word)))
|
||||
|
||||
|
||||
def get_fragments(filename):
|
||||
'''
|
||||
Converts a text file at the given path to a generator
|
||||
of Fragment instances.
|
||||
'''
|
||||
from os import path
|
||||
document = path.splitext(path.basename(filename))[0]
|
||||
idx = 0
|
||||
for word, stem in parse_file(filename):
|
||||
idx += 1
|
||||
yield Fragment(document=document, idx=idx, word=word, stem=stem)
|
||||
print('{} - {} words'.format(filename, idx))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
# Load NLTK data if necessary
|
||||
nltk.download('punkt')
|
||||
nltk.download('wordnet')
|
||||
|
||||
# Initialize database
|
||||
db = Database('default')
|
||||
db.create_table(Fragment)
|
||||
|
||||
# Load files from the command line or everything under ebooks/
|
||||
filenames = sys.argv[1:] or glob('ebooks/*.txt')
|
||||
for filename in filenames:
|
||||
db.insert(get_fragments(filename), batch_size=100000)
|
||||
@@ -0,0 +1,16 @@
|
||||
from datastore_orm import *
|
||||
|
||||
|
||||
class Fragment(Model):
|
||||
|
||||
language = LowCardinalityField(StringField(), default='EN')
|
||||
document = LowCardinalityField(StringField())
|
||||
idx = UInt64Field()
|
||||
word = StringField()
|
||||
stem = StringField()
|
||||
|
||||
# An index for faster search by document and fragment idx
|
||||
index = Index((document, idx), type=Index.minmax(), granularity=1)
|
||||
|
||||
# The primary key allows efficient lookup of stems
|
||||
engine = MergeTree(order_by=(stem, document, idx), partition_key=('language',))
|
||||
@@ -0,0 +1,4 @@
|
||||
datastore_orm
|
||||
nltk
|
||||
requests
|
||||
colorama
|
||||
@@ -0,0 +1,90 @@
|
||||
import sys
|
||||
from colorama import init, Fore, Back, Style
|
||||
from nltk.stem.porter import PorterStemmer
|
||||
from datastore_orm import Database, F
|
||||
from models import Fragment
|
||||
from load import trim_punctuation
|
||||
|
||||
|
||||
# The wildcard character
|
||||
WILDCARD = '*'
|
||||
|
||||
|
||||
def prepare_search_terms(text):
|
||||
'''
|
||||
Convert the text to search into a list of stemmed words.
|
||||
'''
|
||||
stemmer = PorterStemmer()
|
||||
stems = []
|
||||
for word in text.split():
|
||||
if word == WILDCARD:
|
||||
stems.append(WILDCARD)
|
||||
else:
|
||||
stems.append(stemmer.stem(trim_punctuation(word)))
|
||||
return stems
|
||||
|
||||
|
||||
def build_query(db, stems):
|
||||
'''
|
||||
Returns a queryset instance for finding sequences of Fragment instances
|
||||
that matche the list of stemmed words.
|
||||
'''
|
||||
# Start by searching for the first stemmed word
|
||||
all_fragments = Fragment.objects_in(db)
|
||||
query = all_fragments.filter(stem=stems[0]).only(Fragment.document, Fragment.idx)
|
||||
# Add the following words to the queryset
|
||||
for i, stem in enumerate(stems):
|
||||
# Skip the first word (it's already in the query), and wildcards
|
||||
if i == 0 or stem == WILDCARD:
|
||||
continue
|
||||
# Create a subquery that finds instances of the i'th word
|
||||
subquery = all_fragments.filter(stem=stem).only(Fragment.document, Fragment.idx)
|
||||
# Add it to the query, requiring that it will appear i places away from the first word
|
||||
query = query.filter(F.isIn((Fragment.document, Fragment.idx + i), subquery))
|
||||
# Sort the results
|
||||
query = query.order_by(Fragment.document, Fragment.idx)
|
||||
return query
|
||||
|
||||
|
||||
def get_matching_text(db, document, from_idx, to_idx, extra=5):
|
||||
'''
|
||||
Reconstructs the document text between the given indexes (inclusive),
|
||||
plus `extra` words before and after the match. The words that are
|
||||
included in the given range are highlighted in green.
|
||||
'''
|
||||
text = []
|
||||
conds = (Fragment.document == document) & (Fragment.idx >= from_idx - extra) & (Fragment.idx <= to_idx + extra)
|
||||
for fragment in Fragment.objects_in(db).filter(conds).order_by('document', 'idx'):
|
||||
word = fragment.word
|
||||
if fragment.idx == from_idx:
|
||||
word = Fore.GREEN + word
|
||||
if fragment.idx == to_idx:
|
||||
word = word + Style.RESET_ALL
|
||||
text.append(word)
|
||||
return ' '.join(text)
|
||||
|
||||
|
||||
def find(db, text):
|
||||
'''
|
||||
Performs the search for the given text, and prints out the matches.
|
||||
'''
|
||||
stems = prepare_search_terms(text)
|
||||
query = build_query(db, stems)
|
||||
print('\n' + Fore.MAGENTA + str(query) + Style.RESET_ALL + '\n')
|
||||
for match in query:
|
||||
text = get_matching_text(db, match.document, match.idx, match.idx + len(stems) - 1)
|
||||
print(Fore.CYAN + match.document + ':' + Style.RESET_ALL, text)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
# Initialize colored output
|
||||
init()
|
||||
|
||||
# Initialize database
|
||||
db = Database('default')
|
||||
|
||||
# Search
|
||||
text = ' '.join(sys.argv[1:])
|
||||
if text:
|
||||
find(db, text)
|
||||
+3
-4
@@ -35,10 +35,9 @@ Usage:
|
||||
generate_all
|
||||
------------
|
||||
Does everything:
|
||||
|
||||
- Generates the class reference using generate_ref
|
||||
- Generates the table of contents using generate_toc
|
||||
- Converts to HTML for visual inspection using docs2html
|
||||
- Generates the class reference using generate_ref
|
||||
- Generates the table of contents using generate_toc
|
||||
- Converts to HTML for visual inspection using docs2html
|
||||
|
||||
Usage:
|
||||
|
||||
|
||||
@@ -6,5 +6,8 @@ find ./ -iname "*.md" -type f -exec sh -c 'echo "Converting ${0}"; pandoc "${0}"
|
||||
echo "Converting README.md"
|
||||
pandoc ../README.md -s -o "../htmldocs/README.html"
|
||||
|
||||
echo "Converting CHANGELOG.md"
|
||||
pandoc ../CHANGELOG.md -s -o "../htmldocs/CHANGELOG.html"
|
||||
|
||||
echo "Fixing links"
|
||||
sed -i 's/\.md/\.html/g' ../htmldocs/*.html
|
||||
|
||||
+34
-29
@@ -51,7 +51,8 @@ def get_method_sig(method):
|
||||
for arg in argspec.args:
|
||||
default_arg = _get_default_arg(argspec.args, argspec.defaults, arg_index)
|
||||
if default_arg.has_default:
|
||||
args.append("%s=%s" % (arg, default_arg.default_value))
|
||||
val = default_arg.default_value
|
||||
args.append("%s=%s" % (arg, val))
|
||||
else:
|
||||
args.append(arg)
|
||||
arg_index += 1
|
||||
@@ -70,47 +71,47 @@ def docstring(obj):
|
||||
indentation = min(len(line) - len(line.lstrip()) for line in lines if line.strip())
|
||||
# Output the lines without the indentation
|
||||
for line in lines:
|
||||
print line[indentation:]
|
||||
print
|
||||
print(line[indentation:])
|
||||
print()
|
||||
|
||||
|
||||
def class_doc(cls, list_methods=True):
|
||||
bases = ', '.join([b.__name__ for b in cls.__bases__])
|
||||
print '###', cls.__name__
|
||||
print
|
||||
print('###', cls.__name__)
|
||||
print()
|
||||
if bases != 'object':
|
||||
print 'Extends', bases
|
||||
print
|
||||
print('Extends', bases)
|
||||
print()
|
||||
docstring(cls)
|
||||
for name, method in inspect.getmembers(cls, inspect.ismethod):
|
||||
for name, method in inspect.getmembers(cls, lambda m: inspect.ismethod(m) or inspect.isfunction(m)):
|
||||
if name == '__init__':
|
||||
# Initializer
|
||||
print '####', get_method_sig(method).replace(name, cls.__name__)
|
||||
print('####', get_method_sig(method).replace(name, cls.__name__))
|
||||
elif name[0] == '_':
|
||||
# Private method
|
||||
continue
|
||||
elif method.__self__ == cls:
|
||||
elif hasattr(method, '__self__') and method.__self__ == cls:
|
||||
# Class method
|
||||
if not list_methods:
|
||||
continue
|
||||
print '#### %s.%s' % (cls.__name__, get_method_sig(method))
|
||||
print('#### %s.%s' % (cls.__name__, get_method_sig(method)))
|
||||
else:
|
||||
# Regular method
|
||||
if not list_methods:
|
||||
continue
|
||||
print '####', get_method_sig(method)
|
||||
print
|
||||
print('####', get_method_sig(method))
|
||||
print()
|
||||
docstring(method)
|
||||
print
|
||||
print()
|
||||
|
||||
|
||||
def module_doc(classes, list_methods=True):
|
||||
mdl = classes[0].__module__
|
||||
print mdl
|
||||
print '-' * len(mdl)
|
||||
print
|
||||
print(mdl)
|
||||
print('-' * len(mdl))
|
||||
print()
|
||||
for cls in classes:
|
||||
class_doc(cls, list_methods)
|
||||
class_doc(cls, list_methods)
|
||||
|
||||
|
||||
def all_subclasses(cls):
|
||||
@@ -119,17 +120,21 @@ def all_subclasses(cls):
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
from infi.clickhouse_orm import database
|
||||
from infi.clickhouse_orm import fields
|
||||
from infi.clickhouse_orm import engines
|
||||
from infi.clickhouse_orm import models
|
||||
from infi.clickhouse_orm import query
|
||||
from datastore_orm import database
|
||||
from datastore_orm import fields
|
||||
from datastore_orm import engines
|
||||
from datastore_orm import models
|
||||
from datastore_orm import query
|
||||
from datastore_orm import funcs
|
||||
from datastore_orm import system_models
|
||||
|
||||
print 'Class Reference'
|
||||
print '==============='
|
||||
print
|
||||
print('Class Reference')
|
||||
print('===============')
|
||||
print()
|
||||
module_doc([database.Database, database.DatabaseException])
|
||||
module_doc([models.Model, models.BufferModel])
|
||||
module_doc([fields.Field] + all_subclasses(fields.Field), False)
|
||||
module_doc([models.Model, models.BufferModel, models.MergeModel, models.DistributedModel, models.Constraint, models.Index])
|
||||
module_doc(sorted([fields.Field] + all_subclasses(fields.Field), key=lambda x: x.__name__), False)
|
||||
module_doc([engines.Engine] + all_subclasses(engines.Engine), False)
|
||||
module_doc([query.QuerySet])
|
||||
module_doc([query.QuerySet, query.AggregateQuerySet, query.Q])
|
||||
module_doc([funcs.F])
|
||||
module_doc([system_models.SystemPart])
|
||||
|
||||
@@ -9,6 +9,7 @@ printf "# Table of Contents\n\n" > toc.md
|
||||
generate_one "index.md"
|
||||
generate_one "models_and_databases.md"
|
||||
generate_one "querysets.md"
|
||||
generate_one "field_options.md"
|
||||
generate_one "field_types.md"
|
||||
generate_one "table_engines.md"
|
||||
generate_one "schema_migrations.md"
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from HTMLParser import HTMLParser
|
||||
from html.parser import HTMLParser
|
||||
import sys
|
||||
|
||||
|
||||
@@ -17,8 +17,8 @@ class HeadersToMarkdownParser(HTMLParser):
|
||||
def handle_endtag(self, tag):
|
||||
if tag.lower() in HEADER_TAGS:
|
||||
indent = ' ' * int(self.inside[1])
|
||||
fragment = self.text.lower().replace(' ', '-')
|
||||
print '%s* [%s](%s#%s)' % (indent, self.text, sys.argv[1], fragment)
|
||||
fragment = self.text.lower().replace(' ', '-').replace('.', '')
|
||||
print('%s* [%s](%s#%s)' % (indent, self.text, sys.argv[1], fragment))
|
||||
self.inside = None
|
||||
self.text = ''
|
||||
|
||||
@@ -28,4 +28,4 @@ class HeadersToMarkdownParser(HTMLParser):
|
||||
|
||||
|
||||
HeadersToMarkdownParser().feed(sys.stdin.read())
|
||||
print
|
||||
print('')
|
||||
|
||||
@@ -6,14 +6,14 @@ SETUP_INFO = dict(
|
||||
author_email = '${infi.recipe.template.version:author_email}',
|
||||
|
||||
url = ${infi.recipe.template.version:homepage},
|
||||
license = 'PSF',
|
||||
license = 'BSD',
|
||||
description = """${project:description}""",
|
||||
|
||||
# http://pypi.python.org/pypi?%3Aaction=list_classifiers
|
||||
classifiers = [
|
||||
"Intended Audience :: Developers",
|
||||
"Intended Audience :: System Administrators",
|
||||
"License :: OSI Approved :: Python Software Foundation License",
|
||||
"License :: OSI Approved :: BSD License",
|
||||
"Operating System :: OS Independent",
|
||||
"Programming Language :: Python",
|
||||
"Programming Language :: Python :: 2.7",
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
|
||||
SETUP_INFO = dict(
|
||||
name = 'datastore-orm',
|
||||
version = '2.1.0.post20',
|
||||
author = 'Hanzo AI, Inc.',
|
||||
author_email = 'dev@hanzo.ai',
|
||||
|
||||
url = 'https://github.com/hanzoai/datastore-orm',
|
||||
license = 'BSD',
|
||||
description = """Python ORM for Hanzo Datastore""",
|
||||
|
||||
# http://pypi.python.org/pypi?%3Aaction=list_classifiers
|
||||
classifiers = [
|
||||
"Intended Audience :: Developers",
|
||||
"Intended Audience :: System Administrators",
|
||||
"License :: OSI Approved :: BSD License",
|
||||
"Operating System :: OS Independent",
|
||||
"Programming Language :: Python",
|
||||
"Programming Language :: Python :: 2.7",
|
||||
"Programming Language :: Python :: 3.4",
|
||||
"Topic :: Software Development :: Libraries :: Python Modules",
|
||||
"Topic :: Database"
|
||||
],
|
||||
|
||||
install_requires = [
|
||||
'iso8601 >= 0.1.12',
|
||||
'pytz',
|
||||
'requests',
|
||||
'setuptools'
|
||||
],
|
||||
|
||||
package_dir = {'': 'src'},
|
||||
package_data = {'': []},
|
||||
include_package_data = True,
|
||||
zip_safe = False,
|
||||
|
||||
entry_points = dict(
|
||||
console_scripts = [],
|
||||
gui_scripts = [],
|
||||
),
|
||||
)
|
||||
|
||||
if SETUP_INFO['url'] is None:
|
||||
_ = SETUP_INFO.pop('url')
|
||||
|
||||
def setup():
|
||||
from setuptools import setup as _setup
|
||||
from setuptools import find_packages
|
||||
SETUP_INFO['packages'] = find_packages('src')
|
||||
_setup(**SETUP_INFO)
|
||||
|
||||
if __name__ == '__main__':
|
||||
setup()
|
||||
@@ -0,0 +1,13 @@
|
||||
__import__("pkg_resources").declare_namespace(__name__)
|
||||
|
||||
from datastore_orm.database import *
|
||||
from datastore_orm.engines import *
|
||||
from datastore_orm.fields import *
|
||||
from datastore_orm.funcs import *
|
||||
from datastore_orm.migrations import *
|
||||
from datastore_orm.models import *
|
||||
from datastore_orm.query import *
|
||||
from datastore_orm.system_models import *
|
||||
|
||||
from inspect import isclass
|
||||
__all__ = [c.__name__ for c in locals().values() if isclass(c)]
|
||||
@@ -0,0 +1,452 @@
|
||||
from __future__ import unicode_literals
|
||||
|
||||
import re
|
||||
import requests
|
||||
from collections import namedtuple
|
||||
from .models import ModelBase
|
||||
from .utils import escape, parse_tsv, import_submodules
|
||||
from math import ceil
|
||||
import datetime
|
||||
from string import Template
|
||||
import pytz
|
||||
|
||||
import logging
|
||||
logger = logging.getLogger('clickhouse_orm')
|
||||
|
||||
|
||||
Page = namedtuple('Page', 'objects number_of_objects pages_total number page_size')
|
||||
|
||||
|
||||
class DatabaseException(Exception):
|
||||
'''
|
||||
Raised when a database operation fails.
|
||||
'''
|
||||
pass
|
||||
|
||||
|
||||
class ServerError(DatabaseException):
|
||||
"""
|
||||
Raised when a server returns an error.
|
||||
"""
|
||||
def __init__(self, message):
|
||||
self.code = None
|
||||
processed = self.get_error_code_msg(message)
|
||||
if processed:
|
||||
self.code, self.message = processed
|
||||
else:
|
||||
# just skip custom init
|
||||
# if non-standard message format
|
||||
self.message = message
|
||||
super(ServerError, self).__init__(message)
|
||||
|
||||
ERROR_PATTERNS = (
|
||||
# ClickHouse prior to v19.3.3
|
||||
re.compile(r'''
|
||||
Code:\ (?P<code>\d+),
|
||||
\ e\.displayText\(\)\ =\ (?P<type1>[^ \n]+):\ (?P<msg>.+?),
|
||||
\ e.what\(\)\ =\ (?P<type2>[^ \n]+)
|
||||
''', re.VERBOSE | re.DOTALL),
|
||||
# ClickHouse v19.3.3+
|
||||
re.compile(r'''
|
||||
Code:\ (?P<code>\d+),
|
||||
\ e\.displayText\(\)\ =\ (?P<type1>[^ \n]+):\ (?P<msg>.+)
|
||||
''', re.VERBOSE | re.DOTALL),
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def get_error_code_msg(cls, full_error_message):
|
||||
"""
|
||||
Extract the code and message of the exception that clickhouse-server generated.
|
||||
|
||||
See the list of error codes here:
|
||||
https://github.com/yandex/ClickHouse/blob/master/dbms/src/Common/ErrorCodes.cpp
|
||||
"""
|
||||
for pattern in cls.ERROR_PATTERNS:
|
||||
match = pattern.match(full_error_message)
|
||||
if match:
|
||||
# assert match.group('type1') == match.group('type2')
|
||||
return int(match.group('code')), match.group('msg').strip()
|
||||
|
||||
return 0, full_error_message
|
||||
|
||||
def __str__(self):
|
||||
if self.code is not None:
|
||||
return "{} ({})".format(self.message, self.code)
|
||||
|
||||
|
||||
class Database(object):
|
||||
'''
|
||||
Database instances connect to a specific ClickHouse database for running queries,
|
||||
inserting data and other operations.
|
||||
'''
|
||||
|
||||
def __init__(self, db_name, db_url='http://localhost:8123/',
|
||||
username=None, password=None, cluster=None,
|
||||
readonly=False, autocreate=True,
|
||||
timeout=60, verify_ssl_cert=True, log_statements=False, *, randomize_replica_paths=False,
|
||||
trust_env=True):
|
||||
'''
|
||||
Initializes a database instance. Unless it's readonly, the database will be
|
||||
created on the ClickHouse server if it does not already exist.
|
||||
|
||||
- `db_name`: name of the database to connect to.
|
||||
- `db_url`: URL of the ClickHouse server.
|
||||
- `username`: optional connection credentials.
|
||||
- `password`: optional connection credentials.
|
||||
- `cluster`: optional cluster to create tables on
|
||||
- `readonly`: use a read-only connection.
|
||||
- `autocreate`: automatically create the database if it does not exist (unless in readonly mode).
|
||||
- `timeout`: the connection timeout in seconds.
|
||||
- `verify_ssl_cert`: whether to verify the server's certificate when connecting via HTTPS.
|
||||
- `log_statements`: when True, all database statements are logged.
|
||||
- `randomize_replica_paths`: when True, a random integer is appended to table replica paths.
|
||||
This way replicated tables (such as `MigrationHistoryReplicated`) can be dropped and recreated without causing
|
||||
a conflict in Zookeeper. This shouldn't be used in production though.
|
||||
- `trust_env`: when True, the request session will use environment variables for proxy configuration etc.
|
||||
'''
|
||||
self.db_name = db_name
|
||||
self.db_url = db_url
|
||||
self.cluster = cluster
|
||||
self.readonly = False
|
||||
self.timeout = timeout
|
||||
self.request_session = requests.Session()
|
||||
self.request_session.verify = verify_ssl_cert
|
||||
self.request_session.trust_env = trust_env
|
||||
if username:
|
||||
self.request_session.auth = (username, password or '')
|
||||
self.log_statements = log_statements
|
||||
self.randomize_replica_paths = randomize_replica_paths
|
||||
self.settings = {}
|
||||
self.db_exists = False # this is required before running _is_existing_database
|
||||
self.db_exists = self._is_existing_database()
|
||||
if readonly:
|
||||
if not self.db_exists:
|
||||
raise DatabaseException('Database does not exist, and cannot be created under readonly connection')
|
||||
self.connection_readonly = self._is_connection_readonly()
|
||||
self.readonly = True
|
||||
elif autocreate and not self.db_exists:
|
||||
self.create_database()
|
||||
self.server_version = self._get_server_version()
|
||||
# Versions 1.1.53981 and below don't have timezone function
|
||||
self.server_timezone = self._get_server_timezone() if self.server_version > (1, 1, 53981) else pytz.utc
|
||||
# Versions 19.1.16 and above support codec compression
|
||||
self.has_codec_support = self.server_version >= (19, 1, 16)
|
||||
# Version 19.0 and above support LowCardinality
|
||||
self.has_low_cardinality_support = self.server_version >= (19, 0)
|
||||
|
||||
def create_database(self):
|
||||
'''
|
||||
Creates the database on the ClickHouse server if it does not already exist.
|
||||
'''
|
||||
self._send('CREATE DATABASE IF NOT EXISTS `%s`' % self.db_name)
|
||||
self.db_exists = True
|
||||
|
||||
def drop_database(self):
|
||||
'''
|
||||
Deletes the database on the ClickHouse server.
|
||||
'''
|
||||
self._send('DROP DATABASE `%s`' % self.db_name)
|
||||
self.db_exists = False
|
||||
|
||||
def create_table(self, model_class):
|
||||
'''
|
||||
Creates a table for the given model class, if it does not exist already.
|
||||
'''
|
||||
if model_class.is_system_model():
|
||||
raise DatabaseException("You can't create system table")
|
||||
if getattr(model_class, 'engine') is None:
|
||||
raise DatabaseException("%s class must define an engine" % model_class.__name__)
|
||||
self._send(model_class.create_table_sql(self))
|
||||
|
||||
def drop_table(self, model_class):
|
||||
'''
|
||||
Drops the database table of the given model class, if it exists.
|
||||
'''
|
||||
if model_class.is_system_model():
|
||||
raise DatabaseException("You can't drop system table")
|
||||
self._send(model_class.drop_table_sql(self))
|
||||
|
||||
def does_table_exist(self, model_class):
|
||||
'''
|
||||
Checks whether a table for the given model class already exists.
|
||||
Note that this only checks for existence of a table with the expected name.
|
||||
'''
|
||||
sql = "SELECT count() FROM system.tables WHERE database = '%s' AND name = '%s'"
|
||||
r = self._send(sql % (self.db_name, model_class.table_name()))
|
||||
return r.text.strip() == '1'
|
||||
|
||||
def get_model_for_table(self, table_name, system_table=False):
|
||||
'''
|
||||
Generates a model class from an existing table in the database.
|
||||
This can be used for querying tables which don't have a corresponding model class,
|
||||
for example system tables.
|
||||
|
||||
- `table_name`: the table to create a model for
|
||||
- `system_table`: whether the table is a system table, or belongs to the current database
|
||||
'''
|
||||
db_name = 'system' if system_table else self.db_name
|
||||
sql = "DESCRIBE `%s`.`%s` FORMAT TSV" % (db_name, table_name)
|
||||
lines = self._send(sql).iter_lines()
|
||||
fields = [parse_tsv(line)[:2] for line in lines]
|
||||
model = ModelBase.create_ad_hoc_model(fields, table_name)
|
||||
if system_table:
|
||||
model._system = model._readonly = True
|
||||
return model
|
||||
|
||||
def add_setting(self, name, value):
|
||||
'''
|
||||
Adds a database setting that will be sent with every request.
|
||||
For example, `db.add_setting("max_execution_time", 10)` will
|
||||
limit query execution time to 10 seconds.
|
||||
The name must be string, and the value is converted to string in case
|
||||
it isn't. To remove a setting, pass `None` as the value.
|
||||
'''
|
||||
assert isinstance(name, str), 'Setting name must be a string'
|
||||
if value is None:
|
||||
self.settings.pop(name, None)
|
||||
else:
|
||||
self.settings[name] = str(value)
|
||||
|
||||
def insert(self, model_instances, batch_size=1000):
|
||||
'''
|
||||
Insert records into the database.
|
||||
|
||||
- `model_instances`: any iterable containing instances of a single model class.
|
||||
- `batch_size`: number of records to send per chunk (use a lower number if your records are very large).
|
||||
'''
|
||||
from io import BytesIO
|
||||
i = iter(model_instances)
|
||||
try:
|
||||
first_instance = next(i)
|
||||
except StopIteration:
|
||||
return # model_instances is empty
|
||||
model_class = first_instance.__class__
|
||||
|
||||
if first_instance.is_read_only() or first_instance.is_system_model():
|
||||
raise DatabaseException("You can't insert into read only and system tables")
|
||||
|
||||
fields_list = ','.join(
|
||||
['`%s`' % name for name in first_instance.fields(writable=True)])
|
||||
fmt = 'TSKV' if model_class.has_funcs_as_defaults() else 'TabSeparated'
|
||||
query = 'INSERT INTO $table (%s) FORMAT %s\n' % (fields_list, fmt)
|
||||
|
||||
def gen():
|
||||
buf = BytesIO()
|
||||
buf.write(self._substitute(query, model_class).encode('utf-8'))
|
||||
first_instance.set_database(self)
|
||||
buf.write(first_instance.to_db_string())
|
||||
# Collect lines in batches of batch_size
|
||||
lines = 2
|
||||
for instance in i:
|
||||
instance.set_database(self)
|
||||
buf.write(instance.to_db_string())
|
||||
lines += 1
|
||||
if lines >= batch_size:
|
||||
# Return the current batch of lines
|
||||
yield buf.getvalue()
|
||||
# Start a new batch
|
||||
buf = BytesIO()
|
||||
lines = 0
|
||||
# Return any remaining lines in partial batch
|
||||
if lines:
|
||||
yield buf.getvalue()
|
||||
self._send(gen())
|
||||
|
||||
def count(self, model_class, conditions=None):
|
||||
'''
|
||||
Counts the number of records in the model's table.
|
||||
|
||||
- `model_class`: the model to count.
|
||||
- `conditions`: optional SQL conditions (contents of the WHERE clause).
|
||||
'''
|
||||
from datastore_orm.query import Q
|
||||
query = 'SELECT count() FROM $table'
|
||||
if conditions:
|
||||
if isinstance(conditions, Q):
|
||||
conditions = conditions.to_sql(model_class)
|
||||
query += ' WHERE ' + str(conditions)
|
||||
query = self._substitute(query, model_class)
|
||||
r = self._send(query)
|
||||
return int(r.text) if r.text else 0
|
||||
|
||||
def select(self, query, model_class=None, settings=None):
|
||||
'''
|
||||
Performs a query and returns a generator of model instances.
|
||||
|
||||
- `query`: the SQL query to execute.
|
||||
- `model_class`: the model class matching the query's table,
|
||||
or `None` for getting back instances of an ad-hoc model.
|
||||
- `settings`: query settings to send as HTTP GET parameters
|
||||
'''
|
||||
query += ' FORMAT TabSeparatedWithNamesAndTypes'
|
||||
query = self._substitute(query, model_class)
|
||||
r = self._send(query, settings, True)
|
||||
lines = r.iter_lines()
|
||||
field_names = parse_tsv(next(lines))
|
||||
field_types = parse_tsv(next(lines))
|
||||
model_class = model_class or ModelBase.create_ad_hoc_model(zip(field_names, field_types))
|
||||
for line in lines:
|
||||
# skip blank line left by WITH TOTALS modifier
|
||||
if line:
|
||||
yield model_class.from_tsv(line, field_names, self.server_timezone, self)
|
||||
|
||||
def raw(self, query, settings=None, stream=False):
|
||||
'''
|
||||
Performs a query and returns its output as text.
|
||||
|
||||
- `query`: the SQL query to execute.
|
||||
- `settings`: query settings to send as HTTP GET parameters
|
||||
- `stream`: if true, the HTTP response from ClickHouse will be streamed.
|
||||
'''
|
||||
query = self._substitute(query, None)
|
||||
return self._send(query, settings=settings, stream=stream).text
|
||||
|
||||
def paginate(self, model_class, order_by, page_num=1, page_size=100, conditions=None, settings=None):
|
||||
'''
|
||||
Selects records and returns a single page of model instances.
|
||||
|
||||
- `model_class`: the model class matching the query's table,
|
||||
or `None` for getting back instances of an ad-hoc model.
|
||||
- `order_by`: columns to use for sorting the query (contents of the ORDER BY clause).
|
||||
- `page_num`: the page number (1-based), or -1 to get the last page.
|
||||
- `page_size`: number of records to return per page.
|
||||
- `conditions`: optional SQL conditions (contents of the WHERE clause).
|
||||
- `settings`: query settings to send as HTTP GET parameters
|
||||
|
||||
The result is a namedtuple containing `objects` (list), `number_of_objects`,
|
||||
`pages_total`, `number` (of the current page), and `page_size`.
|
||||
'''
|
||||
from datastore_orm.query import Q
|
||||
count = self.count(model_class, conditions)
|
||||
pages_total = int(ceil(count / float(page_size)))
|
||||
if page_num == -1:
|
||||
page_num = max(pages_total, 1)
|
||||
elif page_num < 1:
|
||||
raise ValueError('Invalid page number: %d' % page_num)
|
||||
offset = (page_num - 1) * page_size
|
||||
query = 'SELECT * FROM $table'
|
||||
if conditions:
|
||||
if isinstance(conditions, Q):
|
||||
conditions = conditions.to_sql(model_class)
|
||||
query += ' WHERE ' + str(conditions)
|
||||
query += ' ORDER BY %s' % order_by
|
||||
query += ' LIMIT %d, %d' % (offset, page_size)
|
||||
query = self._substitute(query, model_class)
|
||||
return Page(
|
||||
objects=list(self.select(query, model_class, settings)) if count else [],
|
||||
number_of_objects=count,
|
||||
pages_total=pages_total,
|
||||
number=page_num,
|
||||
page_size=page_size
|
||||
)
|
||||
|
||||
def migrate(self, migrations_package_name, up_to=9999, replicated=False):
|
||||
'''
|
||||
Executes schema migrations.
|
||||
|
||||
- `migrations_package_name` - fully qualified name of the Python package
|
||||
containing the migrations.
|
||||
- `up_to` - number of the last migration to apply.
|
||||
'''
|
||||
from .migrations import MigrationHistory
|
||||
logger = logging.getLogger('migrations')
|
||||
applied_migrations = self._get_applied_migrations_and_create_tables(migrations_package_name, replicated=replicated)
|
||||
modules = import_submodules(migrations_package_name)
|
||||
unapplied_migrations = set(modules.keys()) - applied_migrations
|
||||
for name in sorted(unapplied_migrations):
|
||||
logger.info('Applying migration %s...', name)
|
||||
for operation in modules[name].operations:
|
||||
operation.apply(self)
|
||||
self.insert([MigrationHistory(package_name=migrations_package_name, module_name=name, applied=datetime.date.today())])
|
||||
if int(name[:4]) >= up_to:
|
||||
break
|
||||
|
||||
def _get_applied_migrations_and_create_tables(self, migrations_package_name, replicated, allow_missing_tables=True):
|
||||
if not allow_missing_tables:
|
||||
return self._get_applied_migrations(migrations_package_name, replicated)
|
||||
|
||||
try:
|
||||
return self._get_applied_migrations(migrations_package_name, replicated)
|
||||
except ServerError:
|
||||
from .migrations import MigrationHistory, MigrationHistoryReplicated, MigrationHistoryDistributed
|
||||
if replicated:
|
||||
self.create_table(MigrationHistoryReplicated)
|
||||
self.create_table(MigrationHistoryDistributed)
|
||||
else:
|
||||
self.create_table(MigrationHistory)
|
||||
|
||||
return self._get_applied_migrations_and_create_tables(migrations_package_name, replicated, allow_missing_tables=False)
|
||||
|
||||
|
||||
def _get_applied_migrations(self, migrations_package_name, replicated):
|
||||
from .migrations import MigrationHistory, MigrationHistoryDistributed
|
||||
query = "SELECT DISTINCT module_name FROM $table WHERE package_name = '%s'" % migrations_package_name
|
||||
query = self._substitute(query, MigrationHistoryDistributed if replicated else MigrationHistory)
|
||||
|
||||
return set(obj.module_name for obj in self.select(query))
|
||||
|
||||
def _send(self, data, settings=None, stream=False):
|
||||
if isinstance(data, str):
|
||||
data = data.encode('utf-8')
|
||||
if self.log_statements:
|
||||
logger.info(data)
|
||||
params = self._build_params(settings)
|
||||
r = self.request_session.post(self.db_url, params=params, data=data, stream=stream, timeout=self.timeout)
|
||||
if r.status_code != 200:
|
||||
raise ServerError(r.text)
|
||||
return r
|
||||
|
||||
def _build_params(self, settings):
|
||||
params = dict(settings or {})
|
||||
params.update(self.settings)
|
||||
if self.db_exists:
|
||||
params['database'] = self.db_name
|
||||
# Send the readonly flag, unless the connection is already readonly (to prevent db error)
|
||||
if self.readonly and not self.connection_readonly:
|
||||
params['readonly'] = '1'
|
||||
return params
|
||||
|
||||
def _substitute(self, query, model_class=None):
|
||||
'''
|
||||
Replaces $db and $table placeholders in the query.
|
||||
'''
|
||||
if '$' in query:
|
||||
mapping = dict(db="`%s`" % self.db_name)
|
||||
if model_class:
|
||||
if model_class.is_system_model():
|
||||
mapping['table'] = "`system`.`%s`" % model_class.table_name()
|
||||
else:
|
||||
mapping['table'] = "`%s`.`%s`" % (self.db_name, model_class.table_name())
|
||||
query = Template(query).safe_substitute(mapping)
|
||||
return query
|
||||
|
||||
def _get_server_timezone(self):
|
||||
try:
|
||||
r = self._send('SELECT timezone()')
|
||||
return pytz.timezone(r.text.strip())
|
||||
except ServerError as e:
|
||||
logger.exception('Cannot determine server timezone (%s), assuming UTC', e)
|
||||
return pytz.utc
|
||||
|
||||
def _get_server_version(self, as_tuple=True):
|
||||
try:
|
||||
r = self._send('SELECT version();')
|
||||
ver = r.text
|
||||
except ServerError as e:
|
||||
logger.exception('Cannot determine server version (%s), assuming 1.1.0', e)
|
||||
ver = '1.1.0'
|
||||
# :TRICKY: Altinity cloud uses a non-numeric suffix for the version, which this removes.
|
||||
ver = re.sub(r"[.\D]+$", '', ver)
|
||||
return tuple(int(n) for n in ver.split('.')) if as_tuple else ver
|
||||
|
||||
def _is_existing_database(self):
|
||||
r = self._send("SELECT count() FROM system.databases WHERE name = '%s'" % self.db_name)
|
||||
return r.text.strip() == '1'
|
||||
|
||||
def _is_connection_readonly(self):
|
||||
r = self._send("SELECT value FROM system.settings WHERE name = 'readonly'")
|
||||
return r.text.strip() != '0'
|
||||
|
||||
|
||||
# Expose only relevant classes in import *
|
||||
__all__ = [c.__name__ for c in [Page, DatabaseException, ServerError, Database]]
|
||||
@@ -0,0 +1,277 @@
|
||||
from __future__ import unicode_literals
|
||||
|
||||
import logging
|
||||
import random
|
||||
|
||||
from .utils import comma_join, get_subclass_names
|
||||
|
||||
logger = logging.getLogger('clickhouse_orm')
|
||||
|
||||
|
||||
class Engine(object):
|
||||
|
||||
def create_table_sql(self, db):
|
||||
raise NotImplementedError() # pragma: no cover
|
||||
|
||||
|
||||
class TinyLog(Engine):
|
||||
|
||||
def create_table_sql(self, db):
|
||||
return 'TinyLog'
|
||||
|
||||
|
||||
class Log(Engine):
|
||||
|
||||
def create_table_sql(self, db):
|
||||
return 'Log'
|
||||
|
||||
|
||||
class Memory(Engine):
|
||||
|
||||
def create_table_sql(self, db):
|
||||
return 'Memory'
|
||||
|
||||
|
||||
class MergeTree(Engine):
|
||||
|
||||
def __init__(self, date_col=None, order_by=(), sampling_expr=None,
|
||||
index_granularity=8192, replica_table_path=None, replica_name=None, partition_key=None,
|
||||
primary_key=None):
|
||||
assert type(order_by) in (list, tuple), 'order_by must be a list or tuple'
|
||||
assert date_col is None or isinstance(date_col, str), 'date_col must be string if present'
|
||||
assert primary_key is None or type(primary_key) in (list, tuple), 'primary_key must be a list or tuple'
|
||||
assert partition_key is None or type(partition_key) in (list, tuple),\
|
||||
'partition_key must be tuple or list if present'
|
||||
assert (replica_table_path is None) == (replica_name is None), \
|
||||
'both replica_table_path and replica_name must be specified'
|
||||
|
||||
# These values conflict with each other (old and new syntax of table engines.
|
||||
# So let's control only one of them is given.
|
||||
assert date_col or partition_key, "You must set either date_col or partition_key"
|
||||
self.date_col = date_col
|
||||
self.partition_key = partition_key if partition_key else ('toYYYYMM(`%s`)' % date_col,)
|
||||
self.primary_key = primary_key
|
||||
|
||||
self.order_by = order_by
|
||||
self.sampling_expr = sampling_expr
|
||||
self.index_granularity = index_granularity
|
||||
self.replica_table_path = replica_table_path
|
||||
self.replica_name = replica_name
|
||||
|
||||
# I changed field name for new reality and syntax
|
||||
@property
|
||||
def key_cols(self):
|
||||
logger.warning('`key_cols` attribute is deprecated and may be removed in future. Use `order_by` attribute instead')
|
||||
return self.order_by
|
||||
|
||||
@key_cols.setter
|
||||
def key_cols(self, value):
|
||||
logger.warning('`key_cols` attribute is deprecated and may be removed in future. Use `order_by` attribute instead')
|
||||
self.order_by = value
|
||||
|
||||
def create_table_sql(self, db):
|
||||
name = self.__class__.__name__
|
||||
if self.replica_name:
|
||||
name = 'Replicated' + name
|
||||
|
||||
# In ClickHouse 1.1.54310 custom partitioning key was introduced
|
||||
# https://clickhouse.tech/docs/en/table_engines/custom_partitioning_key/
|
||||
# Let's check version and use new syntax if available
|
||||
if db.server_version >= (1, 1, 54310):
|
||||
partition_sql = "PARTITION BY (%s) ORDER BY (%s)" \
|
||||
% (comma_join(self.partition_key, stringify=True),
|
||||
comma_join(self.order_by, stringify=True))
|
||||
|
||||
if self.primary_key:
|
||||
partition_sql += " PRIMARY KEY (%s)" % comma_join(self.primary_key, stringify=True)
|
||||
|
||||
if self.sampling_expr:
|
||||
partition_sql += " SAMPLE BY %s" % self.sampling_expr
|
||||
|
||||
partition_sql += " SETTINGS index_granularity=%d" % self.index_granularity
|
||||
|
||||
elif not self.date_col:
|
||||
# Can't import it globally due to circular import
|
||||
from datastore_orm.database import DatabaseException
|
||||
raise DatabaseException("Custom partitioning is not supported before ClickHouse 1.1.54310. "
|
||||
"Please update your server or use date_col syntax."
|
||||
"https://clickhouse.tech/docs/en/table_engines/custom_partitioning_key/")
|
||||
else:
|
||||
partition_sql = ''
|
||||
|
||||
params = self._build_sql_params(db)
|
||||
return '%s(%s) %s' % (name, comma_join(params), partition_sql)
|
||||
|
||||
def _build_sql_params(self, db):
|
||||
params = []
|
||||
if self.replica_name:
|
||||
final_replica_table_path = self.replica_table_path
|
||||
if db.randomize_replica_paths:
|
||||
final_replica_table_path += f'/{random.randint(0, 100000000)}'
|
||||
params += ["'%s'" % final_replica_table_path, "'%s'" % self.replica_name]
|
||||
|
||||
# In ClickHouse 1.1.54310 custom partitioning key was introduced
|
||||
# https://clickhouse.tech/docs/en/table_engines/custom_partitioning_key/
|
||||
# These parameters are process in create_table_sql directly.
|
||||
# In previous ClickHouse versions this this syntax does not work.
|
||||
if db.server_version < (1, 1, 54310):
|
||||
params.append(self.date_col)
|
||||
if self.sampling_expr:
|
||||
params.append(self.sampling_expr)
|
||||
params.append('(%s)' % comma_join(self.order_by, stringify=True))
|
||||
params.append(str(self.index_granularity))
|
||||
|
||||
return params
|
||||
|
||||
|
||||
class CollapsingMergeTree(MergeTree):
|
||||
|
||||
def __init__(self, date_col=None, order_by=(), sign_col='sign', sampling_expr=None,
|
||||
index_granularity=8192, replica_table_path=None, replica_name=None, partition_key=None,
|
||||
primary_key=None):
|
||||
super(CollapsingMergeTree, self).__init__(date_col, order_by, sampling_expr, index_granularity,
|
||||
replica_table_path, replica_name, partition_key, primary_key)
|
||||
self.sign_col = sign_col
|
||||
|
||||
def _build_sql_params(self, db):
|
||||
params = super(CollapsingMergeTree, self)._build_sql_params(db)
|
||||
params.append(self.sign_col)
|
||||
return params
|
||||
|
||||
|
||||
class SummingMergeTree(MergeTree):
|
||||
|
||||
def __init__(self, date_col=None, order_by=(), summing_cols=None, sampling_expr=None,
|
||||
index_granularity=8192, replica_table_path=None, replica_name=None, partition_key=None,
|
||||
primary_key=None):
|
||||
super(SummingMergeTree, self).__init__(date_col, order_by, sampling_expr, index_granularity, replica_table_path,
|
||||
replica_name, partition_key, primary_key)
|
||||
assert type is None or type(summing_cols) in (list, tuple), 'summing_cols must be a list or tuple'
|
||||
self.summing_cols = summing_cols
|
||||
|
||||
def _build_sql_params(self, db):
|
||||
params = super(SummingMergeTree, self)._build_sql_params(db)
|
||||
if self.summing_cols:
|
||||
params.append('(%s)' % comma_join(self.summing_cols))
|
||||
return params
|
||||
|
||||
|
||||
class ReplacingMergeTree(MergeTree):
|
||||
|
||||
def __init__(self, date_col=None, order_by=(), ver_col=None, sampling_expr=None,
|
||||
index_granularity=8192, replica_table_path=None, replica_name=None, partition_key=None,
|
||||
primary_key=None):
|
||||
super(ReplacingMergeTree, self).__init__(date_col, order_by, sampling_expr, index_granularity,
|
||||
replica_table_path, replica_name, partition_key, primary_key)
|
||||
self.ver_col = ver_col
|
||||
|
||||
def _build_sql_params(self, db):
|
||||
params = super(ReplacingMergeTree, self)._build_sql_params(db)
|
||||
if self.ver_col:
|
||||
params.append(self.ver_col)
|
||||
return params
|
||||
|
||||
|
||||
class Buffer(Engine):
|
||||
"""
|
||||
Buffers the data to write in RAM, periodically flushing it to another table.
|
||||
Must be used in conjuction with a `BufferModel`.
|
||||
Read more [here](https://clickhouse.tech/docs/en/engines/table-engines/special/buffer/).
|
||||
"""
|
||||
|
||||
#Buffer(database, table, num_layers, min_time, max_time, min_rows, max_rows, min_bytes, max_bytes)
|
||||
def __init__(self, main_model, num_layers=16, min_time=10, max_time=100, min_rows=10000, max_rows=1000000,
|
||||
min_bytes=10000000, max_bytes=100000000):
|
||||
self.main_model = main_model
|
||||
self.num_layers = num_layers
|
||||
self.min_time = min_time
|
||||
self.max_time = max_time
|
||||
self.min_rows = min_rows
|
||||
self.max_rows = max_rows
|
||||
self.min_bytes = min_bytes
|
||||
self.max_bytes = max_bytes
|
||||
|
||||
def create_table_sql(self, db):
|
||||
# Overriden create_table_sql example:
|
||||
# sql = 'ENGINE = Buffer(merge, hits, 16, 10, 100, 10000, 1000000, 10000000, 100000000)'
|
||||
sql = 'ENGINE = Buffer(`%s`, `%s`, %d, %d, %d, %d, %d, %d, %d)' % (
|
||||
db.db_name, self.main_model.table_name(), self.num_layers,
|
||||
self.min_time, self.max_time, self.min_rows,
|
||||
self.max_rows, self.min_bytes, self.max_bytes
|
||||
)
|
||||
return sql
|
||||
|
||||
|
||||
class Merge(Engine):
|
||||
"""
|
||||
The Merge engine (not to be confused with MergeTree) does not store data itself,
|
||||
but allows reading from any number of other tables simultaneously.
|
||||
Writing to a table is not supported
|
||||
https://clickhouse.tech/docs/en/engines/table-engines/special/merge/
|
||||
"""
|
||||
|
||||
def __init__(self, table_regex):
|
||||
assert isinstance(table_regex, str), "'table_regex' parameter must be string"
|
||||
self.table_regex = table_regex
|
||||
|
||||
def create_table_sql(self, db):
|
||||
return "Merge(`%s`, '%s')" % (db.db_name, self.table_regex)
|
||||
|
||||
|
||||
class Distributed(Engine):
|
||||
"""
|
||||
The Distributed engine by itself does not store data,
|
||||
but allows distributed query processing on multiple servers.
|
||||
Reading is automatically parallelized.
|
||||
During a read, the table indexes on remote servers are used, if there are any.
|
||||
|
||||
See full documentation here
|
||||
https://clickhouse.tech/docs/en/engines/table-engines/special/distributed/
|
||||
"""
|
||||
def __init__(self, cluster=None, table=None, sharding_key=None):
|
||||
"""
|
||||
- `cluster`: what cluster to access data from. Defaults to db.cluster
|
||||
- `table`: underlying table that actually stores data.
|
||||
If you are not specifying any table here, ensure that it can be inferred
|
||||
from your model's superclass (see models.DistributedModel.fix_engine_table)
|
||||
- `sharding_key`: how to distribute data among shards when inserting
|
||||
straightly into Distributed table, optional
|
||||
"""
|
||||
self.cluster = cluster
|
||||
self.table = table
|
||||
self.sharding_key = sharding_key
|
||||
|
||||
@property
|
||||
def table_name(self):
|
||||
# TODO: circular import is bad
|
||||
from .models import ModelBase
|
||||
|
||||
table = self.table
|
||||
|
||||
if isinstance(table, ModelBase):
|
||||
return table.table_name()
|
||||
|
||||
return table
|
||||
|
||||
def create_table_sql(self, db):
|
||||
name = self.__class__.__name__
|
||||
params = self._build_sql_params(db)
|
||||
return '%s(%s)' % (name, ', '.join(params))
|
||||
|
||||
def _build_sql_params(self, db):
|
||||
if self.table_name is None:
|
||||
raise ValueError("Cannot create {} engine: specify an underlying table".format(
|
||||
self.__class__.__name__))
|
||||
|
||||
if self.cluster is None and db.cluster is None:
|
||||
raise ValueError("Cannot create engine: specify a cluster")
|
||||
|
||||
cluster = self.cluster if self.cluster is not None else db.cluster
|
||||
params = ["`%s`" % p for p in [cluster, db.db_name, self.table_name]]
|
||||
if self.sharding_key:
|
||||
params.append(self.sharding_key)
|
||||
return params
|
||||
|
||||
|
||||
# Expose only relevant classes in import *
|
||||
__all__ = get_subclass_names(locals(), Engine)
|
||||
@@ -0,0 +1,674 @@
|
||||
from __future__ import unicode_literals
|
||||
import datetime
|
||||
import iso8601
|
||||
import pytz
|
||||
from calendar import timegm
|
||||
from decimal import Decimal, localcontext
|
||||
from uuid import UUID
|
||||
from logging import getLogger
|
||||
from pytz import BaseTzInfo
|
||||
from .utils import escape, parse_array, comma_join, string_or_func, get_subclass_names
|
||||
from .funcs import F, FunctionOperatorsMixin
|
||||
from ipaddress import IPv4Address, IPv6Address
|
||||
|
||||
logger = getLogger('clickhouse_orm')
|
||||
|
||||
|
||||
class Field(FunctionOperatorsMixin):
|
||||
'''
|
||||
Abstract base class for all field types.
|
||||
'''
|
||||
name = None # this is set by the parent model
|
||||
parent = None # this is set by the parent model
|
||||
creation_counter = 0 # used for keeping the model fields ordered
|
||||
class_default = 0 # should be overridden by concrete subclasses
|
||||
db_type = None # should be overridden by concrete subclasses
|
||||
|
||||
def __init__(self, default=None, alias=None, materialized=None, readonly=None, codec=None):
|
||||
assert [default, alias, materialized].count(None) >= 2, \
|
||||
"Only one of default, alias and materialized parameters can be given"
|
||||
assert alias is None or isinstance(alias, F) or isinstance(alias, str) and alias != "",\
|
||||
"Alias parameter must be a string or function object, if given"
|
||||
assert materialized is None or isinstance(materialized, F) or isinstance(materialized, str) and materialized != "",\
|
||||
"Materialized parameter must be a string or function object, if given"
|
||||
assert readonly is None or type(readonly) is bool, "readonly parameter must be bool if given"
|
||||
assert codec is None or isinstance(codec, str) and codec != "", \
|
||||
"Codec field must be string, if given"
|
||||
|
||||
self.creation_counter = Field.creation_counter
|
||||
Field.creation_counter += 1
|
||||
self.default = self.class_default if default is None else default
|
||||
self.alias = alias
|
||||
self.materialized = materialized
|
||||
self.readonly = bool(self.alias or self.materialized or readonly)
|
||||
self.codec = codec
|
||||
|
||||
def __str__(self):
|
||||
return self.name
|
||||
|
||||
def __repr__(self):
|
||||
return '<%s>' % self.__class__.__name__
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
'''
|
||||
Converts the input value into the expected Python data type, raising ValueError if the
|
||||
data can't be converted. Returns the converted value. Subclasses should override this.
|
||||
The timezone_in_use parameter should be consulted when parsing datetime fields.
|
||||
'''
|
||||
return value # pragma: no cover
|
||||
|
||||
def validate(self, value):
|
||||
'''
|
||||
Called after to_python to validate that the value is suitable for the field's database type.
|
||||
Subclasses should override this.
|
||||
'''
|
||||
pass
|
||||
|
||||
def _range_check(self, value, min_value, max_value):
|
||||
'''
|
||||
Utility method to check that the given value is between min_value and max_value.
|
||||
'''
|
||||
if value < min_value or value > max_value:
|
||||
raise ValueError('%s out of range - %s is not between %s and %s' % (self.__class__.__name__, value, min_value, max_value))
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
'''
|
||||
Returns the field's value prepared for writing to the database.
|
||||
When quote is true, strings are surrounded by single quotes.
|
||||
'''
|
||||
return escape(value, quote)
|
||||
|
||||
def get_sql(self, with_default_expression=True, db=None):
|
||||
'''
|
||||
Returns an SQL expression describing the field (e.g. for CREATE TABLE).
|
||||
|
||||
- `with_default_expression`: If True, adds default value to sql.
|
||||
It doesn't affect fields with alias and materialized values.
|
||||
- `db`: Database, used for checking supported features.
|
||||
'''
|
||||
sql = self.db_type
|
||||
args = self.get_db_type_args()
|
||||
if args:
|
||||
sql += '(%s)' % comma_join(args)
|
||||
if with_default_expression:
|
||||
sql += self._extra_params(db)
|
||||
return sql
|
||||
|
||||
def get_db_type_args(self):
|
||||
"""Returns field type arguments"""
|
||||
return []
|
||||
|
||||
def _extra_params(self, db):
|
||||
sql = ''
|
||||
if self.alias:
|
||||
sql += ' ALIAS %s' % string_or_func(self.alias)
|
||||
elif self.materialized:
|
||||
sql += ' MATERIALIZED %s' % string_or_func(self.materialized)
|
||||
elif isinstance(self.default, F):
|
||||
sql += ' DEFAULT %s' % self.default.to_sql()
|
||||
elif self.default:
|
||||
default = self.to_db_string(self.default)
|
||||
sql += ' DEFAULT %s' % default
|
||||
if self.codec and db and db.has_codec_support:
|
||||
sql += ' CODEC(%s)' % self.codec
|
||||
return sql
|
||||
|
||||
def isinstance(self, types):
|
||||
"""
|
||||
Checks if the instance if one of the types provided or if any of the inner_field child is one of the types
|
||||
provided, returns True if field or any inner_field is one of ths provided, False otherwise
|
||||
|
||||
- `types`: Iterable of types to check inclusion of instance
|
||||
|
||||
Returns: Boolean
|
||||
"""
|
||||
if isinstance(self, types):
|
||||
return True
|
||||
inner_field = getattr(self, 'inner_field', None)
|
||||
while inner_field:
|
||||
if isinstance(inner_field, types):
|
||||
return True
|
||||
inner_field = getattr(inner_field, 'inner_field', None)
|
||||
return False
|
||||
|
||||
|
||||
class StringField(Field):
|
||||
|
||||
class_default = ''
|
||||
db_type = 'String'
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, str):
|
||||
return value
|
||||
if isinstance(value, bytes):
|
||||
return value.decode('UTF-8')
|
||||
raise ValueError('Invalid value for %s: %r' % (self.__class__.__name__, value))
|
||||
|
||||
|
||||
class FixedStringField(StringField):
|
||||
|
||||
def __init__(self, length, default=None, alias=None, materialized=None, readonly=None):
|
||||
self._length = length
|
||||
self.db_type = 'FixedString(%d)' % length
|
||||
super(FixedStringField, self).__init__(default, alias, materialized, readonly)
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
value = super(FixedStringField, self).to_python(value, timezone_in_use)
|
||||
return value.rstrip('\0')
|
||||
|
||||
def validate(self, value):
|
||||
if isinstance(value, str):
|
||||
value = value.encode('UTF-8')
|
||||
if len(value) > self._length:
|
||||
raise ValueError('Value of %d bytes is too long for FixedStringField(%d)' % (len(value), self._length))
|
||||
|
||||
|
||||
class DateField(Field):
|
||||
|
||||
min_value = datetime.date(1970, 1, 1)
|
||||
max_value = datetime.date(2105, 12, 31)
|
||||
class_default = min_value
|
||||
db_type = 'Date'
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, datetime.datetime):
|
||||
return value.astimezone(pytz.utc).date() if value.tzinfo else value.date()
|
||||
if isinstance(value, datetime.date):
|
||||
return value
|
||||
if isinstance(value, int):
|
||||
return DateField.class_default + datetime.timedelta(days=value)
|
||||
if isinstance(value, str):
|
||||
if value == '0000-00-00':
|
||||
return DateField.min_value
|
||||
return datetime.datetime.strptime(value, '%Y-%m-%d').date()
|
||||
raise ValueError('Invalid value for %s - %r' % (self.__class__.__name__, value))
|
||||
|
||||
def validate(self, value):
|
||||
self._range_check(value, DateField.min_value, DateField.max_value)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
return escape(value.isoformat(), quote)
|
||||
|
||||
|
||||
class DateTimeField(Field):
|
||||
|
||||
class_default = datetime.datetime.fromtimestamp(0, pytz.utc)
|
||||
db_type = 'DateTime'
|
||||
|
||||
def __init__(self, default=None, alias=None, materialized=None, readonly=None, codec=None,
|
||||
timezone=None):
|
||||
super().__init__(default, alias, materialized, readonly, codec)
|
||||
# assert not timezone, 'Temporarily field timezone is not supported'
|
||||
if timezone:
|
||||
timezone = timezone if isinstance(timezone, BaseTzInfo) else pytz.timezone(timezone)
|
||||
self.timezone = timezone
|
||||
|
||||
def get_db_type_args(self):
|
||||
args = []
|
||||
if self.timezone:
|
||||
args.append(escape(self.timezone.zone))
|
||||
return args
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, datetime.datetime):
|
||||
return value if value.tzinfo else value.replace(tzinfo=pytz.utc)
|
||||
if isinstance(value, datetime.date):
|
||||
return datetime.datetime(value.year, value.month, value.day, tzinfo=pytz.utc)
|
||||
if isinstance(value, int):
|
||||
return datetime.datetime.utcfromtimestamp(value).replace(tzinfo=pytz.utc)
|
||||
if isinstance(value, str):
|
||||
if value == '0000-00-00 00:00:00':
|
||||
return self.class_default
|
||||
if len(value) == 10:
|
||||
try:
|
||||
value = int(value)
|
||||
return datetime.datetime.utcfromtimestamp(value).replace(tzinfo=pytz.utc)
|
||||
except ValueError:
|
||||
pass
|
||||
try:
|
||||
# left the date naive in case of no tzinfo set
|
||||
dt = iso8601.parse_date(value, default_timezone=None)
|
||||
except iso8601.ParseError as e:
|
||||
raise ValueError(str(e))
|
||||
|
||||
# convert naive to aware
|
||||
if dt.tzinfo is None or dt.tzinfo.utcoffset(dt) is None:
|
||||
dt = timezone_in_use.localize(dt)
|
||||
return dt
|
||||
raise ValueError('Invalid value for %s - %r' % (self.__class__.__name__, value))
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
return escape('%010d' % timegm(value.utctimetuple()), quote)
|
||||
|
||||
|
||||
class DateTime64Field(DateTimeField):
|
||||
db_type = 'DateTime64'
|
||||
|
||||
def __init__(self, default=None, alias=None, materialized=None, readonly=None, codec=None,
|
||||
timezone=None, precision=6):
|
||||
super().__init__(default, alias, materialized, readonly, codec, timezone)
|
||||
assert precision is None or isinstance(precision, int), 'Precision must be int type'
|
||||
self.precision = precision
|
||||
|
||||
def get_db_type_args(self):
|
||||
args = [str(self.precision)]
|
||||
if self.timezone:
|
||||
args.append(escape(self.timezone.zone))
|
||||
return args
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
"""
|
||||
Returns the field's value prepared for writing to the database
|
||||
|
||||
Returns string in 0000000000.000000 format, where remainder digits count is equal to precision
|
||||
"""
|
||||
return escape(
|
||||
'{timestamp:0{width}.{precision}f}'.format(
|
||||
timestamp=value.timestamp(),
|
||||
width=11 + self.precision,
|
||||
precision=self.precision),
|
||||
quote
|
||||
)
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
try:
|
||||
return super().to_python(value, timezone_in_use)
|
||||
except ValueError:
|
||||
if isinstance(value, (int, float)):
|
||||
return datetime.datetime.utcfromtimestamp(value).replace(tzinfo=pytz.utc)
|
||||
if isinstance(value, str):
|
||||
left_part = value.split('.')[0]
|
||||
if left_part == '0000-00-00 00:00:00':
|
||||
return self.class_default
|
||||
if len(left_part) == 10:
|
||||
try:
|
||||
value = float(value)
|
||||
return datetime.datetime.utcfromtimestamp(value).replace(tzinfo=pytz.utc)
|
||||
except ValueError:
|
||||
pass
|
||||
raise
|
||||
|
||||
|
||||
class BaseIntField(Field):
|
||||
'''
|
||||
Abstract base class for all integer-type fields.
|
||||
'''
|
||||
def to_python(self, value, timezone_in_use):
|
||||
try:
|
||||
return int(value)
|
||||
except:
|
||||
raise ValueError('Invalid value for %s - %r' % (self.__class__.__name__, value))
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
# There's no need to call escape since numbers do not contain
|
||||
# special characters, and never need quoting
|
||||
return str(value)
|
||||
|
||||
def validate(self, value):
|
||||
self._range_check(value, self.min_value, self.max_value)
|
||||
|
||||
|
||||
class UInt8Field(BaseIntField):
|
||||
|
||||
min_value = 0
|
||||
max_value = 2**8 - 1
|
||||
db_type = 'UInt8'
|
||||
|
||||
|
||||
class UInt16Field(BaseIntField):
|
||||
|
||||
min_value = 0
|
||||
max_value = 2**16 - 1
|
||||
db_type = 'UInt16'
|
||||
|
||||
|
||||
class UInt32Field(BaseIntField):
|
||||
|
||||
min_value = 0
|
||||
max_value = 2**32 - 1
|
||||
db_type = 'UInt32'
|
||||
|
||||
|
||||
class UInt64Field(BaseIntField):
|
||||
|
||||
min_value = 0
|
||||
max_value = 2**64 - 1
|
||||
db_type = 'UInt64'
|
||||
|
||||
|
||||
class Int8Field(BaseIntField):
|
||||
|
||||
min_value = -2**7
|
||||
max_value = 2**7 - 1
|
||||
db_type = 'Int8'
|
||||
|
||||
|
||||
class Int16Field(BaseIntField):
|
||||
|
||||
min_value = -2**15
|
||||
max_value = 2**15 - 1
|
||||
db_type = 'Int16'
|
||||
|
||||
|
||||
class Int32Field(BaseIntField):
|
||||
|
||||
min_value = -2**31
|
||||
max_value = 2**31 - 1
|
||||
db_type = 'Int32'
|
||||
|
||||
|
||||
class Int64Field(BaseIntField):
|
||||
|
||||
min_value = -2**63
|
||||
max_value = 2**63 - 1
|
||||
db_type = 'Int64'
|
||||
|
||||
|
||||
class BaseFloatField(Field):
|
||||
'''
|
||||
Abstract base class for all float-type fields.
|
||||
'''
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
try:
|
||||
return float(value)
|
||||
except:
|
||||
raise ValueError('Invalid value for %s - %r' % (self.__class__.__name__, value))
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
# There's no need to call escape since numbers do not contain
|
||||
# special characters, and never need quoting
|
||||
return str(value)
|
||||
|
||||
|
||||
class Float32Field(BaseFloatField):
|
||||
|
||||
db_type = 'Float32'
|
||||
|
||||
|
||||
class Float64Field(BaseFloatField):
|
||||
|
||||
db_type = 'Float64'
|
||||
|
||||
|
||||
class DecimalField(Field):
|
||||
'''
|
||||
Base class for all decimal fields. Can also be used directly.
|
||||
'''
|
||||
|
||||
def __init__(self, precision, scale, default=None, alias=None, materialized=None, readonly=None):
|
||||
assert 1 <= precision <= 38, 'Precision must be between 1 and 38'
|
||||
assert 0 <= scale <= precision, 'Scale must be between 0 and the given precision'
|
||||
self.precision = precision
|
||||
self.scale = scale
|
||||
self.db_type = 'Decimal(%d,%d)' % (self.precision, self.scale)
|
||||
with localcontext() as ctx:
|
||||
ctx.prec = 38
|
||||
self.exp = Decimal(10) ** -self.scale # for rounding to the required scale
|
||||
self.max_value = Decimal(10 ** (self.precision - self.scale)) - self.exp
|
||||
self.min_value = -self.max_value
|
||||
super(DecimalField, self).__init__(default, alias, materialized, readonly)
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if not isinstance(value, Decimal):
|
||||
try:
|
||||
value = Decimal(value)
|
||||
except:
|
||||
raise ValueError('Invalid value for %s - %r' % (self.__class__.__name__, value))
|
||||
if not value.is_finite():
|
||||
raise ValueError('Non-finite value for %s - %r' % (self.__class__.__name__, value))
|
||||
return self._round(value)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
# There's no need to call escape since numbers do not contain
|
||||
# special characters, and never need quoting
|
||||
return str(value)
|
||||
|
||||
def _round(self, value):
|
||||
return value.quantize(self.exp)
|
||||
|
||||
def validate(self, value):
|
||||
self._range_check(value, self.min_value, self.max_value)
|
||||
|
||||
|
||||
class Decimal32Field(DecimalField):
|
||||
|
||||
def __init__(self, scale, default=None, alias=None, materialized=None, readonly=None):
|
||||
super(Decimal32Field, self).__init__(9, scale, default, alias, materialized, readonly)
|
||||
self.db_type = 'Decimal32(%d)' % scale
|
||||
|
||||
|
||||
class Decimal64Field(DecimalField):
|
||||
|
||||
def __init__(self, scale, default=None, alias=None, materialized=None, readonly=None):
|
||||
super(Decimal64Field, self).__init__(18, scale, default, alias, materialized, readonly)
|
||||
self.db_type = 'Decimal64(%d)' % scale
|
||||
|
||||
|
||||
class Decimal128Field(DecimalField):
|
||||
|
||||
def __init__(self, scale, default=None, alias=None, materialized=None, readonly=None):
|
||||
super(Decimal128Field, self).__init__(38, scale, default, alias, materialized, readonly)
|
||||
self.db_type = 'Decimal128(%d)' % scale
|
||||
|
||||
|
||||
class BaseEnumField(Field):
|
||||
'''
|
||||
Abstract base class for all enum-type fields.
|
||||
'''
|
||||
|
||||
def __init__(self, enum_cls, default=None, alias=None, materialized=None, readonly=None, codec=None):
|
||||
self.enum_cls = enum_cls
|
||||
if default is None:
|
||||
default = list(enum_cls)[0]
|
||||
super(BaseEnumField, self).__init__(default, alias, materialized, readonly, codec)
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, self.enum_cls):
|
||||
return value
|
||||
try:
|
||||
if isinstance(value, str):
|
||||
try:
|
||||
return self.enum_cls[value]
|
||||
except Exception:
|
||||
return self.enum_cls(value)
|
||||
if isinstance(value, bytes):
|
||||
decoded = value.decode('UTF-8')
|
||||
try:
|
||||
return self.enum_cls[decoded]
|
||||
except Exception:
|
||||
return self.enum_cls(decoded)
|
||||
if isinstance(value, int):
|
||||
return self.enum_cls(value)
|
||||
except (KeyError, ValueError):
|
||||
pass
|
||||
raise ValueError('Invalid value for %s: %r' % (self.enum_cls.__name__, value))
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
return escape(value.name, quote)
|
||||
|
||||
def get_db_type_args(self):
|
||||
return ['%s = %d' % (escape(item.name), item.value) for item in self.enum_cls]
|
||||
|
||||
@classmethod
|
||||
def create_ad_hoc_field(cls, db_type):
|
||||
'''
|
||||
Give an SQL column description such as "Enum8('apple' = 1, 'banana' = 2, 'orange' = 3)"
|
||||
this method returns a matching enum field.
|
||||
'''
|
||||
import re
|
||||
from enum import Enum
|
||||
members = {}
|
||||
for match in re.finditer(r"'([\w ]+)' = (-?\d+)", db_type):
|
||||
members[match.group(1)] = int(match.group(2))
|
||||
enum_cls = Enum('AdHocEnum', members)
|
||||
field_class = Enum8Field if db_type.startswith('Enum8') else Enum16Field
|
||||
return field_class(enum_cls)
|
||||
|
||||
|
||||
class Enum8Field(BaseEnumField):
|
||||
|
||||
db_type = 'Enum8'
|
||||
|
||||
|
||||
class Enum16Field(BaseEnumField):
|
||||
|
||||
db_type = 'Enum16'
|
||||
|
||||
|
||||
class ArrayField(Field):
|
||||
|
||||
class_default = []
|
||||
|
||||
def __init__(self, inner_field, default=None, alias=None, materialized=None, readonly=None, codec=None):
|
||||
assert isinstance(inner_field, Field), "The first argument of ArrayField must be a Field instance"
|
||||
assert not isinstance(inner_field, ArrayField), "Multidimensional array fields are not supported by the ORM"
|
||||
self.inner_field = inner_field
|
||||
super(ArrayField, self).__init__(default, alias, materialized, readonly, codec)
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, str):
|
||||
value = parse_array(value)
|
||||
elif isinstance(value, bytes):
|
||||
value = parse_array(value.decode('UTF-8'))
|
||||
elif not isinstance(value, (list, tuple)):
|
||||
raise ValueError('ArrayField expects list or tuple, not %s' % type(value))
|
||||
return [self.inner_field.to_python(v, timezone_in_use) for v in value]
|
||||
|
||||
def validate(self, value):
|
||||
for v in value:
|
||||
self.inner_field.validate(v)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
array = [self.inner_field.to_db_string(v, quote=True) for v in value]
|
||||
return '[' + comma_join(array) + ']'
|
||||
|
||||
def get_sql(self, with_default_expression=True, db=None):
|
||||
sql = 'Array(%s)' % self.inner_field.get_sql(with_default_expression=False, db=db)
|
||||
if with_default_expression and self.codec and db and db.has_codec_support:
|
||||
sql+= ' CODEC(%s)' % self.codec
|
||||
return sql
|
||||
|
||||
|
||||
class UUIDField(Field):
|
||||
|
||||
class_default = UUID(int=0)
|
||||
db_type = 'UUID'
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, UUID):
|
||||
return value
|
||||
elif isinstance(value, bytes):
|
||||
return UUID(bytes=value)
|
||||
elif isinstance(value, str):
|
||||
return UUID(value)
|
||||
elif isinstance(value, int):
|
||||
return UUID(int=value)
|
||||
elif isinstance(value, tuple):
|
||||
return UUID(fields=value)
|
||||
else:
|
||||
raise ValueError('Invalid value for UUIDField: %r' % value)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
return escape(str(value), quote)
|
||||
|
||||
|
||||
class IPv4Field(Field):
|
||||
|
||||
class_default = 0
|
||||
db_type = 'IPv4'
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, IPv4Address):
|
||||
return value
|
||||
elif isinstance(value, (bytes, str, int)):
|
||||
return IPv4Address(value)
|
||||
else:
|
||||
raise ValueError('Invalid value for IPv4Address: %r' % value)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
return escape(str(value), quote)
|
||||
|
||||
|
||||
class IPv6Field(Field):
|
||||
|
||||
class_default = 0
|
||||
db_type = 'IPv6'
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, IPv6Address):
|
||||
return value
|
||||
elif isinstance(value, (bytes, str, int)):
|
||||
return IPv6Address(value)
|
||||
else:
|
||||
raise ValueError('Invalid value for IPv6Address: %r' % value)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
return escape(str(value), quote)
|
||||
|
||||
|
||||
class NullableField(Field):
|
||||
|
||||
class_default = None
|
||||
|
||||
def __init__(self, inner_field, default=None, alias=None, materialized=None,
|
||||
extra_null_values=None, codec=None):
|
||||
assert isinstance(inner_field, Field), "The first argument of NullableField must be a Field instance. Not: {}".format(inner_field)
|
||||
self.inner_field = inner_field
|
||||
self._null_values = [None]
|
||||
if extra_null_values:
|
||||
self._null_values.extend(extra_null_values)
|
||||
super(NullableField, self).__init__(default, alias, materialized, readonly=None, codec=codec)
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if value == '\\N' or value in self._null_values:
|
||||
return None
|
||||
return self.inner_field.to_python(value, timezone_in_use)
|
||||
|
||||
def validate(self, value):
|
||||
value in self._null_values or self.inner_field.validate(value)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
if value in self._null_values:
|
||||
return '\\N'
|
||||
return self.inner_field.to_db_string(value, quote=quote)
|
||||
|
||||
def get_sql(self, with_default_expression=True, db=None):
|
||||
sql = 'Nullable(%s)' % self.inner_field.get_sql(with_default_expression=False, db=db)
|
||||
if with_default_expression:
|
||||
sql += self._extra_params(db)
|
||||
return sql
|
||||
|
||||
|
||||
class LowCardinalityField(Field):
|
||||
|
||||
def __init__(self, inner_field, default=None, alias=None, materialized=None, readonly=None, codec=None):
|
||||
assert isinstance(inner_field, Field), "The first argument of LowCardinalityField must be a Field instance. Not: {}".format(inner_field)
|
||||
assert not isinstance(inner_field, LowCardinalityField), "LowCardinality inner fields are not supported by the ORM"
|
||||
assert not isinstance(inner_field, ArrayField), "Array field inside LowCardinality are not supported by the ORM. Use Array(LowCardinality) instead"
|
||||
self.inner_field = inner_field
|
||||
self.class_default = self.inner_field.class_default
|
||||
super(LowCardinalityField, self).__init__(default, alias, materialized, readonly, codec)
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
return self.inner_field.to_python(value, timezone_in_use)
|
||||
|
||||
def validate(self, value):
|
||||
self.inner_field.validate(value)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
return self.inner_field.to_db_string(value, quote=quote)
|
||||
|
||||
def get_sql(self, with_default_expression=True, db=None):
|
||||
if db and db.has_low_cardinality_support:
|
||||
sql = 'LowCardinality(%s)' % self.inner_field.get_sql(with_default_expression=False)
|
||||
else:
|
||||
sql = self.inner_field.get_sql(with_default_expression=False)
|
||||
logger.warning('LowCardinalityField not supported on clickhouse-server version < 19.0 using {} as fallback'.format(self.inner_field.__class__.__name__))
|
||||
if with_default_expression:
|
||||
sql += self._extra_params(db)
|
||||
return sql
|
||||
|
||||
|
||||
# Expose only relevant classes in import *
|
||||
__all__ = get_subclass_names(locals(), Field)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,326 @@
|
||||
import logging
|
||||
|
||||
from .engines import Distributed, MergeTree
|
||||
from .fields import DateField, StringField
|
||||
from .models import BufferModel, Model
|
||||
from .utils import escape, get_subclass_names
|
||||
|
||||
logger = logging.getLogger("migrations")
|
||||
|
||||
|
||||
class Operation:
|
||||
"""
|
||||
Base class for migration operations.
|
||||
"""
|
||||
|
||||
def apply(self, database):
|
||||
raise NotImplementedError() # pragma: no cover
|
||||
|
||||
|
||||
class ModelOperation(Operation):
|
||||
"""
|
||||
Base class for migration operations that work on a specific model.
|
||||
"""
|
||||
|
||||
def __init__(self, model_class):
|
||||
"""
|
||||
Initializer.
|
||||
"""
|
||||
self.model_class = model_class
|
||||
self.table_name = model_class.table_name()
|
||||
|
||||
def _alter_table(self, database, cmd):
|
||||
"""
|
||||
Utility for running ALTER TABLE commands.
|
||||
"""
|
||||
cmd = "ALTER TABLE $db.`%s` %s" % (self.table_name, cmd)
|
||||
logger.debug(cmd)
|
||||
database.raw(cmd)
|
||||
|
||||
|
||||
class CreateTable(ModelOperation):
|
||||
"""
|
||||
A migration operation that creates a table for a given model class.
|
||||
"""
|
||||
|
||||
def apply(self, database):
|
||||
logger.info(" Create table %s", self.table_name)
|
||||
if issubclass(self.model_class, BufferModel):
|
||||
database.create_table(self.model_class.engine.main_model)
|
||||
database.create_table(self.model_class)
|
||||
|
||||
|
||||
class AlterTable(ModelOperation):
|
||||
"""
|
||||
A migration operation that compares the table of a given model class to
|
||||
the model's fields, and alters the table to match the model. The operation can:
|
||||
- add new columns
|
||||
- drop obsolete columns
|
||||
- modify column types
|
||||
Default values are not altered by this operation.
|
||||
"""
|
||||
|
||||
def _get_table_fields(self, database):
|
||||
query = "DESC `%s`.`%s`" % (database.db_name, self.table_name)
|
||||
return [(row.name, row.type) for row in database.select(query)]
|
||||
|
||||
def apply(self, database):
|
||||
logger.info(" Alter table %s", self.table_name)
|
||||
|
||||
# Note that MATERIALIZED and ALIAS fields are always at the end of the DESC,
|
||||
# ADD COLUMN ... AFTER doesn't affect it
|
||||
table_fields = dict(self._get_table_fields(database))
|
||||
|
||||
# Identify fields that were deleted from the model
|
||||
deleted_fields = set(table_fields.keys()) - set(self.model_class.fields())
|
||||
for name in deleted_fields:
|
||||
logger.info(" Drop column %s", name)
|
||||
self._alter_table(database, "DROP COLUMN %s" % name)
|
||||
del table_fields[name]
|
||||
|
||||
# Identify fields that were added to the model
|
||||
prev_name = None
|
||||
for name, field in self.model_class.fields().items():
|
||||
is_regular_field = not (field.materialized or field.alias)
|
||||
if name not in table_fields:
|
||||
logger.info(" Add column %s", name)
|
||||
assert prev_name, "Cannot add a column to the beginning of the table"
|
||||
cmd = "ADD COLUMN %s %s" % (name, field.get_sql(db=database))
|
||||
if is_regular_field:
|
||||
cmd += " AFTER %s" % prev_name
|
||||
self._alter_table(database, cmd)
|
||||
|
||||
if is_regular_field:
|
||||
# ALIAS and MATERIALIZED fields are not stored in the database, and raise DatabaseError
|
||||
# (no AFTER column). So we will skip them
|
||||
prev_name = name
|
||||
|
||||
# Identify fields whose type was changed
|
||||
# The order of class attributes can be changed any time, so we can't count on it
|
||||
# Secondly, MATERIALIZED and ALIAS fields are always at the end of the DESC, so we can't expect them to save
|
||||
# attribute position. Watch https://github.com/Infinidat/datastore_orm/issues/47
|
||||
model_fields = {
|
||||
name: field.get_sql(with_default_expression=False, db=database)
|
||||
for name, field in self.model_class.fields().items()
|
||||
}
|
||||
for field_name, field_sql in self._get_table_fields(database):
|
||||
# All fields must have been created and dropped by this moment
|
||||
assert (
|
||||
field_name in model_fields
|
||||
), "Model fields and table columns in disagreement"
|
||||
|
||||
if field_sql != model_fields[field_name]:
|
||||
logger.info(
|
||||
" Change type of column %s from %s to %s",
|
||||
field_name,
|
||||
field_sql,
|
||||
model_fields[field_name],
|
||||
)
|
||||
self._alter_table(
|
||||
database,
|
||||
"MODIFY COLUMN %s %s" % (field_name, model_fields[field_name]),
|
||||
)
|
||||
|
||||
|
||||
class AlterTableWithBuffer(ModelOperation):
|
||||
"""
|
||||
A migration operation for altering a buffer table and its underlying on-disk table.
|
||||
The buffer table is dropped, the on-disk table is altered, and then the buffer table
|
||||
is re-created.
|
||||
"""
|
||||
|
||||
def apply(self, database):
|
||||
if issubclass(self.model_class, BufferModel):
|
||||
DropTable(self.model_class).apply(database)
|
||||
AlterTable(self.model_class.engine.main_model).apply(database)
|
||||
CreateTable(self.model_class).apply(database)
|
||||
else:
|
||||
AlterTable(self.model_class).apply(database)
|
||||
|
||||
|
||||
class DropTable(ModelOperation):
|
||||
"""
|
||||
A migration operation that drops the table of a given model class.
|
||||
"""
|
||||
|
||||
def apply(self, database):
|
||||
logger.info(" Drop table %s", self.table_name)
|
||||
database.drop_table(self.model_class)
|
||||
|
||||
|
||||
class AlterConstraints(ModelOperation):
|
||||
"""
|
||||
A migration operation that adds new constraints from the model to the database
|
||||
table, and drops obsolete ones. Constraints are identified by their names, so
|
||||
a change in an existing constraint will not be detected unless its name was changed too.
|
||||
ClickHouse does not check that the constraints hold for existing data in the table.
|
||||
"""
|
||||
|
||||
def apply(self, database):
|
||||
logger.info(" Alter constraints for %s", self.table_name)
|
||||
existing = self._get_constraint_names(database)
|
||||
# Go over constraints in the model
|
||||
for constraint in self.model_class._constraints.values():
|
||||
# Check if it's a new constraint
|
||||
if constraint.name not in existing:
|
||||
logger.info(" Add constraint %s", constraint.name)
|
||||
self._alter_table(database, "ADD %s" % constraint.create_table_sql())
|
||||
else:
|
||||
existing.remove(constraint.name)
|
||||
# Remaining constraints in `existing` are obsolete
|
||||
for name in existing:
|
||||
logger.info(" Drop constraint %s", name)
|
||||
self._alter_table(database, "DROP CONSTRAINT `%s`" % name)
|
||||
|
||||
def _get_constraint_names(self, database):
|
||||
"""
|
||||
Returns a set containing the names of existing constraints in the table.
|
||||
"""
|
||||
import re
|
||||
|
||||
table_def = database.raw("SHOW CREATE TABLE $db.`%s`" % self.table_name)
|
||||
matches = re.findall(r"\sCONSTRAINT\s+`?(.+?)`?\s+CHECK\s", table_def)
|
||||
return set(matches)
|
||||
|
||||
|
||||
class AlterIndexes(ModelOperation):
|
||||
"""
|
||||
A migration operation that adds new indexes from the model to the database
|
||||
table, and drops obsolete ones. Indexes are identified by their names, so
|
||||
a change in an existing index will not be detected unless its name was changed too.
|
||||
"""
|
||||
|
||||
def __init__(self, model_class, reindex=False):
|
||||
"""
|
||||
Initializer.
|
||||
By default ClickHouse does not build indexes over existing data, only for
|
||||
new data. Passing `reindex=True` will run `OPTIMIZE TABLE` in order to build
|
||||
the indexes over the existing data.
|
||||
"""
|
||||
super().__init__(model_class)
|
||||
self.reindex = reindex
|
||||
|
||||
def apply(self, database):
|
||||
logger.info(" Alter indexes for %s", self.table_name)
|
||||
existing = self._get_index_names(database)
|
||||
logger.info(existing)
|
||||
# Go over indexes in the model
|
||||
for index in self.model_class._indexes.values():
|
||||
# Check if it's a new index
|
||||
if index.name not in existing:
|
||||
logger.info(" Add index %s", index.name)
|
||||
self._alter_table(database, "ADD %s" % index.create_table_sql())
|
||||
else:
|
||||
existing.remove(index.name)
|
||||
# Remaining indexes in `existing` are obsolete
|
||||
for name in existing:
|
||||
logger.info(" Drop index %s", name)
|
||||
self._alter_table(database, "DROP INDEX `%s`" % name)
|
||||
# Reindex
|
||||
if self.reindex:
|
||||
logger.info(" Build indexes on table")
|
||||
database.raw("OPTIMIZE TABLE $db.`%s` FINAL" % self.table_name)
|
||||
|
||||
def _get_index_names(self, database):
|
||||
"""
|
||||
Returns a set containing the names of existing indexes in the table.
|
||||
"""
|
||||
import re
|
||||
|
||||
table_def = database.raw("SHOW CREATE TABLE $db.`%s`" % self.table_name)
|
||||
matches = re.findall(r"\sINDEX\s+`?(.+?)`?\s+", table_def)
|
||||
return set(matches)
|
||||
|
||||
|
||||
class RunPython(Operation):
|
||||
"""
|
||||
A migration operation that executes a Python function.
|
||||
"""
|
||||
|
||||
def __init__(self, func):
|
||||
"""
|
||||
Initializer. The given Python function will be called with a single
|
||||
argument - the Database instance to apply the migration to.
|
||||
"""
|
||||
assert callable(func), "'func' argument must be function"
|
||||
self._func = func
|
||||
|
||||
def apply(self, database):
|
||||
logger.info(" Executing python operation %s", self._func.__name__)
|
||||
self._func(database)
|
||||
|
||||
|
||||
class RunSQL(Operation):
|
||||
"""
|
||||
A migration operation that executes arbitrary SQL statements.
|
||||
"""
|
||||
|
||||
def __init__(self, sql):
|
||||
"""
|
||||
Initializer. The given sql argument must be a valid SQL statement or
|
||||
list of statements.
|
||||
"""
|
||||
if isinstance(sql, str):
|
||||
sql = [sql]
|
||||
assert isinstance(sql, list), "'sql' argument must be string or list of strings"
|
||||
self._sql = sql
|
||||
|
||||
def apply(self, database):
|
||||
logger.info(" Executing raw SQL operations")
|
||||
for item in self._sql:
|
||||
database.raw(item)
|
||||
|
||||
|
||||
class MigrationHistory(Model):
|
||||
"""
|
||||
A model for storing which migrations were already applied to the containing database.
|
||||
"""
|
||||
|
||||
package_name = StringField()
|
||||
module_name = StringField()
|
||||
applied = DateField()
|
||||
|
||||
engine = MergeTree('applied', ('package_name', 'module_name'))
|
||||
|
||||
@classmethod
|
||||
def table_name(cls):
|
||||
return "infi_clickhouse_orm_migrations"
|
||||
|
||||
class MigrationHistoryReplicated(Model):
|
||||
"""
|
||||
A model for storing which migrations were already applied to the containing database.
|
||||
"""
|
||||
|
||||
package_name = StringField()
|
||||
module_name = StringField()
|
||||
applied = DateField()
|
||||
|
||||
engine = MergeTree(
|
||||
"applied",
|
||||
("package_name", "module_name"),
|
||||
replica_table_path="/clickhouse/prod/tables/noshard/{database}/{table}",
|
||||
replica_name="{replica}-{shard}",
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def table_name(cls):
|
||||
return "infi_clickhouse_orm_migrations"
|
||||
|
||||
class MigrationHistoryDistributed(Model):
|
||||
"""
|
||||
Distributed table for storing which migrations are applied to the containing database
|
||||
"""
|
||||
|
||||
package_name = StringField()
|
||||
module_name = StringField()
|
||||
applied = DateField()
|
||||
|
||||
engine = Distributed(table="infi_clickhouse_orm_migrations", sharding_key="rand()")
|
||||
|
||||
@classmethod
|
||||
def table_name(cls):
|
||||
return "infi_clickhouse_orm_migrations_distributed"
|
||||
|
||||
|
||||
# Expose only relevant classes in import *
|
||||
__all__ = get_subclass_names(locals(), Operation)
|
||||
@@ -0,0 +1,609 @@
|
||||
from __future__ import unicode_literals
|
||||
import sys
|
||||
from collections import OrderedDict
|
||||
from itertools import chain
|
||||
from logging import getLogger
|
||||
|
||||
import pytz
|
||||
|
||||
from .fields import Field, StringField
|
||||
from .utils import on_cluster, parse_tsv, NO_VALUE, get_subclass_names, arg_to_sql, unescape
|
||||
from .query import QuerySet
|
||||
from .funcs import F
|
||||
from .engines import Merge, Distributed
|
||||
|
||||
logger = getLogger('clickhouse_orm')
|
||||
|
||||
|
||||
|
||||
class Constraint:
|
||||
'''
|
||||
Defines a model constraint.
|
||||
'''
|
||||
|
||||
name = None # this is set by the parent model
|
||||
parent = None # this is set by the parent model
|
||||
|
||||
def __init__(self, expr):
|
||||
'''
|
||||
Initializer. Expects an expression that ClickHouse will verify when inserting data.
|
||||
'''
|
||||
self.expr = expr
|
||||
|
||||
def create_table_sql(self):
|
||||
'''
|
||||
Returns the SQL statement for defining this constraint during table creation.
|
||||
'''
|
||||
return 'CONSTRAINT `%s` CHECK %s' % (self.name, arg_to_sql(self.expr))
|
||||
|
||||
|
||||
class Index:
|
||||
'''
|
||||
Defines a data-skipping index.
|
||||
'''
|
||||
|
||||
name = None # this is set by the parent model
|
||||
parent = None # this is set by the parent model
|
||||
|
||||
def __init__(self, expr, type, granularity):
|
||||
'''
|
||||
Initializer.
|
||||
|
||||
- `expr` - a column, expression, or tuple of columns and expressions to index.
|
||||
- `type` - the index type. Use one of the following methods to specify the type:
|
||||
`Index.minmax`, `Index.set`, `Index.ngrambf_v1`, `Index.tokenbf_v1` or `Index.bloom_filter`.
|
||||
- `granularity` - index block size (number of multiples of the `index_granularity` defined by the engine).
|
||||
'''
|
||||
self.expr = expr
|
||||
self.type = type
|
||||
self.granularity = granularity
|
||||
|
||||
def create_table_sql(self):
|
||||
'''
|
||||
Returns the SQL statement for defining this index during table creation.
|
||||
'''
|
||||
return 'INDEX `%s` %s TYPE %s GRANULARITY %d' % (self.name, arg_to_sql(self.expr), self.type, self.granularity)
|
||||
|
||||
@staticmethod
|
||||
def minmax():
|
||||
'''
|
||||
An index that stores extremes of the specified expression (if the expression is tuple, then it stores
|
||||
extremes for each element of tuple). The stored info is used for skipping blocks of data like the primary key.
|
||||
'''
|
||||
return 'minmax'
|
||||
|
||||
@staticmethod
|
||||
def set(max_rows):
|
||||
'''
|
||||
An index that stores unique values of the specified expression (no more than max_rows rows,
|
||||
or unlimited if max_rows=0). Uses the values to check if the WHERE expression is not satisfiable
|
||||
on a block of data.
|
||||
'''
|
||||
return 'set(%d)' % max_rows
|
||||
|
||||
@staticmethod
|
||||
def ngrambf_v1(n, size_of_bloom_filter_in_bytes, number_of_hash_functions, random_seed):
|
||||
'''
|
||||
An index that stores a Bloom filter containing all ngrams from a block of data.
|
||||
Works only with strings. Can be used for optimization of equals, like and in expressions.
|
||||
|
||||
- `n` — ngram size
|
||||
- `size_of_bloom_filter_in_bytes` — Bloom filter size in bytes (you can use large values here,
|
||||
for example 256 or 512, because it can be compressed well).
|
||||
- `number_of_hash_functions` — The number of hash functions used in the Bloom filter.
|
||||
- `random_seed` — The seed for Bloom filter hash functions.
|
||||
'''
|
||||
return 'ngrambf_v1(%d, %d, %d, %d)' % (n, size_of_bloom_filter_in_bytes, number_of_hash_functions, random_seed)
|
||||
|
||||
@staticmethod
|
||||
def tokenbf_v1(size_of_bloom_filter_in_bytes, number_of_hash_functions, random_seed):
|
||||
'''
|
||||
An index that stores a Bloom filter containing string tokens. Tokens are sequences
|
||||
separated by non-alphanumeric characters.
|
||||
|
||||
- `size_of_bloom_filter_in_bytes` — Bloom filter size in bytes (you can use large values here,
|
||||
for example 256 or 512, because it can be compressed well).
|
||||
- `number_of_hash_functions` — The number of hash functions used in the Bloom filter.
|
||||
- `random_seed` — The seed for Bloom filter hash functions.
|
||||
'''
|
||||
return 'tokenbf_v1(%d, %d, %d)' % (size_of_bloom_filter_in_bytes, number_of_hash_functions, random_seed)
|
||||
|
||||
@staticmethod
|
||||
def bloom_filter(false_positive=0.025):
|
||||
'''
|
||||
An index that stores a Bloom filter containing values of the index expression.
|
||||
|
||||
- `false_positive` - the probability (between 0 and 1) of receiving a false positive
|
||||
response from the filter
|
||||
'''
|
||||
return 'bloom_filter(%f)' % false_positive
|
||||
|
||||
|
||||
class ModelBase(type):
|
||||
'''
|
||||
A metaclass for ORM models. It adds the _fields list to model classes.
|
||||
'''
|
||||
|
||||
ad_hoc_model_cache = {}
|
||||
|
||||
def __new__(cls, name, bases, attrs):
|
||||
|
||||
# Collect fields, constraints and indexes from parent classes
|
||||
fields = {}
|
||||
constraints = {}
|
||||
indexes = {}
|
||||
for base in bases:
|
||||
if isinstance(base, ModelBase):
|
||||
fields.update(base._fields)
|
||||
constraints.update(base._constraints)
|
||||
indexes.update(base._indexes)
|
||||
|
||||
# Add fields, constraints and indexes from this class
|
||||
for n, obj in attrs.items():
|
||||
if isinstance(obj, Field):
|
||||
fields[n] = obj
|
||||
elif isinstance(obj, Constraint):
|
||||
constraints[n] = obj
|
||||
elif isinstance(obj, Index):
|
||||
indexes[n] = obj
|
||||
|
||||
# Convert fields to a list of (name, field) tuples, in the order they were listed in the class
|
||||
fields = sorted(fields.items(), key=lambda item: item[1].creation_counter)
|
||||
|
||||
# Build a dictionary of default values
|
||||
defaults = {}
|
||||
has_funcs_as_defaults = False
|
||||
for n, f in fields:
|
||||
if f.alias or f.materialized:
|
||||
defaults[n] = NO_VALUE
|
||||
elif isinstance(f.default, F):
|
||||
defaults[n] = NO_VALUE
|
||||
has_funcs_as_defaults = True
|
||||
else:
|
||||
defaults[n] = f.to_python(f.default, pytz.UTC)
|
||||
|
||||
# Create the model class
|
||||
attrs = dict(
|
||||
attrs,
|
||||
_fields=OrderedDict(fields),
|
||||
_constraints=constraints,
|
||||
_indexes=indexes,
|
||||
_writable_fields=OrderedDict([f for f in fields if not f[1].readonly]),
|
||||
_defaults=defaults,
|
||||
_has_funcs_as_defaults=has_funcs_as_defaults
|
||||
)
|
||||
model = super(ModelBase, cls).__new__(cls, str(name), bases, attrs)
|
||||
|
||||
# Let each field, constraint and index know its parent and its own name
|
||||
for n, obj in chain(fields, constraints.items(), indexes.items()):
|
||||
setattr(obj, 'parent', model)
|
||||
setattr(obj, 'name', n)
|
||||
|
||||
return model
|
||||
|
||||
@classmethod
|
||||
def create_ad_hoc_model(cls, fields, model_name='AdHocModel'):
|
||||
# fields is a list of tuples (name, db_type)
|
||||
# Check if model exists in cache
|
||||
fields = list(fields)
|
||||
cache_key = model_name + ' ' + str(fields)
|
||||
if cache_key in cls.ad_hoc_model_cache:
|
||||
return cls.ad_hoc_model_cache[cache_key]
|
||||
# Create an ad hoc model class
|
||||
attrs = {}
|
||||
for name, db_type in fields:
|
||||
attrs[name] = cls.create_ad_hoc_field(db_type)
|
||||
model_class = cls.__new__(cls, model_name, (Model,), attrs)
|
||||
# Add the model class to the cache
|
||||
cls.ad_hoc_model_cache[cache_key] = model_class
|
||||
return model_class
|
||||
|
||||
@classmethod
|
||||
def create_ad_hoc_field(cls, db_type):
|
||||
import datastore_orm.fields as orm_fields
|
||||
# Enums
|
||||
if db_type.startswith('Enum'):
|
||||
return orm_fields.BaseEnumField.create_ad_hoc_field(db_type)
|
||||
# DateTime with timezone
|
||||
if db_type.startswith('DateTime('):
|
||||
timezone = db_type[9:-1]
|
||||
return orm_fields.DateTimeField(
|
||||
timezone=timezone[1:-1] if timezone else None
|
||||
)
|
||||
# DateTime64
|
||||
if db_type.startswith('DateTime64('):
|
||||
precision, *timezone = [s.strip() for s in db_type[11:-1].split(',')]
|
||||
return orm_fields.DateTime64Field(
|
||||
precision=int(precision),
|
||||
timezone=timezone[0][1:-1] if timezone else None
|
||||
)
|
||||
# Arrays
|
||||
if db_type.startswith('Array'):
|
||||
inner_field = cls.create_ad_hoc_field(db_type[6 : -1])
|
||||
return orm_fields.ArrayField(inner_field)
|
||||
# Tuples (poor man's version - convert to array)
|
||||
if db_type.startswith('Tuple'):
|
||||
types = [s.strip() for s in db_type[6 : -1].split(',')]
|
||||
assert len(set(types)) == 1, 'No support for mixed types in tuples - ' + db_type
|
||||
inner_field = cls.create_ad_hoc_field(types[0])
|
||||
return orm_fields.ArrayField(inner_field)
|
||||
# FixedString
|
||||
if db_type.startswith('FixedString'):
|
||||
length = int(db_type[12 : -1])
|
||||
return orm_fields.FixedStringField(length)
|
||||
# Decimal / Decimal32 / Decimal64 / Decimal128
|
||||
if db_type.startswith('Decimal'):
|
||||
p = db_type.index('(')
|
||||
args = [int(n.strip()) for n in db_type[p + 1 : -1].split(',')]
|
||||
field_class = getattr(orm_fields, db_type[:p] + 'Field')
|
||||
return field_class(*args)
|
||||
# Nullable
|
||||
if db_type.startswith('Nullable'):
|
||||
inner_field = cls.create_ad_hoc_field(db_type[9 : -1])
|
||||
return orm_fields.NullableField(inner_field)
|
||||
# LowCardinality
|
||||
if db_type.startswith('LowCardinality'):
|
||||
inner_field = cls.create_ad_hoc_field(db_type[15 : -1])
|
||||
return orm_fields.LowCardinalityField(inner_field)
|
||||
# Simple fields
|
||||
name = db_type + 'Field'
|
||||
if not hasattr(orm_fields, name):
|
||||
raise NotImplementedError('No field class for %s' % db_type)
|
||||
return getattr(orm_fields, name)()
|
||||
|
||||
|
||||
class Model(metaclass=ModelBase):
|
||||
'''
|
||||
A base class for ORM models. Each model class represent a ClickHouse table. For example:
|
||||
|
||||
class CPUStats(Model):
|
||||
timestamp = DateTimeField()
|
||||
cpu_id = UInt16Field()
|
||||
cpu_percent = Float32Field()
|
||||
engine = Memory()
|
||||
'''
|
||||
|
||||
engine = None
|
||||
|
||||
# Insert operations are restricted for read only models
|
||||
_readonly = False
|
||||
|
||||
# Create table, drop table, insert operations are restricted for system models
|
||||
_system = False
|
||||
|
||||
_database = None
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
'''
|
||||
Creates a model instance, using keyword arguments as field values.
|
||||
Since values are immediately converted to their Pythonic type,
|
||||
invalid values will cause a `ValueError` to be raised.
|
||||
Unrecognized field names will cause an `AttributeError`.
|
||||
'''
|
||||
super(Model, self).__init__()
|
||||
# Assign default values
|
||||
self.__dict__.update(self._defaults)
|
||||
# Assign field values from keyword arguments
|
||||
for name, value in kwargs.items():
|
||||
field = self.get_field(name)
|
||||
if field:
|
||||
setattr(self, name, value)
|
||||
else:
|
||||
raise AttributeError('%s does not have a field called %s' % (self.__class__.__name__, name))
|
||||
|
||||
def __setattr__(self, name, value):
|
||||
'''
|
||||
When setting a field value, converts the value to its Pythonic type and validates it.
|
||||
This may raise a `ValueError`.
|
||||
'''
|
||||
field = self.get_field(name)
|
||||
if field and (value != NO_VALUE):
|
||||
try:
|
||||
value = field.to_python(value, pytz.utc)
|
||||
field.validate(value)
|
||||
except ValueError:
|
||||
tp, v, tb = sys.exc_info()
|
||||
new_msg = "{} (field '{}')".format(v, name)
|
||||
raise tp.with_traceback(tp(new_msg), tb)
|
||||
super(Model, self).__setattr__(name, value)
|
||||
|
||||
def set_database(self, db):
|
||||
'''
|
||||
Sets the `Database` that this model instance belongs to.
|
||||
This is done automatically when the instance is read from the database or written to it.
|
||||
'''
|
||||
# This can not be imported globally due to circular import
|
||||
from .database import Database
|
||||
assert isinstance(db, Database), "database must be database.Database instance"
|
||||
self._database = db
|
||||
|
||||
def get_database(self):
|
||||
'''
|
||||
Gets the `Database` that this model instance belongs to.
|
||||
Returns `None` unless the instance was read from the database or written to it.
|
||||
'''
|
||||
return self._database
|
||||
|
||||
def get_field(self, name):
|
||||
'''
|
||||
Gets a `Field` instance given its name, or `None` if not found.
|
||||
'''
|
||||
return self._fields.get(name)
|
||||
|
||||
@classmethod
|
||||
def table_name(cls):
|
||||
'''
|
||||
Returns the model's database table name. By default this is the
|
||||
class name converted to lowercase. Override this if you want to use
|
||||
a different table name.
|
||||
'''
|
||||
return cls.__name__.lower()
|
||||
|
||||
@classmethod
|
||||
def has_funcs_as_defaults(cls):
|
||||
'''
|
||||
Return True if some of the model's fields use a function expression
|
||||
as a default value. This requires special handling when inserting instances.
|
||||
'''
|
||||
return cls._has_funcs_as_defaults
|
||||
|
||||
@classmethod
|
||||
def create_table_sql(cls, db):
|
||||
'''
|
||||
Returns the SQL statement for creating a table for this model.
|
||||
'''
|
||||
parts = ['CREATE TABLE IF NOT EXISTS `%s`.`%s` %s (' % (db.db_name, cls.table_name(), on_cluster(db))]
|
||||
# Fields
|
||||
items = []
|
||||
for name, field in cls.fields().items():
|
||||
items.append(' %s %s' % (name, field.get_sql(db=db)))
|
||||
# Constraints
|
||||
for c in cls._constraints.values():
|
||||
items.append(' %s' % c.create_table_sql())
|
||||
# Indexes
|
||||
for i in cls._indexes.values():
|
||||
items.append(' %s' % i.create_table_sql())
|
||||
parts.append(',\n'.join(items))
|
||||
# Engine
|
||||
parts.append(')')
|
||||
parts.append('ENGINE = ' + cls.engine.create_table_sql(db))
|
||||
return '\n'.join(parts)
|
||||
|
||||
@classmethod
|
||||
def drop_table_sql(cls, db):
|
||||
'''
|
||||
Returns the SQL command for deleting this model's table.
|
||||
'''
|
||||
return 'DROP TABLE IF EXISTS `%s`.`%s`' % (db.db_name, cls.table_name())
|
||||
|
||||
@classmethod
|
||||
def from_tsv(cls, line, field_names, timezone_in_use=pytz.utc, database=None):
|
||||
'''
|
||||
Create a model instance from a tab-separated line. The line may or may not include a newline.
|
||||
The `field_names` list must match the fields defined in the model, but does not have to include all of them.
|
||||
|
||||
- `line`: the TSV-formatted data.
|
||||
- `field_names`: names of the model fields in the data.
|
||||
- `timezone_in_use`: the timezone to use when parsing dates and datetimes. Some fields use their own timezones.
|
||||
- `database`: if given, sets the database that this instance belongs to.
|
||||
'''
|
||||
values = iter(parse_tsv(line))
|
||||
kwargs = {}
|
||||
for name in field_names:
|
||||
field = getattr(cls, name)
|
||||
field_timezone = getattr(field, 'timezone', None) or timezone_in_use
|
||||
kwargs[name] = field.to_python(next(values), field_timezone)
|
||||
|
||||
obj = cls(**kwargs)
|
||||
if database is not None:
|
||||
obj.set_database(database)
|
||||
|
||||
return obj
|
||||
|
||||
def to_tsv(self, include_readonly=True):
|
||||
'''
|
||||
Returns the instance's column values as a tab-separated line. A newline is not included.
|
||||
|
||||
- `include_readonly`: if false, returns only fields that can be inserted into database.
|
||||
'''
|
||||
data = self.__dict__
|
||||
fields = self.fields(writable=not include_readonly)
|
||||
return '\t'.join(field.to_db_string(data[name], quote=False) for name, field in fields.items())
|
||||
|
||||
def to_tskv(self, include_readonly=True):
|
||||
'''
|
||||
Returns the instance's column keys and values as a tab-separated line. A newline is not included.
|
||||
Fields that were not assigned a value are omitted.
|
||||
|
||||
- `include_readonly`: if false, returns only fields that can be inserted into database.
|
||||
'''
|
||||
data = self.__dict__
|
||||
fields = self.fields(writable=not include_readonly)
|
||||
parts = []
|
||||
for name, field in fields.items():
|
||||
if data[name] != NO_VALUE:
|
||||
parts.append(name + '=' + field.to_db_string(data[name], quote=False))
|
||||
return '\t'.join(parts)
|
||||
|
||||
def to_db_string(self):
|
||||
'''
|
||||
Returns the instance as a bytestring ready to be inserted into the database.
|
||||
'''
|
||||
s = self.to_tskv(False) if self._has_funcs_as_defaults else self.to_tsv(False)
|
||||
s += '\n'
|
||||
return s.encode('utf-8')
|
||||
|
||||
def to_dict(self, include_readonly=True, field_names=None):
|
||||
'''
|
||||
Returns the instance's column values as a dict.
|
||||
|
||||
- `include_readonly`: if false, returns only fields that can be inserted into database.
|
||||
- `field_names`: an iterable of field names to return (optional)
|
||||
'''
|
||||
fields = self.fields(writable=not include_readonly)
|
||||
|
||||
if field_names is not None:
|
||||
fields = [f for f in fields if f in field_names]
|
||||
|
||||
data = self.__dict__
|
||||
return {name: data[name] for name in fields}
|
||||
|
||||
@classmethod
|
||||
def objects_in(cls, database):
|
||||
'''
|
||||
Returns a `QuerySet` for selecting instances of this model class.
|
||||
'''
|
||||
return QuerySet(cls, database)
|
||||
|
||||
@classmethod
|
||||
def fields(cls, writable=False):
|
||||
'''
|
||||
Returns an `OrderedDict` of the model's fields (from name to `Field` instance).
|
||||
If `writable` is true, only writable fields are included.
|
||||
Callers should not modify the dictionary.
|
||||
'''
|
||||
# noinspection PyProtectedMember,PyUnresolvedReferences
|
||||
return cls._writable_fields if writable else cls._fields
|
||||
|
||||
@classmethod
|
||||
def is_read_only(cls):
|
||||
'''
|
||||
Returns true if the model is marked as read only.
|
||||
'''
|
||||
return cls._readonly
|
||||
|
||||
@classmethod
|
||||
def is_system_model(cls):
|
||||
'''
|
||||
Returns true if the model represents a system table.
|
||||
'''
|
||||
return cls._system
|
||||
|
||||
|
||||
class BufferModel(Model):
|
||||
|
||||
@classmethod
|
||||
def create_table_sql(cls, db):
|
||||
'''
|
||||
Returns the SQL statement for creating a table for this model.
|
||||
'''
|
||||
parts = ['CREATE TABLE IF NOT EXISTS `%s`.`%s` AS `%s`.`%s`' % (db.db_name, cls.table_name(), db.db_name,
|
||||
cls.engine.main_model.table_name())]
|
||||
engine_str = cls.engine.create_table_sql(db)
|
||||
parts.append(engine_str)
|
||||
return ' '.join(parts)
|
||||
|
||||
|
||||
class MergeModel(Model):
|
||||
'''
|
||||
Model for Merge engine
|
||||
Predefines virtual _table column an controls that rows can't be inserted to this table type
|
||||
https://clickhouse.tech/docs/en/single/index.html#document-table_engines/merge
|
||||
'''
|
||||
readonly = True
|
||||
|
||||
# Virtual fields can't be inserted into database
|
||||
_table = StringField(readonly=True)
|
||||
|
||||
@classmethod
|
||||
def create_table_sql(cls, db):
|
||||
'''
|
||||
Returns the SQL statement for creating a table for this model.
|
||||
'''
|
||||
assert isinstance(cls.engine, Merge), "engine must be an instance of engines.Merge"
|
||||
parts = ['CREATE TABLE IF NOT EXISTS `%s`.`%s` (' % (db.db_name, cls.table_name())]
|
||||
cols = []
|
||||
for name, field in cls.fields().items():
|
||||
if name != '_table':
|
||||
cols.append(' %s %s' % (name, field.get_sql(db=db)))
|
||||
parts.append(',\n'.join(cols))
|
||||
parts.append(')')
|
||||
parts.append('ENGINE = ' + cls.engine.create_table_sql(db))
|
||||
return '\n'.join(parts)
|
||||
|
||||
# TODO: base class for models that require specific engine
|
||||
|
||||
|
||||
class DistributedModel(Model):
|
||||
"""
|
||||
Model class for use with a `Distributed` engine.
|
||||
"""
|
||||
|
||||
def set_database(self, db):
|
||||
'''
|
||||
Sets the `Database` that this model instance belongs to.
|
||||
This is done automatically when the instance is read from the database or written to it.
|
||||
'''
|
||||
assert isinstance(self.engine, Distributed), "engine must be an instance of engines.Distributed"
|
||||
res = super(DistributedModel, self).set_database(db)
|
||||
return res
|
||||
|
||||
@classmethod
|
||||
def fix_engine_table(cls):
|
||||
"""
|
||||
Remember: Distributed table does not store any data, just provides distributed access to it.
|
||||
|
||||
So if we define a model with engine that has no defined table for data storage
|
||||
(see FooDistributed below), that table cannot be successfully created.
|
||||
This routine can automatically fix engine's storage table by finding the first
|
||||
non-distributed model among your model's superclasses.
|
||||
|
||||
>>> class Foo(Model):
|
||||
... id = UInt8Field(1)
|
||||
...
|
||||
>>> class FooDistributed(Foo, DistributedModel):
|
||||
... engine = Distributed('my_cluster')
|
||||
...
|
||||
>>> FooDistributed.engine.table
|
||||
None
|
||||
>>> FooDistributed.fix_engine()
|
||||
>>> FooDistributed.engine.table
|
||||
<class '__main__.Foo'>
|
||||
|
||||
However if you prefer more explicit way of doing things,
|
||||
you can always mention the Foo model twice without bothering with any fixes:
|
||||
|
||||
>>> class FooDistributedVerbose(Foo, DistributedModel):
|
||||
... engine = Distributed('my_cluster', Foo)
|
||||
>>> FooDistributedVerbose.engine.table
|
||||
<class '__main__.Foo'>
|
||||
|
||||
See tests.test_engines:DistributedTestCase for more examples
|
||||
"""
|
||||
|
||||
# apply only when engine has no table defined
|
||||
if cls.engine.table_name:
|
||||
return
|
||||
|
||||
# find out all the superclasses of the Model that store any data
|
||||
storage_models = [b for b in cls.__bases__ if issubclass(b, Model)
|
||||
and not issubclass(b, DistributedModel)]
|
||||
if not storage_models:
|
||||
raise TypeError("When defining Distributed engine without the table_name "
|
||||
"ensure that your model has a parent model")
|
||||
|
||||
if len(storage_models) > 1:
|
||||
raise TypeError("When defining Distributed engine without the table_name "
|
||||
"ensure that your model has exactly one non-distributed superclass")
|
||||
|
||||
# enable correct SQL for engine
|
||||
cls.engine.table = storage_models[0]
|
||||
|
||||
@classmethod
|
||||
def create_table_sql(cls, db):
|
||||
'''
|
||||
Returns the SQL statement for creating a table for this model.
|
||||
'''
|
||||
assert isinstance(cls.engine, Distributed), "engine must be engines.Distributed instance"
|
||||
|
||||
cls.fix_engine_table()
|
||||
|
||||
parts = [
|
||||
'CREATE TABLE IF NOT EXISTS `{0}`.`{1}` AS `{0}`.`{2}`'.format(
|
||||
db.db_name, cls.table_name(), cls.engine.table_name),
|
||||
'ENGINE = ' + cls.engine.create_table_sql(db)]
|
||||
return '\n'.join(parts)
|
||||
|
||||
|
||||
# Expose only relevant classes in import *
|
||||
__all__ = get_subclass_names(locals(), (Model, Constraint, Index))
|
||||
@@ -0,0 +1,689 @@
|
||||
from __future__ import unicode_literals
|
||||
|
||||
import pytz
|
||||
from copy import copy, deepcopy
|
||||
from math import ceil
|
||||
from datetime import date, datetime
|
||||
from .utils import comma_join, string_or_func, arg_to_sql
|
||||
|
||||
|
||||
# TODO
|
||||
# - check that field names are valid
|
||||
|
||||
class Operator(object):
|
||||
"""
|
||||
Base class for filtering operators.
|
||||
"""
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
"""
|
||||
Subclasses should implement this method. It returns an SQL string
|
||||
that applies this operator on the given field and value.
|
||||
"""
|
||||
raise NotImplementedError # pragma: no cover
|
||||
|
||||
def _value_to_sql(self, field, value, quote=True):
|
||||
from datastore_orm.funcs import F
|
||||
if isinstance(value, F):
|
||||
return value.to_sql()
|
||||
return field.to_db_string(field.to_python(value, pytz.utc), quote)
|
||||
|
||||
|
||||
class SimpleOperator(Operator):
|
||||
"""
|
||||
A simple binary operator such as a=b, a<b, a>b etc.
|
||||
"""
|
||||
|
||||
def __init__(self, sql_operator, sql_for_null=None):
|
||||
self._sql_operator = sql_operator
|
||||
self._sql_for_null = sql_for_null
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
field = getattr(model_cls, field_name)
|
||||
value = self._value_to_sql(field, value)
|
||||
if value == '\\N' and self._sql_for_null is not None:
|
||||
return ' '.join([field_name, self._sql_for_null])
|
||||
return ' '.join([field_name, self._sql_operator, value])
|
||||
|
||||
|
||||
class InOperator(Operator):
|
||||
"""
|
||||
An operator that implements IN.
|
||||
Accepts 3 different types of values:
|
||||
- a list or tuple of simple values
|
||||
- a string (used verbatim as the contents of the parenthesis)
|
||||
- a queryset (subquery)
|
||||
"""
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
field = getattr(model_cls, field_name)
|
||||
if isinstance(value, QuerySet):
|
||||
value = value.as_sql()
|
||||
elif isinstance(value, str):
|
||||
pass
|
||||
else:
|
||||
value = comma_join([self._value_to_sql(field, v) for v in value])
|
||||
return '%s IN (%s)' % (field_name, value)
|
||||
|
||||
|
||||
class LikeOperator(Operator):
|
||||
"""
|
||||
A LIKE operator that matches the field to a given pattern. Can be
|
||||
case sensitive or insensitive.
|
||||
"""
|
||||
|
||||
def __init__(self, pattern, case_sensitive=True):
|
||||
self._pattern = pattern
|
||||
self._case_sensitive = case_sensitive
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
field = getattr(model_cls, field_name)
|
||||
value = self._value_to_sql(field, value, quote=False)
|
||||
value = value.replace('\\', '\\\\').replace('%', '\\\\%').replace('_', '\\\\_')
|
||||
pattern = self._pattern.format(value)
|
||||
if self._case_sensitive:
|
||||
return '%s LIKE \'%s\'' % (field_name, pattern)
|
||||
else:
|
||||
return 'lowerUTF8(%s) LIKE lowerUTF8(\'%s\')' % (field_name, pattern)
|
||||
|
||||
|
||||
class IExactOperator(Operator):
|
||||
"""
|
||||
An operator for case insensitive string comparison.
|
||||
"""
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
field = getattr(model_cls, field_name)
|
||||
value = self._value_to_sql(field, value)
|
||||
return 'lowerUTF8(%s) = lowerUTF8(%s)' % (field_name, value)
|
||||
|
||||
|
||||
class NotOperator(Operator):
|
||||
"""
|
||||
A wrapper around another operator, which negates it.
|
||||
"""
|
||||
|
||||
def __init__(self, base_operator):
|
||||
self._base_operator = base_operator
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
# Negate the base operator
|
||||
return 'NOT (%s)' % self._base_operator.to_sql(model_cls, field_name, value)
|
||||
|
||||
|
||||
class BetweenOperator(Operator):
|
||||
"""
|
||||
An operator that implements BETWEEN.
|
||||
Accepts list or tuple of two elements and generates sql condition:
|
||||
- 'BETWEEN value[0] AND value[1]' if value[0] and value[1] are not None and not empty
|
||||
Then imitations of BETWEEN, where one of two limits is missing
|
||||
- '>= value[0]' if value[1] is None or empty
|
||||
- '<= value[1]' if value[0] is None or empty
|
||||
"""
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
field = getattr(model_cls, field_name)
|
||||
value0 = self._value_to_sql(field, value[0]) if value[0] is not None or len(str(value[0])) > 0 else None
|
||||
value1 = self._value_to_sql(field, value[1]) if value[1] is not None or len(str(value[1])) > 0 else None
|
||||
if value0 and value1:
|
||||
return '%s BETWEEN %s AND %s' % (field_name, value0, value1)
|
||||
if value0 and not value1:
|
||||
return ' '.join([field_name, '>=', value0])
|
||||
if value1 and not value0:
|
||||
return ' '.join([field_name, '<=', value1])
|
||||
|
||||
# Define the set of builtin operators
|
||||
|
||||
_operators = {}
|
||||
|
||||
def register_operator(name, sql):
|
||||
_operators[name] = sql
|
||||
|
||||
register_operator('eq', SimpleOperator('=', 'IS NULL'))
|
||||
register_operator('ne', SimpleOperator('!=', 'IS NOT NULL'))
|
||||
register_operator('gt', SimpleOperator('>'))
|
||||
register_operator('gte', SimpleOperator('>='))
|
||||
register_operator('lt', SimpleOperator('<'))
|
||||
register_operator('lte', SimpleOperator('<='))
|
||||
register_operator('between', BetweenOperator())
|
||||
register_operator('in', InOperator())
|
||||
register_operator('not_in', NotOperator(InOperator()))
|
||||
register_operator('contains', LikeOperator('%{}%'))
|
||||
register_operator('startswith', LikeOperator('{}%'))
|
||||
register_operator('endswith', LikeOperator('%{}'))
|
||||
register_operator('icontains', LikeOperator('%{}%', False))
|
||||
register_operator('istartswith', LikeOperator('{}%', False))
|
||||
register_operator('iendswith', LikeOperator('%{}', False))
|
||||
register_operator('iexact', IExactOperator())
|
||||
|
||||
|
||||
class Cond(object):
|
||||
"""
|
||||
An abstract object for storing a single query condition Field + Operator + Value.
|
||||
"""
|
||||
|
||||
def to_sql(self, model_cls):
|
||||
raise NotImplementedError
|
||||
|
||||
|
||||
class FieldCond(Cond):
|
||||
"""
|
||||
A single query condition made up of Field + Operator + Value.
|
||||
"""
|
||||
def __init__(self, field_name, operator, value):
|
||||
self._field_name = field_name
|
||||
self._operator = _operators.get(operator)
|
||||
if self._operator is None:
|
||||
# The field name contains __ like my__field
|
||||
self._field_name = field_name + '__' + operator
|
||||
self._operator = _operators['eq']
|
||||
self._value = value
|
||||
|
||||
def to_sql(self, model_cls):
|
||||
return self._operator.to_sql(model_cls, self._field_name, self._value)
|
||||
|
||||
def __deepcopy__(self, memodict={}):
|
||||
res = copy(self)
|
||||
res._value = deepcopy(self._value)
|
||||
return res
|
||||
|
||||
|
||||
class Q(object):
|
||||
|
||||
AND_MODE = 'AND'
|
||||
OR_MODE = 'OR'
|
||||
|
||||
def __init__(self, *filter_funcs, **filter_fields):
|
||||
self._conds = list(filter_funcs) + [self._build_cond(k, v) for k, v in filter_fields.items()]
|
||||
self._children = []
|
||||
self._negate = False
|
||||
self._mode = self.AND_MODE
|
||||
|
||||
@property
|
||||
def is_empty(self):
|
||||
"""
|
||||
Checks if there are any conditions in Q object
|
||||
Returns: Boolean
|
||||
"""
|
||||
return not bool(self._conds or self._children)
|
||||
|
||||
@classmethod
|
||||
def _construct_from(cls, l_child, r_child, mode):
|
||||
if mode == l_child._mode:
|
||||
q = deepcopy(l_child)
|
||||
q._children.append(deepcopy(r_child))
|
||||
elif mode == r_child._mode:
|
||||
q = deepcopy(r_child)
|
||||
q._children.append(deepcopy(l_child))
|
||||
else:
|
||||
# Different modes
|
||||
q = Q()
|
||||
q._children = [l_child, r_child]
|
||||
q._mode = mode # AND/OR
|
||||
|
||||
return q
|
||||
|
||||
def _build_cond(self, key, value):
|
||||
if '__' in key:
|
||||
field_name, operator = key.rsplit('__', 1)
|
||||
else:
|
||||
field_name, operator = key, 'eq'
|
||||
return FieldCond(field_name, operator, value)
|
||||
|
||||
def to_sql(self, model_cls):
|
||||
condition_sql = []
|
||||
|
||||
if self._conds:
|
||||
condition_sql.extend([cond.to_sql(model_cls) for cond in self._conds])
|
||||
|
||||
if self._children:
|
||||
condition_sql.extend([child.to_sql(model_cls) for child in self._children if child])
|
||||
|
||||
if not condition_sql:
|
||||
# Empty Q() object returns everything
|
||||
sql = '1'
|
||||
elif len(condition_sql) == 1:
|
||||
# Skip not needed brackets over single condition
|
||||
sql = condition_sql[0]
|
||||
else:
|
||||
# Each condition must be enclosed in brackets, or order of operations may be wrong
|
||||
sql = '(%s)' % ') {} ('.format(self._mode).join(condition_sql)
|
||||
|
||||
if self._negate:
|
||||
sql = 'NOT (%s)' % sql
|
||||
|
||||
return sql
|
||||
|
||||
def __or__(self, other):
|
||||
return Q._construct_from(self, other, self.OR_MODE)
|
||||
|
||||
def __and__(self, other):
|
||||
return Q._construct_from(self, other, self.AND_MODE)
|
||||
|
||||
def __invert__(self):
|
||||
q = copy(self)
|
||||
q._negate = True
|
||||
return q
|
||||
|
||||
def __bool__(self):
|
||||
return not self.is_empty
|
||||
|
||||
def __deepcopy__(self, memodict={}):
|
||||
q = Q()
|
||||
q._conds = [deepcopy(cond) for cond in self._conds]
|
||||
q._negate = self._negate
|
||||
q._mode = self._mode
|
||||
|
||||
if self._children:
|
||||
q._children = [deepcopy(child) for child in self._children]
|
||||
|
||||
return q
|
||||
|
||||
|
||||
class QuerySet(object):
|
||||
"""
|
||||
A queryset is an object that represents a database query using a specific `Model`.
|
||||
It is lazy, meaning that it does not hit the database until you iterate over its
|
||||
matching rows (model instances).
|
||||
"""
|
||||
|
||||
def __init__(self, model_cls, database):
|
||||
"""
|
||||
Initializer. It is possible to create a queryset like this, but the standard
|
||||
way is to use `MyModel.objects_in(database)`.
|
||||
"""
|
||||
self._model_cls = model_cls
|
||||
self._database = database
|
||||
self._order_by = []
|
||||
self._where_q = Q()
|
||||
self._prewhere_q = Q()
|
||||
self._grouping_fields = []
|
||||
self._grouping_with_totals = False
|
||||
self._fields = model_cls.fields().keys()
|
||||
self._limits = None
|
||||
self._limit_by = None
|
||||
self._limit_by_fields = None
|
||||
self._distinct = False
|
||||
self._final = False
|
||||
|
||||
def __iter__(self):
|
||||
"""
|
||||
Iterates over the model instances matching this queryset
|
||||
"""
|
||||
return self._database.select(self.as_sql(), self._model_cls)
|
||||
|
||||
def __bool__(self):
|
||||
"""
|
||||
Returns true if this queryset matches any rows.
|
||||
"""
|
||||
return bool(self.count())
|
||||
|
||||
def __nonzero__(self): # Python 2 compatibility
|
||||
return type(self).__bool__(self)
|
||||
|
||||
def __str__(self):
|
||||
return self.as_sql()
|
||||
|
||||
def __getitem__(self, s):
|
||||
if isinstance(s, int):
|
||||
# Single index
|
||||
assert s >= 0, 'negative indexes are not supported'
|
||||
qs = copy(self)
|
||||
qs._limits = (s, 1)
|
||||
return next(iter(qs))
|
||||
else:
|
||||
# Slice
|
||||
assert s.step in (None, 1), 'step is not supported in slices'
|
||||
start = s.start or 0
|
||||
stop = s.stop or 2**63 - 1
|
||||
assert start >= 0 and stop >= 0, 'negative indexes are not supported'
|
||||
assert start <= stop, 'start of slice cannot be smaller than its end'
|
||||
qs = copy(self)
|
||||
qs._limits = (start, stop - start)
|
||||
return qs
|
||||
|
||||
def limit_by(self, offset_limit, *fields_or_expr):
|
||||
"""
|
||||
Adds a LIMIT BY clause to the query.
|
||||
- `offset_limit`: either an integer specifying the limit, or a tuple of integers (offset, limit).
|
||||
- `fields_or_expr`: the field names or expressions to use in the clause.
|
||||
"""
|
||||
if isinstance(offset_limit, int):
|
||||
# Single limit
|
||||
offset_limit = (0, offset_limit)
|
||||
offset = offset_limit[0]
|
||||
limit = offset_limit[1]
|
||||
assert offset >= 0 and limit >= 0, 'negative limits are not supported'
|
||||
qs = copy(self)
|
||||
qs._limit_by = (offset, limit)
|
||||
qs._limit_by_fields = fields_or_expr
|
||||
return qs
|
||||
|
||||
def select_fields_as_sql(self):
|
||||
"""
|
||||
Returns the selected fields or expressions as a SQL string.
|
||||
"""
|
||||
fields = '*'
|
||||
if self._fields:
|
||||
fields = comma_join('`%s`' % field for field in self._fields)
|
||||
return fields
|
||||
|
||||
def as_sql(self):
|
||||
"""
|
||||
Returns the whole query as a SQL string.
|
||||
"""
|
||||
distinct = 'DISTINCT ' if self._distinct else ''
|
||||
final = ' FINAL' if self._final else ''
|
||||
table_name = '`%s`' % self._model_cls.table_name()
|
||||
if self._model_cls.is_system_model():
|
||||
table_name = '`system`.' + table_name
|
||||
params = (distinct, self.select_fields_as_sql(), table_name, final)
|
||||
sql = u'SELECT %s%s\nFROM %s%s' % params
|
||||
|
||||
if self._prewhere_q and not self._prewhere_q.is_empty:
|
||||
sql += '\nPREWHERE ' + self.conditions_as_sql(prewhere=True)
|
||||
|
||||
if self._where_q and not self._where_q.is_empty:
|
||||
sql += '\nWHERE ' + self.conditions_as_sql(prewhere=False)
|
||||
|
||||
if self._grouping_fields:
|
||||
sql += '\nGROUP BY %s' % comma_join('`%s`' % field for field in self._grouping_fields)
|
||||
|
||||
if self._grouping_with_totals:
|
||||
sql += ' WITH TOTALS'
|
||||
|
||||
if self._order_by:
|
||||
sql += '\nORDER BY ' + self.order_by_as_sql()
|
||||
|
||||
if self._limit_by:
|
||||
sql += '\nLIMIT %d, %d' % self._limit_by
|
||||
sql += ' BY %s' % comma_join(string_or_func(field) for field in self._limit_by_fields)
|
||||
|
||||
if self._limits:
|
||||
sql += '\nLIMIT %d, %d' % self._limits
|
||||
|
||||
return sql
|
||||
|
||||
def order_by_as_sql(self):
|
||||
"""
|
||||
Returns the contents of the query's `ORDER BY` clause as a string.
|
||||
"""
|
||||
return comma_join([
|
||||
'%s DESC' % field[1:] if isinstance(field, str) and field[0] == '-' else str(field)
|
||||
for field in self._order_by
|
||||
])
|
||||
|
||||
def conditions_as_sql(self, prewhere=False):
|
||||
"""
|
||||
Returns the contents of the query's `WHERE` or `PREWHERE` clause as a string.
|
||||
"""
|
||||
q_object = self._prewhere_q if prewhere else self._where_q
|
||||
return q_object.to_sql(self._model_cls)
|
||||
|
||||
def count(self):
|
||||
"""
|
||||
Returns the number of matching model instances.
|
||||
"""
|
||||
if self._distinct or self._limits:
|
||||
# Use a subquery, since a simple count won't be accurate
|
||||
sql = u'SELECT count() FROM (%s)' % self.as_sql()
|
||||
raw = self._database.raw(sql)
|
||||
return int(raw) if raw else 0
|
||||
|
||||
# Simple case
|
||||
conditions = (self._where_q & self._prewhere_q).to_sql(self._model_cls)
|
||||
return self._database.count(self._model_cls, conditions)
|
||||
|
||||
def order_by(self, *field_names):
|
||||
"""
|
||||
Returns a copy of this queryset with the ordering changed.
|
||||
"""
|
||||
qs = copy(self)
|
||||
qs._order_by = field_names
|
||||
return qs
|
||||
|
||||
def only(self, *field_names):
|
||||
"""
|
||||
Returns a copy of this queryset limited to the specified field names.
|
||||
Useful when there are large fields that are not needed,
|
||||
or for creating a subquery to use with an IN operator.
|
||||
"""
|
||||
qs = copy(self)
|
||||
qs._fields = field_names
|
||||
return qs
|
||||
|
||||
def _filter_or_exclude(self, *q, **kwargs):
|
||||
from .funcs import F
|
||||
|
||||
inverse = kwargs.pop('_inverse', False)
|
||||
prewhere = kwargs.pop('prewhere', False)
|
||||
|
||||
qs = copy(self)
|
||||
|
||||
condition = Q()
|
||||
for arg in q:
|
||||
if isinstance(arg, Q):
|
||||
condition &= arg
|
||||
elif isinstance(arg, F):
|
||||
condition &= Q(arg)
|
||||
else:
|
||||
raise TypeError('Invalid argument "%r" to queryset filter' % arg)
|
||||
|
||||
if kwargs:
|
||||
condition &= Q(**kwargs)
|
||||
|
||||
if inverse:
|
||||
condition = ~condition
|
||||
|
||||
condition = copy(self._prewhere_q if prewhere else self._where_q) & condition
|
||||
if prewhere:
|
||||
qs._prewhere_q = condition
|
||||
else:
|
||||
qs._where_q = condition
|
||||
|
||||
return qs
|
||||
|
||||
def filter(self, *q, **kwargs):
|
||||
"""
|
||||
Returns a copy of this queryset that includes only rows matching the conditions.
|
||||
Pass `prewhere=True` to apply the conditions as PREWHERE instead of WHERE.
|
||||
"""
|
||||
return self._filter_or_exclude(*q, **kwargs)
|
||||
|
||||
def exclude(self, *q, **kwargs):
|
||||
"""
|
||||
Returns a copy of this queryset that excludes all rows matching the conditions.
|
||||
Pass `prewhere=True` to apply the conditions as PREWHERE instead of WHERE.
|
||||
"""
|
||||
return self._filter_or_exclude(*q, _inverse=True, **kwargs)
|
||||
|
||||
def paginate(self, page_num=1, page_size=100):
|
||||
"""
|
||||
Returns a single page of model instances that match the queryset.
|
||||
Note that `order_by` should be used first, to ensure a correct
|
||||
partitioning of records into pages.
|
||||
|
||||
- `page_num`: the page number (1-based), or -1 to get the last page.
|
||||
- `page_size`: number of records to return per page.
|
||||
|
||||
The result is a namedtuple containing `objects` (list), `number_of_objects`,
|
||||
`pages_total`, `number` (of the current page), and `page_size`.
|
||||
"""
|
||||
from .database import Page
|
||||
count = self.count()
|
||||
pages_total = int(ceil(count / float(page_size)))
|
||||
if page_num == -1:
|
||||
page_num = pages_total
|
||||
elif page_num < 1:
|
||||
raise ValueError('Invalid page number: %d' % page_num)
|
||||
offset = (page_num - 1) * page_size
|
||||
return Page(
|
||||
objects=list(self[offset : offset + page_size]),
|
||||
number_of_objects=count,
|
||||
pages_total=pages_total,
|
||||
number=page_num,
|
||||
page_size=page_size
|
||||
)
|
||||
|
||||
def distinct(self):
|
||||
"""
|
||||
Adds a DISTINCT clause to the query, meaning that any duplicate rows
|
||||
in the results will be omitted.
|
||||
"""
|
||||
qs = copy(self)
|
||||
qs._distinct = True
|
||||
return qs
|
||||
|
||||
def final(self):
|
||||
"""
|
||||
Adds a FINAL modifier to table, meaning data will be collapsed to final version.
|
||||
Can be used with the `CollapsingMergeTree` and `ReplacingMergeTree` engines only.
|
||||
"""
|
||||
from .engines import CollapsingMergeTree, ReplacingMergeTree
|
||||
if not isinstance(self._model_cls.engine, (CollapsingMergeTree, ReplacingMergeTree)):
|
||||
raise TypeError('final() method can be used only with the CollapsingMergeTree and ReplacingMergeTree engines')
|
||||
|
||||
qs = copy(self)
|
||||
qs._final = True
|
||||
return qs
|
||||
|
||||
def delete(self):
|
||||
"""
|
||||
Deletes all records matched by this queryset's conditions.
|
||||
Note that ClickHouse performs deletions in the background, so they are not immediate.
|
||||
"""
|
||||
self._verify_mutation_allowed()
|
||||
conditions = (self._where_q & self._prewhere_q).to_sql(self._model_cls)
|
||||
sql = 'ALTER TABLE $db.`%s` DELETE WHERE %s' % (self._model_cls.table_name(), conditions)
|
||||
self._database.raw(sql)
|
||||
return self
|
||||
|
||||
def update(self, **kwargs):
|
||||
"""
|
||||
Updates all records matched by this queryset's conditions.
|
||||
Keyword arguments specify the field names and expressions to use for the update.
|
||||
Note that ClickHouse performs updates in the background, so they are not immediate.
|
||||
"""
|
||||
assert kwargs, 'No fields specified for update'
|
||||
self._verify_mutation_allowed()
|
||||
fields = comma_join('`%s` = %s' % (name, arg_to_sql(expr)) for name, expr in kwargs.items())
|
||||
conditions = (self._where_q & self._prewhere_q).to_sql(self._model_cls)
|
||||
sql = 'ALTER TABLE $db.`%s` UPDATE %s WHERE %s' % (self._model_cls.table_name(), fields, conditions)
|
||||
self._database.raw(sql)
|
||||
return self
|
||||
|
||||
def _verify_mutation_allowed(self):
|
||||
'''
|
||||
Checks that the queryset's state allows mutations. Raises an AssertionError if not.
|
||||
'''
|
||||
assert not self._limits, 'Mutations are not allowed after slicing the queryset'
|
||||
assert not self._limit_by, 'Mutations are not allowed after calling limit_by(...)'
|
||||
assert not self._distinct, 'Mutations are not allowed after calling distinct()'
|
||||
assert not self._final, 'Mutations are not allowed after calling final()'
|
||||
|
||||
def aggregate(self, *args, **kwargs):
|
||||
"""
|
||||
Returns an `AggregateQuerySet` over this query, with `args` serving as
|
||||
grouping fields and `kwargs` serving as calculated fields. At least one
|
||||
calculated field is required. For example:
|
||||
```
|
||||
Event.objects_in(database).filter(date__gt='2017-08-01').aggregate('event_type', count='count()')
|
||||
```
|
||||
is equivalent to:
|
||||
```
|
||||
SELECT event_type, count() AS count FROM event
|
||||
WHERE data > '2017-08-01'
|
||||
GROUP BY event_type
|
||||
```
|
||||
"""
|
||||
return AggregateQuerySet(self, args, kwargs)
|
||||
|
||||
|
||||
class AggregateQuerySet(QuerySet):
|
||||
"""
|
||||
A queryset used for aggregation.
|
||||
"""
|
||||
|
||||
def __init__(self, base_qs, grouping_fields, calculated_fields):
|
||||
"""
|
||||
Initializer. Normally you should not call this but rather use `QuerySet.aggregate()`.
|
||||
|
||||
The grouping fields should be a list/tuple of field names from the model. For example:
|
||||
```
|
||||
('event_type', 'event_subtype')
|
||||
```
|
||||
The calculated fields should be a mapping from name to a ClickHouse aggregation function. For example:
|
||||
```
|
||||
{'weekday': 'toDayOfWeek(event_date)', 'number_of_events': 'count()'}
|
||||
```
|
||||
At least one calculated field is required.
|
||||
"""
|
||||
super(AggregateQuerySet, self).__init__(base_qs._model_cls, base_qs._database)
|
||||
assert calculated_fields, 'No calculated fields specified for aggregation'
|
||||
self._fields = grouping_fields
|
||||
self._grouping_fields = grouping_fields
|
||||
self._calculated_fields = calculated_fields
|
||||
self._order_by = list(base_qs._order_by)
|
||||
self._where_q = base_qs._where_q
|
||||
self._prewhere_q = base_qs._prewhere_q
|
||||
self._limits = base_qs._limits
|
||||
self._distinct = base_qs._distinct
|
||||
|
||||
def group_by(self, *args):
|
||||
"""
|
||||
This method lets you specify the grouping fields explicitly. The `args` must
|
||||
be names of grouping fields or calculated fields that this queryset was
|
||||
created with.
|
||||
"""
|
||||
for name in args:
|
||||
assert name in self._fields or name in self._calculated_fields, \
|
||||
'Cannot group by `%s` since it is not included in the query' % name
|
||||
qs = copy(self)
|
||||
qs._grouping_fields = args
|
||||
return qs
|
||||
|
||||
def only(self, *field_names):
|
||||
"""
|
||||
This method is not supported on `AggregateQuerySet`.
|
||||
"""
|
||||
raise NotImplementedError('Cannot use "only" with AggregateQuerySet')
|
||||
|
||||
def aggregate(self, *args, **kwargs):
|
||||
"""
|
||||
This method is not supported on `AggregateQuerySet`.
|
||||
"""
|
||||
raise NotImplementedError('Cannot re-aggregate an AggregateQuerySet')
|
||||
|
||||
def select_fields_as_sql(self):
|
||||
"""
|
||||
Returns the selected fields or expressions as a SQL string.
|
||||
"""
|
||||
return comma_join([str(f) for f in self._fields] + ['%s AS %s' % (v, k) for k, v in self._calculated_fields.items()])
|
||||
|
||||
def __iter__(self):
|
||||
return self._database.select(self.as_sql()) # using an ad-hoc model
|
||||
|
||||
def count(self):
|
||||
"""
|
||||
Returns the number of rows after aggregation.
|
||||
"""
|
||||
sql = u'SELECT count() FROM (%s)' % self.as_sql()
|
||||
raw = self._database.raw(sql)
|
||||
return int(raw) if raw else 0
|
||||
|
||||
def with_totals(self):
|
||||
"""
|
||||
Adds WITH TOTALS modifier ot GROUP BY, making query return extra row
|
||||
with aggregate function calculated across all the rows. More information:
|
||||
https://clickhouse.tech/docs/en/query_language/select/#with-totals-modifier
|
||||
"""
|
||||
qs = copy(self)
|
||||
qs._grouping_with_totals = True
|
||||
return qs
|
||||
|
||||
def _verify_mutation_allowed(self):
|
||||
raise AssertionError('Cannot mutate an AggregateQuerySet')
|
||||
|
||||
|
||||
# Expose only relevant classes in import *
|
||||
__all__ = [c.__name__ for c in [Q, QuerySet, AggregateQuerySet]]
|
||||
@@ -1,28 +1,35 @@
|
||||
"""
|
||||
This file contains system readonly models that can be got from database
|
||||
https://clickhouse.yandex/reference_en.html#System tables
|
||||
This file contains system readonly models that can be got from the database
|
||||
https://clickhouse.tech/docs/en/system_tables/
|
||||
"""
|
||||
from __future__ import unicode_literals
|
||||
|
||||
from .database import Database
|
||||
from .fields import *
|
||||
from .models import Model
|
||||
from .utils import comma_join
|
||||
|
||||
|
||||
class SystemPart(Model):
|
||||
"""
|
||||
Contains information about parts of a table in the MergeTree family.
|
||||
This model operates only fields, described in the reference. Other fields are ignored.
|
||||
https://clickhouse.yandex/reference_en.html#system.parts
|
||||
https://clickhouse.tech/docs/en/system_tables/system.parts/
|
||||
"""
|
||||
OPERATIONS = frozenset({'DETACH', 'DROP', 'ATTACH', 'FREEZE', 'FETCH'})
|
||||
|
||||
readonly = True
|
||||
_readonly = True
|
||||
_system = True
|
||||
|
||||
database = StringField() # Name of the database where the table that this part belongs to is located.
|
||||
table = StringField() # Name of the table that this part belongs to.
|
||||
engine = StringField() # Name of the table engine, without parameters.
|
||||
partition = StringField() # Name of the partition, in the format YYYYMM.
|
||||
name = StringField() # Name of the part.
|
||||
replicated = UInt8Field() # Whether the part belongs to replicated data.
|
||||
|
||||
# This field is present in the docs (https://clickhouse.tech/docs/en/single/index.html#system-parts),
|
||||
# but is absent in ClickHouse (in version 1.1.54245)
|
||||
# replicated = UInt8Field() # Whether the part belongs to replicated data.
|
||||
|
||||
# Whether the part is used in a table, or is no longer needed and will be deleted soon.
|
||||
# Inactive parts remain after merging.
|
||||
@@ -44,23 +51,26 @@ class SystemPart(Model):
|
||||
|
||||
@classmethod
|
||||
def table_name(cls):
|
||||
return 'system.parts'
|
||||
return 'parts'
|
||||
|
||||
"""
|
||||
Next methods return SQL for some operations, which can be done with partitions
|
||||
https://clickhouse.yandex/reference_en.html#Manipulations with partitions and parts
|
||||
https://clickhouse.tech/docs/en/query_language/queries/#manipulations-with-partitions-and-parts
|
||||
"""
|
||||
def _partition_operation_sql(self, operation, settings=None, from_part=None):
|
||||
"""
|
||||
Performs some operation over partition
|
||||
:param db: Database object to execute operation on
|
||||
:param operation: Operation to execute from SystemPart.OPERATIONS set
|
||||
:param settings: Settings for executing request to ClickHouse over db.raw() method
|
||||
:return: Operation execution result
|
||||
|
||||
- `db`: Database object to execute operation on
|
||||
- `operation`: Operation to execute from SystemPart.OPERATIONS set
|
||||
- `settings`: Settings for executing request to ClickHouse over db.raw() method
|
||||
|
||||
Returns: Operation execution result
|
||||
"""
|
||||
operation = operation.upper()
|
||||
assert operation in self.OPERATIONS, "operation must be in [%s]" % ', '.join(self.OPERATIONS)
|
||||
sql = "ALTER TABLE `%s`.`%s` %s PARTITION '%s'" % (self._database.db_name, self.table, operation, self.partition)
|
||||
assert operation in self.OPERATIONS, "operation must be in [%s]" % comma_join(self.OPERATIONS)
|
||||
|
||||
sql = "ALTER TABLE `%s`.`%s` %s PARTITION %s" % (self._database.db_name, self.table, operation, self.partition)
|
||||
if from_part is not None:
|
||||
sql += " FROM %s" % from_part
|
||||
self._database.raw(sql, settings=settings, stream=False)
|
||||
@@ -68,41 +78,51 @@ class SystemPart(Model):
|
||||
def detach(self, settings=None):
|
||||
"""
|
||||
Move a partition to the 'detached' directory and forget it.
|
||||
:param settings: Settings for executing request to ClickHouse over db.raw() method
|
||||
:return: SQL Query
|
||||
|
||||
- `settings`: Settings for executing request to ClickHouse over db.raw() method
|
||||
|
||||
Returns: SQL Query
|
||||
"""
|
||||
return self._partition_operation_sql('DETACH', settings=settings)
|
||||
|
||||
def drop(self, settings=None):
|
||||
"""
|
||||
Delete a partition
|
||||
:param settings: Settings for executing request to ClickHouse over db.raw() method
|
||||
:return: SQL Query
|
||||
|
||||
- `settings`: Settings for executing request to ClickHouse over db.raw() method
|
||||
|
||||
Returns: SQL Query
|
||||
"""
|
||||
return self._partition_operation_sql('DROP', settings=settings)
|
||||
|
||||
def attach(self, settings=None):
|
||||
"""
|
||||
Add a new part or partition from the 'detached' directory to the table.
|
||||
:param settings: Settings for executing request to ClickHouse over db.raw() method
|
||||
:return: SQL Query
|
||||
|
||||
- `settings`: Settings for executing request to ClickHouse over db.raw() method
|
||||
|
||||
Returns: SQL Query
|
||||
"""
|
||||
return self._partition_operation_sql('ATTACH', settings=settings)
|
||||
|
||||
def freeze(self, settings=None):
|
||||
"""
|
||||
Create a backup of a partition.
|
||||
:param settings: Settings for executing request to ClickHouse over db.raw() method
|
||||
:return: SQL Query
|
||||
|
||||
- `settings`: Settings for executing request to ClickHouse over db.raw() method
|
||||
|
||||
Returns: SQL Query
|
||||
"""
|
||||
return self._partition_operation_sql('FREEZE', settings=settings)
|
||||
|
||||
def fetch(self, zookeeper_path, settings=None):
|
||||
"""
|
||||
Download a partition from another server.
|
||||
:param zookeeper_path: Path in zookeeper to fetch from
|
||||
:param settings: Settings for executing request to ClickHouse over db.raw() method
|
||||
:return: SQL Query
|
||||
|
||||
- `zookeeper_path`: Path in zookeeper to fetch from
|
||||
- `settings`: Settings for executing request to ClickHouse over db.raw() method
|
||||
|
||||
Returns: SQL Query
|
||||
"""
|
||||
return self._partition_operation_sql('FETCH', settings=settings, from_part=zookeeper_path)
|
||||
|
||||
@@ -110,27 +130,35 @@ class SystemPart(Model):
|
||||
def get(cls, database, conditions=""):
|
||||
"""
|
||||
Get all data from system.parts table
|
||||
:param database: A database object to fetch data from.
|
||||
:param conditions: WHERE clause conditions. Database condition is added automatically
|
||||
:return: A list of SystemPart objects
|
||||
|
||||
- `database`: A database object to fetch data from.
|
||||
- `conditions`: WHERE clause conditions. Database condition is added automatically
|
||||
|
||||
Returns: A list of SystemPart objects
|
||||
"""
|
||||
assert isinstance(database, Database), "database must be database.Database class instance"
|
||||
assert isinstance(conditions, str), "conditions must be a string"
|
||||
if conditions:
|
||||
conditions += " AND"
|
||||
field_names = ','.join([f[0] for f in cls._fields])
|
||||
return database.select("SELECT %s FROM %s WHERE %s database='%s'" %
|
||||
(field_names, cls.table_name(), conditions, database.db_name), model_class=cls)
|
||||
field_names = ','.join(cls.fields())
|
||||
return database.select("SELECT %s FROM `system`.%s WHERE %s database='%s'" %
|
||||
(field_names, cls.table_name(), conditions, database.db_name), model_class=cls)
|
||||
|
||||
@classmethod
|
||||
def get_active(cls, database, conditions=""):
|
||||
"""
|
||||
Gets active data from system.parts table
|
||||
:param database: A database object to fetch data from.
|
||||
:param conditions: WHERE clause conditions. Database and active conditions are added automatically
|
||||
:return: A list of SystemPart objects
|
||||
|
||||
- `database`: A database object to fetch data from.
|
||||
- `conditions`: WHERE clause conditions. Database and active conditions are added automatically
|
||||
|
||||
Returns: A list of SystemPart objects
|
||||
"""
|
||||
if conditions:
|
||||
conditions += ' AND '
|
||||
conditions += 'active'
|
||||
return SystemPart.get(database, conditions=conditions)
|
||||
|
||||
|
||||
# Expose only relevant classes in import *
|
||||
__all__ = [c.__name__ for c in [SystemPart]]
|
||||
@@ -0,0 +1,173 @@
|
||||
import codecs
|
||||
import re
|
||||
from datetime import date, datetime, tzinfo, timedelta
|
||||
|
||||
|
||||
SPECIAL_CHARS = {
|
||||
"\b" : "\\b",
|
||||
"\f" : "\\f",
|
||||
"\r" : "\\r",
|
||||
"\n" : "\\n",
|
||||
"\t" : "\\t",
|
||||
"\0" : "\\0",
|
||||
"\\" : "\\\\",
|
||||
"'" : "\\'"
|
||||
}
|
||||
|
||||
SPECIAL_CHARS_REGEX = re.compile("[" + ''.join(SPECIAL_CHARS.values()) + "]")
|
||||
|
||||
|
||||
|
||||
def escape(value, quote=True):
|
||||
'''
|
||||
If the value is a string, escapes any special characters and optionally
|
||||
surrounds it with single quotes. If the value is not a string (e.g. a number),
|
||||
converts it to one.
|
||||
'''
|
||||
def escape_one(match):
|
||||
return SPECIAL_CHARS[match.group(0)]
|
||||
|
||||
if isinstance(value, str):
|
||||
value = SPECIAL_CHARS_REGEX.sub(escape_one, value)
|
||||
if quote:
|
||||
value = "'" + value + "'"
|
||||
return str(value)
|
||||
|
||||
|
||||
def unescape(value):
|
||||
return codecs.escape_decode(value)[0].decode('utf-8')
|
||||
|
||||
|
||||
def string_or_func(obj):
|
||||
return obj.to_sql() if hasattr(obj, 'to_sql') else obj
|
||||
|
||||
|
||||
def arg_to_sql(arg):
|
||||
"""
|
||||
Converts a function argument to SQL string according to its type.
|
||||
Supports functions, model fields, strings, dates, datetimes, timedeltas, booleans,
|
||||
None, numbers, timezones, arrays/iterables.
|
||||
"""
|
||||
from datastore_orm import Field, StringField, DateTimeField, DateField, F, QuerySet
|
||||
if isinstance(arg, F):
|
||||
return arg.to_sql()
|
||||
if isinstance(arg, Field):
|
||||
return "`%s`" % arg
|
||||
if isinstance(arg, str):
|
||||
return StringField().to_db_string(arg)
|
||||
if isinstance(arg, datetime):
|
||||
return "toDateTime(%s)" % DateTimeField().to_db_string(arg)
|
||||
if isinstance(arg, date):
|
||||
return "toDate('%s')" % arg.isoformat()
|
||||
if isinstance(arg, timedelta):
|
||||
return "toIntervalSecond(%d)" % int(arg.total_seconds())
|
||||
if isinstance(arg, bool):
|
||||
return str(int(arg))
|
||||
if isinstance(arg, tzinfo):
|
||||
return StringField().to_db_string(arg.tzname(None))
|
||||
if arg is None:
|
||||
return 'NULL'
|
||||
if isinstance(arg, QuerySet):
|
||||
return "(%s)" % arg
|
||||
if isinstance(arg, tuple):
|
||||
return '(' + comma_join(arg_to_sql(x) for x in arg) + ')'
|
||||
if is_iterable(arg):
|
||||
return '[' + comma_join(arg_to_sql(x) for x in arg) + ']'
|
||||
return str(arg)
|
||||
|
||||
|
||||
def parse_tsv(line):
|
||||
if isinstance(line, bytes):
|
||||
line = line.decode()
|
||||
if line and line[-1] == '\n':
|
||||
line = line[:-1]
|
||||
return [unescape(value) for value in line.split(str('\t'))]
|
||||
|
||||
|
||||
def parse_array(array_string):
|
||||
"""
|
||||
Parse an array or tuple string as returned by clickhouse. For example:
|
||||
"['hello', 'world']" ==> ["hello", "world"]
|
||||
"(1,2,3)" ==> [1, 2, 3]
|
||||
"""
|
||||
# Sanity check
|
||||
if len(array_string) < 2 or array_string[0] not in '[(' or array_string[-1] not in '])':
|
||||
raise ValueError('Invalid array string: "%s"' % array_string)
|
||||
# Drop opening brace
|
||||
array_string = array_string[1:]
|
||||
# Go over the string, lopping off each value at the beginning until nothing is left
|
||||
values = []
|
||||
while True:
|
||||
if array_string in '])':
|
||||
# End of array
|
||||
return values
|
||||
elif array_string[0] in ', ':
|
||||
# In between values
|
||||
array_string = array_string[1:]
|
||||
elif array_string[0] == "'":
|
||||
# Start of quoted value, find its end
|
||||
match = re.search(r"[^\\]'", array_string)
|
||||
if match is None:
|
||||
raise ValueError('Missing closing quote: "%s"' % array_string)
|
||||
values.append(array_string[1 : match.start() + 1])
|
||||
array_string = array_string[match.end():]
|
||||
else:
|
||||
# Start of non-quoted value, find its end
|
||||
match = re.search(r",|\]", array_string)
|
||||
values.append(array_string[0 : match.start()])
|
||||
array_string = array_string[match.end() - 1:]
|
||||
|
||||
|
||||
def import_submodules(package_name):
|
||||
"""
|
||||
Import all submodules of a module.
|
||||
"""
|
||||
import importlib, pkgutil
|
||||
package = importlib.import_module(package_name)
|
||||
return {
|
||||
name: importlib.import_module(package_name + '.' + name)
|
||||
for _, name, _ in pkgutil.iter_modules(package.__path__)
|
||||
}
|
||||
|
||||
|
||||
def comma_join(items, stringify=False):
|
||||
"""
|
||||
Joins an iterable of strings with commas.
|
||||
"""
|
||||
if stringify:
|
||||
return ', '.join(str(item) for item in items)
|
||||
else:
|
||||
return ', '.join(items)
|
||||
|
||||
|
||||
def is_iterable(obj):
|
||||
"""
|
||||
Checks if the given object is iterable.
|
||||
"""
|
||||
try:
|
||||
iter(obj)
|
||||
return True
|
||||
except TypeError:
|
||||
return False
|
||||
|
||||
|
||||
def get_subclass_names(locals, base_class):
|
||||
from inspect import isclass
|
||||
return [c.__name__ for c in locals.values() if isclass(c) and issubclass(c, base_class)]
|
||||
|
||||
def on_cluster(db):
|
||||
if db.cluster is not None:
|
||||
return "ON CLUSTER '{}'".format(db.cluster)
|
||||
else:
|
||||
return ''
|
||||
|
||||
|
||||
class NoValue:
|
||||
'''
|
||||
A sentinel for fields with an expression for a default value,
|
||||
that were not assigned a value yet.
|
||||
'''
|
||||
def __repr__(self):
|
||||
return 'NO_VALUE'
|
||||
|
||||
NO_VALUE = NoValue()
|
||||
@@ -1 +0,0 @@
|
||||
__import__("pkg_resources").declare_namespace(__name__)
|
||||
@@ -1 +0,0 @@
|
||||
__import__("pkg_resources").declare_namespace(__name__)
|
||||
@@ -1,277 +0,0 @@
|
||||
import requests
|
||||
from collections import namedtuple
|
||||
from .models import ModelBase
|
||||
from .utils import escape, parse_tsv, import_submodules
|
||||
from math import ceil
|
||||
import datetime
|
||||
from string import Template
|
||||
from six import PY3, string_types
|
||||
import pytz
|
||||
|
||||
import logging
|
||||
logger = logging.getLogger('clickhouse_orm')
|
||||
|
||||
|
||||
Page = namedtuple('Page', 'objects number_of_objects pages_total number page_size')
|
||||
|
||||
|
||||
class DatabaseException(Exception):
|
||||
'''
|
||||
Raised when a database operation fails.
|
||||
'''
|
||||
pass
|
||||
|
||||
|
||||
class Database(object):
|
||||
'''
|
||||
Database instances connect to a specific ClickHouse database for running queries,
|
||||
inserting data and other operations.
|
||||
'''
|
||||
|
||||
def __init__(self, db_name, db_url='http://localhost:8123/', username=None, password=None, readonly=False):
|
||||
'''
|
||||
Initializes a database instance. Unless it's readonly, the database will be
|
||||
created on the ClickHouse server if it does not already exist.
|
||||
|
||||
- `db_name`: name of the database to connect to.
|
||||
- `db_url`: URL of the ClickHouse server.
|
||||
- `username`: optional connection credentials.
|
||||
- `password`: optional connection credentials.
|
||||
- `readonly`: use a read-only connection.
|
||||
'''
|
||||
self.db_name = db_name
|
||||
self.db_url = db_url
|
||||
self.username = username
|
||||
self.password = password
|
||||
self.readonly = False
|
||||
if readonly:
|
||||
self.connection_readonly = self._is_connection_readonly()
|
||||
self.readonly = True
|
||||
else:
|
||||
self.create_database()
|
||||
self.server_timezone = self._get_server_timezone()
|
||||
|
||||
def create_database(self):
|
||||
'''
|
||||
Creates the database on the ClickHouse server if it does not already exist.
|
||||
'''
|
||||
self._send('CREATE DATABASE IF NOT EXISTS `%s`' % self.db_name)
|
||||
|
||||
def drop_database(self):
|
||||
'''
|
||||
Deletes the database on the ClickHouse server.
|
||||
'''
|
||||
self._send('DROP DATABASE `%s`' % self.db_name)
|
||||
|
||||
def create_table(self, model_class):
|
||||
'''
|
||||
Creates a table for the given model class, if it does not exist already.
|
||||
'''
|
||||
# TODO check that model has an engine
|
||||
if model_class.readonly:
|
||||
raise DatabaseException("You can't create read only table")
|
||||
self._send(model_class.create_table_sql(self.db_name))
|
||||
|
||||
def drop_table(self, model_class):
|
||||
'''
|
||||
Drops the database table of the given model class, if it exists.
|
||||
'''
|
||||
if model_class.readonly:
|
||||
raise DatabaseException("You can't drop read only table")
|
||||
self._send(model_class.drop_table_sql(self.db_name))
|
||||
|
||||
def insert(self, model_instances, batch_size=1000):
|
||||
'''
|
||||
Insert records into the database.
|
||||
|
||||
- `model_instances`: any iterable containing instances of a single model class.
|
||||
- `batch_size`: number of records to send per chunk (use a lower number if your records are very large).
|
||||
'''
|
||||
from six import next
|
||||
from io import BytesIO
|
||||
i = iter(model_instances)
|
||||
try:
|
||||
first_instance = next(i)
|
||||
except StopIteration:
|
||||
return # model_instances is empty
|
||||
model_class = first_instance.__class__
|
||||
|
||||
if first_instance.readonly:
|
||||
raise DatabaseException("You can't insert into read only table")
|
||||
|
||||
def gen():
|
||||
buf = BytesIO()
|
||||
buf.write(self._substitute('INSERT INTO $table FORMAT TabSeparated\n', model_class).encode('utf-8'))
|
||||
first_instance.set_database(self)
|
||||
buf.write(first_instance.to_tsv(include_readonly=False).encode('utf-8'))
|
||||
buf.write('\n'.encode('utf-8'))
|
||||
# Collect lines in batches of batch_size
|
||||
lines = 2
|
||||
for instance in i:
|
||||
instance.set_database(self)
|
||||
buf.write(instance.to_tsv(include_readonly=False).encode('utf-8'))
|
||||
buf.write('\n'.encode('utf-8'))
|
||||
lines += 1
|
||||
if lines >= batch_size:
|
||||
# Return the current batch of lines
|
||||
yield buf.getvalue()
|
||||
# Start a new batch
|
||||
buf = BytesIO()
|
||||
lines = 0
|
||||
# Return any remaining lines in partial batch
|
||||
if lines:
|
||||
yield buf.getvalue()
|
||||
self._send(gen())
|
||||
|
||||
def count(self, model_class, conditions=None):
|
||||
'''
|
||||
Counts the number of records in the model's table.
|
||||
|
||||
- `model_class`: the model to count.
|
||||
- `conditions`: optional SQL conditions (contents of the WHERE clause).
|
||||
'''
|
||||
query = 'SELECT count() FROM $table'
|
||||
if conditions:
|
||||
query += ' WHERE ' + conditions
|
||||
query = self._substitute(query, model_class)
|
||||
r = self._send(query)
|
||||
return int(r.text) if r.text else 0
|
||||
|
||||
def select(self, query, model_class=None, settings=None):
|
||||
'''
|
||||
Performs a query and returns a generator of model instances.
|
||||
|
||||
- `query`: the SQL query to execute.
|
||||
- `model_class`: the model class matching the query's table,
|
||||
or `None` for getting back instances of an ad-hoc model.
|
||||
- `settings`: query settings to send as HTTP GET parameters
|
||||
'''
|
||||
query += ' FORMAT TabSeparatedWithNamesAndTypes'
|
||||
query = self._substitute(query, model_class)
|
||||
r = self._send(query, settings, True)
|
||||
lines = r.iter_lines()
|
||||
field_names = parse_tsv(next(lines))
|
||||
field_types = parse_tsv(next(lines))
|
||||
model_class = model_class or ModelBase.create_ad_hoc_model(zip(field_names, field_types))
|
||||
for line in lines:
|
||||
# skip blank line left by WITH TOTALS modifier
|
||||
if line:
|
||||
yield model_class.from_tsv(line, field_names, self.server_timezone, self)
|
||||
|
||||
def raw(self, query, settings=None, stream=False):
|
||||
'''
|
||||
Performs a query and returns its output as text.
|
||||
|
||||
- `query`: the SQL query to execute.
|
||||
- `settings`: query settings to send as HTTP GET parameters
|
||||
- `stream`: if true, the HTTP response from ClickHouse will be streamed.
|
||||
'''
|
||||
query = self._substitute(query, None)
|
||||
return self._send(query, settings=settings, stream=stream).text
|
||||
|
||||
def paginate(self, model_class, order_by, page_num=1, page_size=100, conditions=None, settings=None):
|
||||
'''
|
||||
Selects records and returns a single page of model instances.
|
||||
|
||||
- `model_class`: the model class matching the query's table,
|
||||
or `None` for getting back instances of an ad-hoc model.
|
||||
- `order_by`: columns to use for sorting the query (contents of the ORDER BY clause).
|
||||
- `page_num`: the page number (1-based), or -1 to get the last page.
|
||||
- `page_size`: number of records to return per page.
|
||||
- `conditions`: optional SQL conditions (contents of the WHERE clause).
|
||||
- `settings`: query settings to send as HTTP GET parameters
|
||||
|
||||
The result is a namedtuple containing `objects` (list), `number_of_objects`,
|
||||
`pages_total`, `number` (of the current page), and `page_size`.
|
||||
'''
|
||||
count = self.count(model_class, conditions)
|
||||
pages_total = int(ceil(count / float(page_size)))
|
||||
if page_num == -1:
|
||||
page_num = pages_total
|
||||
elif page_num < 1:
|
||||
raise ValueError('Invalid page number: %d' % page_num)
|
||||
offset = (page_num - 1) * page_size
|
||||
query = 'SELECT * FROM $table'
|
||||
if conditions:
|
||||
query += ' WHERE ' + conditions
|
||||
query += ' ORDER BY %s' % order_by
|
||||
query += ' LIMIT %d, %d' % (offset, page_size)
|
||||
query = self._substitute(query, model_class)
|
||||
return Page(
|
||||
objects=list(self.select(query, model_class, settings)),
|
||||
number_of_objects=count,
|
||||
pages_total=pages_total,
|
||||
number=page_num,
|
||||
page_size=page_size
|
||||
)
|
||||
|
||||
def migrate(self, migrations_package_name, up_to=9999):
|
||||
'''
|
||||
Executes schema migrations.
|
||||
|
||||
- `migrations_package_name` - fully qualified name of the Python package
|
||||
containing the migrations.
|
||||
- `up_to` - number of the last migration to apply.
|
||||
'''
|
||||
from .migrations import MigrationHistory
|
||||
logger = logging.getLogger('migrations')
|
||||
applied_migrations = self._get_applied_migrations(migrations_package_name)
|
||||
modules = import_submodules(migrations_package_name)
|
||||
unapplied_migrations = set(modules.keys()) - applied_migrations
|
||||
for name in sorted(unapplied_migrations):
|
||||
logger.info('Applying migration %s...', name)
|
||||
for operation in modules[name].operations:
|
||||
operation.apply(self)
|
||||
self.insert([MigrationHistory(package_name=migrations_package_name, module_name=name, applied=datetime.date.today())])
|
||||
if int(name[:4]) >= up_to:
|
||||
break
|
||||
|
||||
def _get_applied_migrations(self, migrations_package_name):
|
||||
from .migrations import MigrationHistory
|
||||
self.create_table(MigrationHistory)
|
||||
query = "SELECT module_name from $table WHERE package_name = '%s'" % migrations_package_name
|
||||
query = self._substitute(query, MigrationHistory)
|
||||
return set(obj.module_name for obj in self.select(query))
|
||||
|
||||
def _send(self, data, settings=None, stream=False):
|
||||
if isinstance(data, string_types):
|
||||
data = data.encode('utf-8')
|
||||
params = self._build_params(settings)
|
||||
r = requests.post(self.db_url, params=params, data=data, stream=stream)
|
||||
if r.status_code != 200:
|
||||
raise DatabaseException(r.text)
|
||||
return r
|
||||
|
||||
def _build_params(self, settings):
|
||||
params = dict(settings or {})
|
||||
if self.username:
|
||||
params['user'] = self.username
|
||||
if self.password:
|
||||
params['password'] = self.password
|
||||
# Send the readonly flag, unless the connection is already readonly (to prevent db error)
|
||||
if self.readonly and not self.connection_readonly:
|
||||
params['readonly'] = '1'
|
||||
return params
|
||||
|
||||
def _substitute(self, query, model_class=None):
|
||||
'''
|
||||
Replaces $db and $table placeholders in the query.
|
||||
'''
|
||||
if '$' in query:
|
||||
mapping = dict(db="`%s`" % self.db_name)
|
||||
if model_class:
|
||||
mapping['table'] = "`%s`.`%s`" % (self.db_name, model_class.table_name())
|
||||
query = Template(query).substitute(mapping)
|
||||
return query
|
||||
|
||||
def _get_server_timezone(self):
|
||||
try:
|
||||
r = self._send('SELECT timezone()')
|
||||
return pytz.timezone(r.text.strip())
|
||||
except DatabaseException:
|
||||
logger.exception('Cannot determine server timezone, assuming UTC')
|
||||
return pytz.utc
|
||||
|
||||
def _is_connection_readonly(self):
|
||||
r = self._send("SELECT value FROM system.settings WHERE name = 'readonly'")
|
||||
return r.text.strip() != '0'
|
||||
@@ -1,127 +0,0 @@
|
||||
|
||||
class Engine(object):
|
||||
|
||||
def create_table_sql(self):
|
||||
raise NotImplementedError()
|
||||
|
||||
|
||||
class TinyLog(Engine):
|
||||
|
||||
def create_table_sql(self):
|
||||
return 'TinyLog'
|
||||
|
||||
|
||||
class Log(Engine):
|
||||
|
||||
def create_table_sql(self):
|
||||
return 'Log'
|
||||
|
||||
|
||||
class Memory(Engine):
|
||||
|
||||
def create_table_sql(self):
|
||||
return 'Memory'
|
||||
|
||||
|
||||
class MergeTree(Engine):
|
||||
|
||||
def __init__(self, date_col, key_cols, sampling_expr=None,
|
||||
index_granularity=8192, replica_table_path=None, replica_name=None):
|
||||
assert type(key_cols) in (list, tuple), 'key_cols must be a list or tuple'
|
||||
self.date_col = date_col
|
||||
self.key_cols = key_cols
|
||||
self.sampling_expr = sampling_expr
|
||||
self.index_granularity = index_granularity
|
||||
self.replica_table_path = replica_table_path
|
||||
self.replica_name = replica_name
|
||||
# TODO verify that both replica fields are either present or missing
|
||||
|
||||
def create_table_sql(self):
|
||||
name = self.__class__.__name__
|
||||
if self.replica_name:
|
||||
name = 'Replicated' + name
|
||||
params = self._build_sql_params()
|
||||
return '%s(%s)' % (name, ', '.join(params))
|
||||
|
||||
def _build_sql_params(self):
|
||||
params = []
|
||||
if self.replica_name:
|
||||
params += ["'%s'" % self.replica_table_path, "'%s'" % self.replica_name]
|
||||
params.append(self.date_col)
|
||||
if self.sampling_expr:
|
||||
params.append(self.sampling_expr)
|
||||
params.append('(%s)' % ', '.join(self.key_cols))
|
||||
params.append(str(self.index_granularity))
|
||||
return params
|
||||
|
||||
|
||||
class CollapsingMergeTree(MergeTree):
|
||||
|
||||
def __init__(self, date_col, key_cols, sign_col, sampling_expr=None,
|
||||
index_granularity=8192, replica_table_path=None, replica_name=None):
|
||||
super(CollapsingMergeTree, self).__init__(date_col, key_cols, sampling_expr, index_granularity, replica_table_path, replica_name)
|
||||
self.sign_col = sign_col
|
||||
|
||||
def _build_sql_params(self):
|
||||
params = super(CollapsingMergeTree, self)._build_sql_params()
|
||||
params.append(self.sign_col)
|
||||
return params
|
||||
|
||||
|
||||
class SummingMergeTree(MergeTree):
|
||||
|
||||
def __init__(self, date_col, key_cols, summing_cols=None, sampling_expr=None,
|
||||
index_granularity=8192, replica_table_path=None, replica_name=None):
|
||||
super(SummingMergeTree, self).__init__(date_col, key_cols, sampling_expr, index_granularity, replica_table_path, replica_name)
|
||||
assert type is None or type(summing_cols) in (list, tuple), 'summing_cols must be a list or tuple'
|
||||
self.summing_cols = summing_cols
|
||||
|
||||
def _build_sql_params(self):
|
||||
params = super(SummingMergeTree, self)._build_sql_params()
|
||||
if self.summing_cols:
|
||||
params.append('(%s)' % ', '.join(self.summing_cols))
|
||||
return params
|
||||
|
||||
|
||||
class ReplacingMergeTree(MergeTree):
|
||||
|
||||
def __init__(self, date_col, key_cols, ver_col=None, sampling_expr=None,
|
||||
index_granularity=8192, replica_table_path=None, replica_name=None):
|
||||
super(ReplacingMergeTree, self).__init__(date_col, key_cols, sampling_expr, index_granularity, replica_table_path, replica_name)
|
||||
self.ver_col = ver_col
|
||||
|
||||
def _build_sql_params(self):
|
||||
params = super(ReplacingMergeTree, self)._build_sql_params()
|
||||
if self.ver_col:
|
||||
params.append(self.ver_col)
|
||||
return params
|
||||
|
||||
|
||||
class Buffer(Engine):
|
||||
"""
|
||||
Buffers the data to write in RAM, periodically flushing it to another table.
|
||||
Must be used in conjuction with a `BufferModel`.
|
||||
Read more [here](https://clickhouse.yandex/reference_en.html#Buffer).
|
||||
"""
|
||||
|
||||
#Buffer(database, table, num_layers, min_time, max_time, min_rows, max_rows, min_bytes, max_bytes)
|
||||
def __init__(self, main_model, num_layers=16, min_time=10, max_time=100, min_rows=10000, max_rows=1000000, min_bytes=10000000, max_bytes=100000000):
|
||||
self.main_model = main_model
|
||||
self.num_layers = num_layers
|
||||
self.min_time = min_time
|
||||
self.max_time = max_time
|
||||
self.min_rows = min_rows
|
||||
self.max_rows = max_rows
|
||||
self.min_bytes = min_bytes
|
||||
self.max_bytes = max_bytes
|
||||
|
||||
|
||||
def create_table_sql(self, db_name):
|
||||
# Overriden create_table_sql example:
|
||||
#sql = 'ENGINE = Buffer(merge, hits, 16, 10, 100, 10000, 1000000, 10000000, 100000000)'
|
||||
sql = 'ENGINE = Buffer(`%s`, `%s`, %d, %d, %d, %d, %d, %d, %d)' % (
|
||||
db_name, self.main_model.table_name(), self.num_layers,
|
||||
self.min_time, self.max_time, self.min_rows,
|
||||
self.max_rows, self.min_bytes, self.max_bytes
|
||||
)
|
||||
return sql
|
||||
@@ -1,364 +0,0 @@
|
||||
from six import string_types, text_type, binary_type
|
||||
import datetime
|
||||
import pytz
|
||||
import time
|
||||
from calendar import timegm
|
||||
|
||||
from .utils import escape, parse_array
|
||||
|
||||
|
||||
class Field(object):
|
||||
'''
|
||||
Abstract base class for all field types.
|
||||
'''
|
||||
creation_counter = 0
|
||||
class_default = 0
|
||||
db_type = None
|
||||
|
||||
def __init__(self, default=None, alias=None, materialized=None):
|
||||
assert (None, None) in {(default, alias), (alias, materialized), (default, materialized)}, \
|
||||
"Only one of default, alias and materialized parameters can be given"
|
||||
assert alias is None or isinstance(alias, str) and alias != "",\
|
||||
"Alias field must be string field name, if given"
|
||||
assert materialized is None or isinstance(materialized, str) and alias != "",\
|
||||
"Materialized field must be string, if given"
|
||||
|
||||
self.creation_counter = Field.creation_counter
|
||||
Field.creation_counter += 1
|
||||
self.default = self.class_default if default is None else default
|
||||
self.alias = alias
|
||||
self.materialized = materialized
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
'''
|
||||
Converts the input value into the expected Python data type, raising ValueError if the
|
||||
data can't be converted. Returns the converted value. Subclasses should override this.
|
||||
The timezone_in_use parameter should be consulted when parsing datetime fields.
|
||||
'''
|
||||
return value
|
||||
|
||||
def validate(self, value):
|
||||
'''
|
||||
Called after to_python to validate that the value is suitable for the field's database type.
|
||||
Subclasses should override this.
|
||||
'''
|
||||
pass
|
||||
|
||||
def _range_check(self, value, min_value, max_value):
|
||||
'''
|
||||
Utility method to check that the given value is between min_value and max_value.
|
||||
'''
|
||||
if value < min_value or value > max_value:
|
||||
raise ValueError('%s out of range - %s is not between %s and %s' % (self.__class__.__name__, value, min_value, max_value))
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
'''
|
||||
Returns the field's value prepared for writing to the database.
|
||||
When quote is true, strings are surrounded by single quotes.
|
||||
'''
|
||||
return escape(value, quote)
|
||||
|
||||
def get_sql(self, with_default=True):
|
||||
'''
|
||||
Returns an SQL expression describing the field (e.g. for CREATE TABLE).
|
||||
:param with_default: If True, adds default value to sql.
|
||||
It doesn't affect fields with alias and materialized values.
|
||||
'''
|
||||
if self.alias:
|
||||
return '%s ALIAS %s' % (self.db_type, self.alias)
|
||||
elif self.materialized:
|
||||
return '%s MATERIALIZED %s' % (self.db_type, self.materialized)
|
||||
elif with_default:
|
||||
default = self.to_db_string(self.default)
|
||||
return '%s DEFAULT %s' % (self.db_type, default)
|
||||
else:
|
||||
return self.db_type
|
||||
|
||||
@property
|
||||
def readonly(self):
|
||||
return bool(self.alias or self.materialized)
|
||||
|
||||
|
||||
class StringField(Field):
|
||||
|
||||
class_default = ''
|
||||
db_type = 'String'
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, text_type):
|
||||
return value
|
||||
if isinstance(value, binary_type):
|
||||
return value.decode('UTF-8')
|
||||
raise ValueError('Invalid value for %s: %r' % (self.__class__.__name__, value))
|
||||
|
||||
|
||||
class FixedStringField(StringField):
|
||||
|
||||
def __init__(self, length, default=None, alias=None, materialized=None):
|
||||
self._length = length
|
||||
self.db_type = 'FixedString(%d)' % length
|
||||
super(FixedStringField, self).__init__(default, alias, materialized)
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
value = super(FixedStringField, self).to_python(value, timezone_in_use)
|
||||
return value.rstrip('\0')
|
||||
|
||||
def validate(self, value):
|
||||
if isinstance(value, text_type):
|
||||
value = value.encode('UTF-8')
|
||||
if len(value) > self._length:
|
||||
raise ValueError('Value of %d bytes is too long for FixedStringField(%d)' % (len(value), self._length))
|
||||
|
||||
|
||||
class DateField(Field):
|
||||
|
||||
min_value = datetime.date(1970, 1, 1)
|
||||
max_value = datetime.date(2038, 1, 19)
|
||||
class_default = min_value
|
||||
db_type = 'Date'
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, datetime.datetime):
|
||||
return value.astimezone(pytz.utc).date() if value.tzinfo else value.date()
|
||||
if isinstance(value, datetime.date):
|
||||
return value
|
||||
if isinstance(value, int):
|
||||
return DateField.class_default + datetime.timedelta(days=value)
|
||||
if isinstance(value, string_types):
|
||||
if value == '0000-00-00':
|
||||
return DateField.min_value
|
||||
return datetime.datetime.strptime(value, '%Y-%m-%d').date()
|
||||
raise ValueError('Invalid value for %s - %r' % (self.__class__.__name__, value))
|
||||
|
||||
def validate(self, value):
|
||||
self._range_check(value, DateField.min_value, DateField.max_value)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
return escape(value.isoformat(), quote)
|
||||
|
||||
|
||||
class DateTimeField(Field):
|
||||
|
||||
class_default = datetime.datetime.fromtimestamp(0, pytz.utc)
|
||||
db_type = 'DateTime'
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, datetime.datetime):
|
||||
return value.astimezone(pytz.utc) if value.tzinfo else value.replace(tzinfo=pytz.utc)
|
||||
if isinstance(value, datetime.date):
|
||||
return datetime.datetime(value.year, value.month, value.day, tzinfo=pytz.utc)
|
||||
if isinstance(value, int):
|
||||
return datetime.datetime.utcfromtimestamp(value).replace(tzinfo=pytz.utc)
|
||||
if isinstance(value, string_types):
|
||||
if value == '0000-00-00 00:00:00':
|
||||
return self.class_default
|
||||
if len(value) == 10:
|
||||
try:
|
||||
value = int(value)
|
||||
return datetime.datetime.utcfromtimestamp(value).replace(tzinfo=pytz.utc)
|
||||
except ValueError:
|
||||
pass
|
||||
dt = datetime.datetime.strptime(value, '%Y-%m-%d %H:%M:%S')
|
||||
return timezone_in_use.localize(dt).astimezone(pytz.utc)
|
||||
raise ValueError('Invalid value for %s - %r' % (self.__class__.__name__, value))
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
return escape('%010d' % timegm(value.utctimetuple()), quote)
|
||||
|
||||
|
||||
class BaseIntField(Field):
|
||||
'''
|
||||
Abstract base class for all integer-type fields.
|
||||
'''
|
||||
def to_python(self, value, timezone_in_use):
|
||||
try:
|
||||
return int(value)
|
||||
except:
|
||||
raise ValueError('Invalid value for %s - %r' % (self.__class__.__name__, value))
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
# There's no need to call escape since numbers do not contain
|
||||
# special characters, and never need quoting
|
||||
return text_type(value)
|
||||
|
||||
def validate(self, value):
|
||||
self._range_check(value, self.min_value, self.max_value)
|
||||
|
||||
|
||||
class UInt8Field(BaseIntField):
|
||||
|
||||
min_value = 0
|
||||
max_value = 2**8 - 1
|
||||
db_type = 'UInt8'
|
||||
|
||||
|
||||
class UInt16Field(BaseIntField):
|
||||
|
||||
min_value = 0
|
||||
max_value = 2**16 - 1
|
||||
db_type = 'UInt16'
|
||||
|
||||
|
||||
class UInt32Field(BaseIntField):
|
||||
|
||||
min_value = 0
|
||||
max_value = 2**32 - 1
|
||||
db_type = 'UInt32'
|
||||
|
||||
|
||||
class UInt64Field(BaseIntField):
|
||||
|
||||
min_value = 0
|
||||
max_value = 2**64 - 1
|
||||
db_type = 'UInt64'
|
||||
|
||||
|
||||
class Int8Field(BaseIntField):
|
||||
|
||||
min_value = -2**7
|
||||
max_value = 2**7 - 1
|
||||
db_type = 'Int8'
|
||||
|
||||
|
||||
class Int16Field(BaseIntField):
|
||||
|
||||
min_value = -2**15
|
||||
max_value = 2**15 - 1
|
||||
db_type = 'Int16'
|
||||
|
||||
|
||||
class Int32Field(BaseIntField):
|
||||
|
||||
min_value = -2**31
|
||||
max_value = 2**31 - 1
|
||||
db_type = 'Int32'
|
||||
|
||||
|
||||
class Int64Field(BaseIntField):
|
||||
|
||||
min_value = -2**63
|
||||
max_value = 2**63 - 1
|
||||
db_type = 'Int64'
|
||||
|
||||
|
||||
class BaseFloatField(Field):
|
||||
'''
|
||||
Abstract base class for all float-type fields.
|
||||
'''
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
try:
|
||||
return float(value)
|
||||
except:
|
||||
raise ValueError('Invalid value for %s - %r' % (self.__class__.__name__, value))
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
# There's no need to call escape since numbers do not contain
|
||||
# special characters, and never need quoting
|
||||
return text_type(value)
|
||||
|
||||
|
||||
class Float32Field(BaseFloatField):
|
||||
|
||||
db_type = 'Float32'
|
||||
|
||||
|
||||
class Float64Field(BaseFloatField):
|
||||
|
||||
db_type = 'Float64'
|
||||
|
||||
|
||||
class BaseEnumField(Field):
|
||||
'''
|
||||
Abstract base class for all enum-type fields.
|
||||
'''
|
||||
|
||||
def __init__(self, enum_cls, default=None, alias=None, materialized=None):
|
||||
self.enum_cls = enum_cls
|
||||
if default is None:
|
||||
default = list(enum_cls)[0]
|
||||
super(BaseEnumField, self).__init__(default, alias, materialized)
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, self.enum_cls):
|
||||
return value
|
||||
try:
|
||||
if isinstance(value, text_type):
|
||||
return self.enum_cls[value]
|
||||
if isinstance(value, binary_type):
|
||||
return self.enum_cls[value.decode('UTF-8')]
|
||||
if isinstance(value, int):
|
||||
return self.enum_cls(value)
|
||||
except (KeyError, ValueError):
|
||||
pass
|
||||
raise ValueError('Invalid value for %s: %r' % (self.enum_cls.__name__, value))
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
return escape(value.name, quote)
|
||||
|
||||
def get_sql(self, with_default=True):
|
||||
values = ['%s = %d' % (escape(item.name), item.value) for item in self.enum_cls]
|
||||
sql = '%s(%s)' % (self.db_type, ' ,'.join(values))
|
||||
if with_default:
|
||||
default = self.to_db_string(self.default)
|
||||
sql = '%s DEFAULT %s' % (sql, default)
|
||||
return sql
|
||||
|
||||
@classmethod
|
||||
def create_ad_hoc_field(cls, db_type):
|
||||
'''
|
||||
Give an SQL column description such as "Enum8('apple' = 1, 'banana' = 2, 'orange' = 3)"
|
||||
this method returns a matching enum field.
|
||||
'''
|
||||
import re
|
||||
try:
|
||||
Enum # exists in Python 3.4+
|
||||
except NameError:
|
||||
from enum import Enum # use the enum34 library instead
|
||||
members = {}
|
||||
for match in re.finditer("'(\w+)' = (\d+)", db_type):
|
||||
members[match.group(1)] = int(match.group(2))
|
||||
enum_cls = Enum('AdHocEnum', members)
|
||||
field_class = Enum8Field if db_type.startswith('Enum8') else Enum16Field
|
||||
return field_class(enum_cls)
|
||||
|
||||
|
||||
class Enum8Field(BaseEnumField):
|
||||
|
||||
db_type = 'Enum8'
|
||||
|
||||
|
||||
class Enum16Field(BaseEnumField):
|
||||
|
||||
db_type = 'Enum16'
|
||||
|
||||
|
||||
class ArrayField(Field):
|
||||
|
||||
class_default = []
|
||||
|
||||
def __init__(self, inner_field, default=None, alias=None, materialized=None):
|
||||
self.inner_field = inner_field
|
||||
super(ArrayField, self).__init__(default, alias, materialized)
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
if isinstance(value, text_type):
|
||||
value = parse_array(value)
|
||||
elif isinstance(value, binary_type):
|
||||
value = parse_array(value.decode('UTF-8'))
|
||||
elif not isinstance(value, (list, tuple)):
|
||||
raise ValueError('ArrayField expects list or tuple, not %s' % type(value))
|
||||
return [self.inner_field.to_python(v, timezone_in_use) for v in value]
|
||||
|
||||
def validate(self, value):
|
||||
for v in value:
|
||||
self.inner_field.validate(v)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
array = [self.inner_field.to_db_string(v, quote=True) for v in value]
|
||||
return '[' + ', '.join(array) + ']'
|
||||
|
||||
def get_sql(self, with_default=True):
|
||||
from .utils import escape
|
||||
return 'Array(%s)' % self.inner_field.get_sql(with_default=False)
|
||||
|
||||
@@ -1,109 +0,0 @@
|
||||
from .models import Model
|
||||
from .fields import DateField, StringField
|
||||
from .engines import MergeTree
|
||||
from .utils import escape
|
||||
|
||||
from six.moves import zip
|
||||
|
||||
import logging
|
||||
logger = logging.getLogger('migrations')
|
||||
|
||||
|
||||
class Operation(object):
|
||||
'''
|
||||
Base class for migration operations.
|
||||
'''
|
||||
|
||||
def apply(self, database):
|
||||
raise NotImplementedError()
|
||||
|
||||
|
||||
class CreateTable(Operation):
|
||||
'''
|
||||
A migration operation that creates a table for a given model class.
|
||||
'''
|
||||
|
||||
def __init__(self, model_class):
|
||||
self.model_class = model_class
|
||||
|
||||
def apply(self, database):
|
||||
logger.info(' Create table %s', self.model_class.table_name())
|
||||
database.create_table(self.model_class)
|
||||
|
||||
|
||||
class AlterTable(Operation):
|
||||
'''
|
||||
A migration operation that compares the table of a given model class to
|
||||
the model's fields, and alters the table to match the model. The operation can:
|
||||
- add new columns
|
||||
- drop obsolete columns
|
||||
- modify column types
|
||||
Default values are not altered by this operation.
|
||||
'''
|
||||
|
||||
def __init__(self, model_class):
|
||||
self.model_class = model_class
|
||||
|
||||
def _get_table_fields(self, database):
|
||||
query = "DESC `%s`.`%s`" % (database.db_name, self.model_class.table_name())
|
||||
return [(row.name, row.type) for row in database.select(query)]
|
||||
|
||||
def _alter_table(self, database, cmd):
|
||||
cmd = "ALTER TABLE `%s`.`%s` %s" % (database.db_name, self.model_class.table_name(), cmd)
|
||||
logger.debug(cmd)
|
||||
database._send(cmd)
|
||||
|
||||
def apply(self, database):
|
||||
logger.info(' Alter table %s', self.model_class.table_name())
|
||||
table_fields = dict(self._get_table_fields(database))
|
||||
# Identify fields that were deleted from the model
|
||||
deleted_fields = set(table_fields.keys()) - set(name for name, field in self.model_class._fields)
|
||||
for name in deleted_fields:
|
||||
logger.info(' Drop column %s', name)
|
||||
self._alter_table(database, 'DROP COLUMN %s' % name)
|
||||
del table_fields[name]
|
||||
# Identify fields that were added to the model
|
||||
prev_name = None
|
||||
for name, field in self.model_class._fields:
|
||||
if name not in table_fields:
|
||||
logger.info(' Add column %s', name)
|
||||
assert prev_name, 'Cannot add a column to the beginning of the table'
|
||||
cmd = 'ADD COLUMN %s %s AFTER %s' % (name, field.get_sql(), prev_name)
|
||||
self._alter_table(database, cmd)
|
||||
prev_name = name
|
||||
# Identify fields whose type was changed
|
||||
model_fields = [(name, field.get_sql(with_default=False)) for name, field in self.model_class._fields]
|
||||
for model_field, table_field in zip(model_fields, self._get_table_fields(database)):
|
||||
assert model_field[0] == table_field[0], 'Model fields and table columns in disagreement'
|
||||
if model_field[1] != table_field[1]:
|
||||
logger.info(' Change type of column %s from %s to %s', table_field[0], table_field[1], model_field[1])
|
||||
self._alter_table(database, 'MODIFY COLUMN %s %s' % model_field)
|
||||
|
||||
|
||||
class DropTable(Operation):
|
||||
'''
|
||||
A migration operation that drops the table of a given model class.
|
||||
'''
|
||||
|
||||
def __init__(self, model_class):
|
||||
self.model_class = model_class
|
||||
|
||||
def apply(self, database):
|
||||
logger.info(' Drop table %s', self.model_class.__name__)
|
||||
database.drop_table(self.model_class)
|
||||
|
||||
|
||||
class MigrationHistory(Model):
|
||||
'''
|
||||
A model for storing which migrations were already applied to the containing database.
|
||||
'''
|
||||
|
||||
package_name = StringField()
|
||||
module_name = StringField()
|
||||
applied = DateField()
|
||||
|
||||
engine = MergeTree('applied', ('package_name', 'module_name'))
|
||||
|
||||
@classmethod
|
||||
def table_name(cls):
|
||||
return 'infi_clickhouse_orm_migrations'
|
||||
@@ -1,243 +0,0 @@
|
||||
from logging import getLogger
|
||||
|
||||
from six import with_metaclass
|
||||
import pytz
|
||||
|
||||
from .fields import Field
|
||||
from .utils import parse_tsv
|
||||
from .query import QuerySet
|
||||
|
||||
logger = getLogger('clickhouse_orm')
|
||||
|
||||
|
||||
class ModelBase(type):
|
||||
'''
|
||||
A metaclass for ORM models. It adds the _fields list to model classes.
|
||||
'''
|
||||
|
||||
ad_hoc_model_cache = {}
|
||||
|
||||
def __new__(cls, name, bases, attrs):
|
||||
new_cls = super(ModelBase, cls).__new__(cls, name, bases, attrs)
|
||||
# Collect fields from parent classes
|
||||
base_fields = []
|
||||
for base in bases:
|
||||
if isinstance(base, ModelBase):
|
||||
base_fields += base._fields
|
||||
# Build a list of fields, in the order they were listed in the class
|
||||
fields = base_fields + [item for item in attrs.items() if isinstance(item[1], Field)]
|
||||
fields.sort(key=lambda item: item[1].creation_counter)
|
||||
setattr(new_cls, '_fields', fields)
|
||||
setattr(new_cls, '_writable_fields', [f for f in fields if not f[1].readonly])
|
||||
return new_cls
|
||||
|
||||
@classmethod
|
||||
def create_ad_hoc_model(cls, fields):
|
||||
# fields is a list of tuples (name, db_type)
|
||||
# Check if model exists in cache
|
||||
fields = list(fields)
|
||||
cache_key = str(fields)
|
||||
if cache_key in cls.ad_hoc_model_cache:
|
||||
return cls.ad_hoc_model_cache[cache_key]
|
||||
# Create an ad hoc model class
|
||||
attrs = {}
|
||||
for name, db_type in fields:
|
||||
attrs[name] = cls.create_ad_hoc_field(db_type)
|
||||
model_class = cls.__new__(cls, 'AdHocModel', (Model,), attrs)
|
||||
# Add the model class to the cache
|
||||
cls.ad_hoc_model_cache[cache_key] = model_class
|
||||
return model_class
|
||||
|
||||
@classmethod
|
||||
def create_ad_hoc_field(cls, db_type):
|
||||
import infi.clickhouse_orm.fields as orm_fields
|
||||
# Enums
|
||||
if db_type.startswith('Enum'):
|
||||
return orm_fields.BaseEnumField.create_ad_hoc_field(db_type)
|
||||
# Arrays
|
||||
if db_type.startswith('Array'):
|
||||
inner_field = cls.create_ad_hoc_field(db_type[6 : -1])
|
||||
return orm_fields.ArrayField(inner_field)
|
||||
# FixedString
|
||||
if db_type.startswith('FixedString'):
|
||||
length = int(db_type[12 : -1])
|
||||
return orm_fields.FixedStringField(length)
|
||||
# Simple fields
|
||||
name = db_type + 'Field'
|
||||
if not hasattr(orm_fields, name):
|
||||
raise NotImplementedError('No field class for %s' % db_type)
|
||||
return getattr(orm_fields, name)()
|
||||
|
||||
|
||||
class Model(with_metaclass(ModelBase)):
|
||||
'''
|
||||
A base class for ORM models. Each model class represent a ClickHouse table. For example:
|
||||
|
||||
class CPUStats(Model):
|
||||
timestamp = DateTimeField()
|
||||
cpu_id = UInt16Field()
|
||||
cpu_percent = Float32Field()
|
||||
engine = Memory()
|
||||
'''
|
||||
|
||||
engine = None
|
||||
readonly = False
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
'''
|
||||
Creates a model instance, using keyword arguments as field values.
|
||||
Since values are immediately converted to their Pythonic type,
|
||||
invalid values will cause a `ValueError` to be raised.
|
||||
Unrecognized field names will cause an `AttributeError`.
|
||||
'''
|
||||
super(Model, self).__init__()
|
||||
|
||||
self._database = None
|
||||
|
||||
# Assign field values from keyword arguments
|
||||
for name, value in kwargs.items():
|
||||
field = self.get_field(name)
|
||||
if field:
|
||||
setattr(self, name, value)
|
||||
else:
|
||||
raise AttributeError('%s does not have a field called %s' % (self.__class__.__name__, name))
|
||||
# Assign default values for fields not included in the keyword arguments
|
||||
for name, field in self._fields:
|
||||
if name not in kwargs:
|
||||
setattr(self, name, field.default)
|
||||
|
||||
def __setattr__(self, name, value):
|
||||
'''
|
||||
When setting a field value, converts the value to its Pythonic type and validates it.
|
||||
This may raise a `ValueError`.
|
||||
'''
|
||||
field = self.get_field(name)
|
||||
if field:
|
||||
value = field.to_python(value, pytz.utc)
|
||||
field.validate(value)
|
||||
super(Model, self).__setattr__(name, value)
|
||||
|
||||
def set_database(self, db):
|
||||
'''
|
||||
Sets the `Database` that this model instance belongs to.
|
||||
This is done automatically when the instance is read from the database or written to it.
|
||||
'''
|
||||
# This can not be imported globally due to circular import
|
||||
from .database import Database
|
||||
assert isinstance(db, Database), "database must be database.Database instance"
|
||||
self._database = db
|
||||
|
||||
def get_database(self):
|
||||
'''
|
||||
Gets the `Database` that this model instance belongs to.
|
||||
Returns `None` unless the instance was read from the database or written to it.
|
||||
'''
|
||||
return self._database
|
||||
|
||||
def get_field(self, name):
|
||||
'''
|
||||
Gets a `Field` instance given its name, or `None` if not found.
|
||||
'''
|
||||
field = getattr(self.__class__, name, None)
|
||||
return field if isinstance(field, Field) else None
|
||||
|
||||
@classmethod
|
||||
def table_name(cls):
|
||||
'''
|
||||
Returns the model's database table name. By default this is the
|
||||
class name converted to lowercase. Override this if you want to use
|
||||
a different table name.
|
||||
'''
|
||||
return cls.__name__.lower()
|
||||
|
||||
@classmethod
|
||||
def create_table_sql(cls, db_name):
|
||||
'''
|
||||
Returns the SQL command for creating a table for this model.
|
||||
'''
|
||||
parts = ['CREATE TABLE IF NOT EXISTS `%s`.`%s` (' % (db_name, cls.table_name())]
|
||||
cols = []
|
||||
for name, field in cls._fields:
|
||||
cols.append(' %s %s' % (name, field.get_sql()))
|
||||
parts.append(',\n'.join(cols))
|
||||
parts.append(')')
|
||||
parts.append('ENGINE = ' + cls.engine.create_table_sql())
|
||||
return '\n'.join(parts)
|
||||
|
||||
@classmethod
|
||||
def drop_table_sql(cls, db_name):
|
||||
'''
|
||||
Returns the SQL command for deleting this model's table.
|
||||
'''
|
||||
return 'DROP TABLE IF EXISTS `%s`.`%s`' % (db_name, cls.table_name())
|
||||
|
||||
@classmethod
|
||||
def from_tsv(cls, line, field_names=None, timezone_in_use=pytz.utc, database=None):
|
||||
'''
|
||||
Create a model instance from a tab-separated line. The line may or may not include a newline.
|
||||
The `field_names` list must match the fields defined in the model, but does not have to include all of them.
|
||||
If omitted, it is assumed to be the names of all fields in the model, in order of definition.
|
||||
|
||||
- `line`: the TSV-formatted data.
|
||||
- `field_names`: names of the model fields in the data.
|
||||
- `timezone_in_use`: the timezone to use when parsing dates and datetimes.
|
||||
- `database`: if given, sets the database that this instance belongs to.
|
||||
'''
|
||||
from six import next
|
||||
field_names = field_names or [name for name, field in cls._fields]
|
||||
values = iter(parse_tsv(line))
|
||||
kwargs = {}
|
||||
for name in field_names:
|
||||
field = getattr(cls, name)
|
||||
kwargs[name] = field.to_python(next(values), timezone_in_use)
|
||||
|
||||
obj = cls(**kwargs)
|
||||
if database is not None:
|
||||
obj.set_database(database)
|
||||
|
||||
return obj
|
||||
|
||||
def to_tsv(self, include_readonly=True):
|
||||
'''
|
||||
Returns the instance's column values as a tab-separated line. A newline is not included.
|
||||
|
||||
- `include_readonly`: if false, returns only fields that can be inserted into database.
|
||||
'''
|
||||
data = self.__dict__
|
||||
fields = self._fields if include_readonly else self._writable_fields
|
||||
return '\t'.join(field.to_db_string(data[name], quote=False) for name, field in fields)
|
||||
|
||||
def to_dict(self, include_readonly=True, field_names=None):
|
||||
'''
|
||||
Returns the instance's column values as a dict.
|
||||
|
||||
- `include_readonly`: if false, returns only fields that can be inserted into database.
|
||||
- `field_names`: an iterable of field names to return (optional)
|
||||
'''
|
||||
fields = self._fields if include_readonly else self._writable_fields
|
||||
|
||||
if field_names is not None:
|
||||
fields = [f for f in fields if f[0] in field_names]
|
||||
|
||||
data = self.__dict__
|
||||
return {name: data[name] for name, field in fields}
|
||||
|
||||
@classmethod
|
||||
def objects_in(cls, database):
|
||||
'''
|
||||
Returns a `QuerySet` for selecting instances of this model class.
|
||||
'''
|
||||
return QuerySet(cls, database)
|
||||
|
||||
|
||||
class BufferModel(Model):
|
||||
|
||||
@classmethod
|
||||
def create_table_sql(cls, db_name):
|
||||
'''
|
||||
Returns the SQL command for creating a table for this model.
|
||||
'''
|
||||
parts = ['CREATE TABLE IF NOT EXISTS `%s`.`%s` AS `%s`.`%s`' % (db_name, cls.table_name(), db_name, cls.engine.main_model.table_name())]
|
||||
engine_str = cls.engine.create_table_sql(db_name)
|
||||
parts.append(engine_str)
|
||||
return ' '.join(parts)
|
||||
@@ -1,256 +0,0 @@
|
||||
import six
|
||||
import pytz
|
||||
from copy import copy
|
||||
|
||||
|
||||
# TODO
|
||||
# - and/or between Q objects
|
||||
# - check that field names are valid
|
||||
# - qs slicing
|
||||
# - operators for arrays: length, has, empty
|
||||
|
||||
class Operator(object):
|
||||
"""
|
||||
Base class for filtering operators.
|
||||
"""
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
"""
|
||||
Subclasses should implement this method. It returns an SQL string
|
||||
that applies this operator on the given field and value.
|
||||
"""
|
||||
raise NotImplementedError
|
||||
|
||||
|
||||
class SimpleOperator(Operator):
|
||||
"""
|
||||
A simple binary operator such as a=b, a<b, a>b etc.
|
||||
"""
|
||||
|
||||
def __init__(self, sql_operator):
|
||||
self._sql_operator = sql_operator
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
field = getattr(model_cls, field_name)
|
||||
value = field.to_db_string(field.to_python(value, pytz.utc))
|
||||
return ' '.join([field_name, self._sql_operator, value])
|
||||
|
||||
|
||||
class InOperator(Operator):
|
||||
"""
|
||||
An operator that implements IN.
|
||||
Accepts 3 different types of values:
|
||||
- a list or tuple of simple values
|
||||
- a string (used verbatim as the contents of the parenthesis)
|
||||
- a queryset (subquery)
|
||||
"""
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
field = getattr(model_cls, field_name)
|
||||
if isinstance(value, QuerySet):
|
||||
value = value.as_sql()
|
||||
elif isinstance(value, six.string_types):
|
||||
pass
|
||||
else:
|
||||
value = ', '.join([field.to_db_string(field.to_python(v, pytz.utc)) for v in value])
|
||||
return '%s IN (%s)' % (field_name, value)
|
||||
|
||||
|
||||
class LikeOperator(Operator):
|
||||
"""
|
||||
A LIKE operator that matches the field to a given pattern. Can be
|
||||
case sensitive or insensitive.
|
||||
"""
|
||||
|
||||
def __init__(self, pattern, case_sensitive=True):
|
||||
self._pattern = pattern
|
||||
self._case_sensitive = case_sensitive
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
field = getattr(model_cls, field_name)
|
||||
value = field.to_db_string(field.to_python(value, pytz.utc), quote=False)
|
||||
value = value.replace('\\', '\\\\').replace('%', '\\\\%').replace('_', '\\\\_')
|
||||
pattern = self._pattern.format(value)
|
||||
if self._case_sensitive:
|
||||
return '%s LIKE \'%s\'' % (field_name, pattern)
|
||||
else:
|
||||
return 'lowerUTF8(%s) LIKE lowerUTF8(\'%s\')' % (field_name, pattern)
|
||||
|
||||
|
||||
class IExactOperator(Operator):
|
||||
"""
|
||||
An operator for case insensitive string comparison.
|
||||
"""
|
||||
|
||||
def to_sql(self, model_cls, field_name, value):
|
||||
field = getattr(model_cls, field_name)
|
||||
value = field.to_db_string(field.to_python(value, pytz.utc))
|
||||
return 'lowerUTF8(%s) = lowerUTF8(%s)' % (field_name, value)
|
||||
|
||||
|
||||
# Define the set of builtin operators
|
||||
|
||||
_operators = {}
|
||||
|
||||
def register_operator(name, sql):
|
||||
_operators[name] = sql
|
||||
|
||||
register_operator('eq', SimpleOperator('='))
|
||||
register_operator('gt', SimpleOperator('>'))
|
||||
register_operator('gte', SimpleOperator('>='))
|
||||
register_operator('lt', SimpleOperator('<'))
|
||||
register_operator('lte', SimpleOperator('<='))
|
||||
register_operator('in', InOperator())
|
||||
register_operator('contains', LikeOperator('%{}%'))
|
||||
register_operator('startswith', LikeOperator('{}%'))
|
||||
register_operator('endswith', LikeOperator('%{}'))
|
||||
register_operator('icontains', LikeOperator('%{}%', False))
|
||||
register_operator('istartswith', LikeOperator('{}%', False))
|
||||
register_operator('iendswith', LikeOperator('%{}', False))
|
||||
register_operator('iexact', IExactOperator())
|
||||
|
||||
|
||||
class FOV(object):
|
||||
"""
|
||||
An object for storing Field + Operator + Value.
|
||||
"""
|
||||
|
||||
def __init__(self, field_name, operator, value):
|
||||
self._field_name = field_name
|
||||
self._operator = _operators[operator]
|
||||
self._value = value
|
||||
|
||||
def to_sql(self, model_cls):
|
||||
return self._operator.to_sql(model_cls, self._field_name, self._value)
|
||||
|
||||
|
||||
class Q(object):
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
self._fovs = [self._build_fov(k, v) for k, v in six.iteritems(kwargs)]
|
||||
self._negate = False
|
||||
|
||||
def _build_fov(self, key, value):
|
||||
if '__' in key:
|
||||
field_name, operator = key.rsplit('__', 1)
|
||||
else:
|
||||
field_name, operator = key, 'eq'
|
||||
return FOV(field_name, operator, value)
|
||||
|
||||
def to_sql(self, model_cls):
|
||||
if not self._fovs:
|
||||
return '1'
|
||||
sql = ' AND '.join(fov.to_sql(model_cls) for fov in self._fovs)
|
||||
if self._negate:
|
||||
sql = 'NOT (%s)' % sql
|
||||
return sql
|
||||
|
||||
def __invert__(self):
|
||||
q = copy(self)
|
||||
q._negate = True
|
||||
return q
|
||||
|
||||
|
||||
class QuerySet(object):
|
||||
"""
|
||||
A queryset is an object that represents a database query using a specific `Model`.
|
||||
It is lazy, meaning that it does not hit the database until you iterate over its
|
||||
matching rows (model instances).
|
||||
"""
|
||||
|
||||
def __init__(self, model_cls, database):
|
||||
"""
|
||||
Initializer. It is possible to create a queryset like this, but the standard
|
||||
way is to use `MyModel.objects_in(database)`.
|
||||
"""
|
||||
self._model_cls = model_cls
|
||||
self._database = database
|
||||
self._order_by = [f[0] for f in model_cls._fields]
|
||||
self._q = []
|
||||
self._fields = []
|
||||
|
||||
def __iter__(self):
|
||||
"""
|
||||
Iterates over the model instances matching this queryset
|
||||
"""
|
||||
return self._database.select(self.as_sql(), self._model_cls)
|
||||
|
||||
def __bool__(self):
|
||||
"""
|
||||
Returns true if this queryset matches any rows.
|
||||
"""
|
||||
return bool(self.count())
|
||||
|
||||
def __nonzero__(self): # Python 2 compatibility
|
||||
return type(self).__bool__(self)
|
||||
|
||||
def __unicode__(self):
|
||||
return self.as_sql()
|
||||
|
||||
def as_sql(self):
|
||||
"""
|
||||
Returns the whole query as a SQL string.
|
||||
"""
|
||||
fields = '*'
|
||||
if self._fields:
|
||||
fields = ', '.join('`%s`' % field for field in self._fields)
|
||||
params = (fields, self._database.db_name, self._model_cls.table_name(), self.conditions_as_sql(), self.order_by_as_sql())
|
||||
return u'SELECT %s\nFROM `%s`.`%s`\nWHERE %s\nORDER BY %s' % params
|
||||
|
||||
def order_by_as_sql(self):
|
||||
"""
|
||||
Returns the contents of the query's `ORDER BY` clause as a string.
|
||||
"""
|
||||
return u', '.join([
|
||||
'%s DESC' % field[1:] if field[0] == '-' else field
|
||||
for field in self._order_by
|
||||
])
|
||||
|
||||
def conditions_as_sql(self):
|
||||
"""
|
||||
Returns the contents of the query's `WHERE` clause as a string.
|
||||
"""
|
||||
if self._q:
|
||||
return u' AND '.join([q.to_sql(self._model_cls) for q in self._q])
|
||||
else:
|
||||
return u'1'
|
||||
|
||||
def count(self):
|
||||
"""
|
||||
Returns the number of matching model instances.
|
||||
"""
|
||||
return self._database.count(self._model_cls, self.conditions_as_sql())
|
||||
|
||||
def order_by(self, *field_names):
|
||||
"""
|
||||
Returns a new `QuerySet` instance with the ordering changed.
|
||||
"""
|
||||
qs = copy(self)
|
||||
qs._order_by = field_names
|
||||
return qs
|
||||
|
||||
def only(self, *field_names):
|
||||
"""
|
||||
Returns a new `QuerySet` instance limited to the specified field names.
|
||||
Useful when there are large fields that are not needed,
|
||||
or for creating a subquery to use with an IN operator.
|
||||
"""
|
||||
qs = copy(self)
|
||||
qs._fields = field_names
|
||||
return qs
|
||||
|
||||
def filter(self, **kwargs):
|
||||
"""
|
||||
Returns a new `QuerySet` instance that includes only rows matching the conditions.
|
||||
"""
|
||||
qs = copy(self)
|
||||
qs._q = list(self._q) + [Q(**kwargs)]
|
||||
return qs
|
||||
|
||||
def exclude(self, **kwargs):
|
||||
"""
|
||||
Returns a new `QuerySet` instance that excludes all rows matching the conditions.
|
||||
"""
|
||||
qs = copy(self)
|
||||
qs._q = list(self._q) + [~Q(**kwargs)]
|
||||
return qs
|
||||
@@ -1,92 +0,0 @@
|
||||
from six import string_types, binary_type, text_type, PY3
|
||||
import codecs
|
||||
import re
|
||||
|
||||
|
||||
SPECIAL_CHARS = {
|
||||
"\b" : "\\b",
|
||||
"\f" : "\\f",
|
||||
"\r" : "\\r",
|
||||
"\n" : "\\n",
|
||||
"\t" : "\\t",
|
||||
"\0" : "\\0",
|
||||
"\\" : "\\\\",
|
||||
"'" : "\\'"
|
||||
}
|
||||
|
||||
SPECIAL_CHARS_REGEX = re.compile("[" + ''.join(SPECIAL_CHARS.values()) + "]")
|
||||
|
||||
|
||||
|
||||
def escape(value, quote=True):
|
||||
'''
|
||||
If the value is a string, escapes any special characters and optionally
|
||||
surrounds it with single quotes. If the value is not a string (e.g. a number),
|
||||
converts it to one.
|
||||
'''
|
||||
def escape_one(match):
|
||||
return SPECIAL_CHARS[match.group(0)]
|
||||
|
||||
if isinstance(value, string_types):
|
||||
value = SPECIAL_CHARS_REGEX.sub(escape_one, value)
|
||||
if quote:
|
||||
value = "'" + value + "'"
|
||||
return text_type(value)
|
||||
|
||||
|
||||
def unescape(value):
|
||||
return codecs.escape_decode(value)[0].decode('utf-8')
|
||||
|
||||
|
||||
def parse_tsv(line):
|
||||
if PY3 and isinstance(line, binary_type):
|
||||
line = line.decode()
|
||||
if line and line[-1] == '\n':
|
||||
line = line[:-1]
|
||||
return [unescape(value) for value in line.split('\t')]
|
||||
|
||||
|
||||
def parse_array(array_string):
|
||||
"""
|
||||
Parse an array string as returned by clickhouse. For example:
|
||||
"['hello', 'world']" ==> ["hello", "world"]
|
||||
"[1,2,3]" ==> [1, 2, 3]
|
||||
"""
|
||||
# Sanity check
|
||||
if len(array_string) < 2 or array_string[0] != '[' or array_string[-1] != ']':
|
||||
raise ValueError('Invalid array string: "%s"' % array_string)
|
||||
# Drop opening brace
|
||||
array_string = array_string[1:]
|
||||
# Go over the string, lopping off each value at the beginning until nothing is left
|
||||
values = []
|
||||
while True:
|
||||
if array_string == ']':
|
||||
# End of array
|
||||
return values
|
||||
elif array_string[0] in ', ':
|
||||
# In between values
|
||||
array_string = array_string[1:]
|
||||
elif array_string[0] == "'":
|
||||
# Start of quoted value, find its end
|
||||
match = re.search(r"[^\\]'", array_string)
|
||||
if match is None:
|
||||
raise ValueError('Missing closing quote: "%s"' % array_string)
|
||||
values.append(array_string[1 : match.start() + 1])
|
||||
array_string = array_string[match.end():]
|
||||
else:
|
||||
# Start of non-quoted value, find its end
|
||||
match = re.search(r",|\]", array_string)
|
||||
values.append(array_string[0 : match.start()])
|
||||
array_string = array_string[match.end() - 1:]
|
||||
|
||||
|
||||
def import_submodules(package_name):
|
||||
"""
|
||||
Import all submodules of a module.
|
||||
"""
|
||||
import importlib, pkgutil
|
||||
package = importlib.import_module(package_name)
|
||||
return {
|
||||
name: importlib.import_module(package_name + '.' + name)
|
||||
for _, name, _ in pkgutil.iter_modules(package.__path__)
|
||||
}
|
||||
@@ -1,11 +1,10 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import unittest
|
||||
|
||||
from infi.clickhouse_orm.database import Database
|
||||
from infi.clickhouse_orm.models import Model
|
||||
from infi.clickhouse_orm.fields import *
|
||||
from infi.clickhouse_orm.engines import *
|
||||
from datastore_orm.database import Database
|
||||
from datastore_orm.models import Model
|
||||
from datastore_orm.fields import *
|
||||
from datastore_orm.engines import *
|
||||
|
||||
import logging
|
||||
logging.getLogger("requests").setLevel(logging.WARNING)
|
||||
@@ -14,37 +13,47 @@ logging.getLogger("requests").setLevel(logging.WARNING)
|
||||
class TestCaseWithData(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.database = Database('test-db')
|
||||
self.database = Database('test-db', log_statements=True)
|
||||
self.database.create_table(Person)
|
||||
|
||||
def tearDown(self):
|
||||
self.database.drop_table(Person)
|
||||
self.database.drop_database()
|
||||
|
||||
def _insert_and_check(self, data, count):
|
||||
self.database.insert(data)
|
||||
self.assertEquals(count, self.database.count(Person))
|
||||
def _insert_all(self):
|
||||
self.database.insert(self._sample_data())
|
||||
self.assertTrue(self.database.count(Person))
|
||||
|
||||
def _insert_and_check(self, data, count, batch_size=1000):
|
||||
self.database.insert(data, batch_size=batch_size)
|
||||
self.assertEqual(count, self.database.count(Person))
|
||||
for instance in data:
|
||||
self.assertEquals(self.database, instance.get_database())
|
||||
self.assertEqual(self.database, instance.get_database())
|
||||
|
||||
def _sample_data(self):
|
||||
for entry in data:
|
||||
yield Person(**entry)
|
||||
|
||||
|
||||
|
||||
class Person(Model):
|
||||
|
||||
first_name = StringField()
|
||||
last_name = StringField()
|
||||
last_name = LowCardinalityField(StringField())
|
||||
birthday = DateField()
|
||||
height = Float32Field()
|
||||
passport = NullableField(UInt32Field())
|
||||
|
||||
engine = MergeTree('birthday', ('first_name', 'last_name', 'birthday'))
|
||||
|
||||
|
||||
data = [
|
||||
{"first_name": "Abdul", "last_name": "Hester", "birthday": "1970-12-02", "height": "1.63"},
|
||||
{"first_name": "Adam", "last_name": "Goodman", "birthday": "1986-01-07", "height": "1.74"},
|
||||
{"first_name": "Abdul", "last_name": "Hester", "birthday": "1970-12-02", "height": "1.63",
|
||||
"passport": 35052255},
|
||||
|
||||
{"first_name": "Adam", "last_name": "Goodman", "birthday": "1986-01-07", "height": "1.74",
|
||||
"passport": 36052255},
|
||||
|
||||
{"first_name": "Adena", "last_name": "Norman", "birthday": "1979-05-14", "height": "1.66"},
|
||||
{"first_name": "Aline", "last_name": "Crane", "birthday": "1988-05-01", "height": "1.62"},
|
||||
{"first_name": "Althea", "last_name": "Barrett", "birthday": "2004-07-28", "height": "1.71"},
|
||||
@@ -143,4 +152,4 @@ data = [
|
||||
{"first_name": "Whitney", "last_name": "Scott", "birthday": "1971-07-04", "height": "1.70"},
|
||||
{"first_name": "Wynter", "last_name": "Garcia", "birthday": "1975-01-10", "height": "1.69"},
|
||||
{"first_name": "Yolanda", "last_name": "Duke", "birthday": "1997-02-25", "height": "1.74"}
|
||||
];
|
||||
]
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from infi.clickhouse_orm import migrations
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from infi.clickhouse_orm import migrations
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from infi.clickhouse_orm import migrations
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from infi.clickhouse_orm import migrations
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from infi.clickhouse_orm import migrations
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from infi.clickhouse_orm import migrations
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from infi.clickhouse_orm import migrations
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from infi.clickhouse_orm import migrations
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from infi.clickhouse_orm import migrations
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
migrations.CreateTable(Model4Buffer)
|
||||
]
|
||||
@@ -0,0 +1,6 @@
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
migrations.AlterTableWithBuffer(Model4Buffer_changed)
|
||||
]
|
||||
@@ -0,0 +1,9 @@
|
||||
from datastore_orm import migrations
|
||||
|
||||
operations = [
|
||||
migrations.RunSQL("INSERT INTO `mig` (date, f1, f3, f4) VALUES ('2016-01-01', 1, 1, 'test') "),
|
||||
migrations.RunSQL([
|
||||
"INSERT INTO `mig` (date, f1, f3, f4) VALUES ('2016-01-02', 2, 2, 'test2') ",
|
||||
"INSERT INTO `mig` (date, f1, f3, f4) VALUES ('2016-01-03', 3, 3, 'test3') ",
|
||||
])
|
||||
]
|
||||
@@ -0,0 +1,15 @@
|
||||
import datetime
|
||||
|
||||
from datastore_orm import migrations
|
||||
from test_migrations import Model3
|
||||
|
||||
|
||||
def forward(database):
|
||||
database.insert([
|
||||
Model3(date=datetime.date(2016, 1, 4), f1=4, f3=1, f4='test4')
|
||||
])
|
||||
|
||||
|
||||
operations = [
|
||||
migrations.RunPython(forward)
|
||||
]
|
||||
@@ -0,0 +1,7 @@
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
migrations.AlterTable(MaterializedModel1),
|
||||
migrations.AlterTable(AliasModel1)
|
||||
]
|
||||
@@ -0,0 +1,7 @@
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
migrations.AlterTable(Model4_compressed),
|
||||
migrations.AlterTable(Model2LowCardinality)
|
||||
]
|
||||
@@ -0,0 +1,6 @@
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
migrations.CreateTable(ModelWithConstraints)
|
||||
]
|
||||
@@ -0,0 +1,6 @@
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
migrations.AlterConstraints(ModelWithConstraints2)
|
||||
]
|
||||
@@ -0,0 +1,6 @@
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
migrations.CreateTable(ModelWithIndex)
|
||||
]
|
||||
@@ -0,0 +1,6 @@
|
||||
from datastore_orm import migrations
|
||||
from ..test_migrations import *
|
||||
|
||||
operations = [
|
||||
migrations.AlterIndexes(ModelWithIndex2, reindex=True)
|
||||
]
|
||||
+26
-17
@@ -1,16 +1,17 @@
|
||||
import unittest
|
||||
from datetime import date
|
||||
|
||||
from infi.clickhouse_orm.database import Database
|
||||
from infi.clickhouse_orm.models import Model
|
||||
from infi.clickhouse_orm.fields import *
|
||||
from infi.clickhouse_orm.engines import *
|
||||
from datastore_orm.database import Database
|
||||
from datastore_orm.models import Model, NO_VALUE
|
||||
from datastore_orm.fields import *
|
||||
from datastore_orm.engines import *
|
||||
from datastore_orm.funcs import F
|
||||
|
||||
|
||||
class MaterializedFieldsTest(unittest.TestCase):
|
||||
class AliasFieldsTest(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.database = Database('test-db')
|
||||
self.database = Database('test-db', log_statements=True)
|
||||
self.database.create_table(ModelWithAliasFields)
|
||||
|
||||
def tearDown(self):
|
||||
@@ -24,17 +25,18 @@ class MaterializedFieldsTest(unittest.TestCase):
|
||||
)
|
||||
self.database.insert([instance])
|
||||
# We can't select * from table, as it doesn't select materialized and alias fields
|
||||
query = 'SELECT date_field, int_field, str_field, alias_int, alias_date, alias_str' \
|
||||
query = 'SELECT date_field, int_field, str_field, alias_int, alias_date, alias_str, alias_func' \
|
||||
' FROM $db.%s ORDER BY alias_date' % ModelWithAliasFields.table_name()
|
||||
for model_cls in (ModelWithAliasFields, None):
|
||||
results = list(self.database.select(query, model_cls))
|
||||
self.assertEquals(len(results), 1)
|
||||
self.assertEquals(results[0].date_field, instance.date_field)
|
||||
self.assertEquals(results[0].int_field, instance.int_field)
|
||||
self.assertEquals(results[0].str_field, instance.str_field)
|
||||
self.assertEquals(results[0].alias_int, instance.int_field)
|
||||
self.assertEquals(results[0].alias_str, instance.str_field)
|
||||
self.assertEquals(results[0].alias_date, instance.date_field)
|
||||
self.assertEqual(len(results), 1)
|
||||
self.assertEqual(results[0].date_field, instance.date_field)
|
||||
self.assertEqual(results[0].int_field, instance.int_field)
|
||||
self.assertEqual(results[0].str_field, instance.str_field)
|
||||
self.assertEqual(results[0].alias_int, instance.int_field)
|
||||
self.assertEqual(results[0].alias_str, instance.str_field)
|
||||
self.assertEqual(results[0].alias_date, instance.date_field)
|
||||
self.assertEqual(results[0].alias_func, 201608)
|
||||
|
||||
def test_assignment_error(self):
|
||||
# I can't prevent assigning at all, in case db.select statements with model provided sets model fields.
|
||||
@@ -54,16 +56,23 @@ class MaterializedFieldsTest(unittest.TestCase):
|
||||
with self.assertRaises(AssertionError):
|
||||
StringField(alias='str_field', materialized='str_field')
|
||||
|
||||
def test_default_value(self):
|
||||
instance = ModelWithAliasFields()
|
||||
self.assertEqual(instance.alias_str, NO_VALUE)
|
||||
# Check that NO_VALUE can be assigned to a field
|
||||
instance.str_field = NO_VALUE
|
||||
# Check that NO_VALUE can be assigned when creating a new instance
|
||||
instance2 = ModelWithAliasFields(**instance.to_dict())
|
||||
|
||||
|
||||
class ModelWithAliasFields(Model):
|
||||
int_field = Int32Field()
|
||||
date_field = DateField()
|
||||
str_field = StringField()
|
||||
|
||||
alias_str = StringField(alias='str_field')
|
||||
alias_str = StringField(alias=u'str_field')
|
||||
alias_int = Int32Field(alias='int_field')
|
||||
alias_date = DateField(alias='date_field')
|
||||
alias_func = Int32Field(alias=F.toYYYYMM(date_field))
|
||||
|
||||
engine = MergeTree('date_field', ('date_field',))
|
||||
|
||||
|
||||
|
||||
+30
-26
@@ -1,16 +1,16 @@
|
||||
import unittest
|
||||
from datetime import date
|
||||
|
||||
from infi.clickhouse_orm.database import Database
|
||||
from infi.clickhouse_orm.models import Model
|
||||
from infi.clickhouse_orm.fields import *
|
||||
from infi.clickhouse_orm.engines import *
|
||||
from datastore_orm.database import Database
|
||||
from datastore_orm.models import Model
|
||||
from datastore_orm.fields import *
|
||||
from datastore_orm.engines import *
|
||||
|
||||
|
||||
class ArrayFieldsTest(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.database = Database('test-db')
|
||||
self.database = Database('test-db', log_statements=True)
|
||||
self.database.create_table(ModelWithArrays)
|
||||
|
||||
def tearDown(self):
|
||||
@@ -18,27 +18,27 @@ class ArrayFieldsTest(unittest.TestCase):
|
||||
|
||||
def test_insert_and_select(self):
|
||||
instance = ModelWithArrays(
|
||||
date_field='2016-08-30',
|
||||
arr_str=['goodbye,', 'cruel', 'world', 'special chars: ,"\\\'` \n\t\\[]'],
|
||||
arr_date=['2010-01-01']
|
||||
date_field='2016-08-30',
|
||||
arr_str=['goodbye,', 'cruel', 'world', 'special chars: ,"\\\'` \n\t\\[]'],
|
||||
arr_date=['2010-01-01'],
|
||||
)
|
||||
self.database.insert([instance])
|
||||
query = 'SELECT * from $db.modelwitharrays ORDER BY date_field'
|
||||
for model_cls in (ModelWithArrays, None):
|
||||
results = list(self.database.select(query, model_cls))
|
||||
self.assertEquals(len(results), 1)
|
||||
self.assertEquals(results[0].arr_str, instance.arr_str)
|
||||
self.assertEquals(results[0].arr_int, instance.arr_int)
|
||||
self.assertEquals(results[0].arr_date, instance.arr_date)
|
||||
self.assertEqual(len(results), 1)
|
||||
self.assertEqual(results[0].arr_str, instance.arr_str)
|
||||
self.assertEqual(results[0].arr_int, instance.arr_int)
|
||||
self.assertEqual(results[0].arr_date, instance.arr_date)
|
||||
|
||||
def test_conversion(self):
|
||||
instance = ModelWithArrays(
|
||||
arr_int=('1', '2', '3'),
|
||||
arr_date=['2010-01-01']
|
||||
)
|
||||
self.assertEquals(instance.arr_str, [])
|
||||
self.assertEquals(instance.arr_int, [1, 2, 3])
|
||||
self.assertEquals(instance.arr_date, [date(2010, 1, 1)])
|
||||
self.assertEqual(instance.arr_str, [])
|
||||
self.assertEqual(instance.arr_int, [1, 2, 3])
|
||||
self.assertEqual(instance.arr_date, [date(2010, 1, 1)])
|
||||
|
||||
def test_assignment_error(self):
|
||||
instance = ModelWithArrays()
|
||||
@@ -47,20 +47,25 @@ class ArrayFieldsTest(unittest.TestCase):
|
||||
instance.arr_int = value
|
||||
|
||||
def test_parse_array(self):
|
||||
from infi.clickhouse_orm.utils import parse_array, unescape
|
||||
self.assertEquals(parse_array("[]"), [])
|
||||
self.assertEquals(parse_array("[1, 2, 395, -44]"), ["1", "2", "395", "-44"])
|
||||
self.assertEquals(parse_array("['big','mouse','','!']"), ["big", "mouse", "", "!"])
|
||||
self.assertEquals(parse_array(unescape("['\\r\\n\\0\\t\\b']")), ["\r\n\0\t\b"])
|
||||
for s in ("",
|
||||
"[",
|
||||
"]",
|
||||
"[1, 2",
|
||||
"3, 4]",
|
||||
from datastore_orm.utils import parse_array, unescape
|
||||
self.assertEqual(parse_array("[]"), [])
|
||||
self.assertEqual(parse_array("[1, 2, 395, -44]"), ["1", "2", "395", "-44"])
|
||||
self.assertEqual(parse_array("['big','mouse','','!']"), ["big", "mouse", "", "!"])
|
||||
self.assertEqual(parse_array(unescape("['\\r\\n\\0\\t\\b']")), ["\r\n\0\t\b"])
|
||||
for s in ("",
|
||||
"[",
|
||||
"]",
|
||||
"[1, 2",
|
||||
"3, 4]",
|
||||
"['aaa', 'aaa]"):
|
||||
with self.assertRaises(ValueError):
|
||||
parse_array(s)
|
||||
|
||||
def test_invalid_inner_field(self):
|
||||
for x in (DateField, None, "", ArrayField(Int32Field())):
|
||||
with self.assertRaises(AssertionError):
|
||||
ArrayField(x)
|
||||
|
||||
|
||||
class ModelWithArrays(Model):
|
||||
|
||||
@@ -70,4 +75,3 @@ class ModelWithArrays(Model):
|
||||
arr_date = ArrayField(DateField())
|
||||
|
||||
engine = MergeTree('date_field', ('date_field',))
|
||||
|
||||
|
||||
@@ -1,9 +1,8 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import unittest
|
||||
|
||||
from infi.clickhouse_orm.models import BufferModel
|
||||
from infi.clickhouse_orm.engines import *
|
||||
from datastore_orm.models import BufferModel
|
||||
from datastore_orm.engines import *
|
||||
from .base_test_with_data import *
|
||||
|
||||
|
||||
@@ -11,7 +10,7 @@ class BufferTestCase(TestCaseWithData):
|
||||
|
||||
def _insert_and_check_buffer(self, data, count):
|
||||
self.database.insert(data)
|
||||
self.assertEquals(count, self.database.count(PersonBuffer))
|
||||
self.assertEqual(count, self.database.count(PersonBuffer))
|
||||
|
||||
def _sample_buffer_data(self):
|
||||
for entry in data:
|
||||
@@ -23,7 +22,5 @@ class BufferTestCase(TestCaseWithData):
|
||||
|
||||
|
||||
class PersonBuffer(BufferModel, Person):
|
||||
|
||||
engine = Buffer(Person)
|
||||
|
||||
|
||||
engine = Buffer(Person)
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
import unittest
|
||||
import datetime
|
||||
import pytz
|
||||
|
||||
from datastore_orm.database import Database
|
||||
from datastore_orm.models import Model, NO_VALUE
|
||||
from datastore_orm.fields import *
|
||||
from datastore_orm.engines import *
|
||||
from datastore_orm.utils import parse_tsv
|
||||
|
||||
|
||||
class CompressedFieldsTestCase(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.database = Database('test-db', log_statements=True)
|
||||
self.database.create_table(CompressedModel)
|
||||
|
||||
def tearDown(self):
|
||||
self.database.drop_database()
|
||||
|
||||
def test_defaults(self):
|
||||
# Check that all fields have their explicit or implicit defaults
|
||||
instance = CompressedModel()
|
||||
self.database.insert([instance])
|
||||
self.assertEqual(instance.date_field, datetime.date(1970, 1, 1))
|
||||
self.assertEqual(instance.datetime_field, datetime.datetime(1970, 1, 1, tzinfo=pytz.utc))
|
||||
self.assertEqual(instance.string_field, 'dozo')
|
||||
self.assertEqual(instance.int64_field, 42)
|
||||
self.assertEqual(instance.float_field, 0)
|
||||
self.assertEqual(instance.nullable_field, None)
|
||||
self.assertEqual(instance.array_field, [])
|
||||
|
||||
def test_assignment(self):
|
||||
# Check that all fields are assigned during construction
|
||||
kwargs = dict(
|
||||
uint64_field=217,
|
||||
date_field=datetime.date(1973, 12, 6),
|
||||
datetime_field=datetime.datetime(2000, 5, 24, 10, 22, tzinfo=pytz.utc),
|
||||
string_field='aloha',
|
||||
int64_field=-50,
|
||||
float_field=3.14,
|
||||
nullable_field=-2.718281,
|
||||
array_field=['123456789123456','','a']
|
||||
)
|
||||
instance = CompressedModel(**kwargs)
|
||||
self.database.insert([instance])
|
||||
for name, value in kwargs.items():
|
||||
self.assertEqual(kwargs[name], getattr(instance, name))
|
||||
|
||||
def test_string_conversion(self):
|
||||
# Check field conversion from string during construction
|
||||
instance = CompressedModel(date_field='1973-12-06', int64_field='100', float_field='7', nullable_field=None, array_field='[a,b,c]')
|
||||
self.assertEqual(instance.date_field, datetime.date(1973, 12, 6))
|
||||
self.assertEqual(instance.int64_field, 100)
|
||||
self.assertEqual(instance.float_field, 7)
|
||||
self.assertEqual(instance.nullable_field, None)
|
||||
self.assertEqual(instance.array_field, ['a', 'b', 'c'])
|
||||
# Check field conversion from string during assignment
|
||||
instance.int64_field = '99'
|
||||
self.assertEqual(instance.int64_field, 99)
|
||||
|
||||
def test_to_dict(self):
|
||||
instance = CompressedModel(date_field='1973-12-06', int64_field='100', float_field='7', array_field='[a,b,c]')
|
||||
self.assertDictEqual(instance.to_dict(), {
|
||||
"date_field": datetime.date(1973, 12, 6),
|
||||
"int64_field": 100,
|
||||
"float_field": 7.0,
|
||||
"datetime_field": datetime.datetime(1970, 1, 1, 0, 0, 0, tzinfo=pytz.utc),
|
||||
"alias_field": NO_VALUE,
|
||||
'string_field': 'dozo',
|
||||
'nullable_field': None,
|
||||
'uint64_field': 0,
|
||||
'array_field': ['a','b','c']
|
||||
})
|
||||
self.assertDictEqual(instance.to_dict(include_readonly=False), {
|
||||
"date_field": datetime.date(1973, 12, 6),
|
||||
"int64_field": 100,
|
||||
"float_field": 7.0,
|
||||
"datetime_field": datetime.datetime(1970, 1, 1, 0, 0, 0, tzinfo=pytz.utc),
|
||||
'string_field': 'dozo',
|
||||
'nullable_field': None,
|
||||
'uint64_field': 0,
|
||||
'array_field': ['a', 'b', 'c']
|
||||
})
|
||||
self.assertDictEqual(
|
||||
instance.to_dict(include_readonly=False, field_names=('int64_field', 'alias_field', 'datetime_field')), {
|
||||
"int64_field": 100,
|
||||
"datetime_field": datetime.datetime(1970, 1, 1, 0, 0, 0, tzinfo=pytz.utc)
|
||||
})
|
||||
|
||||
def test_confirm_compression_codec(self):
|
||||
if self.database.server_version < (19, 17):
|
||||
raise unittest.SkipTest('ClickHouse version too old')
|
||||
instance = CompressedModel(date_field='1973-12-06', int64_field='100', float_field='7', array_field='[a,b,c]')
|
||||
self.database.insert([instance])
|
||||
r = self.database.raw("select name, compression_codec from system.columns where table = '{}' and database='{}' FORMAT TabSeparatedWithNamesAndTypes".format(instance.table_name(), self.database.db_name))
|
||||
lines = r.splitlines()
|
||||
field_names = parse_tsv(lines[0])
|
||||
field_types = parse_tsv(lines[1])
|
||||
data = [tuple(parse_tsv(line)) for line in lines[2:]]
|
||||
self.assertListEqual(data, [('uint64_field', 'CODEC(ZSTD(10))'),
|
||||
('datetime_field', 'CODEC(Delta(4), ZSTD(1))'),
|
||||
('date_field', 'CODEC(Delta(4), ZSTD(22))'),
|
||||
('int64_field', 'CODEC(LZ4)'),
|
||||
('string_field', 'CODEC(LZ4HC(10))'),
|
||||
('nullable_field', 'CODEC(ZSTD(1))'),
|
||||
('array_field', 'CODEC(Delta(2), LZ4HC(0))'),
|
||||
('float_field', 'CODEC(NONE)'),
|
||||
('alias_field', 'CODEC(ZSTD(4))')])
|
||||
|
||||
|
||||
class CompressedModel(Model):
|
||||
uint64_field = UInt64Field(codec='ZSTD(10)')
|
||||
datetime_field = DateTimeField(codec='Delta,ZSTD')
|
||||
date_field = DateField(codec='Delta(4),ZSTD(22)')
|
||||
int64_field = Int64Field(default=42, codec='LZ4')
|
||||
string_field = StringField(default='dozo', codec='LZ4HC(10)')
|
||||
nullable_field = NullableField(Float32Field(), codec='ZSTD')
|
||||
array_field = ArrayField(FixedStringField(length=15), codec='Delta(2),LZ4HC')
|
||||
float_field = Float32Field(codec='NONE')
|
||||
alias_field = Float32Field(alias='float_field', codec='ZSTD(4)')
|
||||
|
||||
engine = MergeTree('datetime_field', ('uint64_field', 'datetime_field'))
|
||||
@@ -0,0 +1,44 @@
|
||||
import unittest
|
||||
|
||||
from datastore_orm import *
|
||||
from .base_test_with_data import Person
|
||||
|
||||
|
||||
class ConstraintsTest(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.database = Database('test-db', log_statements=True)
|
||||
if self.database.server_version < (19, 14, 3, 3):
|
||||
raise unittest.SkipTest('ClickHouse version too old')
|
||||
self.database.create_table(PersonWithConstraints)
|
||||
|
||||
def tearDown(self):
|
||||
self.database.drop_database()
|
||||
|
||||
def test_insert_valid_values(self):
|
||||
self.database.insert([
|
||||
PersonWithConstraints(first_name="Mike", last_name="Caruzo", birthday="2000-01-01", height=1.66)
|
||||
])
|
||||
|
||||
def test_insert_invalid_values(self):
|
||||
with self.assertRaises(ServerError) as e:
|
||||
self.database.insert([
|
||||
PersonWithConstraints(first_name="Mike", last_name="Caruzo", birthday="2100-01-01", height=1.66)
|
||||
])
|
||||
self.assertEqual(e.code, 469)
|
||||
self.assertTrue('Constraint `birthday_in_the_past`' in e.message)
|
||||
|
||||
with self.assertRaises(ServerError) as e:
|
||||
self.database.insert([
|
||||
PersonWithConstraints(first_name="Mike", last_name="Caruzo", birthday="1970-01-01", height=3)
|
||||
])
|
||||
self.assertEqual(e.code, 469)
|
||||
self.assertTrue('Constraint `max_height`' in e.message)
|
||||
|
||||
|
||||
class PersonWithConstraints(Person):
|
||||
|
||||
birthday_in_the_past = Constraint(Person.birthday <= F.today())
|
||||
max_height = Constraint(Person.height <= 2.75)
|
||||
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
import unittest
|
||||
from datastore_orm.database import Database
|
||||
from datastore_orm.fields import Field, Int16Field
|
||||
from datastore_orm.models import Model
|
||||
from datastore_orm.engines import Memory
|
||||
|
||||
|
||||
class CustomFieldsTest(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.database = Database('test-db', log_statements=True)
|
||||
|
||||
def tearDown(self):
|
||||
self.database.drop_database()
|
||||
|
||||
def test_boolean_field(self):
|
||||
# Create a model
|
||||
class TestModel(Model):
|
||||
i = Int16Field()
|
||||
f = BooleanField()
|
||||
engine = Memory()
|
||||
self.database.create_table(TestModel)
|
||||
# Check valid values
|
||||
for index, value in enumerate([1, '1', True, 0, '0', False]):
|
||||
rec = TestModel(i=index, f=value)
|
||||
self.database.insert([rec])
|
||||
self.assertEqual([rec.f for rec in TestModel.objects_in(self.database).order_by('i')],
|
||||
[True, True, True, False, False, False])
|
||||
# Check invalid values
|
||||
for value in [None, 'zzz', -5, 7]:
|
||||
with self.assertRaises(ValueError):
|
||||
TestModel(i=1, f=value)
|
||||
|
||||
|
||||
class BooleanField(Field):
|
||||
|
||||
# The ClickHouse column type to use
|
||||
db_type = 'UInt8'
|
||||
|
||||
# The default value if empty
|
||||
class_default = False
|
||||
|
||||
def to_python(self, value, timezone_in_use):
|
||||
# Convert valid values to bool
|
||||
if value in (1, '1', True):
|
||||
return True
|
||||
elif value in (0, '0', False):
|
||||
return False
|
||||
else:
|
||||
raise ValueError('Invalid value for BooleanField: %r' % value)
|
||||
|
||||
def to_db_string(self, value, quote=True):
|
||||
# The value was already converted by to_python, so it's a bool
|
||||
return '1' if value else '0'
|
||||
|
||||
+186
-32
@@ -1,8 +1,13 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import unittest
|
||||
import datetime
|
||||
|
||||
from infi.clickhouse_orm.database import Database, DatabaseException
|
||||
from datastore_orm.database import ServerError, DatabaseException
|
||||
from datastore_orm.models import Model
|
||||
from datastore_orm.engines import Memory
|
||||
from datastore_orm.fields import *
|
||||
from datastore_orm.funcs import F
|
||||
from datastore_orm.query import Q
|
||||
from .base_test_with_data import *
|
||||
|
||||
|
||||
@@ -20,34 +25,64 @@ class DatabaseTestCase(TestCaseWithData):
|
||||
def test_insert__empty(self):
|
||||
self._insert_and_check([], 0)
|
||||
|
||||
def test_insert__small_batches(self):
|
||||
self._insert_and_check(self._sample_data(), len(data), batch_size=10)
|
||||
|
||||
def test_insert__medium_batches(self):
|
||||
self._insert_and_check(self._sample_data(), len(data), batch_size=100)
|
||||
|
||||
def test_insert__funcs_as_default_values(self):
|
||||
if self.database.server_version < (20, 1, 2, 4):
|
||||
raise unittest.SkipTest('Buggy in server versions before 20.1.2.4')
|
||||
class TestModel(Model):
|
||||
a = DateTimeField(default=datetime.datetime(2020, 1, 1))
|
||||
b = DateField(default=F.toDate(a))
|
||||
c = Int32Field(default=7)
|
||||
d = Int32Field(default=c * 5)
|
||||
engine = Memory()
|
||||
self.database.create_table(TestModel)
|
||||
self.database.insert([TestModel()])
|
||||
t = TestModel.objects_in(self.database)[0]
|
||||
self.assertEqual(str(t.b), '2020-01-01')
|
||||
self.assertEqual(t.d, 35)
|
||||
|
||||
def test_count(self):
|
||||
self.database.insert(self._sample_data())
|
||||
self.assertEquals(self.database.count(Person), 100)
|
||||
self.assertEquals(self.database.count(Person, "first_name = 'Courtney'"), 2)
|
||||
self.assertEquals(self.database.count(Person, "birthday > '2000-01-01'"), 22)
|
||||
self.assertEquals(self.database.count(Person, "birthday < '1970-03-01'"), 0)
|
||||
self.assertEqual(self.database.count(Person), 100)
|
||||
# Conditions as string
|
||||
self.assertEqual(self.database.count(Person, "first_name = 'Courtney'"), 2)
|
||||
self.assertEqual(self.database.count(Person, "birthday > '2000-01-01'"), 22)
|
||||
self.assertEqual(self.database.count(Person, "birthday < '1970-03-01'"), 0)
|
||||
# Conditions as expression
|
||||
self.assertEqual(self.database.count(Person, Person.birthday > datetime.date(2000, 1, 1)), 22)
|
||||
# Conditions as Q object
|
||||
self.assertEqual(self.database.count(Person, Q(birthday__gt=datetime.date(2000, 1, 1))), 22)
|
||||
|
||||
def test_select(self):
|
||||
self._insert_and_check(self._sample_data(), len(data))
|
||||
query = "SELECT * FROM `test-db`.person WHERE first_name = 'Whitney' ORDER BY last_name"
|
||||
results = list(self.database.select(query, Person))
|
||||
self.assertEquals(len(results), 2)
|
||||
self.assertEquals(results[0].last_name, 'Durham')
|
||||
self.assertEquals(results[0].height, 1.72)
|
||||
self.assertEquals(results[1].last_name, 'Scott')
|
||||
self.assertEquals(results[1].height, 1.70)
|
||||
self.assertEqual(len(results), 2)
|
||||
self.assertEqual(results[0].last_name, 'Durham')
|
||||
self.assertEqual(results[0].height, 1.72)
|
||||
self.assertEqual(results[1].last_name, 'Scott')
|
||||
self.assertEqual(results[1].height, 1.70)
|
||||
self.assertEqual(results[0].get_database(), self.database)
|
||||
self.assertEqual(results[1].get_database(), self.database)
|
||||
|
||||
def test_dollar_in_select(self):
|
||||
query = "SELECT * FROM $table WHERE first_name = '$utm_source'"
|
||||
list(self.database.select(query, Person))
|
||||
|
||||
def test_select_partial_fields(self):
|
||||
self._insert_and_check(self._sample_data(), len(data))
|
||||
query = "SELECT first_name, last_name FROM `test-db`.person WHERE first_name = 'Whitney' ORDER BY last_name"
|
||||
results = list(self.database.select(query, Person))
|
||||
self.assertEquals(len(results), 2)
|
||||
self.assertEquals(results[0].last_name, 'Durham')
|
||||
self.assertEquals(results[0].height, 0) # default value
|
||||
self.assertEquals(results[1].last_name, 'Scott')
|
||||
self.assertEquals(results[1].height, 0) # default value
|
||||
self.assertEqual(len(results), 2)
|
||||
self.assertEqual(results[0].last_name, 'Durham')
|
||||
self.assertEqual(results[0].height, 0) # default value
|
||||
self.assertEqual(results[1].last_name, 'Scott')
|
||||
self.assertEqual(results[1].height, 0) # default value
|
||||
self.assertEqual(results[0].get_database(), self.database)
|
||||
self.assertEqual(results[1].get_database(), self.database)
|
||||
|
||||
@@ -55,12 +90,12 @@ class DatabaseTestCase(TestCaseWithData):
|
||||
self._insert_and_check(self._sample_data(), len(data))
|
||||
query = "SELECT * FROM `test-db`.person WHERE first_name = 'Whitney' ORDER BY last_name"
|
||||
results = list(self.database.select(query))
|
||||
self.assertEquals(len(results), 2)
|
||||
self.assertEquals(results[0].__class__.__name__, 'AdHocModel')
|
||||
self.assertEquals(results[0].last_name, 'Durham')
|
||||
self.assertEquals(results[0].height, 1.72)
|
||||
self.assertEquals(results[1].last_name, 'Scott')
|
||||
self.assertEquals(results[1].height, 1.70)
|
||||
self.assertEqual(len(results), 2)
|
||||
self.assertEqual(results[0].__class__.__name__, 'AdHocModel')
|
||||
self.assertEqual(results[0].last_name, 'Durham')
|
||||
self.assertEqual(results[0].height, 1.72)
|
||||
self.assertEqual(results[1].last_name, 'Scott')
|
||||
self.assertEqual(results[1].height, 1.70)
|
||||
self.assertEqual(results[0].get_database(), self.database)
|
||||
self.assertEqual(results[1].get_database(), self.database)
|
||||
|
||||
@@ -71,7 +106,7 @@ class DatabaseTestCase(TestCaseWithData):
|
||||
total = sum(r.height for r in results[:-1])
|
||||
# Last line has an empty last name, and total of all heights
|
||||
self.assertFalse(results[-1].last_name)
|
||||
self.assertEquals(total, results[-1].height)
|
||||
self.assertEqual(total, results[-1].height)
|
||||
|
||||
def test_pagination(self):
|
||||
self._insert_and_check(self._sample_data(), len(data))
|
||||
@@ -82,14 +117,14 @@ class DatabaseTestCase(TestCaseWithData):
|
||||
instances = set()
|
||||
while True:
|
||||
page = self.database.paginate(Person, 'first_name, last_name', page_num, page_size)
|
||||
self.assertEquals(page.number_of_objects, len(data))
|
||||
self.assertEqual(page.number_of_objects, len(data))
|
||||
self.assertGreater(page.pages_total, 0)
|
||||
[instances.add(obj.to_tsv()) for obj in page.objects]
|
||||
if page.pages_total == page_num:
|
||||
break
|
||||
page_num += 1
|
||||
# Verify that all instances were returned
|
||||
self.assertEquals(len(instances), len(data))
|
||||
self.assertEqual(len(instances), len(data))
|
||||
|
||||
def test_pagination_last_page(self):
|
||||
self._insert_and_check(self._sample_data(), len(data))
|
||||
@@ -98,10 +133,18 @@ class DatabaseTestCase(TestCaseWithData):
|
||||
# Ask for the last page in two different ways and verify equality
|
||||
page_a = self.database.paginate(Person, 'first_name, last_name', -1, page_size)
|
||||
page_b = self.database.paginate(Person, 'first_name, last_name', page_a.pages_total, page_size)
|
||||
self.assertEquals(page_a[1:], page_b[1:])
|
||||
self.assertEquals([obj.to_tsv() for obj in page_a.objects],
|
||||
self.assertEqual(page_a[1:], page_b[1:])
|
||||
self.assertEqual([obj.to_tsv() for obj in page_a.objects],
|
||||
[obj.to_tsv() for obj in page_b.objects])
|
||||
|
||||
def test_pagination_empty_page(self):
|
||||
for page_num in (-1, 1, 2):
|
||||
page = self.database.paginate(Person, 'first_name, last_name', page_num, 10, conditions="first_name = 'Ziggy'")
|
||||
self.assertEqual(page.number_of_objects, 0)
|
||||
self.assertEqual(page.objects, [])
|
||||
self.assertEqual(page.pages_total, 0)
|
||||
self.assertEqual(page.number, max(page_num, 1))
|
||||
|
||||
def test_pagination_invalid_page(self):
|
||||
self._insert_and_check(self._sample_data(), len(data))
|
||||
for page_num in (0, -2, -100):
|
||||
@@ -110,22 +153,133 @@ class DatabaseTestCase(TestCaseWithData):
|
||||
|
||||
def test_pagination_with_conditions(self):
|
||||
self._insert_and_check(self._sample_data(), len(data))
|
||||
# Conditions as string
|
||||
page = self.database.paginate(Person, 'first_name, last_name', 1, 100, conditions="first_name < 'Ava'")
|
||||
self.assertEquals(page.number_of_objects, 10)
|
||||
self.assertEqual(page.number_of_objects, 10)
|
||||
# Conditions as expression
|
||||
page = self.database.paginate(Person, 'first_name, last_name', 1, 100, conditions=Person.first_name < 'Ava')
|
||||
self.assertEqual(page.number_of_objects, 10)
|
||||
# Conditions as Q object
|
||||
page = self.database.paginate(Person, 'first_name, last_name', 1, 100, conditions=Q(first_name__lt='Ava'))
|
||||
self.assertEqual(page.number_of_objects, 10)
|
||||
|
||||
def test_special_chars(self):
|
||||
s = u'אבגד \\\'"`,.;éåäöšž\n\t\0\b\r'
|
||||
p = Person(first_name=s)
|
||||
self.database.insert([p])
|
||||
p = list(self.database.select("SELECT * from $table", Person))[0]
|
||||
self.assertEquals(p.first_name, s)
|
||||
self.assertEqual(p.first_name, s)
|
||||
|
||||
def test_raw(self):
|
||||
self._insert_and_check(self._sample_data(), len(data))
|
||||
query = "SELECT * FROM `test-db`.person WHERE first_name = 'Whitney' ORDER BY last_name"
|
||||
results = self.database.raw(query)
|
||||
self.assertEqual(results, "Whitney\tDurham\t1977-09-15\t1.72\nWhitney\tScott\t1971-07-04\t1.7\n")
|
||||
self.assertEqual(results, "Whitney\tDurham\t1977-09-15\t1.72\t\\N\nWhitney\tScott\t1971-07-04\t1.7\t\\N\n")
|
||||
|
||||
def test_invalid_user(self):
|
||||
with self.assertRaises(DatabaseException):
|
||||
Database(self.database.db_name, username='default', password='wrong')
|
||||
with self.assertRaises(ServerError) as cm:
|
||||
Database(self.database.db_name, username='default', password='wrong')
|
||||
|
||||
exc = cm.exception
|
||||
if exc.code == 193: # ClickHouse version < 20.3
|
||||
self.assertTrue(exc.message.startswith('Wrong password for user default'))
|
||||
elif exc.code == 516: # ClickHouse version >= 20.3
|
||||
self.assertTrue(exc.message.startswith('default: Authentication failed'))
|
||||
else:
|
||||
raise Exception('Unexpected error code - %s' % exc.code)
|
||||
|
||||
def test_nonexisting_db(self):
|
||||
db = Database('db_not_here', autocreate=False)
|
||||
with self.assertRaises(ServerError) as cm:
|
||||
db.create_table(Person)
|
||||
exc = cm.exception
|
||||
self.assertEqual(exc.code, 81)
|
||||
self.assertTrue(exc.message.startswith("Database db_not_here doesn't exist"))
|
||||
# Create and delete the db twice, to ensure db_exists gets updated
|
||||
for i in range(2):
|
||||
# Now create the database - should succeed
|
||||
db.create_database()
|
||||
self.assertTrue(db.db_exists)
|
||||
db.create_table(Person)
|
||||
# Drop the database
|
||||
db.drop_database()
|
||||
self.assertFalse(db.db_exists)
|
||||
|
||||
def test_preexisting_db(self):
|
||||
db = Database(self.database.db_name, autocreate=False)
|
||||
db.count(Person)
|
||||
|
||||
def test_missing_engine(self):
|
||||
class EnginelessModel(Model):
|
||||
float_field = Float32Field()
|
||||
with self.assertRaises(DatabaseException) as cm:
|
||||
self.database.create_table(EnginelessModel)
|
||||
self.assertEqual(str(cm.exception), 'EnginelessModel class must define an engine')
|
||||
|
||||
def test_potentially_problematic_field_names(self):
|
||||
class Model1(Model):
|
||||
system = StringField()
|
||||
readonly = StringField()
|
||||
engine = Memory()
|
||||
instance = Model1(system='s', readonly='r')
|
||||
self.assertEqual(instance.to_dict(), dict(system='s', readonly='r'))
|
||||
self.database.create_table(Model1)
|
||||
self.database.insert([instance])
|
||||
instance = Model1.objects_in(self.database)[0]
|
||||
self.assertEqual(instance.to_dict(), dict(system='s', readonly='r'))
|
||||
|
||||
def test_does_table_exist(self):
|
||||
class Person2(Person):
|
||||
pass
|
||||
self.assertTrue(self.database.does_table_exist(Person))
|
||||
self.assertFalse(self.database.does_table_exist(Person2))
|
||||
|
||||
def test_add_setting(self):
|
||||
# Non-string setting name should not be accepted
|
||||
with self.assertRaises(AssertionError):
|
||||
self.database.add_setting(0, 1)
|
||||
# Add a setting and see that it makes the query fail
|
||||
self.database.add_setting('max_columns_to_read', 1)
|
||||
with self.assertRaises(ServerError):
|
||||
list(self.database.select('SELECT * from system.tables'))
|
||||
# Remove the setting and see that now it works
|
||||
self.database.add_setting('max_columns_to_read', None)
|
||||
list(self.database.select('SELECT * from system.tables'))
|
||||
|
||||
def test_create_ad_hoc_field(self):
|
||||
# Tests that create_ad_hoc_field works for all column types in the database
|
||||
from datastore_orm.models import ModelBase
|
||||
query = "SELECT DISTINCT type FROM system.columns"
|
||||
for row in self.database.select(query):
|
||||
ModelBase.create_ad_hoc_field(row.type)
|
||||
|
||||
def test_get_model_for_table(self):
|
||||
# Tests that get_model_for_table works for a non-system model
|
||||
model = self.database.get_model_for_table('person')
|
||||
self.assertFalse(model.is_system_model())
|
||||
self.assertFalse(model.is_read_only())
|
||||
self.assertEqual(model.table_name(), 'person')
|
||||
# Read a few records
|
||||
list(model.objects_in(self.database)[:10])
|
||||
# Inserts should work too
|
||||
self.database.insert([
|
||||
model(first_name='aaa', last_name='bbb', height=1.77)
|
||||
])
|
||||
|
||||
def test_get_model_for_table__system(self):
|
||||
# Tests that get_model_for_table works for all system tables
|
||||
query = "SELECT name FROM system.tables WHERE database='system'"
|
||||
for row in self.database.select(query):
|
||||
print(row.name)
|
||||
model = self.database.get_model_for_table(row.name, system_table=True)
|
||||
self.assertTrue(model.is_system_model())
|
||||
self.assertTrue(model.is_read_only())
|
||||
self.assertEqual(model.table_name(), row.name)
|
||||
# Read a few records
|
||||
try:
|
||||
list(model.objects_in(self.database)[:10])
|
||||
except ServerError as e:
|
||||
if 'Not enough privileges' in e.message:
|
||||
pass
|
||||
else:
|
||||
raise
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
import unittest
|
||||
import datetime
|
||||
import pytz
|
||||
|
||||
from datastore_orm.database import Database
|
||||
from datastore_orm.models import Model
|
||||
from datastore_orm.fields import *
|
||||
from datastore_orm.engines import *
|
||||
|
||||
|
||||
class DateFieldsTest(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.database = Database('test-db', log_statements=True)
|
||||
if self.database.server_version < (20, 1, 2, 4):
|
||||
raise unittest.SkipTest('ClickHouse version too old')
|
||||
self.database.create_table(ModelWithDate)
|
||||
|
||||
def tearDown(self):
|
||||
self.database.drop_database()
|
||||
|
||||
def test_ad_hoc_model(self):
|
||||
self.database.insert([
|
||||
ModelWithDate(
|
||||
date_field='2016-08-30',
|
||||
datetime_field='2016-08-30 03:50:00',
|
||||
datetime64_field='2016-08-30 03:50:00.123456',
|
||||
datetime64_3_field='2016-08-30 03:50:00.123456'
|
||||
),
|
||||
ModelWithDate(
|
||||
date_field='2016-08-31',
|
||||
datetime_field='2016-08-31 01:30:00',
|
||||
datetime64_field='2016-08-31 01:30:00.123456',
|
||||
datetime64_3_field='2016-08-31 01:30:00.123456')
|
||||
])
|
||||
|
||||
# toStartOfHour returns DateTime('Asia/Yekaterinburg') in my case, so I test it here to
|
||||
query = 'SELECT toStartOfHour(datetime_field) as hour_start, * from $db.modelwithdate ORDER BY date_field'
|
||||
results = list(self.database.select(query))
|
||||
self.assertEqual(len(results), 2)
|
||||
self.assertEqual(results[0].date_field, datetime.date(2016, 8, 30))
|
||||
self.assertEqual(results[0].datetime_field, datetime.datetime(2016, 8, 30, 3, 50, 0, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[0].hour_start, datetime.datetime(2016, 8, 30, 3, 0, 0, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[1].date_field, datetime.date(2016, 8, 31))
|
||||
self.assertEqual(results[1].datetime_field, datetime.datetime(2016, 8, 31, 1, 30, 0, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[1].hour_start, datetime.datetime(2016, 8, 31, 1, 0, 0, tzinfo=pytz.UTC))
|
||||
|
||||
self.assertEqual(results[0].datetime64_field, datetime.datetime(2016, 8, 30, 3, 50, 0, 123456, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[0].datetime64_3_field, datetime.datetime(2016, 8, 30, 3, 50, 0, 123000,
|
||||
tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[1].datetime64_field, datetime.datetime(2016, 8, 31, 1, 30, 0, 123456, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[1].datetime64_3_field, datetime.datetime(2016, 8, 31, 1, 30, 0, 123000,
|
||||
tzinfo=pytz.UTC))
|
||||
|
||||
|
||||
class ModelWithDate(Model):
|
||||
date_field = DateField()
|
||||
datetime_field = DateTimeField()
|
||||
datetime64_field = DateTime64Field()
|
||||
datetime64_3_field = DateTime64Field(precision=3)
|
||||
|
||||
engine = MergeTree('date_field', ('date_field',))
|
||||
|
||||
|
||||
class ModelWithTz(Model):
|
||||
datetime_no_tz_field = DateTimeField() # server tz
|
||||
datetime_tz_field = DateTimeField(timezone='Europe/Madrid')
|
||||
datetime64_tz_field = DateTime64Field(timezone='Europe/Madrid')
|
||||
datetime_utc_field = DateTimeField(timezone=pytz.UTC)
|
||||
|
||||
engine = MergeTree('datetime_no_tz_field', ('datetime_no_tz_field',))
|
||||
|
||||
|
||||
class DateTimeFieldWithTzTest(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.database = Database('test-db', log_statements=True)
|
||||
if self.database.server_version < (20, 1, 2, 4):
|
||||
raise unittest.SkipTest('ClickHouse version too old')
|
||||
self.database.create_table(ModelWithTz)
|
||||
|
||||
def tearDown(self):
|
||||
self.database.drop_database()
|
||||
|
||||
def test_ad_hoc_model(self):
|
||||
self.database.insert([
|
||||
ModelWithTz(
|
||||
datetime_no_tz_field='2020-06-11 04:00:00',
|
||||
datetime_tz_field='2020-06-11 04:00:00',
|
||||
datetime64_tz_field='2020-06-11 04:00:00',
|
||||
datetime_utc_field='2020-06-11 04:00:00',
|
||||
),
|
||||
ModelWithTz(
|
||||
datetime_no_tz_field='2020-06-11 07:00:00+0300',
|
||||
datetime_tz_field='2020-06-11 07:00:00+0300',
|
||||
datetime64_tz_field='2020-06-11 07:00:00+0300',
|
||||
datetime_utc_field='2020-06-11 07:00:00+0300',
|
||||
),
|
||||
])
|
||||
query = 'SELECT * from $db.modelwithtz ORDER BY datetime_no_tz_field'
|
||||
results = list(self.database.select(query))
|
||||
|
||||
self.assertEqual(results[0].datetime_no_tz_field, datetime.datetime(2020, 6, 11, 4, 0, 0, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[0].datetime_tz_field, datetime.datetime(2020, 6, 11, 4, 0, 0, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[0].datetime64_tz_field, datetime.datetime(2020, 6, 11, 4, 0, 0, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[0].datetime_utc_field, datetime.datetime(2020, 6, 11, 4, 0, 0, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[1].datetime_no_tz_field, datetime.datetime(2020, 6, 11, 4, 0, 0, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[1].datetime_tz_field, datetime.datetime(2020, 6, 11, 4, 0, 0, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[1].datetime64_tz_field, datetime.datetime(2020, 6, 11, 4, 0, 0, tzinfo=pytz.UTC))
|
||||
self.assertEqual(results[1].datetime_utc_field, datetime.datetime(2020, 6, 11, 4, 0, 0, tzinfo=pytz.UTC))
|
||||
|
||||
self.assertEqual(results[0].datetime_no_tz_field.tzinfo.zone, self.database.server_timezone.zone)
|
||||
self.assertEqual(results[0].datetime_tz_field.tzinfo.zone, pytz.timezone('Europe/Madrid').zone)
|
||||
self.assertEqual(results[0].datetime64_tz_field.tzinfo.zone, pytz.timezone('Europe/Madrid').zone)
|
||||
self.assertEqual(results[0].datetime_utc_field.tzinfo.zone, pytz.timezone('UTC').zone)
|
||||
self.assertEqual(results[1].datetime_no_tz_field.tzinfo.zone, self.database.server_timezone.zone)
|
||||
self.assertEqual(results[1].datetime_tz_field.tzinfo.zone, pytz.timezone('Europe/Madrid').zone)
|
||||
self.assertEqual(results[1].datetime64_tz_field.tzinfo.zone, pytz.timezone('Europe/Madrid').zone)
|
||||
self.assertEqual(results[1].datetime_utc_field.tzinfo.zone, pytz.timezone('UTC').zone)
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user