6 Commits

Author SHA1 Message Date
Bill
5313f5f4ef version update to publish to Pypi 6
Some checks failed
Publish to PyPI / Build wheels on macos-latest (push) Has been cancelled
Publish to PyPI / Build wheels on ubuntu-latest (push) Has been cancelled
Publish to PyPI / Build wheels on windows-latest (push) Has been cancelled
Publish to PyPI / Publish to PyPI (push) Has been cancelled
2025-11-23 13:34:19 -07:00
Bill
72974a9658 version update to publish to Pypi 5
Some checks failed
Publish to PyPI / Build wheels on macos-latest (push) Has been cancelled
Publish to PyPI / Build wheels on ubuntu-latest (push) Has been cancelled
Publish to PyPI / Build wheels on windows-latest (push) Has been cancelled
Publish to PyPI / Publish to PyPI (push) Has been cancelled
2025-11-23 13:27:58 -07:00
Bill
8981b3038d version update to publish to Pypi 5 2025-11-23 13:27:22 -07:00
Bill
751b95a123 version update to publish to Pypi 4
Some checks failed
Publish to PyPI / Build wheels on macos-latest (push) Has been cancelled
Publish to PyPI / Build wheels on ubuntu-latest (push) Has been cancelled
Publish to PyPI / Build wheels on windows-latest (push) Has been cancelled
Publish to PyPI / Publish to PyPI (push) Has been cancelled
2025-11-23 13:07:02 -07:00
Bill
26d1eb631e version update to publish to Pypi 2
Some checks failed
Publish to PyPI / Build wheels on macos-latest (push) Has been cancelled
Publish to PyPI / Build wheels on ubuntu-latest (push) Has been cancelled
Publish to PyPI / Build wheels on windows-latest (push) Has been cancelled
Publish to PyPI / Publish to PyPI (push) Has been cancelled
2025-11-23 12:53:54 -07:00
Bill
d6bd2eee00 version update to publish to Pypi
Some checks failed
Publish to PyPI / Build wheels on macos-latest (push) Has been cancelled
Publish to PyPI / Build wheels on ubuntu-latest (push) Has been cancelled
Publish to PyPI / Build wheels on windows-latest (push) Has been cancelled
Publish to PyPI / Publish to PyPI (push) Has been cancelled
2025-11-23 12:45:31 -07:00
8 changed files with 746 additions and 226 deletions

122
Cargo.lock generated
View File

@@ -86,7 +86,7 @@ dependencies = [
"arrow-data", "arrow-data",
"arrow-schema", "arrow-schema",
"chrono", "chrono",
"chrono-tz", "chrono-tz 0.10.4",
"half", "half",
"hashbrown", "hashbrown",
"num-complex", "num-complex",
@@ -365,6 +365,17 @@ dependencies = [
"windows-link 0.2.1", "windows-link 0.2.1",
] ]
[[package]]
name = "chrono-tz"
version = "0.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "93698b29de5e97ad0ae26447b344c482a7284c737d9ddc5f9e52b74a336671bb"
dependencies = [
"chrono",
"chrono-tz-build",
"phf 0.11.3",
]
[[package]] [[package]]
name = "chrono-tz" name = "chrono-tz"
version = "0.10.4" version = "0.10.4"
@@ -375,6 +386,17 @@ dependencies = [
"phf 0.12.1", "phf 0.12.1",
] ]
[[package]]
name = "chrono-tz-build"
version = "0.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0c088aee841df9c3041febbb73934cfc39708749bf96dc827e3359cd39ef11b1"
dependencies = [
"parse-zoneinfo",
"phf 0.11.3",
"phf_codegen",
]
[[package]] [[package]]
name = "comfy-table" name = "comfy-table"
version = "7.1.2" version = "7.1.2"
@@ -1081,12 +1103,30 @@ dependencies = [
"windows-link 0.2.1", "windows-link 0.2.1",
] ]
[[package]]
name = "parse-zoneinfo"
version = "0.3.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1f2a05b18d44e2957b88f96ba460715e295bc1d7510468a2f3d3b44535d26c24"
dependencies = [
"regex",
]
[[package]] [[package]]
name = "percent-encoding" name = "percent-encoding"
version = "2.3.2" version = "2.3.2"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220"
[[package]]
name = "phf"
version = "0.11.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1fd6780a80ae0c52cc120a26a1a42c1ae51b247a253e4e06113d23d2c2edd078"
dependencies = [
"phf_shared 0.11.3",
]
[[package]] [[package]]
name = "phf" name = "phf"
version = "0.12.1" version = "0.12.1"
@@ -1106,6 +1146,35 @@ dependencies = [
"serde", "serde",
] ]
[[package]]
name = "phf_codegen"
version = "0.11.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "aef8048c789fa5e851558d709946d6d79a8ff88c0440c587967f8e94bfb1216a"
dependencies = [
"phf_generator",
"phf_shared 0.11.3",
]
[[package]]
name = "phf_generator"
version = "0.11.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3c80231409c20246a13fddb31776fb942c38553c51e871f8cbd687a4cfb5843d"
dependencies = [
"phf_shared 0.11.3",
"rand 0.8.5",
]
[[package]]
name = "phf_shared"
version = "0.11.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "67eabc2ef2a60eb7faa00097bd1ffdb5bd28e62bf39990626a582201b7a754e5"
dependencies = [
"siphasher",
]
[[package]] [[package]]
name = "phf_shared" name = "phf_shared"
version = "0.12.1" version = "0.12.1"
@@ -1164,7 +1233,7 @@ dependencies = [
"hmac", "hmac",
"md-5", "md-5",
"memchr", "memchr",
"rand", "rand 0.9.2",
"sha2", "sha2",
"stringprep", "stringprep",
] ]
@@ -1200,12 +1269,12 @@ dependencies = [
[[package]] [[package]]
name = "pyo3" name = "pyo3"
version = "0.27.1" version = "0.27.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "37a6df7eab65fc7bee654a421404947e10a0f7085b6951bf2ea395f4659fb0cf" checksum = "fa8e48c12afdeb26aa4be4e5c49fb5e11c3efa0878db783a960eea2b9ac6dd19"
dependencies = [ dependencies = [
"chrono", "chrono",
"chrono-tz", "chrono-tz 0.10.4",
"indexmap", "indexmap",
"indoc", "indoc",
"libc", "libc",
@@ -1231,7 +1300,7 @@ dependencies = [
"arrow-schema", "arrow-schema",
"arrow-select", "arrow-select",
"chrono", "chrono",
"chrono-tz", "chrono-tz 0.10.4",
"half", "half",
"indexmap", "indexmap",
"numpy", "numpy",
@@ -1241,18 +1310,18 @@ dependencies = [
[[package]] [[package]]
name = "pyo3-build-config" name = "pyo3-build-config"
version = "0.27.1" version = "0.27.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f77d387774f6f6eec64a004eac0ed525aab7fa1966d94b42f743797b3e395afb" checksum = "bc1989dbf2b60852e0782c7487ebf0b4c7f43161ffe820849b56cf05f945cee1"
dependencies = [ dependencies = [
"target-lexicon", "target-lexicon",
] ]
[[package]] [[package]]
name = "pyo3-ffi" name = "pyo3-ffi"
version = "0.27.1" version = "0.27.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2dd13844a4242793e02df3e2ec093f540d948299a6a77ea9ce7afd8623f542be" checksum = "c808286da7500385148930152e54fb6883452033085bf1f857d85d4e82ca905c"
dependencies = [ dependencies = [
"libc", "libc",
"pyo3-build-config", "pyo3-build-config",
@@ -1260,9 +1329,9 @@ dependencies = [
[[package]] [[package]]
name = "pyo3-macros" name = "pyo3-macros"
version = "0.27.1" version = "0.27.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "eaf8f9f1108270b90d3676b8679586385430e5c0bb78bb5f043f95499c821a71" checksum = "83a0543c16be0d86cf0dbf2e2b636ece9fd38f20406bb43c255e0bc368095f92"
dependencies = [ dependencies = [
"proc-macro2", "proc-macro2",
"pyo3-macros-backend", "pyo3-macros-backend",
@@ -1272,9 +1341,9 @@ dependencies = [
[[package]] [[package]]
name = "pyo3-macros-backend" name = "pyo3-macros-backend"
version = "0.27.1" version = "0.27.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "70a3b2274450ba5288bc9b8c1b69ff569d1d61189d4bff38f8d22e03d17f932b" checksum = "2a00da2ce064dcd582448ea24a5a26fa9527e0483103019b741ebcbe632dcd29"
dependencies = [ dependencies = [
"heck", "heck",
"proc-macro2", "proc-macro2",
@@ -1298,6 +1367,15 @@ version = "5.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f"
[[package]]
name = "rand"
version = "0.8.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "34af8d1a0e25924bc5b7c43c079c942339d8f0a8b57c39049bef581b46327404"
dependencies = [
"rand_core 0.6.4",
]
[[package]] [[package]]
name = "rand" name = "rand"
version = "0.9.2" version = "0.9.2"
@@ -1305,7 +1383,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6db2770f06117d490610c7488547d543617b21bfa07796d7a12f6f1bd53850d1" checksum = "6db2770f06117d490610c7488547d543617b21bfa07796d7a12f6f1bd53850d1"
dependencies = [ dependencies = [
"rand_chacha", "rand_chacha",
"rand_core", "rand_core 0.9.3",
] ]
[[package]] [[package]]
@@ -1315,9 +1393,15 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb"
dependencies = [ dependencies = [
"ppv-lite86", "ppv-lite86",
"rand_core", "rand_core 0.9.3",
] ]
[[package]]
name = "rand_core"
version = "0.6.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c"
[[package]] [[package]]
name = "rand_core" name = "rand_core"
version = "0.9.3" version = "0.9.3"
@@ -1694,7 +1778,7 @@ dependencies = [
"pin-project-lite", "pin-project-lite",
"postgres-protocol", "postgres-protocol",
"postgres-types", "postgres-types",
"rand", "rand 0.9.2",
"socket2", "socket2",
"tokio", "tokio",
"tokio-util", "tokio-util",
@@ -1815,7 +1899,7 @@ checksum = "562d481066bde0658276a35467c4af00bdc6ee726305698a55b86e61d7ad82bb"
[[package]] [[package]]
name = "unchecked-io" name = "unchecked-io"
version = "0.1.0" version = "0.1.6"
dependencies = [ dependencies = [
"anyhow", "anyhow",
"arrow", "arrow",
@@ -1823,7 +1907,7 @@ dependencies = [
"byteorder", "byteorder",
"bytes", "bytes",
"chrono", "chrono",
"chrono-tz", "chrono-tz 0.9.0",
"deadpool-postgres", "deadpool-postgres",
"futures-util", "futures-util",
"mimalloc", "mimalloc",

View File

@@ -1,8 +1,8 @@
[package] [package]
name = "unchecked-io" name = "unchecked-io"
version = "0.1.0" version = "0.1.6"
authors = ["Billthemaker"] # Replace with your name or alias authors = ["Billthemaker"] # Replace with your name or alias
license = "Apache-2.0" # Good practice for open-source license = "BSL-1" # Good practice for open-source
edition = "2024" edition = "2024"
[lib] [lib]
@@ -11,8 +11,8 @@ crate-type = ["cdylib", "rlib"]
[dependencies] [dependencies]
# 1. Python Bindings for FFI # 1. Python Bindings for FFI
pyo3 = { version = "0.27.1", features = ["extension-module"] } pyo3 = { version = "=0.27.0", features = ["extension-module", "chrono-tz", "chrono"] }
chrono-tz = "0.10"
# 2. Configuration Parsing (YAML) # 2. Configuration Parsing (YAML)
serde = { version = "1.0", features = ["derive"] } serde = { version = "1.0", features = ["derive"] }
@@ -44,6 +44,7 @@ byteorder = "1.5"
# 10. Timestamp Handling (NEW) # 10. Timestamp Handling (NEW)
chrono = "0.4" chrono = "0.4"
chrono-tz = "=0.9"
# 11. UUID Handling (NEW) # 11. UUID Handling (NEW)
uuid = { version = "1.8", features = ["serde", "v4"] } uuid = { version = "1.8", features = ["serde", "v4"] }

387
LICENSE
View File

@@ -1,70 +1,373 @@
Apache License Mozilla Public License Version 2.0
Version 2.0, January 2004 ==================================
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION 1. Definitions
--------------
1. Definitions. 1.1. "Contributor"
means each individual or legal entity that creates, contributes to
the creation of, or owns Covered Software.
"License" shall mean the terms and conditions for use, reproduction, and distribution as defined by Sections 1 through 9 of this document. 1.2. "Contributor Version"
means the combination of the Contributions of others (if any) used
by a Contributor and that particular Contributor's Contribution.
"Licensor" shall mean the copyright owner or entity authorized by the copyright owner that is granting the License. 1.3. "Contribution"
means Covered Software of a particular Contributor.
"Legal Entity" shall mean the union of the acting entity and all other entities that control, are controlled by, or are under common control with that entity. For the purposes of this definition, "control" means (i) the power, direct or indirect, to cause the direction or management of such entity, whether by contract or otherwise, or (ii) ownership of fifty percent (50%) or more of the outstanding shares, or (iii) beneficial ownership of such entity. 1.4. "Covered Software"
means Source Code Form to which the initial Contributor has attached
the notice in Exhibit A, the Executable Form of such Source Code
Form, and Modifications of such Source Code Form, in each case
including portions thereof.
"You" (or "Your") shall mean an individual or Legal Entity exercising permissions granted by this License. 1.5. "Incompatible With Secondary Licenses"
means
"Source" form shall mean the preferred form for making modifications, including but not limited to software source code, documentation source, and configuration files. (a) that the initial Contributor has attached the notice described
in Exhibit B to the Covered Software; or
"Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types. (b) that the Covered Software was made available under the terms of
version 1.1 or earlier of the License, but not also under the
terms of a Secondary License.
"Work" shall mean the work of authorship, whether in Source or Object form, made available under the License, as indicated by a copyright notice that is included in or attached to the work (an example is provided in the Appendix below). 1.6. "Executable Form"
means any form of the work other than Source Code Form.
"Derivative Works" shall mean any work, whether in Source or Object form, that is based on (or derived from) the Work and for which the editorial revisions, annotations, elaborations, or other modifications represent, as a whole, an original work of authorship. For the purposes of this License, Derivative Works shall not include works that remain separable from, or merely link (or bind by name) to the interfaces of, the Work and Derivative Works thereof. 1.7. "Larger Work"
means a work that combines Covered Software with other material, in
a separate file or files, that is not Covered Software.
"Contribution" shall mean any work of authorship, including the original version of the Work and any modifications or additions to that Work or Derivative Works thereof, that is intentionally submitted to Licensor for inclusion in the Work by the copyright owner or by an individual or Legal Entity authorized to submit on behalf of the copyright owner. For the purposes of this definition, "submitted" means any form of electronic, verbal, or written communication sent to the Licensor or its representatives, including but not limited to communication on electronic mailing lists, source code control systems, and issue tracking systems that are managed by, or on behalf of, the Licensor for the purpose of discussing and improving the Work, but excluding communication that is conspicuously marked or otherwise designated in writing by the copyright owner as "Not a Contribution." 1.8. "License"
means this document.
"Contributor" shall mean Licensor and any individual or Legal Entity on behalf of whom a Contribution has been received by Licensor and subsequently incorporated within the Work. 1.9. "Licensable"
means having the right to grant, to the maximum extent possible,
whether at the time of the initial grant or subsequently, any and
all of the rights conveyed by this License.
2. Grant of Copyright License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable copyright license to reproduce, prepare Derivative Works of, publicly display, publicly perform, sublicense, and distribute the Work and such Derivative Works in Source or Object form. 1.10. "Modifications"
means any of the following:
3. Grant of Patent License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable (except as stated in this section) patent license to make, have made, use, offer to sell, sell, import, and otherwise transfer the Work, where such license applies only to those patent claims licensable by such Contributor that are necessarily infringed by their Contribution(s) alone or by combination of their Contribution(s) with the Work to which such Contribution(s) was submitted. If You institute patent litigation against any entity (including a cross-claim or counterclaim in a lawsuit) alleging that the Work or a Contribution incorporated within the Work constitutes direct or contributory patent infringement, then any patent licenses granted to You under this License for that Work shall terminate as of the date such litigation is filed. (a) any file in Source Code Form that results from an addition to,
deletion from, or modification of the contents of Covered
Software; or
4. Redistribution. You may reproduce and distribute copies of the Work or Derivative Works thereof in any medium, with or without modifications, and in Source or Object form, provided that You meet the following conditions: (b) any new file in Source Code Form that contains any Covered
Software.
You must give any other recipients of the Work or Derivative Works a copy of this License; and 1.11. "Patent Claims" of a Contributor
You must cause any modified files to carry prominent notices stating that You changed the files; and means any patent claim(s), including without limitation, method,
You must retain, in the Source form of any Derivative Works that You distribute, all copyright, patent, trademark, and attribution notices from the Source form of the Work, excluding those notices that do not pertain to any part of the Derivative Works; and process, and apparatus claims, in any patent Licensable by such
If the Work includes a "NOTICE" text file as part of its distribution, then any Derivative Works that You distribute must include a readable copy of the attribution notices contained within such NOTICE file, excluding those notices that do not pertain to any part of the Derivative Works, in at least one of the following places: within a NOTICE text file distributed as part of the Derivative Works; within the Source form or documentation, if provided along with the Derivative Works; or, within a display generated by the Derivative Works, if and wherever such third-party notices normally appear. The contents of the NOTICE file are for informational purposes only and do not modify the License. You may add Your own attribution notices within Derivative Works that You distribute, alongside or as an addendum to the NOTICE text from the Work, provided that such additional attribution notices cannot be construed as modifying the License. Contributor that would be infringed, but for the grant of the
You may add Your own copyright statement to Your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of Your modifications, or for any such Derivative Works as a whole, provided Your use, reproduction, and distribution of the Work otherwise complies with the conditions stated in this License. License, by the making, using, selling, offering for sale, having
made, import, or transfer of either its Contributions or its
Contributor Version.
5. Submission of Contributions. Unless You explicitly state otherwise, any Contribution intentionally submitted for inclusion in the Work by You to the Licensor shall be under the terms and conditions of this License, without any additional terms or conditions. Notwithstanding the above, nothing herein shall supersede or modify the terms of any separate license agreement you may have executed with Licensor regarding such Contributions. 1.12. "Secondary License"
means either the GNU General Public License, Version 2.0, the GNU
Lesser General Public License, Version 2.1, the GNU Affero General
Public License, Version 3.0, or any later versions of those
licenses.
6. Trademarks. This License does not grant permission to use the trade names, trademarks, service marks, or product names of the Licensor, except as required for reasonable and customary use in describing the origin of the Work and reproducing the content of the NOTICE file. 1.13. "Source Code Form"
means the form of the work preferred for making modifications.
7. Disclaimer of Warranty. Unless required by applicable law or agreed to in writing, Licensor provides the Work (and each Contributor provides its Contributions) on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied, including, without limitation, any warranties or conditions of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A PARTICULAR PURPOSE. You are solely responsible for determining the appropriateness of using or redistributing the Work and assume any risks associated with Your exercise of permissions under this License. 1.14. "You" (or "Your")
means an individual or a legal entity exercising rights under this
License. For legal entities, "You" includes any entity that
controls, is controlled by, or is under common control with You. For
purposes of this definition, "control" means (a) the power, direct
or indirect, to cause the direction or management of such entity,
whether by contract or otherwise, or (b) ownership of more than
fifty percent (50%) of the outstanding shares or beneficial
ownership of such entity.
8. Limitation of Liability. In no event and under no legal theory, whether in tort (including negligence), contract, or otherwise, unless required by applicable law (such as deliberate and grossly negligent acts) or agreed to in writing, shall any Contributor be liable to You for damages, including any direct, indirect, special, incidental, or consequential damages of any character arising as a result of this License or out of the use or inability to use the Work (including but not limited to damages for loss of goodwill, work stoppage, computer failure or malfunction, or any and all other commercial damages or losses), even if such Contributor has been advised of the possibility of such damages. 2. License Grants and Conditions
--------------------------------
9. Accepting Warranty or Additional Liability. While redistributing the Work or Derivative Works thereof, You may choose to offer, and charge a fee for, acceptance of support, warranty, indemnity, or other liability obligations and/or rights consistent with this License. However, in accepting such obligations, You may act only on Your own behalf and on Your sole responsibility, not on behalf of any other Contributor, and only if You agree to indemnify, defend, and hold each Contributor harmless for any liability incurred by, or claims asserted against, such Contributor by reason of your accepting any such warranty or additional liability. 2.1. Grants
END OF TERMS AND CONDITIONS Each Contributor hereby grants You a world-wide, royalty-free,
non-exclusive license:
How to apply the Apache License to your work (a) under intellectual property rights (other than patent or trademark)
Include a copy of the Apache License, typically in a file called LICENSE, in your work, and consider also including a NOTICE file that references the License. Licensable by such Contributor to use, reproduce, make available,
modify, display, perform, distribute, and otherwise exploit its
Contributions, either on an unmodified basis, with Modifications, or
as part of a Larger Work; and
To apply the Apache License to specific files in your work, attach the following boilerplate declaration, replacing the fields enclosed by brackets "[]" with your own identifying information. (Don't include the brackets!) Enclose the text in the appropriate comment syntax for the file format. We also recommend that you include a file or class name and description of purpose on the same "printed page" as the copyright notice for easier identification within third-party archives. (b) under Patent Claims of such Contributor to make, use, sell, offer
for sale, have made, import, and otherwise transfer either its
Contributions or its Contributor Version.
Copyright [yyyy] [name of copyright owner] 2.2. Effective Date
Licensed under the Apache License, Version 2.0 (the "License"); The licenses granted in Section 2.1 with respect to any Contribution
you may not use this file except in compliance with the License. become effective for each Contribution on the date the Contributor first
You may obtain a copy of the License at distributes such Contribution.
http://www.apache.org/licenses/LICENSE-2.0 2.3. Limitations on Grant Scope
Unless required by applicable law or agreed to in writing, software The licenses granted in this Section 2 are the only rights granted under
distributed under the License is distributed on an "AS IS" BASIS, this License. No additional rights or licenses will be implied from the
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. distribution or licensing of Covered Software under this License.
See the License for the specific language governing permissions and Notwithstanding Section 2.1(b) above, no patent license is granted by a
limitations under the License. Contributor:
(a) for any code that a Contributor has removed from Covered Software;
or
(b) for infringements caused by: (i) Your and any other third party's
modifications of Covered Software, or (ii) the combination of its
Contributions with other software (except as part of its Contributor
Version); or
(c) under Patent Claims infringed by Covered Software in the absence of
its Contributions.
This License does not grant any rights in the trademarks, service marks,
or logos of any Contributor (except as may be necessary to comply with
the notice requirements in Section 3.4).
2.4. Subsequent Licenses
No Contributor makes additional grants as a result of Your choice to
distribute the Covered Software under a subsequent version of this
License (see Section 10.2) or under the terms of a Secondary License (if
permitted under the terms of Section 3.3).
2.5. Representation
Each Contributor represents that the Contributor believes its
Contributions are its original creation(s) or it has sufficient rights
to grant the rights to its Contributions conveyed by this License.
2.6. Fair Use
This License is not intended to limit any rights You have under
applicable copyright doctrines of fair use, fair dealing, or other
equivalents.
2.7. Conditions
Sections 3.1, 3.2, 3.3, and 3.4 are conditions of the licenses granted
in Section 2.1.
3. Responsibilities
-------------------
3.1. Distribution of Source Form
All distribution of Covered Software in Source Code Form, including any
Modifications that You create or to which You contribute, must be under
the terms of this License. You must inform recipients that the Source
Code Form of the Covered Software is governed by the terms of this
License, and how they can obtain a copy of this License. You may not
attempt to alter or restrict the recipients' rights in the Source Code
Form.
3.2. Distribution of Executable Form
If You distribute Covered Software in Executable Form then:
(a) such Covered Software must also be made available in Source Code
Form, as described in Section 3.1, and You must inform recipients of
the Executable Form how they can obtain a copy of such Source Code
Form by reasonable means in a timely manner, at a charge no more
than the cost of distribution to the recipient; and
(b) You may distribute such Executable Form under the terms of this
License, or sublicense it under different terms, provided that the
license for the Executable Form does not attempt to limit or alter
the recipients' rights in the Source Code Form under this License.
3.3. Distribution of a Larger Work
You may create and distribute a Larger Work under terms of Your choice,
provided that You also comply with the requirements of this License for
the Covered Software. If the Larger Work is a combination of Covered
Software with a work governed by one or more Secondary Licenses, and the
Covered Software is not Incompatible With Secondary Licenses, this
License permits You to additionally distribute such Covered Software
under the terms of such Secondary License(s), so that the recipient of
the Larger Work may, at their option, further distribute the Covered
Software under the terms of either this License or such Secondary
License(s).
3.4. Notices
You may not remove or alter the substance of any license notices
(including copyright notices, patent notices, disclaimers of warranty,
or limitations of liability) contained within the Source Code Form of
the Covered Software, except that You may alter any license notices to
the extent required to remedy known factual inaccuracies.
3.5. Application of Additional Terms
You may choose to offer, and to charge a fee for, warranty, support,
indemnity or liability obligations to one or more recipients of Covered
Software. However, You may do so only on Your own behalf, and not on
behalf of any Contributor. You must make it absolutely clear that any
such warranty, support, indemnity, or liability obligation is offered by
You alone, and You hereby agree to indemnify every Contributor for any
liability incurred by such Contributor as a result of warranty, support,
indemnity or liability terms You offer. You may include additional
disclaimers of warranty and limitations of liability specific to any
jurisdiction.
4. Inability to Comply Due to Statute or Regulation
---------------------------------------------------
If it is impossible for You to comply with any of the terms of this
License with respect to some or all of the Covered Software due to
statute, judicial order, or regulation then You must: (a) comply with
the terms of this License to the maximum extent possible; and (b)
describe the limitations and the code they affect. Such description must
be placed in a text file included with all distributions of the Covered
Software under this License. Except to the extent prohibited by statute
or regulation, such description must be sufficiently detailed for a
recipient of ordinary skill to be able to understand it.
5. Termination
--------------
5.1. The rights granted under this License will terminate automatically
if You fail to comply with any of its terms. However, if You become
compliant, then the rights granted under this License from a particular
Contributor are reinstated (a) provisionally, unless and until such
Contributor explicitly and finally terminates Your grants, and (b) on an
ongoing basis, if such Contributor fails to notify You of the
non-compliance by some reasonable means prior to 60 days after You have
come back into compliance. Moreover, Your grants from a particular
Contributor are reinstated on an ongoing basis if such Contributor
notifies You of the non-compliance by some reasonable means, this is the
first time You have received notice of non-compliance with this License
from such Contributor, and You become compliant prior to 30 days after
Your receipt of the notice.
5.2. If You initiate litigation against any entity by asserting a patent
infringement claim (excluding declaratory judgment actions,
counter-claims, and cross-claims) alleging that a Contributor Version
directly or indirectly infringes any patent, then the rights granted to
You by any and all Contributors for the Covered Software under Section
2.1 of this License shall terminate.
5.3. In the event of termination under Sections 5.1 or 5.2 above, all
end user license agreements (excluding distributors and resellers) which
have been validly granted by You or Your distributors under this License
prior to termination shall survive termination.
************************************************************************
* *
* 6. Disclaimer of Warranty *
* ------------------------- *
* *
* Covered Software is provided under this License on an "as is" *
* basis, without warranty of any kind, either expressed, implied, or *
* statutory, including, without limitation, warranties that the *
* Covered Software is free of defects, merchantable, fit for a *
* particular purpose or non-infringing. The entire risk as to the *
* quality and performance of the Covered Software is with You. *
* Should any Covered Software prove defective in any respect, You *
* (not any Contributor) assume the cost of any necessary servicing, *
* repair, or correction. This disclaimer of warranty constitutes an *
* essential part of this License. No use of any Covered Software is *
* authorized under this License except under this disclaimer. *
* *
************************************************************************
************************************************************************
* *
* 7. Limitation of Liability *
* -------------------------- *
* *
* Under no circumstances and under no legal theory, whether tort *
* (including negligence), contract, or otherwise, shall any *
* Contributor, or anyone who distributes Covered Software as *
* permitted above, be liable to You for any direct, indirect, *
* special, incidental, or consequential damages of any character *
* including, without limitation, damages for lost profits, loss of *
* goodwill, work stoppage, computer failure or malfunction, or any *
* and all other commercial damages or losses, even if such party *
* shall have been informed of the possibility of such damages. This *
* limitation of liability shall not apply to liability for death or *
* personal injury resulting from such party's negligence to the *
* extent applicable law prohibits such limitation. Some *
* jurisdictions do not allow the exclusion or limitation of *
* incidental or consequential damages, so this exclusion and *
* limitation may not apply to You. *
* *
************************************************************************
8. Litigation
-------------
Any litigation relating to this License may be brought only in the
courts of a jurisdiction where the defendant maintains its principal
place of business and such litigation shall be governed by laws of that
jurisdiction, without reference to its conflict-of-law provisions.
Nothing in this Section shall prevent a party's ability to bring
cross-claims or counter-claims.
9. Miscellaneous
----------------
This License represents the complete agreement concerning the subject
matter hereof. If any provision of this License is held to be
unenforceable, such provision shall be reformed only to the extent
necessary to make it enforceable. Any law or regulation which provides
that the language of a contract shall be construed against the drafter
shall not be used to construe this License against a Contributor.
10. Versions of the License
---------------------------
10.1. New Versions
Mozilla Foundation is the license steward. Except as provided in Section
10.3, no one other than the license steward has the right to modify or
publish new versions of this License. Each version will be given a
distinguishing version number.
10.2. Effect of New Versions
You may distribute the Covered Software under the terms of the version
of the License under which You originally received the Covered Software,
or under the terms of any subsequent version published by the license
steward.
10.3. Modified Versions
If you create software not governed by this License, and you want to
create a new license for such software, you may create and use a
modified version of this License if you rename the license and remove
any references to the name of the license steward (except to note that
such modified license differs from this License).
10.4. Distributing Source Code Form that is Incompatible With Secondary
Licenses
If You choose to distribute Source Code Form that is Incompatible With
Secondary Licenses under the terms of this version of the License, the
notice described in Exhibit B of this License must be attached.
Exhibit A - Source Code Form License Notice
-------------------------------------------
This Source Code Form is subject to the terms of the Mozilla Public
License, v. 2.0. If a copy of the MPL was not distributed with this
file, You can obtain one at https://mozilla.org/MPL/2.0/.
If it is not possible or desirable to put the notice in a particular
file, then You may include the notice in a location (such as a LICENSE
file in a relevant directory) where a recipient would be likely to look
for such a notice.
You may add additional accurate notices of copyright ownership.
Exhibit B - "Incompatible With Secondary Licenses" Notice
---------------------------------------------------------
This Source Code Form is "Incompatible With Secondary Licenses", as
defined by the Mozilla Public License, v. 2.0.

View File

@@ -4,34 +4,19 @@ import sqlalchemy
import connectorx as cx import connectorx as cx
import unchecked_io import unchecked_io
import os import os
import yaml import yaml # We need pyyaml for this
import time import time
# --- 1. Define Connection Strings and Query --- # --- 1. Define Connection Strings and Query ---
# These must match your local Docker setup
DB_USER = "postgres" DB_USER = "postgres"
DB_PASS = "mysecretpassword" DB_PASS = "mysecretpassword"
DB_HOST = "localhost" DB_HOST = "localhost"
DB_PORT = "5433" DB_PORT = "5433" # <-- Your local Docker port
DB_NAME = "postgres" DB_NAME = "postgres"
# Global Configuration # Global Configuration
BLAST_RADIUS = 312500 # Rows per parallel task (1M / 62500 = 16 partitions)
# Rows per parallel task (1M / 62500 = 16 partitions)
#BLAST_RADIUS = 1250000 # 16 partitions
#BLAST_RADIUS = 625000 # 32 partitions
#BLAST_RADIUS = 312500 # 64 partitions
#BLAST_RADIUS = 156250 # 128 partitions
#BLAST_RADIUS = 125000 # 160 partitions
#BLAST_RADIUS = 12500 # 1600 partitions
BLAST_RADIUS = 1250 # 16000 partitions
BATCH_SIZE = 262144 # Aggregation Size (Large Memory chunks)
# SQLAlchemy connection string (for Pandas) # SQLAlchemy connection string (for Pandas)
sqlalchemy_conn_str = f"postgresql://{DB_USER}:{DB_PASS}@{DB_HOST}:{DB_PORT}/{DB_NAME}" sqlalchemy_conn_str = f"postgresql://{DB_USER}:{DB_PASS}@{DB_HOST}:{DB_PORT}/{DB_NAME}"
@@ -50,7 +35,6 @@ unchecked_io_query = "COPY (SELECT id::BIGINT, uuid::TEXT, username::TEXT, score
config_data = { config_data = {
'connection_string': connectorx_conn_str, 'connection_string': connectorx_conn_str,
'query': unchecked_io_query, 'query': unchecked_io_query,
'batch_size': BATCH_SIZE, # <--- This explicitly writes your setting to the file
'schema': [ 'schema': [
{'column_name': 'id', 'arrow_type': 'Int64'}, {'column_name': 'id', 'arrow_type': 'Int64'},
{'column_name': 'uuid', 'arrow_type': 'Utf8'}, {'column_name': 'uuid', 'arrow_type': 'Utf8'},
@@ -69,8 +53,6 @@ with open(config_file, 'w') as f:
yaml.dump(config_data, f) yaml.dump(config_data, f)
print(f"UncheckedIO Config: {config_file} (dynamically created for local test)") print(f"UncheckedIO Config: {config_file} (dynamically created for local test)")
print(f" - Blast Radius: {BLAST_RADIUS}")
print(f" - Batch Size: {BATCH_SIZE}")
# --- 3. Define Benchmark Functions --- # --- 3. Define Benchmark Functions ---
@@ -91,6 +73,14 @@ def test_unchecked_io():
# --- 4. Run Benchmarks --- # --- 4. Run Benchmarks ---
run_count = 1 run_count = 1
print(f"Running benchmarks for 20,000,000 rows (average of {run_count} runs)...") print(f"Running benchmarks for 20,000,000 rows (average of {run_count} runs)...")
print(f"Blast Radius: {BLAST_RADIUS} rows per task")
# --- Pandas ---
# print("\nRunning Pandas warmup...")
# _ = test_pandas()
# print("Timing pandas.read_sql...")
# pandas_time = timeit.timeit(test_pandas, number=run_count) / run_count
# print(f"Pandas Average Time: {pandas_time * 1000:.2f} ms")
# --- ConnectorX --- # --- ConnectorX ---
print("\nRunning ConnectorX warmup...") print("\nRunning ConnectorX warmup...")

View File

@@ -1,4 +1,3 @@
batch_size: 524288
connection_string: postgresql://postgres:mysecretpassword@localhost:5433/postgres connection_string: postgresql://postgres:mysecretpassword@localhost:5433/postgres
query: COPY (SELECT id::BIGINT, uuid::TEXT, username::TEXT, score::REAL, is_active::BOOLEAN, query: COPY (SELECT id::BIGINT, uuid::TEXT, username::TEXT, score::REAL, is_active::BOOLEAN,
last_login::TIMESTAMP, notes::TEXT, course_id::INT, start_date::DATE, rating::FLOAT8 last_login::TIMESTAMP, notes::TEXT, course_id::INT, start_date::DATE, rating::FLOAT8

View File

@@ -1 +0,0 @@
I need to refactor src/parser.rs to implement a "Worker Pool" pattern that strictly limits active database queries to the number of CPU cores.The Goal:Currently, the code spawns a Tokio task for every partition immediately. This floods the scheduler and causes database thrashing (too many concurrent COPY commands).I want to spawn a fixed number of workers (e.g., 16) that consume tasks from a queue. This ensures that if I have 1,600 partitions, only 16 COPY commands are active on the database at any given time.Requirements:Dependencies: Use async-channel (assume it's in Cargo.toml).Concurrency Limit:let num_workers = num_cpus::get();Set deadpool_postgres max_size to num_workers.The Queue:Create a channel: let (tx, rx) = async_channel::bounded(num_workers * 2); (Backpressure is good).Spawn a separate "Distributor" task that iterates partitions and sends them into tx.The Workers:Spawn exactly num_workers tasks.Each worker loops: while let Ok(task) = rx.recv().await.Inside the loop:Get a connection from the pool.Run copy_out(task.query).Parse the result.Store the resulting RecordBatch.Output:Collect all RecordBatch results from all workers.Flatten and concatenate them using concat_batches.Why: This ensures Postgres never sees a queue of queries. It only sees active workers. The "queue" lives entirely in Rust memory.

View File

@@ -1,8 +1,10 @@
// src/config.rs // --- External Crates ---
use serde::Deserialize; use serde::Deserialize;
use anyhow::{Context, Result, anyhow}; // Added anyhow macro just in case use anyhow::{Context, Result}; // Removed unused 'anyhow' macro import if not used, but Context/Result likely used
use std::fmt::{self, Display}; use std::fmt::{self, Display};
// --- 1. CUSTOM ERROR DEFINITION ---
#[derive(Debug)] #[derive(Debug)]
pub struct ConfigError(String); pub struct ConfigError(String);
@@ -13,6 +15,10 @@ impl Display for ConfigError {
} }
impl std::error::Error for ConfigError {} impl std::error::Error for ConfigError {}
// --- 2. CONFIGURATION STRUCTS (The "Configured Opinion") ---
// We make these 'pub' (public) so src/lib.rs can use them.
#[derive(Debug, Deserialize, Clone)] #[derive(Debug, Deserialize, Clone)]
pub struct ColumnConfig { pub struct ColumnConfig {
pub column_name: String, pub column_name: String,
@@ -24,10 +30,10 @@ pub struct ConnectorConfig {
pub connection_string: String, pub connection_string: String,
pub query: String, pub query: String,
pub schema: Vec<ColumnConfig>, pub schema: Vec<ColumnConfig>,
// NEW: Optional Batch Size configuration
pub batch_size: Option<usize>,
} }
// --- 3. CONFIG LOADING FUNCTION ---
// This is also 'pub' so src/lib.rs can call it.
pub fn load_and_validate_config(path: &str) -> Result<ConnectorConfig> { pub fn load_and_validate_config(path: &str) -> Result<ConnectorConfig> {
let file_content = std::fs::read_to_string(path) let file_content = std::fs::read_to_string(path)
.context(format!("Failed to read config file at path: {}", path))?; .context(format!("Failed to read config file at path: {}", path))?;
@@ -41,6 +47,7 @@ pub fn load_and_validate_config(path: &str) -> Result<ConnectorConfig> {
if config.query.is_empty() { if config.query.is_empty() {
return Err(ConfigError("Query cannot be empty.".to_string())).map_err(anyhow::Error::from)?; return Err(ConfigError("Query cannot be empty.".to_string())).map_err(anyhow::Error::from)?;
} }
// FIX: Make the COPY check more flexible (allows newlines)
let uppercase_query = config.query.trim().to_uppercase(); let uppercase_query = config.query.trim().to_uppercase();
if !uppercase_query.starts_with("COPY") || !uppercase_query.contains("TO STDOUT") || !uppercase_query.contains("FORMAT BINARY") { if !uppercase_query.starts_with("COPY") || !uppercase_query.contains("TO STDOUT") || !uppercase_query.contains("FORMAT BINARY") {
return Err(ConfigError("Query must be a 'COPY ... TO STDOUT (FORMAT binary)' command.".to_string())).map_err(anyhow::Error::from)?; return Err(ConfigError("Query must be a 'COPY ... TO STDOUT (FORMAT binary)' command.".to_string())).map_err(anyhow::Error::from)?;

View File

@@ -1,9 +1,8 @@
// src/parser.rs
// --- External Crates --- // --- External Crates ---
use std::pin::Pin; use std::pin::Pin;
use std::sync::Arc; use std::sync::Arc;
use tokio_postgres::{NoTls, CopyOutStream, Config as PgConfig}; use tokio_postgres::{NoTls, CopyOutStream, Config as PgConfig};
// Required for the connection pool
use deadpool_postgres::{Pool, Manager, Runtime}; use deadpool_postgres::{Pool, Manager, Runtime};
use anyhow::{Context, Result, anyhow}; use anyhow::{Context, Result, anyhow};
use arrow::array::{ use arrow::array::{
@@ -21,117 +20,143 @@ use byteorder::{BigEndian, ReadBytesExt};
use std::io::{Cursor, Read}; use std::io::{Cursor, Read};
use std::str; use std::str;
use std::str::FromStr; use std::str::FromStr;
use chrono::{NaiveDateTime, NaiveDate};
use std::time::Instant; use std::time::Instant;
use tokio::task::JoinSet; use tokio::task::JoinSet;
use arrow::compute::concat_batches; use arrow::compute::concat_batches;
// NEW: For the Work Stealing Queue
use async_channel; use async_channel;
// NEW: Tracing macros for profiling - Only import if feature is enabled
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
use tracing::{span, Level}; use tracing::{span, Level};
// --- Internal Crates ---
use crate::config::{ConnectorConfig, load_and_validate_config}; use crate::config::{ConnectorConfig, load_and_validate_config};
// --- CONSTANTS --- // --- CONSTANTS ---
// Optimized calculation of epoch delta (2000-01-01 00:00:00 to 1970-01-01 00:00:00)
// 10957 days * 86400 seconds/day * 1,000,000 micros/second = 946684800000000 micros
const POSTGRES_EPOCH_MICROS_OFFSET: i64 = 946684800000000; const POSTGRES_EPOCH_MICROS_OFFSET: i64 = 946684800000000;
// Default batch size (64k is a standard Arrow chunk size)
const DEFAULT_BATCH_SIZE: usize = 65_536;
// --- 1. CORE DATABASE LOGIC --- // --- 1. CORE DATABASE LOGIC (WORKER POOL PATTERN) ---
// This is the fully optimized function using Connection Pooling and Static Dispatch.
pub async fn run_db_logic(config: ConnectorConfig, blast_radius: i64) -> Result<RecordBatch> { pub async fn run_db_logic(config: ConnectorConfig, blast_radius: i64) -> Result<RecordBatch> {
// Start overall timer
let start_total = Instant::now(); let start_total = Instant::now();
let start_phase1 = Instant::now(); let start_phase1 = Instant::now();
// NEW: High-level span for the whole operation
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let root_span = span!(Level::INFO, "UncheckedIO_Run"); let root_span = span!(Level::INFO, "UncheckedIO_Run");
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let _root_guard = root_span.enter(); let _root_guard = root_span.enter();
// --- PHASE 1: SETUP --- // --- PHASE 1: SETUP CONNECTION POOL & STATS ---
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let phase1_span = span!(Level::INFO, "Phase1_Setup"); let phase1_span = span!(Level::INFO, "Phase1_Setup");
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let _p1_guard = phase1_span.enter(); let _p1_guard = phase1_span.enter();
// 1. Calculate Worker Count (Fixed Parallelism)
let num_workers = num_cpus::get(); let num_workers = num_cpus::get();
println!("UncheckedIO: Detected {} logical cores. Spawning {} persistent worker threads.", num_workers, num_workers); println!("UncheckedIO: Detected {} logical cores. Spawning {} worker threads.", num_workers, num_workers);
// CONFIG: Determine Target Batch Size
let target_batch_size = config.batch_size.unwrap_or(DEFAULT_BATCH_SIZE);
println!("UncheckedIO: Worker Aggregation Target = {} rows/batch.", target_batch_size);
// 2. Setup Connection Pool
let pg_config: tokio_postgres::Config = PgConfig::from_str(&config.connection_string) let pg_config: tokio_postgres::Config = PgConfig::from_str(&config.connection_string)
.context("Invalid connection string")?; .context("Invalid connection string in config")?;
let manager = Manager::new(pg_config.clone(), NoTls); let manager = Manager::new(pg_config.clone(), NoTls);
// FIX: Set pool size exactly to num_workers to prevent starvation or waiting
let pool = Pool::builder(manager) let pool = Pool::builder(manager)
.max_size(num_workers) .max_size(num_workers)
.runtime(Runtime::Tokio1) .runtime(Runtime::Tokio1)
.build() .build()
.context("Failed to build connection pool")?; .context("Failed to build connection pool")?;
let client = pool.get().await.context("Failed to get pool connection")?; // 3. Query Table Bounds
// We grab a temporary connection just for this setup phase
let client = pool.get().await.context("Failed to get pool connection for stats query")?;
let partition_key = "id"; let partition_key = "id";
let (base_query, _) = config.query.trim().split_once("TO STDOUT (FORMAT binary)") let (base_query, _) = config.query.trim().split_once("TO STDOUT (FORMAT binary)")
.context("Failed to parse base query")?; .context("Failed to parse base query from config")?;
let base_query_inner = base_query.trim().trim_start_matches("COPY (").trim_end_matches(")"); let base_query_inner = base_query.trim().trim_start_matches("COPY (").trim_end_matches(")");
let stats_query = format!("SELECT MIN({}), MAX({}) FROM ({}) AS subquery", partition_key, partition_key, base_query_inner); let stats_query = format!("SELECT MIN({}), MAX({}) FROM ({}) AS subquery", partition_key, partition_key, base_query_inner);
let row = client.query_one(&stats_query, &[]).await?; let row = client.query_one(&stats_query, &[]).await?;
let min_id: i64 = row.try_get(0).context("Failed to get MIN(id)")?; let min_id: i64 = row.try_get(0).context("Failed to get MIN(id)")?;
let max_id: i64 = row.try_get(1).context("Failed to get MAX(id)")?; let max_id: i64 = row.try_get(1).context("Failed to get MAX(id)")?;
drop(client); drop(client); // Return connection to pool immediately
println!("UncheckedIO: ID Range: {} to {}", min_id, max_id);
// --- NEW: DYNAMIC PARTITION SIZING ---
let total_rows = (max_id - min_id + 1).max(1); let total_rows = (max_id - min_id + 1).max(1);
let calculated_blast_radius = if blast_radius <= 0 { let calculated_blast_radius = if blast_radius <= 0 {
// Auto-tuning: Aim for ~4 chunks per worker to balance load
let target_chunks = (num_workers * 4) as i64; let target_chunks = (num_workers * 4) as i64;
let dynamic_size = total_rows / target_chunks; let dynamic_size = total_rows / target_chunks;
dynamic_size.max(10_000) // Ensure a sane minimum (e.g., don't make chunks of 1 row)
let size = dynamic_size.max(10_000);
println!("UncheckedIO: Auto-tuned partition size to {} rows (Targeting {} chunks).", size, target_chunks);
size
} else { } else {
println!("UncheckedIO: Using user-defined partition size: {} rows.", blast_radius);
blast_radius blast_radius
}; };
println!("UncheckedIO: Auto-tuned partition size to {} rows.", calculated_blast_radius);
// 4. Create Work Queue
// We use a tuple: (index, query, expected_rows) so we can re-sort later
struct PartitionTask { struct PartitionTask {
index: usize, index: usize,
query: String, query: String,
expected_rows: usize expected_rows: usize
} }
// Create an unbounded channel.
// tx = transmitter (main thread), rx = receiver (workers)
let (tx, rx) = async_channel::unbounded::<PartitionTask>(); let (tx, rx) = async_channel::unbounded::<PartitionTask>();
// 5. Populate the Queue (The "Blast Radius" Logic)
let mut current_min = min_id; let mut current_min = min_id;
let mut idx = 0; let mut idx = 0;
let mut total_partitions = 0; let mut total_partitions = 0;
while current_min <= max_id { while current_min <= max_id {
let current_max = (current_min + calculated_blast_radius - 1).min(max_id); let current_max = (current_min + calculated_blast_radius - 1).min(max_id);
let new_query = format!( let new_query = format!(
"COPY (SELECT * FROM ({}) AS sub WHERE {} BETWEEN {} AND {}) TO STDOUT (FORMAT binary)", "COPY (SELECT * FROM ({}) AS sub WHERE {} BETWEEN {} AND {}) TO STDOUT (FORMAT binary)",
base_query_inner, partition_key, current_min, current_max base_query_inner, partition_key, current_min, current_max
); );
let estimated_rows = (current_max - current_min + 1) as usize; let estimated_rows = (current_max - current_min + 1) as usize;
// FIX: Removed 'partitions.push(...)' which caused the error.
// We send directly to the channel now.
let task = PartitionTask { index: idx, query: new_query, expected_rows: estimated_rows }; let task = PartitionTask { index: idx, query: new_query, expected_rows: estimated_rows };
// Send to queue (non-blocking since it's unbounded)
tx.send(task).await.context("Failed to fill work queue")?; tx.send(task).await.context("Failed to fill work queue")?;
current_min += calculated_blast_radius; current_min += calculated_blast_radius;
idx += 1; idx += 1;
total_partitions += 1; total_partitions += 1;
} }
// Close the channel so workers know when to stop
tx.close(); tx.close();
#[cfg(feature = "profiling")] println!("UncheckedIO: Queued {} partitions for processing.", total_partitions);
drop(_p1_guard);
// FIX: Define the duration variable here so it can be used in the print statement later #[cfg(feature = "profiling")]
drop(_p1_guard); // End Phase 1 Span
let duration_phase1 = start_phase1.elapsed(); let duration_phase1 = start_phase1.elapsed();
// --- PHASE 2: PARALLEL EXECUTION (WITH AGGREGATION) --- // --- PHASE 2: PARALLEL EXECUTION (DATA TRANSFER + PARSING) ---
let start_phase2 = Instant::now(); let start_phase2 = Instant::now();
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let phase2_span = span!(Level::INFO, "Phase2_Execution"); let phase2_span = span!(Level::INFO, "Phase2_Execution");
@@ -141,12 +166,14 @@ pub async fn run_db_logic(config: ConnectorConfig, blast_radius: i64) -> Result<
let arrow_schema = Arc::new(build_arrow_schema(&config)?); let arrow_schema = Arc::new(build_arrow_schema(&config)?);
let mut join_set = JoinSet::new(); let mut join_set = JoinSet::new();
// Spawn exactly 'num_workers' long-lived tasks
for worker_id in 0..num_workers { for worker_id in 0..num_workers {
let worker_rx = rx.clone(); let worker_rx = rx.clone();
let worker_pool = pool.clone(); let worker_pool = pool.clone();
let worker_schema = arrow_schema.clone(); let worker_schema = arrow_schema.clone();
join_set.spawn(async move { join_set.spawn(async move {
// VISUALIZATION: Create a "track" for this worker in Tracy
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let worker_span = span!(Level::INFO, "Worker_Thread", id = worker_id); let worker_span = span!(Level::INFO, "Worker_Thread", id = worker_id);
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
@@ -154,114 +181,114 @@ pub async fn run_db_logic(config: ConnectorConfig, blast_radius: i64) -> Result<
let mut worker_batches: Vec<(usize, RecordBatch)> = Vec::new(); let mut worker_batches: Vec<(usize, RecordBatch)> = Vec::new();
// Initialize the Parser ONCE per worker (The Buffer) // Worker Loop: Keep grabbing tasks until the queue is empty and closed
let mut parser = create_parser();
let mut parser_row_count = 0;
let mut client = worker_pool.get().await
.context(format!("Worker {} failed to acquire connection", worker_id))?;
// Worker Loop
while let Ok(task) = worker_rx.recv().await { while let Ok(task) = worker_rx.recv().await {
// VISUALIZATION: Show exactly which partition is being processed
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let task_span = span!(Level::INFO, "Processing_Task", partition_id = task.index); let task_span = span!(Level::INFO, "Processing_Task", partition_id = task.index);
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let _t_guard = task_span.enter(); let _t_guard = task_span.enter();
// Process the task
// We wrap this in an inner block to easily catch errors for Self-Healing
let result = async { let result = async {
let client = worker_pool.get().await.context("Pool exhausted")?;
let copy_stream = client.copy_out(task.query.as_str()).await?; let copy_stream = client.copy_out(task.query.as_str()).await?;
let pinned_stream: Pin<Box<CopyOutStream>> = Box::pin(copy_stream); let pinned_stream: Pin<Box<CopyOutStream>> = Box::pin(copy_stream);
// Parse DIRECTLY into the persistent parser // Call Static Dispatch Parser
parse_binary_stream_static(pinned_stream, &mut parser).await parse_data_with_schema(pinned_stream, worker_schema.clone()).await
}.await; }.await;
match result { match result {
Ok(rows_read) => { Ok((_rows, batch)) => {
parser_row_count += rows_read; worker_batches.push((task.index, batch));
// CHECK FLUSH: Did we hit the batch size?
if parser_row_count >= target_batch_size {
let batch = flush_parser(&mut parser, worker_schema.clone())?;
worker_batches.push((task.index, batch));
parser_row_count = 0;
// Re-init parser builders
parser = create_parser();
}
} }
Err(e) => { Err(e) => {
eprintln!("Worker {} partition {} failed: {}. Handling failure.", worker_id, task.index, e); // ERROR LOGGING
#[cfg(feature = "profiling")]
tracing::error!("Worker {}: Partition {} failed! Error: {}", worker_id, task.index, e);
// SAFETY FLUSH: If we have pending data, flush it first! // --- SELF-HEALING LOGIC ---
if parser_row_count > 0 { // If a partition fails (e.g. bad data), we log it and return NULLs
let batch = flush_parser(&mut parser, worker_schema.clone())?; eprintln!("UncheckedIO Worker {}: Partition {} failed! Error: {}. Filling NULLs.", worker_id, task.index, e);
worker_batches.push((task.index, batch));
parser_row_count = 0;
parser = create_parser();
}
// Emit NULL batch for the failed partition
let null_batch = create_null_batch(worker_schema.clone(), task.expected_rows)?; let null_batch = create_null_batch(worker_schema.clone(), task.expected_rows)?;
worker_batches.push((task.index, null_batch)); worker_batches.push((task.index, null_batch));
} }
} }
} }
// FINAL FLUSH: Handle any remaining rows after queue is empty // Return all batches processed by this worker
if parser_row_count > 0 {
let batch = flush_parser(&mut parser, worker_schema.clone())?;
worker_batches.push((usize::MAX, batch));
}
Ok::<Vec<(usize, RecordBatch)>, anyhow::Error>(worker_batches) Ok::<Vec<(usize, RecordBatch)>, anyhow::Error>(worker_batches)
}); });
} }
// --- PHASE 3: AGGREGATION --- // --- PHASE 3: AGGREGATION ---
let mut all_results = Vec::new(); let mut all_results: Vec<(usize, RecordBatch)> = Vec::with_capacity(total_partitions);
while let Some(res) = join_set.join_next().await {
match res { while let Some(join_result) = join_set.join_next().await {
Ok(Ok(batches)) => all_results.extend(batches), match join_result {
Ok(Err(e)) => return Err(anyhow!("Worker failed: {}", e)), Ok(worker_result) => {
Err(e) => return Err(anyhow!("Worker panic: {}", e)), match worker_result {
Ok(batches) => all_results.extend(batches),
Err(e) => return Err(anyhow!("Worker task failed internally: {}", e)),
}
}
Err(e) => return Err(anyhow!("Worker task panic: {}", e)),
} }
} }
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
drop(_p2_guard); drop(_p2_guard); // End Phase 2 Span
let duration_phase2 = start_phase2.elapsed();
// --- PHASE 4: CONCAT ---
// --- PHASE 3: CONCATENATION AND FINALIZATION ---
let start_phase3 = Instant::now(); let start_phase3 = Instant::now();
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let phase3_span = span!(Level::INFO, "Phase3_Concat"); let phase3_span = span!(Level::INFO, "Phase3_Concat");
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let _p3_guard = phase3_span.enter(); let _p3_guard = phase3_span.enter();
if all_results.is_empty() { return Ok(RecordBatch::new_empty(arrow_schema)); } if all_results.is_empty() {
println!("UncheckedIO: All workers returned empty batches.");
let duration_total = start_total.elapsed();
println!("--- UncheckedIO Internal Timing ---");
println!("Phase 1 (Setup, Query): {:.2?}", duration_phase1);
println!("Phase 2 (I/O, Parsing): {:.2?}", duration_phase2);
println!("Phase 3 (Concatenation): {:.2?}", start_phase3.elapsed());
println!("Total Wall Time: {:.2?}", duration_total);
return Ok(RecordBatch::new_empty(arrow_schema));
}
// Sort by task index to maintain relative order // 1. Sort by index to restore original table order
all_results.sort_by_key(|(index, _)| *index); all_results.sort_by_key(|(index, _)| *index);
let batches: Vec<RecordBatch> = all_results.into_iter().map(|(_, b)| b).collect();
let final_batch = concat_batches(&arrow_schema, &batches)?; // 2. Strip indices
let batches: Vec<RecordBatch> = all_results.into_iter().map(|(_, batch)| batch).collect();
// 3. Final Concatenation
let final_batch = concat_batches(&arrow_schema, &batches)
.context("Failed to stitch final batches")?;
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
drop(_p3_guard); drop(_p3_guard); // End Phase 3 Span
let duration_phase3 = start_phase3.elapsed();
let duration_total = start_total.elapsed(); let duration_total = start_total.elapsed();
// --- REPORT ---
// --- FINAL REPORTING ---
println!("--- UncheckedIO Internal Timing ---"); println!("--- UncheckedIO Internal Timing ---");
// FIX: Using duration_phase1 correctly now println!("Phase 1 (Setup, Query): {:.2?}", duration_phase1);
println!("Phase 1 (Setup): {:.2?}", duration_phase1); println!("Phase 2 (I/O, Parsing): {:.2?}", duration_phase2);
println!("Phase 2 (Execute): {:.2?}", start_phase3.duration_since(start_phase2)); println!("Phase 3 (Concatenation): {:.2?}", duration_phase3);
println!("Phase 3 (Concat): {:.2?}", start_phase3.elapsed()); println!("Total Wall Time: {:.2?}", duration_total);
println!("Total Wall Time: {:.2?}", duration_total);
Ok(final_batch) Ok(final_batch)
} }
// --- HELPER FUNCTIONS ---
fn create_null_batch(schema: Arc<Schema>, num_rows: usize) -> Result<RecordBatch> { fn create_null_batch(schema: Arc<Schema>, num_rows: usize) -> Result<RecordBatch> {
let columns: Vec<ArrayRef> = schema.fields().iter().map(|field| { let columns: Vec<ArrayRef> = schema.fields().iter().map(|field| {
arrow::array::new_null_array(field.data_type(), num_rows) arrow::array::new_null_array(field.data_type(), num_rows)
@@ -269,8 +296,12 @@ fn create_null_batch(schema: Arc<Schema>, num_rows: usize) -> Result<RecordBatch
RecordBatch::try_new(schema, columns).context("Failed to create null placeholder batch") RecordBatch::try_new(schema, columns).context("Failed to create null placeholder batch")
} }
/// Helper function to build the Arrow Schema from the config
fn build_arrow_schema(config: &ConnectorConfig) -> Result<Schema> { fn build_arrow_schema(config: &ConnectorConfig) -> Result<Schema> {
let schema_fields: Vec<Field> = config.schema.iter().map(|col_cfg| { let schema_fields: Vec<Field> = config.schema.iter().map(|col_cfg| {
// FIX: Force all columns to be nullable for safety
let nullable = true;
let arrow_type = match col_cfg.arrow_type.as_str() { let arrow_type = match col_cfg.arrow_type.as_str() {
"Int64" => DataType::Int64, "Int64" => DataType::Int64,
"Int32" => DataType::Int32, "Int32" => DataType::Int32,
@@ -280,15 +311,20 @@ fn build_arrow_schema(config: &ConnectorConfig) -> Result<Schema> {
"Boolean" => DataType::Boolean, "Boolean" => DataType::Boolean,
"Timestamp(Nanosecond, None)" => DataType::Timestamp(arrow::datatypes::TimeUnit::Nanosecond, None), "Timestamp(Nanosecond, None)" => DataType::Timestamp(arrow::datatypes::TimeUnit::Nanosecond, None),
"Date32" => DataType::Date32, "Date32" => DataType::Date32,
_ => return Err(anyhow!("Unsupported type: {}", col_cfg.arrow_type)), _ => return Err(anyhow!("Unsupported type in config: {}", col_cfg.arrow_type)),
}; };
Ok(Field::new(&col_cfg.column_name, arrow_type, true)) Ok(Field::new(&col_cfg.column_name, arrow_type, nullable))
}).collect::<Result<Vec<Field>>>()?; }).collect::<Result<Vec<Field>>>()?;
Ok(Schema::new(schema_fields)) Ok(Schema::new(schema_fields))
} }
// --- PARSER STRUCT ---
// --------------------------------------------------------------------------------
// --- 2. STATIC DISPATCH IMPLEMENTATION (The Fast Parser) ---
// --------------------------------------------------------------------------------
// Struct to hold the builders in a statically-known, fixed order (eliminates DynamicBuilder enum)
struct SchemaParser { struct SchemaParser {
id: Box<Int64Builder>, id: Box<Int64Builder>,
uuid: Box<StringBuilder>, uuid: Box<StringBuilder>,
@@ -302,8 +338,13 @@ struct SchemaParser {
rating: Box<Float64Builder>, rating: Box<Float64Builder>,
} }
fn create_parser() -> SchemaParser { // Helper to construct and parse data using the static SchemaParser
SchemaParser { async fn parse_data_with_schema(
stream: Pin<Box<CopyOutStream>>,
arrow_schema: Arc<Schema>
) -> Result<(usize, RecordBatch)> {
let mut parser = SchemaParser {
id: Box::new(Int64Builder::new()), id: Box::new(Int64Builder::new()),
uuid: Box::new(StringBuilder::new()), uuid: Box::new(StringBuilder::new()),
username: Box::new(StringBuilder::new()), username: Box::new(StringBuilder::new()),
@@ -314,11 +355,11 @@ fn create_parser() -> SchemaParser {
course_id: Box::new(Int32Builder::new()), course_id: Box::new(Int32Builder::new()),
start_date: Box::new(Date32Builder::new()), start_date: Box::new(Date32Builder::new()),
rating: Box::new(Float64Builder::new()), rating: Box::new(Float64Builder::new()),
} };
}
// Helper to flush the parser into a RecordBatch let rows_processed = parse_binary_stream_static(stream, &mut parser).await?;
fn flush_parser(parser: &mut SchemaParser, schema: Arc<Schema>) -> Result<RecordBatch> {
// Collect all final arrays in the correct order (must match struct field order)
let final_columns: Vec<ArrayRef> = vec![ let final_columns: Vec<ArrayRef> = vec![
Arc::new(parser.id.finish()), Arc::new(parser.id.finish()),
Arc::new(parser.uuid.finish()), Arc::new(parser.uuid.finish()),
@@ -331,11 +372,16 @@ fn flush_parser(parser: &mut SchemaParser, schema: Arc<Schema>) -> Result<Record
Arc::new(parser.start_date.finish()), Arc::new(parser.start_date.finish()),
Arc::new(parser.rating.finish()), Arc::new(parser.rating.finish()),
]; ];
RecordBatch::try_new(schema, final_columns).context("Failed to build RecordBatch")
let record_batch = RecordBatch::try_new(
arrow_schema,
final_columns,
).context("Failed to create final Arrow RecordBatch")?;
Ok((rows_processed, record_batch))
} }
// --- STREAMING PARSER --- // The core streaming parser logic - INSTRUMENTED
async fn parse_binary_stream_static( async fn parse_binary_stream_static(
mut stream: Pin<Box<CopyOutStream>>, mut stream: Pin<Box<CopyOutStream>>,
parser: &mut SchemaParser, parser: &mut SchemaParser,
@@ -345,31 +391,35 @@ async fn parse_binary_stream_static(
let mut is_header_parsed: bool = false; let mut is_header_parsed: bool = false;
let mut rows_processed: usize = 0; let mut rows_processed: usize = 0;
// NEW: Refactored loop to visualize Starvation vs Work
'stream_loop: loop { 'stream_loop: loop {
// 1. MEASURE STARVATION (Waiting for Network)
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let wait_span = span!(Level::ERROR, "IO_WAIT"); let wait_span = span!(Level::ERROR, "IO_WAIT_STARVATION");
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let guard = wait_span.enter(); let guard = wait_span.enter();
let next_item = stream.next().await; let next_item = stream.next().await;
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
drop(guard); drop(guard); // Important: Drop guard immediately when data arrives!
match next_item { match next_item {
Some(segment_result) => { Some(segment_result) => {
// 2. MEASURE WORK (CPU Parsing)
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let work_span = span!(Level::INFO, "CPU_Parse"); let work_span = span!(Level::INFO, "CPU_Parse_Chunk");
#[cfg(feature = "profiling")] #[cfg(feature = "profiling")]
let _work_guard = work_span.enter(); let _work_guard = work_span.enter();
let segment: Bytes = segment_result.context("Error reading segment")?; let segment: Bytes = segment_result.context("Error reading segment from CopyOutStream")?;
buffer.extend_from_slice(&segment); buffer.extend_from_slice(&segment);
if !is_header_parsed { if !is_header_parsed {
if buffer.len() < 19 { continue 'stream_loop; } if buffer.len() < 19 { continue 'stream_loop; }
let mut header_cursor = Cursor::new(&buffer[..]); let mut header_cursor = Cursor::new(&buffer[..]);
parse_stream_header(&mut header_cursor)?; parse_stream_header(&mut header_cursor).context("Failed to parse stream header")?;
buffer.advance(19); buffer.advance(19);
is_header_parsed = true; is_header_parsed = true;
} }
@@ -397,109 +447,196 @@ async fn parse_binary_stream_static(
} }
Err(e) if e.kind() == std::io::ErrorKind::UnexpectedEof => { Err(e) if e.kind() == std::io::ErrorKind::UnexpectedEof => {
cursor.set_position(safe_position); cursor.set_position(safe_position);
// Copy remaining bytes back to the buffer for the next chunk
let remaining_slice = &buffer.as_ref()[safe_position as usize..]; let remaining_slice = &buffer.as_ref()[safe_position as usize..];
let mut leftover_vec = Vec::new(); let mut leftover_buffer_vec = Vec::new();
leftover_vec.extend_from_slice(remaining_slice); leftover_buffer_vec.extend_from_slice(remaining_slice);
buffer.clear(); buffer.clear();
buffer.extend_from_slice(&leftover_vec); buffer.extend_from_slice(&leftover_buffer_vec);
break 'parsing_loop; break 'parsing_loop;
} }
Err(e) => return Err(e.into()), Err(e) => {
return Err(e.into());
}
} }
} }
} }
None => break 'stream_loop, None => break 'stream_loop, // End of stream
} }
} }
if !buffer.is_empty() { if !buffer.is_empty() {
return Err(anyhow!("Stream ended with leftover bytes")); return Err(anyhow!("Stream ended with leftover bytes ({}) but no trailer.", buffer.len()));
} }
Ok(rows_processed) Ok(rows_processed)
} }
fn parse_stream_header(cursor: &mut Cursor<&[u8]>) -> Result<()> { fn parse_stream_header(cursor: &mut Cursor<&[u8]>) -> Result<()> {
let mut magic = [0u8; 11]; let mut magic_signature = [0u8; 11];
cursor.read_exact(&mut magic)?; cursor.read_exact(&mut magic_signature).context("Failed to read magic signature")?;
if &magic != b"PGCOPY\n\xff\r\n\0" { return Err(anyhow!("Invalid signature")); } if &magic_signature != b"PGCOPY\n\xff\r\n\0" {
let _ = cursor.read_u32::<BigEndian>()?; return Err(anyhow!("Invalid Postgres COPY binary signature."));
let _ = cursor.read_u32::<BigEndian>()?; }
let _flags = cursor.read_u32::<BigEndian>().context("Failed to read flags")?;
let _header_ext_len = cursor.read_u32::<BigEndian>().context("Failed to read header extension length")?;
Ok(()) Ok(())
} }
// --- STATIC DISPATCH ROW PARSER (The Key Speedup) ---
#[inline(always)] #[inline(always)]
fn parse_row_static( fn parse_row_static(
cursor: &mut Cursor<&[u8]>, cursor: &mut Cursor<&[u8]>,
p: &mut SchemaParser, p: &mut SchemaParser, // The concrete, statically-typed parser struct
current_chunk: &[u8] current_chunk: &[u8]
) -> Result<(), std::io::Error> { ) -> Result<(), std::io::Error> {
// Column 0: id // Column 0: id (BIGINT)
let len = cursor.read_i32::<BigEndian>()?; let len = cursor.read_i32::<BigEndian>()?;
if len == -1 { p.id.append_null() } else { p.id.append_value(cursor.read_i64::<BigEndian>()?) } if len == -1 { p.id.append_null() } else { p.id.append_value(cursor.read_i64::<BigEndian>()?) }
// Column 1: uuid // Column 1: uuid (TEXT)
let len = cursor.read_i32::<BigEndian>()?; let len = cursor.read_i32::<BigEndian>()?;
if len == -1 { p.uuid.append_null() } else { read_string_field(cursor, p.uuid.as_mut(), current_chunk, len as usize)? } if len == -1 { p.uuid.append_null() } else { read_string_field(cursor, p.uuid.as_mut(), current_chunk, len as usize)? }
// Column 2: username // Column 2: username (TEXT)
let len = cursor.read_i32::<BigEndian>()?; let len = cursor.read_i32::<BigEndian>()?;
if len == -1 { p.username.append_null() } else { read_string_field(cursor, p.username.as_mut(), current_chunk, len as usize)? } if len == -1 { p.username.append_null() } else { read_string_field(cursor, p.username.as_mut(), current_chunk, len as usize)? }
// Column 3: score // Column 3: score (REAL/Float32)
let len = cursor.read_i32::<BigEndian>()?; let len = cursor.read_i32::<BigEndian>()?;
if len == -1 { p.score.append_null() } else { p.score.append_value(cursor.read_f32::<BigEndian>()?) } if len == -1 { p.score.append_null() } else { p.score.append_value(cursor.read_f32::<BigEndian>()?) }
// Column 4: is_active // Column 4: is_active (BOOLEAN)
let len = cursor.read_i32::<BigEndian>()?; let len = cursor.read_i32::<BigEndian>()?;
if len == -1 { p.is_active.append_null() } else { p.is_active.append_value(cursor.read_u8()? != 0) } if len == -1 { p.is_active.append_null() } else { p.is_active.append_value(cursor.read_u8()? != 0) }
// Column 5: last_login // Column 5: last_login (TIMESTAMP)
let len = cursor.read_i32::<BigEndian>()?; let len = cursor.read_i32::<BigEndian>()?;
if len == -1 { p.last_login.append_null() } else { if len == -1 { p.last_login.append_null() } else {
let val = cursor.read_i64::<BigEndian>()? + POSTGRES_EPOCH_MICROS_OFFSET; let pg_micros = cursor.read_i64::<BigEndian>()?;
p.last_login.append_value(val * 1000); // Optimization: Constant offset applied
let unix_micros = pg_micros + POSTGRES_EPOCH_MICROS_OFFSET;
p.last_login.append_value(unix_micros * 1000);
} }
// Column 6: notes // Column 6: notes (TEXT)
let len = cursor.read_i32::<BigEndian>()?; let len = cursor.read_i32::<BigEndian>()?;
if len == -1 { p.notes.append_null() } else { read_string_field(cursor, p.notes.as_mut(), current_chunk, len as usize)? } if len == -1 { p.notes.append_null() } else { read_string_field(cursor, p.notes.as_mut(), current_chunk, len as usize)? }
// Column 7: course_id // Column 7: course_id (INT)
let len = cursor.read_i32::<BigEndian>()?; let len = cursor.read_i32::<BigEndian>()?;
if len == -1 { p.course_id.append_null() } else { p.course_id.append_value(cursor.read_i32::<BigEndian>()?) } if len == -1 { p.course_id.append_null() } else { p.course_id.append_value(cursor.read_i32::<BigEndian>()?) }
// Column 8: start_date // Column 8: start_date (DATE)
let len = cursor.read_i32::<BigEndian>()?; let len = cursor.read_i32::<BigEndian>()?;
if len == -1 { p.start_date.append_null() } else { if len == -1 { p.start_date.append_null() } else {
p.start_date.append_value(cursor.read_i32::<BigEndian>()? + 10957); let pg_days = cursor.read_i32::<BigEndian>()?;
// Optimization: 10957 days between 1970 and 2000
p.start_date.append_value(pg_days + 10957);
} }
// Column 9: rating // Column 9: rating (FLOAT8/Float64)
let len = cursor.read_i32::<BigEndian>()?; let len = cursor.read_i32::<BigEndian>()?;
if len == -1 { p.rating.append_null() } else { p.rating.append_value(cursor.read_f64::<BigEndian>()?) } if len == -1 { p.rating.append_null() } else { p.rating.append_value(cursor.read_f64::<BigEndian>()?) }
Ok(()) Ok(())
} }
// Helper function to consolidate zero-copy string reading and boundary checks
fn read_string_field( fn read_string_field(
cursor: &mut Cursor<&[u8]>, cursor: &mut Cursor<&[u8]>,
builder: &mut StringBuilder, builder: &mut StringBuilder,
current_chunk: &[u8], current_chunk: &[u8],
len: usize field_len_usize: usize
) -> Result<(), std::io::Error> { ) -> Result<(), std::io::Error> {
if (cursor.position() as usize + len) > current_chunk.len() {
return Err(std::io::Error::new(std::io::ErrorKind::UnexpectedEof, "Partial string")); if (cursor.position() as usize + field_len_usize) > current_chunk.len() {
return Err(std::io::Error::new(std::io::ErrorKind::UnexpectedEof, "Partial string field read"));
} }
let start = cursor.position() as usize; let start = cursor.position() as usize;
let end = start + len; let end = start + field_len_usize;
let slice = &current_chunk[start..end]; let slice = &current_chunk[start..end];
let val = str::from_utf8(slice).map_err(|e| std::io::Error::new(std::io::ErrorKind::InvalidData, e))?;
builder.append_value(val); let val_str = str::from_utf8(slice)
cursor.set_position(end as u64); .map_err(|e| std::io::Error::new(std::io::ErrorKind::InvalidData, e))?;
builder.append_value(val_str);
cursor.set_position(end as u64); // Manually advance cursor
Ok(()) Ok(())
} }
// Profiler logic omitted for brevity (it remains unchanged from previous version) // --------------------------------------------------------------------------------
pub async fn run_profiler_logic(_: &str) -> Result<String> { Ok("Profiler Placeholder".to_string()) } // --- 3. PROFILER LOGIC (New Feature) ---
// --------------------------------------------------------------------------------
/// Maps a PostgreSQL internal type name to a standard Arrow Type string for config.yaml.
fn map_postgres_to_arrow_type(pg_type_name: &str) -> Option<&'static str> {
match pg_type_name {
"int8" | "bigint" | "serial8" => Some("Int64"),
"int4" | "integer" | "serial" => Some("Int32"),
"float8" | "double precision" => Some("Float64"),
"float4" | "real" => Some("Float32"),
"varchar" | "text" | "uuid" => Some("Utf8"),
"bool" | "boolean" => Some("Boolean"),
"timestamptz" | "timestamp" => Some("Timestamp(Nanosecond, None)"),
"date" => Some("Date32"),
_ => None, // Returns None for unsupported types (like JSON, arrays, etc.)
}
}
pub async fn run_profiler_logic(config_path: &str) -> Result<String> {
// Phase 1: Load config to get connection string and query
let config: ConnectorConfig = load_and_validate_config(config_path)
.context("Failed to load and validate config for profiling")?;
// Use a non-COPY query to get metadata
let (base_query, _) = config.query.trim().split_once("TO STDOUT (FORMAT binary)")
.context("Query in config is malformed or not a COPY command")?;
// We only need the base query for the metadata query
let base_query_inner = base_query.trim().trim_start_matches("COPY (").trim_end_matches(")");
// Construct the metadata query (limit 0 is fastest)
let metadata_query = format!("SELECT * FROM ({}) AS subquery LIMIT 0", base_query_inner);
// Phase 2: Connect and execute the query
let pg_config: PgConfig = PgConfig::from_str(&config.connection_string)?;
let (client, connection) = pg_config.connect(NoTls).await
.context("Profiler: Failed to connect to PostgreSQL")?;
tokio::spawn(async move {
if let Err(e) = connection.await { eprintln!("Profiler connection error: {}", e); }
});
let statement = client.prepare(&metadata_query).await
.context("Profiler: Failed to prepare metadata query")?;
let mut output = String::from("schema:\n");
// Phase 3: Inspect the statement's columns for metadata
for column in statement.columns() {
let pg_type_name = column.type_().name().to_lowercase();
let arrow_type = map_postgres_to_arrow_type(&pg_type_name)
.unwrap_or("UNKNOWN (Review Manually)");
let column_entry = format!(
"- arrow_type: {}\n column_name: {}\n",
arrow_type,
column.name()
);
output.push_str(&column_entry);
}
// Final instructions for the user
output.push_str("\n# NOTE: Paste the 'schema' block above into your config.yaml\n");
output.push_str(
"# REVIEW any UNKNOWN types. PostgreSQL types: (int8, float8, text, bool, timestamp, date, etc.)\n"
);
Ok(output)
}