mirror of
https://github.com/t8y2/dbx.git
synced 2026-10-02 02:34:42 +08:00
fix(sqlserver): keep rows readable on collation decode failure
This commit is contained in:
+3
-1
@@ -37,7 +37,9 @@ postgres-protocol = { git = "https://github.com/t8y2/tokio-postgres-gaussdb.git"
|
||||
# rustls dangerous-mode connections to legacy certificates. Remove after upstream support lands.
|
||||
mysql_async = { git = "https://github.com/zipg/mysql_async.git", rev = "4a9e60a62f5a48e35ef9ccc9fc8277d5012b19dd" }
|
||||
# Preserve readable SQL Server result sets when a row contains an unpaired
|
||||
# UTF-16 surrogate. Remove after the behavior is available in upstream tiberius.
|
||||
# UTF-16 surrogate, or varchar/text bytes that are invalid in the column's
|
||||
# collation (replaced with U+FFFD, no BOM sniffing). Remove after the behavior
|
||||
# is available in upstream tiberius.
|
||||
tiberius = { path = "vendor/tiberius" }
|
||||
|
||||
[profile.release]
|
||||
|
||||
Vendored
+8
-3
@@ -6,14 +6,19 @@ identified by the checksum
|
||||
upstream release tag points to commit
|
||||
`0e2897a276166503ba78fe3e1cee501e9a034021`.
|
||||
|
||||
DBX changes only SQL Server Unicode column-data decoding:
|
||||
DBX changes only SQL Server column-data decoding:
|
||||
|
||||
- NCHAR, NVARCHAR, and NTEXT row values replace unpaired UTF-16 surrogates with
|
||||
U+FFFD, matching the Microsoft JDBC driver's observable behavior.
|
||||
- CHAR, VARCHAR, and TEXT row values under a non-Unicode collation decode
|
||||
lossily: byte sequences that are invalid in the column's collation become
|
||||
U+FFFD instead of failing the whole result set, following the JDBC driver's
|
||||
REPLACE policy. The lossy decoder does no BOM handling, so bytes that merely
|
||||
look like a UTF-8 BOM (e.g. under GB18030) stay ordinary collation data.
|
||||
- Odd byte lengths remain protocol errors, including an explicit NTEXT guard
|
||||
that prevents truncating a trailing byte and desynchronizing the TDS stream.
|
||||
- Metadata, environment tokens, other protocol strings, and non-Unicode
|
||||
codepage decoding retain upstream's strict behavior.
|
||||
- Metadata, environment tokens, and other protocol strings retain upstream's
|
||||
strict behavior.
|
||||
|
||||
The regression tests in `src/tds/codec/column_data.rs` exercise the decoder
|
||||
with raw TDS value frames. The upstream integration fixtures are omitted from
|
||||
|
||||
+63
@@ -758,6 +758,69 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn varchar_replaces_invalid_collation_bytes() {
|
||||
// CP936/GB18030 (sort ID 198): 0x81 starts a two-byte sequence, and
|
||||
// 0x40..0x7e is a valid trail byte, so 0x81 0x20 is invalid. A legacy
|
||||
// varchar column can still hold those bytes when an application wrote
|
||||
// them under a different codepage.
|
||||
let value = decode_wire(
|
||||
VarLenType::BigVarChar,
|
||||
20,
|
||||
Some(Collation::new(0x804, 198)),
|
||||
&short_string_wire(b"ok\x81\x20ok"),
|
||||
)
|
||||
.await
|
||||
.expect("undecodable varchar bytes must not fail the whole result set");
|
||||
|
||||
let decoded = decoded_string(value).expect("varchar row value");
|
||||
assert!(decoded.starts_with("ok"), "readable bytes are preserved: {decoded:?}");
|
||||
assert!(decoded.ends_with("ok"), "bytes after the bad sequence are preserved: {decoded:?}");
|
||||
assert!(decoded.contains('\u{fffd}'), "the bad sequence is replaced: {decoded:?}");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn text_replaces_invalid_collation_bytes() {
|
||||
let value = decode_wire(
|
||||
VarLenType::Text,
|
||||
0,
|
||||
Some(Collation::new(0x804, 198)),
|
||||
&ntext_wire(b"ok\x81\x20ok"),
|
||||
)
|
||||
.await
|
||||
.expect("undecodable text bytes must not fail the whole result set");
|
||||
|
||||
let decoded = decoded_string(value).expect("text row value");
|
||||
assert!(decoded.starts_with("ok"), "readable bytes are preserved: {decoded:?}");
|
||||
assert!(decoded.ends_with("ok"), "bytes after the bad sequence are preserved: {decoded:?}");
|
||||
assert!(decoded.contains('\u{fffd}'), "the bad sequence is replaced: {decoded:?}");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn varchar_gb18030_does_not_sniff_a_utf8_bom() {
|
||||
// EF BB BF is fully valid GB18030 data: EF BB and BF 61 form two
|
||||
// two-byte sequences (锘縜) followed by ASCII "bc". The collation
|
||||
// already names the encoding, so the lossy decoder must not let
|
||||
// encoding_rs sniff those bytes as a UTF-8 BOM and return "abc".
|
||||
let (text, had_errors) = Collation::new(0x804, 198)
|
||||
.codec()
|
||||
.unwrap()
|
||||
.decode_lossy(b"\xef\xbb\xbfabc");
|
||||
assert_eq!(text, "锘縜bc");
|
||||
assert!(!had_errors, "the bytes are valid GB18030");
|
||||
|
||||
let value = decode_wire(
|
||||
VarLenType::BigVarChar,
|
||||
20,
|
||||
Some(Collation::new(0x804, 198)),
|
||||
&short_string_wire(b"\xef\xbb\xbfabc"),
|
||||
)
|
||||
.await
|
||||
.expect("BOM-like varchar bytes must decode as ordinary GB18030 data");
|
||||
|
||||
assert_eq!(decoded_string(value).as_deref(), Some("锘縜bc"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn varchar_cp850_collation_decodes_french_text() {
|
||||
// SQL_1xCompat_CP850_CI_AS: LCID 0x409, sort ID 49. The DOS codepages
|
||||
|
||||
+3
-4
@@ -21,10 +21,9 @@ where
|
||||
// Codepages other than UTF
|
||||
(Some(buf), BigChar) | (Some(buf), BigVarChar) => {
|
||||
let collation = collation.as_ref().unwrap();
|
||||
let s = collation
|
||||
.codec()?
|
||||
.decode(buf.as_ref())
|
||||
.ok_or_else(|| Error::Encoding("invalid sequence".into()))?;
|
||||
// Lossy: a row holding bytes that are invalid in its declared
|
||||
// collation must stay readable instead of failing the result set.
|
||||
let (s, _) = collation.codec()?.decode_lossy(buf.as_ref());
|
||||
|
||||
Ok(Some(s.into()))
|
||||
}
|
||||
|
||||
+4
-3
@@ -31,9 +31,10 @@ where
|
||||
buf.push(src.read_u8().await?);
|
||||
}
|
||||
|
||||
codec
|
||||
.decode(buf.as_ref())
|
||||
.ok_or_else(|| Error::Encoding("invalid sequence".into()))?
|
||||
// Lossy: a row holding bytes that are invalid in its declared
|
||||
// collation must stay readable instead of failing the result set.
|
||||
let (text, _) = codec.decode_lossy(buf.as_ref());
|
||||
text
|
||||
}
|
||||
// NTEXT
|
||||
None => {
|
||||
|
||||
Vendored
+28
@@ -103,6 +103,34 @@ impl CollationCodec {
|
||||
}
|
||||
}
|
||||
|
||||
/// Decode legacy varchar/text data, substituting U+FFFD for byte sequences
|
||||
/// that are invalid in the target encoding.
|
||||
///
|
||||
/// Legacy `varchar`/`text` columns can hold bytes that do not form a valid
|
||||
/// sequence in their declared collation, usually from an application that
|
||||
/// wrote data under a different codepage. Decoding those strictly fails the
|
||||
/// whole result set, so a single bad row makes an otherwise readable table
|
||||
/// impossible to browse, while other clients render the row and replace the
|
||||
/// undecodable bytes. This mirrors the lossy UTF-16 handling that `nchar`,
|
||||
/// `nvarchar`, and `ntext` already use for unpaired surrogates.
|
||||
///
|
||||
/// Returns the decoded string and whether any byte had to be replaced.
|
||||
pub fn decode_lossy(&self, bytes: &[u8]) -> (String, bool) {
|
||||
match self {
|
||||
// No BOM handling: the collation already names the encoding, so a
|
||||
// leading byte sequence that looks like a UTF-8 BOM is ordinary
|
||||
// data (the NVARCHAR path likewise keeps a U+FEFF character).
|
||||
CollationCodec::BuiltIn(encoding) => {
|
||||
let (text, had_errors) = encoding.decode_without_bom_handling(bytes);
|
||||
(text.into_owned(), had_errors)
|
||||
}
|
||||
// Single-byte codecs map every one of the 256 byte values, so they
|
||||
// can never fail and never need a replacement character.
|
||||
CollationCodec::Cp437 => (decode_single_byte(bytes, &CP437_HIGH), false),
|
||||
CollationCodec::Cp850 => (decode_single_byte(bytes, &CP850_HIGH), false),
|
||||
}
|
||||
}
|
||||
|
||||
/// Encode a string for a legacy varchar parameter. Returns `None` when the
|
||||
/// input contains a character that the target encoding cannot represent.
|
||||
pub fn encode_from_utf8(&self, text: &str) -> Option<Vec<u8>> {
|
||||
|
||||
Reference in New Issue
Block a user