From fb7c085bb02f386c2f1227a170f724fdf4c28501 Mon Sep 17 00:00:00 2001 From: muqiuhan Date: Thu, 6 Jun 2024 12:26:58 +0000 Subject: deploy: 1f3b4b7176d57d55dcf85329c213ae5d354fb044 --- .../index.html" | 309 +++++++++++++++++++++ .../index.html" | 289 +++++++++++++++++++ .../index.html" | 283 +++++++++++++++++++ "2023/05/04/C-\347\232\204-Trait/index.html" | 294 ++++++++++++++++++++ .../index.html" | 279 +++++++++++++++++++ .../index.html" | 274 ++++++++++++++++++ .../index.html" | 279 +++++++++++++++++++ 7 files changed, 2007 insertions(+) create mode 100644 "2023/05/01/Rust-\350\231\232\350\241\250\345\270\203\345\261\200\350\247\204\345\210\231\344\273\213\347\273\215/index.html" create mode 100644 "2023/05/02/Rust-Partial-\350\257\255\344\271\211/index.html" create mode 100644 "2023/05/03/Rust-NewType-\346\250\241\345\274\217/index.html" create mode 100644 "2023/05/04/C-\347\232\204-Trait/index.html" create mode 100644 "2023/05/11/C-vector-\347\232\204-push-back-\345\222\214-emplace-back/index.html" create mode 100644 "2023/05/12/C-20-\345\256\236\347\216\260-string-split/index.html" create mode 100644 "2023/05/24/Rust-\351\227\255\345\214\205-lifetime-may-not-live-long-enough-\351\227\256\351\242\230/index.html" (limited to '2023/05') diff --git "a/2023/05/01/Rust-\350\231\232\350\241\250\345\270\203\345\261\200\350\247\204\345\210\231\344\273\213\347\273\215/index.html" "b/2023/05/01/Rust-\350\231\232\350\241\250\345\270\203\345\261\200\350\247\204\345\210\231\344\273\213\347\273\215/index.html" new file mode 100644 index 00000000..4dfa968b --- /dev/null +++ "b/2023/05/01/Rust-\350\231\232\350\241\250\345\270\203\345\261\200\350\247\204\345\210\231\344\273\213\347\273\215/index.html" @@ -0,0 +1,309 @@ + + + + + + + + + + + + + + + + + + + + + +Rust 虚表布局规则介绍 | 暮秋小屋 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ +
+ +
+
+
+ + + +
+
+
+ + +
+
+
+ + +
+ +
+ +
+ +
+
+
+

在 Rust 中,一个指向未知大小对象(!Sized)的引用或指针被实现为一个由两个 usize 大小的域构成的胖指针。这两个域中,其中一个域保存了被引用或被指向的对象的地址,另一个域保存了一个名为 metadata 的数据。对于 slice 的引用或指针来说,其 metadata 为 slice 的长度。对于 trait object 的引用或指针来说,其 metadata 为虚表(vtable)地址。与 C++ 虚表类似,Rust 虚表的存在使得诸多动态语言特性得以实现,例如动态派发(dynamic dispatch)、向上转换(upcasting)、向下转换(downcast)等。本文将对 Rust 中虚表的布局规则进行简要介绍,并在此过程中对 Rust 中若干动态特性的实现方法进行简要介绍。

+
+

注意:Rust 虚表及其结构属于 Rust 语言的内部实现细节,不保证稳定性。本文所介绍的虚表布局仅反映本文创作时最新的 Rust 虚表结构[1],在将来 Rust 虚表结构可能会发生变化。一个 Rust 程序的正确性不应该以任何方式依赖于 Rust 虚表的结构。

+
+

基本结构

Rust 程序中的所有虚表均以一个固定结构的 header 开头。Header 中按顺序包含三个usize 大小的字段:drop_in_place ,size 和 align 。在 header 之后是一系列的 usize 大小字段,其数量以及含义在每个虚表中可能都不同。

+
1
2
3
4
5
6
7
8
9
10
11
12
13
+---------------+
| drop_in_place |
+---------------+
| size |
+---------------+
| align |
+---------------+
| entry1 |
+---------------+
| entry2 |
+---------------+
| entry3 |
+---------------+
+

虚表 header 中的drop_in_place 是一个函数指针,其指向的函数能够原地 drop 当前胖指针所引用的对象。size 和 align 两个域分别给出对象的大小和内存对齐,这两个域共同构成一个 std::alloc::Layout 结构,可用于释放当前胖指针所引用的对象所占据的内存。虚表 header 的存在使得 trait object 总是能被销毁和释放。例如当销毁一个 Box<dyn Trait> 时,Box::<dyn Trait>::drop 会首先调用虚表中的 drop_in_place 函数原地销毁 Box 所引用的对象,然后再调用 dealloc 函数并传递虚表中的 size 和 align 释放堆空间。

+

在虚表 header 之后是一系列的字段。在最普遍的情况下,每个字段代表一个指向 trait 所定义的函数的指针。例如,对于下列 object safe 的 trait:

+
1
2
3
4
5
pub trait Trait {
fn fun1(&self);
fn fun2(&self);
fn fun3(&self);
}
+ +

如果类型T 实现了 Trait,那么为 T 生成的 Trait 虚表的结构为:

+
1
2
3
4
5
6
7
8
9
10
11
12
13
+--------------------------+
| fn drop_in_place(*mut T) |
+--------------------------+
| size of T |
+--------------------------+
| align of T |
+--------------------------+
| fn <T as Trait>::fun1 |
+--------------------------+
| fn <T as Trait>::fun2 |
+--------------------------+
| fn <T as Trait>::fun3 |
+--------------------------+
+ +

Trait 中的函数按照声明顺序依次排列在虚表 header 之后。当通过一个指向 T 对象的 &dyn Trait 调用 fun2 函数时,程序会先从虚表的第 5 个域中得到为 T 实现的 Trait::fun2 函数的地址,然后再调用之。

+

Super Trait

Object safe 的 trait 可以有 super trait。例如:

+
1
2
3
4
5
6
7
8
9
10
11
12
13
pub trait Grand {
fn grand_fun1(&self);
fn grand_fun2(&self);
}

pub trait Parent : Grand {
fn parent_fun1(&self);
fn parent_fun2(&self);
}

pub trait Trait : Parent {
fn fun(&self);
}
+ +

如果类型T 实现了 Trait,那么此时为 T 生成的 Trait 虚表的结构为:

+
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
+-------------------------------+
| fn drop_in_place(*mut T) |
+-------------------------------+
| size of T |
+-------------------------------+
| align of T |
+-------------------------------+
| fn <T as Grand>::grand_fun1 |
+-------------------------------+
| fn <T as Grand>::grand_fun2 |
+-------------------------------+
| fn <T as Parent>::parent_fun1 |
+-------------------------------+
| fn <T as Parent>::parent_fun2 |
+-------------------------------+
| fn <T as Trait>::fun |
+-------------------------------+
+ +

可以看到,此时Trait 以及 Trait 的所有直接或间接父 trait 所定义的所有函数均包含在虚表 header 之后,且顺序为后序(即先排布 Trait 的父 trait 所定义的所有函数,最后再排布 Trait 所定义的所有函数)。这样的排布方式使得在得到 T 类型的 Trait 虚表的同时也同时得到了 T 类型的 Parent 虚表和 Grand 虚表。T 类型的 Grand 虚表恰好由 Trait 虚表的前五个域构成,T 类型的 Parent 虚表恰好由 Trait 虚表的前七项构成。这使得向上转换变得非常简单。

+

所谓向上转换,即 Rust 允许将&dyn Trait 转换为 &dyn Parent 或 &dyn Grand 。在向上转换的过程中,胖指针的对象地址域保持不变,但 metadata 域可能需要进行调整,因为不同的 trait 可能具有不同的虚表地址。但在当前示例中,向上转换不需要调整 metadata 域,因为一个指向 Trait 虚表的指针同时也指向 Parent 虚表和 Grand 虚表。在后文中我们会进一步介绍需要调整 metadata 域的向上转换的情况。

+
+

注意:目前 stable Rust 暂不支持向上转换。要使用向上转换特性,必须使用 nightly 工具链,并向源文件中添加 #![feature(trait_upcasting)] 特性开关。

+
+

多重继承

Trait 可以有多个 super trait。例如:

+
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
pub trait Base {
fn base_fun1(&self);
fn base_fun2(&self);
}

pub trait Left : Base {
fn left_fun1(&self);
fn left_fun2(&self);
}

pub trait Right : Base {
fn right_fun1(&self);
fn right_fun2(&self);
}

pub trait Trait : Left + Right {
fn fun(&self);
}
+ +

如果类型T 实现了 Trait,那么此时为 T 生成的 Trait 虚表的结构为:

+
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
+-----------------------------+
| fn drop_in_place(*mut T) |
+-----------------------------+
| size of T |
+-----------------------------+
| align of T |
+-----------------------------+
| fn <T as Base>::base_fun1 |
+-----------------------------+
| fn <T as Base>::base_fun2 |
+-----------------------------+
| fn <T as Left>::left_fun1 |
+-----------------------------+
| fn <T as Left>::left_fun2 |
+-----------------------------+
| fn <T as Right>::right_fun1 |
+-----------------------------+
| fn <T as Right>::right_fun2 |
+-----------------------------+
| ptr to <T as Right>::vtable |
+-----------------------------+
| fn <T as Trait>::fun |
+-----------------------------+
+ +

可以看到,此时Trait 及其所有直接或间接父 trait 所定义的所有函数仍然包含在虚表内,因此通过 &dyn Trait 调用的函数仍然可以直接从虚表内得到其实际目标函数的地址。另外,Trait 虚表内仍然包含有效的 Base 虚表和 Left 虚表。因此,将 &dyn Trait 向上转换为 &dyn Left 或 &dyn Base 仍然是极其简单的,不需要调整胖指针的 metadata 域。但是,将 &dyn Trait 向上转换为 &dyn Right 就需要调整 metadata 域了,因为 Trait 虚表内并不包含一个有效的 Right 虚表。这也是 Trait 虚表中 ptr to <T as Right>::vtable 域的作用:在执行向上转换时,程序会读取 Trait 虚表的这个域作为得到的 &dyn Right 胖指针的 metadata 。这也是 Rust 向上转换与 C++ 向上转换的一个很大不同:在 C++ 中的向上转换通常并不需要访问虚表(除非需要执行跨虚继承边界的转换),但在 Rust 中向上转换可能需要访问虚表。

+

更加一般地,对于一个 object safe 的 traitTr,将其第一个父 trait、第一个父 trait 的第一个父 trait、…… 这一系列直接或间接父 trait 记为这个 trait 的 PrefixTrait 集合。在将 &dyn Tr 向上转换时,如果转换到的目标 trait 包含在 PrefixTrait 集合内,那么这个向上转换是平凡的:不需要调整胖指针的 metadata 域。否则,这个向上转换需要在 Tr 的虚表内读取目标 trait 的虚表指针作为转换结果的 metadata 。在 Tr 的虚表结构中,位于 PrefixTrait 集合中的父 trait 只需要排布他们所定义的函数即可;对于其他父 trait 还需要额外在虚表内排布一个指向其虚表的指针用于向上转换。

+

向下转换

Rust 提供了一个特殊的 trait:std::any::Any 。该 trait 支持向下转换,即可以将 &dyn Any 转换为 T 。转换过程中会对胖指针所指向的对象的实际类型进行检查,确认其确实是一个 T 类型的对象。Any trait 的虚表结构有一些特殊;在虚表 header 之后,Any 虚表仅包含一个域,这个域直接给出胖指针指向的对象的类型标识(由一个 std::any::TypeId 类型的值表示)。例如,对于任意的 T: ‘static,编译器为其生成的 Any 虚表为:

+
1
2
3
4
5
6
7
8
9
+--------------------------+
| fn drop_in_place(*mut T) |
+--------------------------+
| size of T |
+--------------------------+
| align of T |
+--------------------------+
| TypeId of T |
+--------------------------+
+

在执行向下转换时,程序首先检查转换到的类型是否与虚表中给出的TypeId 所标识的类型一致。若类型检查通过,向下转换操作可以直接返回胖指针中的指针域作为转换结果。

+

REFERENCE

    +
  1. Vtable format to support dyn upcasting coercion https://rust-lang.github.io/dyn-upcasting-coercion-initiative/design-discussions/vtable-layout.*html*
  2. +
+ +
+ + + + + +
+ + + + + + + +
+ + +
+
+
+ + + +
+ + +
+
+
+
+
+
+ +
+
+
+
+
+
+
+
+
+ + + + + + + + + diff --git "a/2023/05/02/Rust-Partial-\350\257\255\344\271\211/index.html" "b/2023/05/02/Rust-Partial-\350\257\255\344\271\211/index.html" new file mode 100644 index 00000000..37356140 --- /dev/null +++ "b/2023/05/02/Rust-Partial-\350\257\255\344\271\211/index.html" @@ -0,0 +1,289 @@ + + + + + + + + + + + + + + + + + + + + + +Rust Partial 语义 | 暮秋小屋 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ +
+ +
+
+
+ + + +
+
+
+ + +
+
+
+ + +
+ +
+ +
+ +
+
+
+
+

在Rust中,PartialEq和PartialOrd trait处理了不是所有值都可以相互比较的情况。

+
+

PartialEq Trait

PartialEq trait用于定义值相等性的比较。它的设计允许类型的值之间进行相等(==)和不等(!=)的比较。与其对应的 Eq trait 确保一个类型的所有值都是可以可靠比较的,即满足等价关系的特性,如自反性、对称性和传递性。

+
1
2
fn eq(&self, other: &Self) -> bool;
fn ne(&self, other: &Self) -> bool;
+ +

在大多数情况下,类型的值都能够完全比较相等性,这时可以实现Eq。然而,对于一些特殊类型的值,如浮点数,由于存在无穷大的正负值和NaN值,导致它们的比较更加复杂。例如,根据IEEE浮点数的标准,NaN与任何值(包括它自己)比较都不相等。

+

PartialOrd Trait

PartialOrd trait用于定义值之间的大小比较。类似于PartialEq,它允许部分比较大小,返回一个Option,表示比较结果可能存在,也可能不存在(即比较无法进行时返回None):

+
1
fn partial_cmp(&self, other: &Self) -> Option<Ordering>;
+ +

在全部比较可能的场景,我们会使用Ord trait,它要求实现cmp方法,总是返回一个Ordering,表示两个值之间的确切比较关系。Ord是在所有值都能够比较时使用的,例如整数和字符串。

+

设计用意和解决的问题

Rust 设计 PartialEq 和 PartialOrd trait 主要出于以下几个理由:

+
    +
  • 非总序理念:并不是所有类型都有一个全局的排序方法。例如,复数之间就没有一个自然的大小顺序。为了避免为这些类型人为地赋予一个排序方法,Rust 提供了一个只需部分实现序列操作的选择。
  • +
  • IEEE 浮点数标准:由于浮点数标准定义了特殊值(NaN, 正负无穷),以及NaN不等于自身的规则,浮点数在一些情况下不能进行相等性或大小比较。
  • +
  • 提升错误处理能力和安全性:通过返回 Option<Ordering>,partial_cmp 方法明确指出了失败的可能性,从而迫使程序员在使用时考虑并处理这种情况,增加了代码的正确性和稳健性。
  • +
  • 表达性和灵活性:这些 trait 允许开发者为自定义类型定义适当的相等性和排序行为,从而加强了 Rust 类型系统的表达性和灵活性。
  • +
+

PartialEq 和 PartialOrd trait 的设计允许程序员选择精准的相等性和排序语义,同时明确了对于某些类型相等性比较和大小排序并不总是可能的事实。通过引入适度的复杂性,让 Rust 的类型系统更加安全。

+ +
+ + + + + +
+ + + + + + + +
+ + +
+
+
+ + + +
+ + +
+
+
+
+
+
+ +
+
+
+
+
+
+
+
+
+ + + + + + + + + diff --git "a/2023/05/03/Rust-NewType-\346\250\241\345\274\217/index.html" "b/2023/05/03/Rust-NewType-\346\250\241\345\274\217/index.html" new file mode 100644 index 00000000..7ad16dd5 --- /dev/null +++ "b/2023/05/03/Rust-NewType-\346\250\241\345\274\217/index.html" @@ -0,0 +1,283 @@ + + + + + + + + + + + + + + + + + + + + + +Rust NewType 模式 | 暮秋小屋 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ +
+ +
+
+
+ + + +
+
+
+ + +
+
+
+ + +
+ +
+ +
+ +
+
+
+

New Type模式是一种软件设计模式,用于在已有类型的基础上创建一个新的类型。在Rust中,这通常是通过定义一个结构体,其中只包含一个单一成员。这个结构体(New Type)对外提供了一个新的、独立的类型,用于对原始类型增加额外的语义或限制。

+

加强类型安全

1
2
3
4
5
6
7
8
9
10
11
12
13
struct Meters(f64);
struct Feet(f64);

let length_in_meters = Meters(100.0);
let length_in_feet = Feet(328.084);

// 编译器会防止以下代码执行,因为类型不匹配
// let wrong_length = Meters(length_in_feet); // 编译错误

// 正确的构造
fn add_lengths(length1: Meters, length2: Meters) -> Meters {
Meters(length1.0 + length2.0)
}
+ +

这个例子使用 newtype 模式避免将原始类型f64用于不同的量度,从而增强了类型的安全性。

+

实现特定 trait

1
2
3
4
5
6
7
8
9
10
struct Kilometers(f64);

impl Kilometers {
fn to_miles(&self) -> f64 {
self.0 * 0.621371
}
}

let distance = Kilometers(10.0);
println!("The distance in miles is {}", distance.to_miles());
+ +

这里,Kilometers有一个方法to_miles,该方法是不会影响其他f64数据的。如果我们有另一个表示温度的f64类型,就不会意外调用到与距离相关的方法。

+

New Type模式同样适用于对Box<dyn SomeTrait>类型的包装,这可以在需要动态分派(动态调用实现了某个接口的不同类型的对象的方法)的时候提供便利。通过创建一个New Type来包装这样的Box<dyn SomeTrait>类型,可以提供自定义的方法或实现更多的trait,同时也可以让API更加清晰和易于使用。

+

零成本抽象

在Rust中,New Type模式不仅是类型安全的,还是一种零成本抽象。这是因为Rust编译器在编译时期会进行足够的优化,以确保New Type的使用没有运行时开销。 Rust的零成本抽象原则确保了抽象不会引入额外的运行时成本。例如,当你使用Meters这样的New Type时,Rust确保:

+
    +
  1. 无额外内存开销:Meters只包含一个f64,在内存中的表现和单独的f64是一样的。
  2. +
  3. 无额外运行时开销:使用Meters时,性能和直接使用f64完全相同。编译器会移除任何关于New Type的包装和解包的代码。
  4. +
+ +
+ + + + + +
+ + + + + + + +
+ + +
+
+
+ + + +
+ + +
+
+
+
+
+
+ +
+
+
+
+
+
+
+
+
+ + + + + + + + + diff --git "a/2023/05/04/C-\347\232\204-Trait/index.html" "b/2023/05/04/C-\347\232\204-Trait/index.html" new file mode 100644 index 00000000..e626505f --- /dev/null +++ "b/2023/05/04/C-\347\232\204-Trait/index.html" @@ -0,0 +1,294 @@ + + + + + + + + + + + + + + + + + + + + + +C++ 的 Trait | 暮秋小屋 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ +
+ +
+
+
+ + + +
+
+
+ + +
+
+
+ + +
+ +
+ +
+ +
+
+
+
+

C++ 的 traits 技术,是一种约定俗称的技术方案,用来为同一类数据(包括自定义数据类型和内置数据类型)提供统一的类型名(traits),这样可以统一的操作函数,例如 advance(), swap(), encode()/decode() 等。

+
+

例如,拥有义类型Foo, Bar,以及编译器自带类型 int, double, string,我们想要为这些不同的类型提供统一的编码函数 decode() 。

+

除了使用 trait 技术之外,函数重载和模板函数 + 内置字段也可以实现,前者每增加一种数据类型就需要重新实现一个函数,而同一类数据(int, unsinged int)可以使用同样的编码方法。我们想要的是针对同一种数据类型,只编写一个函数。后者对于系统自定义变量 int, double 而言,是无法在其内部定义 type 的。

+

traits 技术的关键在于,使用另外的模板类 type_traits 来保存不同数据类型的 type,这样就可以兼容自定义数据类型和内置数据类型:

+
1
2
3
4
5
6
// 定义数据 type 类
enum Type {
TYPE_1,
TYPE_2,
TYPE_3
}
+ +

对于自定义类型,在类内部定义 type,然后在 traits 类中定义同样的 type:

+
1
2
3
4
5
6
7
8
9
10
11
12
13
// 自定义数据类型
class Foo {
public:
Type type = TYPE_1;
};
class Bar {
public:
Type type = TYPE_2;
};
template<typename T>
struct type_traits {
Type type = T::type;
}
+ +

对于内置数据类型,使用模板类的特化为自定义类型生成独有的 type_traits:

+
1
2
3
4
5
6
7
8
9
// 内置数据类型
template<typename int>
struct type_traits {
Type type = Type::TYPE_1;
}
template<typename double>
struct type_traits {
Type type = Type::TYPE_3;
}
+ +

这样就可以为不同数据类型生成统一的模板函数

+
1
2
3
4
5
6
7
8
9
10
// 统一的编码函数
template<typename T>
void decode<const T& data, char* buf) {
if(type_traits<T>::type == Type::TYPE_1) {
...
}
else if(type_traits<T>::type == Type::TYPE_2) {
...
}
}
+ +

总结

+
    +
  • traits 技术的关键在于使用第三方模板类 traits,利用模板特化的功能, 实现对自定义数据和编译器内置数据的统一
  • +
  • 这个例子使用了枚举变量来表示数据类型,而实际操作中通常使用不同的类来表示不同的类型,这样可以在编写模板函数时更好的优化。
  • +
  • tratis 技术常见于标准库的实现中,但对日常开发中降低代码冗余也有很好的借鉴意义
  • +
  • C++20 提供了Concept 的特性,使用Concept 可以使得实现类似的功能更加方便
  • +
+ +
+ + + + + +
+ + + + + + + +
+ + +
+
+
+ + + +
+ + +
+
+
+
+
+
+ +
+
+
+
+
+
+
+
+
+ + + + + + + + + diff --git "a/2023/05/11/C-vector-\347\232\204-push-back-\345\222\214-emplace-back/index.html" "b/2023/05/11/C-vector-\347\232\204-push-back-\345\222\214-emplace-back/index.html" new file mode 100644 index 00000000..ef87129a --- /dev/null +++ "b/2023/05/11/C-vector-\347\232\204-push-back-\345\222\214-emplace-back/index.html" @@ -0,0 +1,279 @@ + + + + + + + + + + + + + + + + + + + + + +C++ vector 的 push_back 和 emplace_back | 暮秋小屋 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ +
+ +
+
+
+ + + +
+
+
+ + +
+
+
+ + +
+ +
+ +
+ +
+
+
+
1
2
3
4
5
6
/// Inserts a new element at the end of the vector, right after its current last element. This new element is constructed in place using args as the arguments for its constructor.
/// This effectively increases the container size by one, which causes an automatic reallocation of the allocated storage space if -and only if- the new vector size surpasses the current vector capacity.
/// The element is constructed in-place by calling allocator_traits::construct with args forwarded.
///A similar member function exists, push_back, which either copies or moves an existing object into the container.
template <class... Args>
void emplace_back (Args&&... args);
+ +

push_back 会构造一个临时对象,这个临时对象会被拷贝或者移入到容器中,然而 emplace_back 会直接根据传入的参数在容器的适当位置进行构造而避免拷贝或者移动。

+

传统观点认为 push_back 会构造一个临时对象,这个临时对象会被移入到 v 中,然而 emplace_back 会直接根据传入的参数在适当位置进行构造而避免拷贝或者移动。从标准库代码的实现角度来说这是对的,但是对于提供了优化的编译器来讲,上面示例中最后两行表达式生成的代码其实没有区别。

+

真正的区别在于,emplace_back 更加强大,它可以调用任何类型的(只要存在)构造函数。而 push_back 会更加严谨,它只调用隐式构造函数。隐式构造函数被认为是安全的。如果能够通过对象 T 隐式构造对象 U,就认为 U 能够完整包含 T 的所有内容,这样将 T 传递给 U 通常是安全的。正确使用隐式构造的例子是用 std::uint32_t 对象构造 std::uint64_t 对象,错误使用隐式构造的例子是用 double 构造 std::uint8_t。

+

如果想要调用显示构造函数,那么就调用 emplace_back。如果只希望调用隐式构造函数,那么请使用更加安全的 push_back:

+
1
2
3
4
std::vector<std::unique_ptr<T>> v;
T a;
v.emplace_back(std::addressof(a)); // compiles
v.push_back(std::addressof(a)); // fails to compile
+ +

std::unique_ptr<T> 包含了显示构造函数通过 T* 进行构造。因为 emplace_back 能够调用显示构造函数,所以传递一个裸指针并不会产生编译错误。然而,当 v 超出了作用域,std::unique_ptr<T> 的析构函数会尝试 delete 类型 T* 的指针,而类型 T* 的指针并不是通过 new 来分配的,因为它保存的是栈对象的地址,因此 delete 行为是未定义的。

+ +
+ + + + + +
+ + + + + + + +
+ + +
+
+
+ + + +
+ + +
+
+
+
+
+
+ +
+
+
+
+
+
+
+
+
+ + + + + + + + + diff --git "a/2023/05/12/C-20-\345\256\236\347\216\260-string-split/index.html" "b/2023/05/12/C-20-\345\256\236\347\216\260-string-split/index.html" new file mode 100644 index 00000000..a2db0de9 --- /dev/null +++ "b/2023/05/12/C-20-\345\256\236\347\216\260-string-split/index.html" @@ -0,0 +1,274 @@ + + + + + + + + + + + + + + + + + + + + + +C++ 20 实现 string split | 暮秋小屋 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ +
+ +
+
+
+ + + +
+
+
+ + +
+
+
+ + +
+ +
+ +
+ +
+
+
+

C++20引入了范围库ranges,其中提供的两个范围适配器std::split、std::lazy_split可以使我们以一种更为优雅的形式实现split:

+
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
#include <concept>
#include <ranges>
#include <algorithm>
#include <format>
#include <iostream>

#define stdr std::ranges
#define stdrv std::ranges::views

template<template<typename> typename Container = std::vector, typename Arg = std::string_view>
auto Split(std::string_view str, std::string_view delimiter)
{
Container<Arg> myCont;
auto temp = str
| stdrv::split(delimiter)
| stdrv::transform([](auto&& r)
{
return Arg(std::addressof(*r.begin()), stdr::distance(r));
});
auto iter = std::inserter(myCont, myCont.end());
stdr::for_each(temp, [&](auto&& x) { iter = {x.begin(), x.end()}; });
return myCont;
}
int main()
{
std::string str = "Hello233C++20233and233New233Spilt";
std::string delimiter = "233";
auto&& strCont = Split<std::list, std::string>(str, delimiter);
stdr::for_each(strCont, [](auto&& x) { std::cout << std::format("{} ", x); });
}
//output: Hello C++20 and New Spilt
+ +

C++20没有提供关键的 ranges::to<container>函数,导致demo中还需要额外封装并手写for_each来写入数据,等到C++23实装了该函数,split的实现会比现在简洁优雅的多,真正做到方便泛用、无需封装:

+
1
2
3
auto&& strCont = str
| stdrv::lazy_split(delimiter)
| stdr ::to<std::vector<std::string>>;
+
+ + + + + +
+ + + + + + + +
+ + +
+
+
+ + + +
+ + +
+
+
+
+
+
+ +
+
+
+
+
+
+
+
+
+ + + + + + + + + diff --git "a/2023/05/24/Rust-\351\227\255\345\214\205-lifetime-may-not-live-long-enough-\351\227\256\351\242\230/index.html" "b/2023/05/24/Rust-\351\227\255\345\214\205-lifetime-may-not-live-long-enough-\351\227\256\351\242\230/index.html" new file mode 100644 index 00000000..3e34ba99 --- /dev/null +++ "b/2023/05/24/Rust-\351\227\255\345\214\205-lifetime-may-not-live-long-enough-\351\227\256\351\242\230/index.html" @@ -0,0 +1,279 @@ + + + + + + + + + + + + + + + + + + + + + +Rust 闭包 lifetime may not live long enough 问题 | 暮秋小屋 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ +
+ +
+
+
+ + + +
+
+
+ + +
+
+
+ + +
+ +
+ +
+ +
+
+
+

代码:

+
1
2
3
4
5
6
7
8
9
10
11
12
13
...
fn handlers(self) -> crate::server::request::Handlers {
vec![(
"/tree",
routing::get(move || async {
(
StatusCode::OK,
Json(json!(self.clone().tree(self.clone().root))),
)
}),
)]
}
...
+ +

编译错误:

+
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
error: lifetime may not live long enough
--> src/storage/filesystem/mod.rs:45:34
|
45 | routing::get(move || async {
| __________________________-------_^
| | | |
| | | return type of closure `{async block@src/storage/filesystem/mod.rs:45:34: 50:14}` contains a lifetime `'2`
| | lifetime `'1` represents this closure's body
46 | | (
47 | | StatusCode::OK,
48 | | Json(json!(self.clone().tree(self.clone().root))),
49 | | )
50 | | }),
| |_____________^ returning this value requires that `'1` must outlive `'2`
|
= note: closure implements `Fn`, so references to captured variables can't escape the closure
+ +

这是因为 handlers 里面的闭包捕获了一个引用,并且尝试返回一个包含该引用的值导致的。

+

细说就是:闭包内部使用了 self.clone() 来获取一个新的实例,然后在异步块中返回一个 JSON 对象,这个 JSON 对象依赖于 self.tree() 的结果。因为闭包捕获了 self 的引用,所以它必须保证 self 在闭包执行完毕后仍然有效。

+

解决这个问题的思路是:确保闭包中的所有引用都在闭包执行完毕之前就不再被使用。
就是说,要将闭包的作用域限制在一个更短的生命周期内,或者使用其他方式来避免闭包捕获长期存在的引用:

+
1
2
3
4
5
6
7
8
9
10

...
fn handlers(self) -> crate::server::request::Handlers {
let tree = json!(self.clone().tree(self.clone().root));
vec![(
"/tree",
routing::get(move || async { (StatusCode::OK, Json(tree)) }),
)]
}
...
+
+ + + + + +
+ + + + + + + +
+ + +
+
+
+ + + +
+ + +
+
+
+
+
+
+ +
+
+
+
+
+
+
+
+
+ + + + + + + + + -- cgit v1.2.3