| author | |
| committer | |
| log | 6d84caf72771cf05997518ae2fa40a94de709de4 |
| tree | 862543ae6537595ef67ced8d4d06da5384f29699 |
| parent | 7bae6d90648e6ef0782c7f5e8a72066742feacaf |
10 files changed, 2033 insertions(+), 2035 deletions(-)
src/Manifest.zig deleted-564| ... | ... | @@ -1,564 +0,0 @@ |
| 1 | pub const max_bytes = 10 * 1024 * 1024; | |
| 2 | pub const basename = "build.zig.zon"; | |
| 3 | pub const Hash = std.crypto.hash.sha2.Sha256; | |
| 4 | pub const Digest = [Hash.digest_length]u8; | |
| 5 | pub const multihash_len = 1 + 1 + Hash.digest_length; | |
| 6 | pub const multihash_hex_digest_len = 2 * multihash_len; | |
| 7 | pub const MultiHashHexDigest = [multihash_hex_digest_len]u8; | |
| 8 | ||
| 9 | pub const Dependency = struct { | |
| 10 | location: Location, | |
| 11 | location_tok: Ast.TokenIndex, | |
| 12 | hash: ?[]const u8, | |
| 13 | hash_tok: Ast.TokenIndex, | |
| 14 | ||
| 15 | pub const Location = union(enum) { | |
| 16 | url: []const u8, | |
| 17 | path: []const u8, | |
| 18 | }; | |
| 19 | }; | |
| 20 | ||
| 21 | pub const ErrorMessage = struct { | |
| 22 | msg: []const u8, | |
| 23 | tok: Ast.TokenIndex, | |
| 24 | off: u32, | |
| 25 | }; | |
| 26 | ||
| 27 | pub const MultihashFunction = enum(u16) { | |
| 28 | identity = 0x00, | |
| 29 | sha1 = 0x11, | |
| 30 | @"sha2-256" = 0x12, | |
| 31 | @"sha2-512" = 0x13, | |
| 32 | @"sha3-512" = 0x14, | |
| 33 | @"sha3-384" = 0x15, | |
| 34 | @"sha3-256" = 0x16, | |
| 35 | @"sha3-224" = 0x17, | |
| 36 | @"sha2-384" = 0x20, | |
| 37 | @"sha2-256-trunc254-padded" = 0x1012, | |
| 38 | @"sha2-224" = 0x1013, | |
| 39 | @"sha2-512-224" = 0x1014, | |
| 40 | @"sha2-512-256" = 0x1015, | |
| 41 | @"blake2b-256" = 0xb220, | |
| 42 | _, | |
| 43 | }; | |
| 44 | ||
| 45 | pub const multihash_function: MultihashFunction = switch (Hash) { | |
| 46 | std.crypto.hash.sha2.Sha256 => .@"sha2-256", | |
| 47 | else => @compileError("unreachable"), | |
| 48 | }; | |
| 49 | comptime { | |
| 50 | // We avoid unnecessary uleb128 code in hexDigest by asserting here the | |
| 51 | // values are small enough to be contained in the one-byte encoding. | |
| 52 | assert(@intFromEnum(multihash_function) < 127); | |
| 53 | assert(Hash.digest_length < 127); | |
| 54 | } | |
| 55 | ||
| 56 | name: []const u8, | |
| 57 | version: std.SemanticVersion, | |
| 58 | dependencies: std.StringArrayHashMapUnmanaged(Dependency), | |
| 59 | paths: std.StringArrayHashMapUnmanaged(void), | |
| 60 | ||
| 61 | errors: []ErrorMessage, | |
| 62 | arena_state: std.heap.ArenaAllocator.State, | |
| 63 | ||
| 64 | pub const ParseOptions = struct { | |
| 65 | allow_missing_paths_field: bool = false, | |
| 66 | }; | |
| 67 | ||
| 68 | pub const Error = Allocator.Error; | |
| 69 | ||
| 70 | pub fn parse(gpa: Allocator, ast: std.zig.Ast, options: ParseOptions) Error!Manifest { | |
| 71 | const node_tags = ast.nodes.items(.tag); | |
| 72 | const node_datas = ast.nodes.items(.data); | |
| 73 | assert(node_tags[0] == .root); | |
| 74 | const main_node_index = node_datas[0].lhs; | |
| 75 | ||
| 76 | var arena_instance = std.heap.ArenaAllocator.init(gpa); | |
| 77 | errdefer arena_instance.deinit(); | |
| 78 | ||
| 79 | var p: Parse = .{ | |
| 80 | .gpa = gpa, | |
| 81 | .ast = ast, | |
| 82 | .arena = arena_instance.allocator(), | |
| 83 | .errors = .{}, | |
| 84 | ||
| 85 | .name = undefined, | |
| 86 | .version = undefined, | |
| 87 | .dependencies = .{}, | |
| 88 | .paths = .{}, | |
| 89 | .allow_missing_paths_field = options.allow_missing_paths_field, | |
| 90 | .buf = .{}, | |
| 91 | }; | |
| 92 | defer p.buf.deinit(gpa); | |
| 93 | defer p.errors.deinit(gpa); | |
| 94 | defer p.dependencies.deinit(gpa); | |
| 95 | defer p.paths.deinit(gpa); | |
| 96 | ||
| 97 | p.parseRoot(main_node_index) catch |err| switch (err) { | |
| 98 | error.ParseFailure => assert(p.errors.items.len > 0), | |
| 99 | else => |e| return e, | |
| 100 | }; | |
| 101 | ||
| 102 | return .{ | |
| 103 | .name = p.name, | |
| 104 | .version = p.version, | |
| 105 | .dependencies = try p.dependencies.clone(p.arena), | |
| 106 | .paths = try p.paths.clone(p.arena), | |
| 107 | .errors = try p.arena.dupe(ErrorMessage, p.errors.items), | |
| 108 | .arena_state = arena_instance.state, | |
| 109 | }; | |
| 110 | } | |
| 111 | ||
| 112 | pub fn deinit(man: *Manifest, gpa: Allocator) void { | |
| 113 | man.arena_state.promote(gpa).deinit(); | |
| 114 | man.* = undefined; | |
| 115 | } | |
| 116 | ||
| 117 | const hex_charset = "0123456789abcdef"; | |
| 118 | ||
| 119 | pub fn hex64(x: u64) [16]u8 { | |
| 120 | var result: [16]u8 = undefined; | |
| 121 | var i: usize = 0; | |
| 122 | while (i < 8) : (i += 1) { | |
| 123 | const byte = @as(u8, @truncate(x >> @as(u6, @intCast(8 * i)))); | |
| 124 | result[i * 2 + 0] = hex_charset[byte >> 4]; | |
| 125 | result[i * 2 + 1] = hex_charset[byte & 15]; | |
| 126 | } | |
| 127 | return result; | |
| 128 | } | |
| 129 | ||
| 130 | test hex64 { | |
| 131 | const s = "[" ++ hex64(0x12345678_abcdef00) ++ "]"; | |
| 132 | try std.testing.expectEqualStrings("[00efcdab78563412]", s); | |
| 133 | } | |
| 134 | ||
| 135 | pub fn hexDigest(digest: Digest) MultiHashHexDigest { | |
| 136 | var result: MultiHashHexDigest = undefined; | |
| 137 | ||
| 138 | result[0] = hex_charset[@intFromEnum(multihash_function) >> 4]; | |
| 139 | result[1] = hex_charset[@intFromEnum(multihash_function) & 15]; | |
| 140 | ||
| 141 | result[2] = hex_charset[Hash.digest_length >> 4]; | |
| 142 | result[3] = hex_charset[Hash.digest_length & 15]; | |
| 143 | ||
| 144 | for (digest, 0..) |byte, i| { | |
| 145 | result[4 + i * 2] = hex_charset[byte >> 4]; | |
| 146 | result[5 + i * 2] = hex_charset[byte & 15]; | |
| 147 | } | |
| 148 | return result; | |
| 149 | } | |
| 150 | ||
| 151 | const Parse = struct { | |
| 152 | gpa: Allocator, | |
| 153 | ast: std.zig.Ast, | |
| 154 | arena: Allocator, | |
| 155 | buf: std.ArrayListUnmanaged(u8), | |
| 156 | errors: std.ArrayListUnmanaged(ErrorMessage), | |
| 157 | ||
| 158 | name: []const u8, | |
| 159 | version: std.SemanticVersion, | |
| 160 | dependencies: std.StringArrayHashMapUnmanaged(Dependency), | |
| 161 | paths: std.StringArrayHashMapUnmanaged(void), | |
| 162 | allow_missing_paths_field: bool, | |
| 163 | ||
| 164 | const InnerError = error{ ParseFailure, OutOfMemory }; | |
| 165 | ||
| 166 | fn parseRoot(p: *Parse, node: Ast.Node.Index) !void { | |
| 167 | const ast = p.ast; | |
| 168 | const main_tokens = ast.nodes.items(.main_token); | |
| 169 | const main_token = main_tokens[node]; | |
| 170 | ||
| 171 | var buf: [2]Ast.Node.Index = undefined; | |
| 172 | const struct_init = ast.fullStructInit(&buf, node) orelse { | |
| 173 | return fail(p, main_token, "expected top level expression to be a struct", .{}); | |
| 174 | }; | |
| 175 | ||
| 176 | var have_name = false; | |
| 177 | var have_version = false; | |
| 178 | var have_included_paths = false; | |
| 179 | ||
| 180 | for (struct_init.ast.fields) |field_init| { | |
| 181 | const name_token = ast.firstToken(field_init) - 2; | |
| 182 | const field_name = try identifierTokenString(p, name_token); | |
| 183 | // We could get fancy with reflection and comptime logic here but doing | |
| 184 | // things manually provides an opportunity to do any additional verification | |
| 185 | // that is desirable on a per-field basis. | |
| 186 | if (mem.eql(u8, field_name, "dependencies")) { | |
| 187 | try parseDependencies(p, field_init); | |
| 188 | } else if (mem.eql(u8, field_name, "paths")) { | |
| 189 | have_included_paths = true; | |
| 190 | try parseIncludedPaths(p, field_init); | |
| 191 | } else if (mem.eql(u8, field_name, "name")) { | |
| 192 | p.name = try parseString(p, field_init); | |
| 193 | have_name = true; | |
| 194 | } else if (mem.eql(u8, field_name, "version")) { | |
| 195 | const version_text = try parseString(p, field_init); | |
| 196 | p.version = std.SemanticVersion.parse(version_text) catch |err| v: { | |
| 197 | try appendError(p, main_tokens[field_init], "unable to parse semantic version: {s}", .{@errorName(err)}); | |
| 198 | break :v undefined; | |
| 199 | }; | |
| 200 | have_version = true; | |
| 201 | } else { | |
| 202 | // Ignore unknown fields so that we can add fields in future zig | |
| 203 | // versions without breaking older zig versions. | |
| 204 | } | |
| 205 | } | |
| 206 | ||
| 207 | if (!have_name) { | |
| 208 | try appendError(p, main_token, "missing top-level 'name' field", .{}); | |
| 209 | } | |
| 210 | ||
| 211 | if (!have_version) { | |
| 212 | try appendError(p, main_token, "missing top-level 'version' field", .{}); | |
| 213 | } | |
| 214 | ||
| 215 | if (!have_included_paths) { | |
| 216 | if (p.allow_missing_paths_field) { | |
| 217 | try p.paths.put(p.gpa, "", {}); | |
| 218 | } else { | |
| 219 | try appendError(p, main_token, "missing top-level 'paths' field", .{}); | |
| 220 | } | |
| 221 | } | |
| 222 | } | |
| 223 | ||
| 224 | fn parseDependencies(p: *Parse, node: Ast.Node.Index) !void { | |
| 225 | const ast = p.ast; | |
| 226 | const main_tokens = ast.nodes.items(.main_token); | |
| 227 | ||
| 228 | var buf: [2]Ast.Node.Index = undefined; | |
| 229 | const struct_init = ast.fullStructInit(&buf, node) orelse { | |
| 230 | const tok = main_tokens[node]; | |
| 231 | return fail(p, tok, "expected dependencies expression to be a struct", .{}); | |
| 232 | }; | |
| 233 | ||
| 234 | for (struct_init.ast.fields) |field_init| { | |
| 235 | const name_token = ast.firstToken(field_init) - 2; | |
| 236 | const dep_name = try identifierTokenString(p, name_token); | |
| 237 | const dep = try parseDependency(p, field_init); | |
| 238 | try p.dependencies.put(p.gpa, dep_name, dep); | |
| 239 | } | |
| 240 | } | |
| 241 | ||
| 242 | fn parseDependency(p: *Parse, node: Ast.Node.Index) !Dependency { | |
| 243 | const ast = p.ast; | |
| 244 | const main_tokens = ast.nodes.items(.main_token); | |
| 245 | ||
| 246 | var buf: [2]Ast.Node.Index = undefined; | |
| 247 | const struct_init = ast.fullStructInit(&buf, node) orelse { | |
| 248 | const tok = main_tokens[node]; | |
| 249 | return fail(p, tok, "expected dependency expression to be a struct", .{}); | |
| 250 | }; | |
| 251 | ||
| 252 | var dep: Dependency = .{ | |
| 253 | .location = undefined, | |
| 254 | .location_tok = 0, | |
| 255 | .hash = null, | |
| 256 | .hash_tok = 0, | |
| 257 | }; | |
| 258 | var has_location = false; | |
| 259 | ||
| 260 | for (struct_init.ast.fields) |field_init| { | |
| 261 | const name_token = ast.firstToken(field_init) - 2; | |
| 262 | const field_name = try identifierTokenString(p, name_token); | |
| 263 | // We could get fancy with reflection and comptime logic here but doing | |
| 264 | // things manually provides an opportunity to do any additional verification | |
| 265 | // that is desirable on a per-field basis. | |
| 266 | if (mem.eql(u8, field_name, "url")) { | |
| 267 | if (has_location) { | |
| 268 | return fail(p, main_tokens[field_init], "dependency should specify only one of 'url' and 'path' fields.", .{}); | |
| 269 | } | |
| 270 | dep.location = .{ | |
| 271 | .url = parseString(p, field_init) catch |err| switch (err) { | |
| 272 | error.ParseFailure => continue, | |
| 273 | else => |e| return e, | |
| 274 | }, | |
| 275 | }; | |
| 276 | has_location = true; | |
| 277 | dep.location_tok = main_tokens[field_init]; | |
| 278 | } else if (mem.eql(u8, field_name, "path")) { | |
| 279 | if (has_location) { | |
| 280 | return fail(p, main_tokens[field_init], "dependency should specify only one of 'url' and 'path' fields.", .{}); | |
| 281 | } | |
| 282 | dep.location = .{ | |
| 283 | .path = parseString(p, field_init) catch |err| switch (err) { | |
| 284 | error.ParseFailure => continue, | |
| 285 | else => |e| return e, | |
| 286 | }, | |
| 287 | }; | |
| 288 | has_location = true; | |
| 289 | dep.location_tok = main_tokens[field_init]; | |
| 290 | } else if (mem.eql(u8, field_name, "hash")) { | |
| 291 | dep.hash = parseHash(p, field_init) catch |err| switch (err) { | |
| 292 | error.ParseFailure => continue, | |
| 293 | else => |e| return e, | |
| 294 | }; | |
| 295 | dep.hash_tok = main_tokens[field_init]; | |
| 296 | } else { | |
| 297 | // Ignore unknown fields so that we can add fields in future zig | |
| 298 | // versions without breaking older zig versions. | |
| 299 | } | |
| 300 | } | |
| 301 | ||
| 302 | if (!has_location) { | |
| 303 | try appendError(p, main_tokens[node], "dependency requires location field, one of 'url' or 'path'.", .{}); | |
| 304 | } | |
| 305 | ||
| 306 | return dep; | |
| 307 | } | |
| 308 | ||
| 309 | fn parseIncludedPaths(p: *Parse, node: Ast.Node.Index) !void { | |
| 310 | const ast = p.ast; | |
| 311 | const main_tokens = ast.nodes.items(.main_token); | |
| 312 | ||
| 313 | var buf: [2]Ast.Node.Index = undefined; | |
| 314 | const array_init = ast.fullArrayInit(&buf, node) orelse { | |
| 315 | const tok = main_tokens[node]; | |
| 316 | return fail(p, tok, "expected paths expression to be a struct", .{}); | |
| 317 | }; | |
| 318 | ||
| 319 | for (array_init.ast.elements) |elem_node| { | |
| 320 | const path_string = try parseString(p, elem_node); | |
| 321 | const normalized = try std.fs.path.resolve(p.arena, &.{path_string}); | |
| 322 | try p.paths.put(p.gpa, normalized, {}); | |
| 323 | } | |
| 324 | } | |
| 325 | ||
| 326 | fn parseString(p: *Parse, node: Ast.Node.Index) ![]const u8 { | |
| 327 | const ast = p.ast; | |
| 328 | const node_tags = ast.nodes.items(.tag); | |
| 329 | const main_tokens = ast.nodes.items(.main_token); | |
| 330 | if (node_tags[node] != .string_literal) { | |
| 331 | return fail(p, main_tokens[node], "expected string literal", .{}); | |
| 332 | } | |
| 333 | const str_lit_token = main_tokens[node]; | |
| 334 | const token_bytes = ast.tokenSlice(str_lit_token); | |
| 335 | p.buf.clearRetainingCapacity(); | |
| 336 | try parseStrLit(p, str_lit_token, &p.buf, token_bytes, 0); | |
| 337 | const duped = try p.arena.dupe(u8, p.buf.items); | |
| 338 | return duped; | |
| 339 | } | |
| 340 | ||
| 341 | fn parseHash(p: *Parse, node: Ast.Node.Index) ![]const u8 { | |
| 342 | const ast = p.ast; | |
| 343 | const main_tokens = ast.nodes.items(.main_token); | |
| 344 | const tok = main_tokens[node]; | |
| 345 | const h = try parseString(p, node); | |
| 346 | ||
| 347 | if (h.len >= 2) { | |
| 348 | const their_multihash_func = std.fmt.parseInt(u8, h[0..2], 16) catch |err| { | |
| 349 | return fail(p, tok, "invalid multihash value: unable to parse hash function: {s}", .{ | |
| 350 | @errorName(err), | |
| 351 | }); | |
| 352 | }; | |
| 353 | if (@as(MultihashFunction, @enumFromInt(their_multihash_func)) != multihash_function) { | |
| 354 | return fail(p, tok, "unsupported hash function: only sha2-256 is supported", .{}); | |
| 355 | } | |
| 356 | } | |
| 357 | ||
| 358 | if (h.len != multihash_hex_digest_len) { | |
| 359 | return fail(p, tok, "wrong hash size. expected: {d}, found: {d}", .{ | |
| 360 | multihash_hex_digest_len, h.len, | |
| 361 | }); | |
| 362 | } | |
| 363 | ||
| 364 | return h; | |
| 365 | } | |
| 366 | ||
| 367 | /// TODO: try to DRY this with AstGen.identifierTokenString | |
| 368 | fn identifierTokenString(p: *Parse, token: Ast.TokenIndex) InnerError![]const u8 { | |
| 369 | const ast = p.ast; | |
| 370 | const token_tags = ast.tokens.items(.tag); | |
| 371 | assert(token_tags[token] == .identifier); | |
| 372 | const ident_name = ast.tokenSlice(token); | |
| 373 | if (!mem.startsWith(u8, ident_name, "@")) { | |
| 374 | return ident_name; | |
| 375 | } | |
| 376 | p.buf.clearRetainingCapacity(); | |
| 377 | try parseStrLit(p, token, &p.buf, ident_name, 1); | |
| 378 | const duped = try p.arena.dupe(u8, p.buf.items); | |
| 379 | return duped; | |
| 380 | } | |
| 381 | ||
| 382 | /// TODO: try to DRY this with AstGen.parseStrLit | |
| 383 | fn parseStrLit( | |
| 384 | p: *Parse, | |
| 385 | token: Ast.TokenIndex, | |
| 386 | buf: *std.ArrayListUnmanaged(u8), | |
| 387 | bytes: []const u8, | |
| 388 | offset: u32, | |
| 389 | ) InnerError!void { | |
| 390 | const raw_string = bytes[offset..]; | |
| 391 | var buf_managed = buf.toManaged(p.gpa); | |
| 392 | const result = std.zig.string_literal.parseWrite(buf_managed.writer(), raw_string); | |
| 393 | buf.* = buf_managed.moveToUnmanaged(); | |
| 394 | switch (try result) { | |
| 395 | .success => {}, | |
| 396 | .failure => |err| try p.appendStrLitError(err, token, bytes, offset), | |
| 397 | } | |
| 398 | } | |
| 399 | ||
| 400 | /// TODO: try to DRY this with AstGen.failWithStrLitError | |
| 401 | fn appendStrLitError( | |
| 402 | p: *Parse, | |
| 403 | err: std.zig.string_literal.Error, | |
| 404 | token: Ast.TokenIndex, | |
| 405 | bytes: []const u8, | |
| 406 | offset: u32, | |
| 407 | ) Allocator.Error!void { | |
| 408 | const raw_string = bytes[offset..]; | |
| 409 | switch (err) { | |
| 410 | .invalid_escape_character => |bad_index| { | |
| 411 | try p.appendErrorOff( | |
| 412 | token, | |
| 413 | offset + @as(u32, @intCast(bad_index)), | |
| 414 | "invalid escape character: '{c}'", | |
| 415 | .{raw_string[bad_index]}, | |
| 416 | ); | |
| 417 | }, | |
| 418 | .expected_hex_digit => |bad_index| { | |
| 419 | try p.appendErrorOff( | |
| 420 | token, | |
| 421 | offset + @as(u32, @intCast(bad_index)), | |
| 422 | "expected hex digit, found '{c}'", | |
| 423 | .{raw_string[bad_index]}, | |
| 424 | ); | |
| 425 | }, | |
| 426 | .empty_unicode_escape_sequence => |bad_index| { | |
| 427 | try p.appendErrorOff( | |
| 428 | token, | |
| 429 | offset + @as(u32, @intCast(bad_index)), | |
| 430 | "empty unicode escape sequence", | |
| 431 | .{}, | |
| 432 | ); | |
| 433 | }, | |
| 434 | .expected_hex_digit_or_rbrace => |bad_index| { | |
| 435 | try p.appendErrorOff( | |
| 436 | token, | |
| 437 | offset + @as(u32, @intCast(bad_index)), | |
| 438 | "expected hex digit or '}}', found '{c}'", | |
| 439 | .{raw_string[bad_index]}, | |
| 440 | ); | |
| 441 | }, | |
| 442 | .invalid_unicode_codepoint => |bad_index| { | |
| 443 | try p.appendErrorOff( | |
| 444 | token, | |
| 445 | offset + @as(u32, @intCast(bad_index)), | |
| 446 | "unicode escape does not correspond to a valid codepoint", | |
| 447 | .{}, | |
| 448 | ); | |
| 449 | }, | |
| 450 | .expected_lbrace => |bad_index| { | |
| 451 | try p.appendErrorOff( | |
| 452 | token, | |
| 453 | offset + @as(u32, @intCast(bad_index)), | |
| 454 | "expected '{{', found '{c}", | |
| 455 | .{raw_string[bad_index]}, | |
| 456 | ); | |
| 457 | }, | |
| 458 | .expected_rbrace => |bad_index| { | |
| 459 | try p.appendErrorOff( | |
| 460 | token, | |
| 461 | offset + @as(u32, @intCast(bad_index)), | |
| 462 | "expected '}}', found '{c}", | |
| 463 | .{raw_string[bad_index]}, | |
| 464 | ); | |
| 465 | }, | |
| 466 | .expected_single_quote => |bad_index| { | |
| 467 | try p.appendErrorOff( | |
| 468 | token, | |
| 469 | offset + @as(u32, @intCast(bad_index)), | |
| 470 | "expected single quote ('), found '{c}", | |
| 471 | .{raw_string[bad_index]}, | |
| 472 | ); | |
| 473 | }, | |
| 474 | .invalid_character => |bad_index| { | |
| 475 | try p.appendErrorOff( | |
| 476 | token, | |
| 477 | offset + @as(u32, @intCast(bad_index)), | |
| 478 | "invalid byte in string or character literal: '{c}'", | |
| 479 | .{raw_string[bad_index]}, | |
| 480 | ); | |
| 481 | }, | |
| 482 | } | |
| 483 | } | |
| 484 | ||
| 485 | fn fail( | |
| 486 | p: *Parse, | |
| 487 | tok: Ast.TokenIndex, | |
| 488 | comptime fmt: []const u8, | |
| 489 | args: anytype, | |
| 490 | ) InnerError { | |
| 491 | try appendError(p, tok, fmt, args); | |
| 492 | return error.ParseFailure; | |
| 493 | } | |
| 494 | ||
| 495 | fn appendError(p: *Parse, tok: Ast.TokenIndex, comptime fmt: []const u8, args: anytype) !void { | |
| 496 | return appendErrorOff(p, tok, 0, fmt, args); | |
| 497 | } | |
| 498 | ||
| 499 | fn appendErrorOff( | |
| 500 | p: *Parse, | |
| 501 | tok: Ast.TokenIndex, | |
| 502 | byte_offset: u32, | |
| 503 | comptime fmt: []const u8, | |
| 504 | args: anytype, | |
| 505 | ) Allocator.Error!void { | |
| 506 | try p.errors.append(p.gpa, .{ | |
| 507 | .msg = try std.fmt.allocPrint(p.arena, fmt, args), | |
| 508 | .tok = tok, | |
| 509 | .off = byte_offset, | |
| 510 | }); | |
| 511 | } | |
| 512 | }; | |
| 513 | ||
| 514 | const Manifest = @This(); | |
| 515 | const std = @import("std"); | |
| 516 | const mem = std.mem; | |
| 517 | const Allocator = std.mem.Allocator; | |
| 518 | const assert = std.debug.assert; | |
| 519 | const Ast = std.zig.Ast; | |
| 520 | const testing = std.testing; | |
| 521 | ||
| 522 | test "basic" { | |
| 523 | const gpa = testing.allocator; | |
| 524 | ||
| 525 | const example = | |
| 526 | \\.{ | |
| 527 | \\ .name = "foo", | |
| 528 | \\ .version = "3.2.1", | |
| 529 | \\ .dependencies = .{ | |
| 530 | \\ .bar = .{ | |
| 531 | \\ .url = "https://example.com/baz.tar.gz", | |
| 532 | \\ .hash = "1220f1b680b6065fcfc94fe777f22e73bcb7e2767e5f4d99d4255fe76ded69c7a35f", | |
| 533 | \\ }, | |
| 534 | \\ }, | |
| 535 | \\} | |
| 536 | ; | |
| 537 | ||
| 538 | var ast = try std.zig.Ast.parse(gpa, example, .zon); | |
| 539 | defer ast.deinit(gpa); | |
| 540 | ||
| 541 | try testing.expect(ast.errors.len == 0); | |
| 542 | ||
| 543 | var manifest = try Manifest.parse(gpa, ast); | |
| 544 | defer manifest.deinit(gpa); | |
| 545 | ||
| 546 | try testing.expectEqualStrings("foo", manifest.name); | |
| 547 | ||
| 548 | try testing.expectEqual(@as(std.SemanticVersion, .{ | |
| 549 | .major = 3, | |
| 550 | .minor = 2, | |
| 551 | .patch = 1, | |
| 552 | }), manifest.version); | |
| 553 | ||
| 554 | try testing.expect(manifest.dependencies.count() == 1); | |
| 555 | try testing.expectEqualStrings("bar", manifest.dependencies.keys()[0]); | |
| 556 | try testing.expectEqualStrings( | |
| 557 | "https://example.com/baz.tar.gz", | |
| 558 | manifest.dependencies.values()[0].url, | |
| 559 | ); | |
| 560 | try testing.expectEqualStrings( | |
| 561 | "1220f1b680b6065fcfc94fe777f22e73bcb7e2767e5f4d99d4255fe76ded69c7a35f", | |
| 562 | manifest.dependencies.values()[0].hash orelse return error.TestFailed, | |
| 563 | ); | |
| 564 | } |
src/Package.zig+1-1| ... | ... | @@ -1,7 +1,7 @@ |
| 1 | 1 | pub const Module = @import("Package/Module.zig"); |
| 2 | 2 | pub const Fetch = @import("Package/Fetch.zig"); |
| 3 | 3 | pub const build_zig_basename = "build.zig"; |
| 4 | pub const Manifest = @import("Manifest.zig"); | |
| 4 | pub const Manifest = @import("Package/Manifest.zig"); | |
| 5 | 5 | |
| 6 | 6 | pub const Path = struct { |
| 7 | 7 | root_dir: Cache.Directory, |
src/Package/Fetch.zig+2-2| ... | ... | @@ -1449,9 +1449,9 @@ const Allocator = std.mem.Allocator; |
| 1449 | 1449 | const Cache = std.Build.Cache; |
| 1450 | 1450 | const ThreadPool = std.Thread.Pool; |
| 1451 | 1451 | const WaitGroup = std.Thread.WaitGroup; |
| 1452 | const Manifest = @import("../Manifest.zig"); | |
| 1453 | 1452 | const Fetch = @This(); |
| 1454 | 1453 | const main = @import("../main.zig"); |
| 1455 | const git = @import("../git.zig"); | |
| 1454 | const git = @import("Fetch/git.zig"); | |
| 1456 | 1455 | const Package = @import("../Package.zig"); |
| 1456 | const Manifest = Package.Manifest; | |
| 1457 | 1457 | const ErrorBundle = std.zig.ErrorBundle; |
src/Package/Fetch/git.zig created+1466| ... | ... | @@ -0,0 +1,1466 @@ |
| 1 | //! Git support for package fetching. | |
| 2 | //! | |
| 3 | //! This is not intended to support all features of Git: it is limited to the | |
| 4 | //! basic functionality needed to clone a repository for the purpose of fetching | |
| 5 | //! a package. | |
| 6 | ||
| 7 | const std = @import("std"); | |
| 8 | const mem = std.mem; | |
| 9 | const testing = std.testing; | |
| 10 | const Allocator = mem.Allocator; | |
| 11 | const Sha1 = std.crypto.hash.Sha1; | |
| 12 | const assert = std.debug.assert; | |
| 13 | ||
| 14 | pub const oid_length = Sha1.digest_length; | |
| 15 | pub const fmt_oid_length = 2 * oid_length; | |
| 16 | /// The ID of a Git object (an SHA-1 hash). | |
| 17 | pub const Oid = [oid_length]u8; | |
| 18 | ||
| 19 | pub fn parseOid(s: []const u8) !Oid { | |
| 20 | if (s.len != fmt_oid_length) return error.InvalidOid; | |
| 21 | var oid: Oid = undefined; | |
| 22 | for (&oid, 0..) |*b, i| { | |
| 23 | b.* = std.fmt.parseUnsigned(u8, s[2 * i ..][0..2], 16) catch return error.InvalidOid; | |
| 24 | } | |
| 25 | return oid; | |
| 26 | } | |
| 27 | ||
| 28 | test parseOid { | |
| 29 | try testing.expectEqualSlices( | |
| 30 | u8, | |
| 31 | &.{ 0xCE, 0x91, 0x9C, 0xCF, 0x45, 0x95, 0x18, 0x56, 0xA7, 0x62, 0xFF, 0xDB, 0x8E, 0xF8, 0x50, 0x30, 0x1C, 0xD8, 0xC5, 0x88 }, | |
| 32 | &try parseOid("ce919ccf45951856a762ffdb8ef850301cd8c588"), | |
| 33 | ); | |
| 34 | try testing.expectError(error.InvalidOid, parseOid("ce919ccf")); | |
| 35 | try testing.expectError(error.InvalidOid, parseOid("master")); | |
| 36 | try testing.expectError(error.InvalidOid, parseOid("HEAD")); | |
| 37 | } | |
| 38 | ||
| 39 | pub const Diagnostics = struct { | |
| 40 | allocator: Allocator, | |
| 41 | errors: std.ArrayListUnmanaged(Error) = .{}, | |
| 42 | ||
| 43 | pub const Error = union(enum) { | |
| 44 | unable_to_create_sym_link: struct { | |
| 45 | code: anyerror, | |
| 46 | file_name: []const u8, | |
| 47 | link_name: []const u8, | |
| 48 | }, | |
| 49 | }; | |
| 50 | ||
| 51 | pub fn deinit(d: *Diagnostics) void { | |
| 52 | for (d.errors.items) |item| { | |
| 53 | switch (item) { | |
| 54 | .unable_to_create_sym_link => |info| { | |
| 55 | d.allocator.free(info.file_name); | |
| 56 | d.allocator.free(info.link_name); | |
| 57 | }, | |
| 58 | } | |
| 59 | } | |
| 60 | d.errors.deinit(d.allocator); | |
| 61 | d.* = undefined; | |
| 62 | } | |
| 63 | }; | |
| 64 | ||
| 65 | pub const Repository = struct { | |
| 66 | odb: Odb, | |
| 67 | ||
| 68 | pub fn init(allocator: Allocator, pack_file: std.fs.File, index_file: std.fs.File) !Repository { | |
| 69 | return .{ .odb = try Odb.init(allocator, pack_file, index_file) }; | |
| 70 | } | |
| 71 | ||
| 72 | pub fn deinit(repository: *Repository) void { | |
| 73 | repository.odb.deinit(); | |
| 74 | repository.* = undefined; | |
| 75 | } | |
| 76 | ||
| 77 | /// Checks out the repository at `commit_oid` to `worktree`. | |
| 78 | pub fn checkout( | |
| 79 | repository: *Repository, | |
| 80 | worktree: std.fs.Dir, | |
| 81 | commit_oid: Oid, | |
| 82 | diagnostics: *Diagnostics, | |
| 83 | ) !void { | |
| 84 | try repository.odb.seekOid(commit_oid); | |
| 85 | const tree_oid = tree_oid: { | |
| 86 | var commit_object = try repository.odb.readObject(); | |
| 87 | if (commit_object.type != .commit) return error.NotACommit; | |
| 88 | break :tree_oid try getCommitTree(commit_object.data); | |
| 89 | }; | |
| 90 | try repository.checkoutTree(worktree, tree_oid, "", diagnostics); | |
| 91 | } | |
| 92 | ||
| 93 | /// Checks out the tree at `tree_oid` to `worktree`. | |
| 94 | fn checkoutTree( | |
| 95 | repository: *Repository, | |
| 96 | dir: std.fs.Dir, | |
| 97 | tree_oid: Oid, | |
| 98 | current_path: []const u8, | |
| 99 | diagnostics: *Diagnostics, | |
| 100 | ) !void { | |
| 101 | try repository.odb.seekOid(tree_oid); | |
| 102 | const tree_object = try repository.odb.readObject(); | |
| 103 | if (tree_object.type != .tree) return error.NotATree; | |
| 104 | // The tree object may be evicted from the object cache while we're | |
| 105 | // iterating over it, so we can make a defensive copy here to make sure | |
| 106 | // it remains valid until we're done with it | |
| 107 | const tree_data = try repository.odb.allocator.dupe(u8, tree_object.data); | |
| 108 | defer repository.odb.allocator.free(tree_data); | |
| 109 | ||
| 110 | var tree_iter: TreeIterator = .{ .data = tree_data }; | |
| 111 | while (try tree_iter.next()) |entry| { | |
| 112 | switch (entry.type) { | |
| 113 | .directory => { | |
| 114 | try dir.makeDir(entry.name); | |
| 115 | var subdir = try dir.openDir(entry.name, .{}); | |
| 116 | defer subdir.close(); | |
| 117 | const sub_path = try std.fs.path.join(repository.odb.allocator, &.{ current_path, entry.name }); | |
| 118 | defer repository.odb.allocator.free(sub_path); | |
| 119 | try repository.checkoutTree(subdir, entry.oid, sub_path, diagnostics); | |
| 120 | }, | |
| 121 | .file => { | |
| 122 | var file = try dir.createFile(entry.name, .{}); | |
| 123 | defer file.close(); | |
| 124 | try repository.odb.seekOid(entry.oid); | |
| 125 | var file_object = try repository.odb.readObject(); | |
| 126 | if (file_object.type != .blob) return error.InvalidFile; | |
| 127 | try file.writeAll(file_object.data); | |
| 128 | try file.sync(); | |
| 129 | }, | |
| 130 | .symlink => { | |
| 131 | try repository.odb.seekOid(entry.oid); | |
| 132 | var symlink_object = try repository.odb.readObject(); | |
| 133 | if (symlink_object.type != .blob) return error.InvalidFile; | |
| 134 | const link_name = symlink_object.data; | |
| 135 | dir.symLink(link_name, entry.name, .{}) catch |e| { | |
| 136 | const file_name = try std.fs.path.join(diagnostics.allocator, &.{ current_path, entry.name }); | |
| 137 | errdefer diagnostics.allocator.free(file_name); | |
| 138 | const link_name_dup = try diagnostics.allocator.dupe(u8, link_name); | |
| 139 | errdefer diagnostics.allocator.free(link_name_dup); | |
| 140 | try diagnostics.errors.append(diagnostics.allocator, .{ .unable_to_create_sym_link = .{ | |
| 141 | .code = e, | |
| 142 | .file_name = file_name, | |
| 143 | .link_name = link_name_dup, | |
| 144 | } }); | |
| 145 | }; | |
| 146 | }, | |
| 147 | .gitlink => { | |
| 148 | // Consistent with git archive behavior, create the directory but | |
| 149 | // do nothing else | |
| 150 | try dir.makeDir(entry.name); | |
| 151 | }, | |
| 152 | } | |
| 153 | } | |
| 154 | } | |
| 155 | ||
| 156 | /// Returns the ID of the tree associated with the given commit (provided as | |
| 157 | /// raw object data). | |
| 158 | fn getCommitTree(commit_data: []const u8) !Oid { | |
| 159 | if (!mem.startsWith(u8, commit_data, "tree ") or | |
| 160 | commit_data.len < "tree ".len + fmt_oid_length + "\n".len or | |
| 161 | commit_data["tree ".len + fmt_oid_length] != '\n') | |
| 162 | { | |
| 163 | return error.InvalidCommit; | |
| 164 | } | |
| 165 | return try parseOid(commit_data["tree ".len..][0..fmt_oid_length]); | |
| 166 | } | |
| 167 | ||
| 168 | const TreeIterator = struct { | |
| 169 | data: []const u8, | |
| 170 | pos: usize = 0, | |
| 171 | ||
| 172 | const Entry = struct { | |
| 173 | type: Type, | |
| 174 | executable: bool, | |
| 175 | name: [:0]const u8, | |
| 176 | oid: Oid, | |
| 177 | ||
| 178 | const Type = enum(u4) { | |
| 179 | directory = 0o4, | |
| 180 | file = 0o10, | |
| 181 | symlink = 0o12, | |
| 182 | gitlink = 0o16, | |
| 183 | }; | |
| 184 | }; | |
| 185 | ||
| 186 | fn next(iterator: *TreeIterator) !?Entry { | |
| 187 | if (iterator.pos == iterator.data.len) return null; | |
| 188 | ||
| 189 | const mode_end = mem.indexOfScalarPos(u8, iterator.data, iterator.pos, ' ') orelse return error.InvalidTree; | |
| 190 | const mode: packed struct { | |
| 191 | permission: u9, | |
| 192 | unused: u3, | |
| 193 | type: u4, | |
| 194 | } = @bitCast(std.fmt.parseUnsigned(u16, iterator.data[iterator.pos..mode_end], 8) catch return error.InvalidTree); | |
| 195 | const @"type" = std.meta.intToEnum(Entry.Type, mode.type) catch return error.InvalidTree; | |
| 196 | const executable = switch (mode.permission) { | |
| 197 | 0 => if (@"type" == .file) return error.InvalidTree else false, | |
| 198 | 0o644 => if (@"type" != .file) return error.InvalidTree else false, | |
| 199 | 0o755 => if (@"type" != .file) return error.InvalidTree else true, | |
| 200 | else => return error.InvalidTree, | |
| 201 | }; | |
| 202 | iterator.pos = mode_end + 1; | |
| 203 | ||
| 204 | const name_end = mem.indexOfScalarPos(u8, iterator.data, iterator.pos, 0) orelse return error.InvalidTree; | |
| 205 | const name = iterator.data[iterator.pos..name_end :0]; | |
| 206 | iterator.pos = name_end + 1; | |
| 207 | ||
| 208 | if (iterator.pos + oid_length > iterator.data.len) return error.InvalidTree; | |
| 209 | const oid = iterator.data[iterator.pos..][0..oid_length].*; | |
| 210 | iterator.pos += oid_length; | |
| 211 | ||
| 212 | return .{ .type = @"type", .executable = executable, .name = name, .oid = oid }; | |
| 213 | } | |
| 214 | }; | |
| 215 | }; | |
| 216 | ||
| 217 | /// A Git object database backed by a packfile. A packfile index is also used | |
| 218 | /// for efficient access to objects in the packfile. | |
| 219 | /// | |
| 220 | /// The format of the packfile and its associated index are documented in | |
| 221 | /// [pack-format](https://git-scm.com/docs/pack-format). | |
| 222 | const Odb = struct { | |
| 223 | pack_file: std.fs.File, | |
| 224 | index_header: IndexHeader, | |
| 225 | index_file: std.fs.File, | |
| 226 | cache: ObjectCache = .{}, | |
| 227 | allocator: Allocator, | |
| 228 | ||
| 229 | /// Initializes the database from open pack and index files. | |
| 230 | fn init(allocator: Allocator, pack_file: std.fs.File, index_file: std.fs.File) !Odb { | |
| 231 | try pack_file.seekTo(0); | |
| 232 | try index_file.seekTo(0); | |
| 233 | const index_header = try IndexHeader.read(index_file.reader()); | |
| 234 | return .{ | |
| 235 | .pack_file = pack_file, | |
| 236 | .index_header = index_header, | |
| 237 | .index_file = index_file, | |
| 238 | .allocator = allocator, | |
| 239 | }; | |
| 240 | } | |
| 241 | ||
| 242 | fn deinit(odb: *Odb) void { | |
| 243 | odb.cache.deinit(odb.allocator); | |
| 244 | odb.* = undefined; | |
| 245 | } | |
| 246 | ||
| 247 | /// Reads the object at the current position in the database. | |
| 248 | fn readObject(odb: *Odb) !Object { | |
| 249 | var base_offset = try odb.pack_file.getPos(); | |
| 250 | var base_header: EntryHeader = undefined; | |
| 251 | var delta_offsets = std.ArrayListUnmanaged(u64){}; | |
| 252 | defer delta_offsets.deinit(odb.allocator); | |
| 253 | const base_object = while (true) { | |
| 254 | if (odb.cache.get(base_offset)) |base_object| break base_object; | |
| 255 | ||
| 256 | base_header = try EntryHeader.read(odb.pack_file.reader()); | |
| 257 | switch (base_header) { | |
| 258 | .ofs_delta => |ofs_delta| { | |
| 259 | try delta_offsets.append(odb.allocator, base_offset); | |
| 260 | base_offset = std.math.sub(u64, base_offset, ofs_delta.offset) catch return error.InvalidFormat; | |
| 261 | try odb.pack_file.seekTo(base_offset); | |
| 262 | }, | |
| 263 | .ref_delta => |ref_delta| { | |
| 264 | try delta_offsets.append(odb.allocator, base_offset); | |
| 265 | try odb.seekOid(ref_delta.base_object); | |
| 266 | base_offset = try odb.pack_file.getPos(); | |
| 267 | }, | |
| 268 | else => { | |
| 269 | const base_data = try readObjectRaw(odb.allocator, odb.pack_file.reader(), base_header.uncompressedLength()); | |
| 270 | errdefer odb.allocator.free(base_data); | |
| 271 | const base_object: Object = .{ .type = base_header.objectType(), .data = base_data }; | |
| 272 | try odb.cache.put(odb.allocator, base_offset, base_object); | |
| 273 | break base_object; | |
| 274 | }, | |
| 275 | } | |
| 276 | }; | |
| 277 | ||
| 278 | const base_data = try resolveDeltaChain( | |
| 279 | odb.allocator, | |
| 280 | odb.pack_file, | |
| 281 | base_object, | |
| 282 | delta_offsets.items, | |
| 283 | &odb.cache, | |
| 284 | ); | |
| 285 | ||
| 286 | return .{ .type = base_object.type, .data = base_data }; | |
| 287 | } | |
| 288 | ||
| 289 | /// Seeks to the beginning of the object with the given ID. | |
| 290 | fn seekOid(odb: *Odb, oid: Oid) !void { | |
| 291 | const key = oid[0]; | |
| 292 | var start_index = if (key > 0) odb.index_header.fan_out_table[key - 1] else 0; | |
| 293 | var end_index = odb.index_header.fan_out_table[key]; | |
| 294 | const found_index = while (start_index < end_index) { | |
| 295 | const mid_index = start_index + (end_index - start_index) / 2; | |
| 296 | try odb.index_file.seekTo(IndexHeader.size + mid_index * oid_length); | |
| 297 | const mid_oid = try odb.index_file.reader().readBytesNoEof(oid_length); | |
| 298 | switch (mem.order(u8, &mid_oid, &oid)) { | |
| 299 | .lt => start_index = mid_index + 1, | |
| 300 | .gt => end_index = mid_index, | |
| 301 | .eq => break mid_index, | |
| 302 | } | |
| 303 | } else return error.ObjectNotFound; | |
| 304 | ||
| 305 | const n_objects = odb.index_header.fan_out_table[255]; | |
| 306 | const offset_values_start = IndexHeader.size + n_objects * (oid_length + 4); | |
| 307 | try odb.index_file.seekTo(offset_values_start + found_index * 4); | |
| 308 | const l1_offset: packed struct { value: u31, big: bool } = @bitCast(try odb.index_file.reader().readIntBig(u32)); | |
| 309 | const pack_offset = pack_offset: { | |
| 310 | if (l1_offset.big) { | |
| 311 | const l2_offset_values_start = offset_values_start + n_objects * 4; | |
| 312 | try odb.index_file.seekTo(l2_offset_values_start + l1_offset.value * 4); | |
| 313 | break :pack_offset try odb.index_file.reader().readIntBig(u64); | |
| 314 | } else { | |
| 315 | break :pack_offset l1_offset.value; | |
| 316 | } | |
| 317 | }; | |
| 318 | ||
| 319 | try odb.pack_file.seekTo(pack_offset); | |
| 320 | } | |
| 321 | }; | |
| 322 | ||
| 323 | const Object = struct { | |
| 324 | type: Type, | |
| 325 | data: []const u8, | |
| 326 | ||
| 327 | const Type = enum { | |
| 328 | commit, | |
| 329 | tree, | |
| 330 | blob, | |
| 331 | tag, | |
| 332 | }; | |
| 333 | }; | |
| 334 | ||
| 335 | /// A cache for object data. | |
| 336 | /// | |
| 337 | /// The purpose of this cache is to speed up resolution of deltas by caching the | |
| 338 | /// results of resolving delta objects, while maintaining a maximum cache size | |
| 339 | /// to avoid excessive memory usage. If the total size of the objects in the | |
| 340 | /// cache exceeds the maximum, the cache will begin evicting the least recently | |
| 341 | /// used objects: when resolving delta chains, the most recently used objects | |
| 342 | /// will likely be more helpful as they will be further along in the chain | |
| 343 | /// (skipping earlier reconstruction steps). | |
| 344 | /// | |
| 345 | /// Object data stored in the cache is managed by the cache. It should not be | |
| 346 | /// freed by the caller at any point after inserting it into the cache. Any | |
| 347 | /// objects remaining in the cache will be freed when the cache itself is freed. | |
| 348 | const ObjectCache = struct { | |
| 349 | objects: std.AutoHashMapUnmanaged(u64, CacheEntry) = .{}, | |
| 350 | lru_nodes: LruList = .{}, | |
| 351 | byte_size: usize = 0, | |
| 352 | ||
| 353 | const max_byte_size = 128 * 1024 * 1024; // 128MiB | |
| 354 | /// A list of offsets stored in the cache, with the most recently used | |
| 355 | /// entries at the end. | |
| 356 | const LruList = std.DoublyLinkedList(u64); | |
| 357 | const CacheEntry = struct { object: Object, lru_node: *LruList.Node }; | |
| 358 | ||
| 359 | fn deinit(cache: *ObjectCache, allocator: Allocator) void { | |
| 360 | var object_iterator = cache.objects.iterator(); | |
| 361 | while (object_iterator.next()) |object| { | |
| 362 | allocator.free(object.value_ptr.object.data); | |
| 363 | allocator.destroy(object.value_ptr.lru_node); | |
| 364 | } | |
| 365 | cache.objects.deinit(allocator); | |
| 366 | cache.* = undefined; | |
| 367 | } | |
| 368 | ||
| 369 | /// Gets an object from the cache, moving it to the most recently used | |
| 370 | /// position if it is present. | |
| 371 | fn get(cache: *ObjectCache, offset: u64) ?Object { | |
| 372 | if (cache.objects.get(offset)) |entry| { | |
| 373 | cache.lru_nodes.remove(entry.lru_node); | |
| 374 | cache.lru_nodes.append(entry.lru_node); | |
| 375 | return entry.object; | |
| 376 | } else { | |
| 377 | return null; | |
| 378 | } | |
| 379 | } | |
| 380 | ||
| 381 | /// Puts an object in the cache, possibly evicting older entries if the | |
| 382 | /// cache exceeds its maximum size. Note that, although old objects may | |
| 383 | /// be evicted, the object just added to the cache with this function | |
| 384 | /// will not be evicted before the next call to `put` or `deinit` even if | |
| 385 | /// it exceeds the maximum cache size. | |
| 386 | fn put(cache: *ObjectCache, allocator: Allocator, offset: u64, object: Object) !void { | |
| 387 | const lru_node = try allocator.create(LruList.Node); | |
| 388 | errdefer allocator.destroy(lru_node); | |
| 389 | lru_node.data = offset; | |
| 390 | ||
| 391 | const gop = try cache.objects.getOrPut(allocator, offset); | |
| 392 | if (gop.found_existing) { | |
| 393 | cache.byte_size -= gop.value_ptr.object.data.len; | |
| 394 | cache.lru_nodes.remove(gop.value_ptr.lru_node); | |
| 395 | allocator.destroy(gop.value_ptr.lru_node); | |
| 396 | allocator.free(gop.value_ptr.object.data); | |
| 397 | } | |
| 398 | gop.value_ptr.* = .{ .object = object, .lru_node = lru_node }; | |
| 399 | cache.byte_size += object.data.len; | |
| 400 | cache.lru_nodes.append(lru_node); | |
| 401 | ||
| 402 | while (cache.byte_size > max_byte_size and cache.lru_nodes.len > 1) { | |
| 403 | // The > 1 check is to make sure that we don't evict the most | |
| 404 | // recently added node, even if it by itself happens to exceed the | |
| 405 | // maximum size of the cache. | |
| 406 | const evict_node = cache.lru_nodes.popFirst().?; | |
| 407 | const evict_offset = evict_node.data; | |
| 408 | allocator.destroy(evict_node); | |
| 409 | const evict_object = cache.objects.get(evict_offset).?.object; | |
| 410 | cache.byte_size -= evict_object.data.len; | |
| 411 | allocator.free(evict_object.data); | |
| 412 | _ = cache.objects.remove(evict_offset); | |
| 413 | } | |
| 414 | } | |
| 415 | }; | |
| 416 | ||
| 417 | /// A single pkt-line in the Git protocol. | |
| 418 | /// | |
| 419 | /// The format of a pkt-line is documented in | |
| 420 | /// [protocol-common](https://git-scm.com/docs/protocol-common). The special | |
| 421 | /// meanings of the delimiter and response-end packets are documented in | |
| 422 | /// [protocol-v2](https://git-scm.com/docs/protocol-v2). | |
| 423 | const Packet = union(enum) { | |
| 424 | flush, | |
| 425 | delimiter, | |
| 426 | response_end, | |
| 427 | data: []const u8, | |
| 428 | ||
| 429 | const max_data_length = 65516; | |
| 430 | ||
| 431 | /// Reads a packet in pkt-line format. | |
| 432 | fn read(reader: anytype, buf: *[max_data_length]u8) !Packet { | |
| 433 | const length = std.fmt.parseUnsigned(u16, &try reader.readBytesNoEof(4), 16) catch return error.InvalidPacket; | |
| 434 | switch (length) { | |
| 435 | 0 => return .flush, | |
| 436 | 1 => return .delimiter, | |
| 437 | 2 => return .response_end, | |
| 438 | 3 => return error.InvalidPacket, | |
| 439 | else => if (length - 4 > max_data_length) return error.InvalidPacket, | |
| 440 | } | |
| 441 | const data = buf[0 .. length - 4]; | |
| 442 | try reader.readNoEof(data); | |
| 443 | return .{ .data = data }; | |
| 444 | } | |
| 445 | ||
| 446 | /// Writes a packet in pkt-line format. | |
| 447 | fn write(packet: Packet, writer: anytype) !void { | |
| 448 | switch (packet) { | |
| 449 | .flush => try writer.writeAll("0000"), | |
| 450 | .delimiter => try writer.writeAll("0001"), | |
| 451 | .response_end => try writer.writeAll("0002"), | |
| 452 | .data => |data| { | |
| 453 | assert(data.len <= max_data_length); | |
| 454 | try writer.print("{x:0>4}", .{data.len + 4}); | |
| 455 | try writer.writeAll(data); | |
| 456 | }, | |
| 457 | } | |
| 458 | } | |
| 459 | }; | |
| 460 | ||
| 461 | /// A client session for the Git protocol, currently limited to an HTTP(S) | |
| 462 | /// transport. Only protocol version 2 is supported, as documented in | |
| 463 | /// [protocol-v2](https://git-scm.com/docs/protocol-v2). | |
| 464 | pub const Session = struct { | |
| 465 | transport: *std.http.Client, | |
| 466 | uri: std.Uri, | |
| 467 | supports_agent: bool = false, | |
| 468 | supports_shallow: bool = false, | |
| 469 | ||
| 470 | const agent = "zig/" ++ @import("builtin").zig_version_string; | |
| 471 | const agent_capability = std.fmt.comptimePrint("agent={s}\n", .{agent}); | |
| 472 | ||
| 473 | /// Discovers server capabilities. This should be called before using any | |
| 474 | /// other client functionality, or the client will be forced to default to | |
| 475 | /// the bare minimum server requirements, which may be considerably less | |
| 476 | /// efficient (e.g. no shallow fetches). | |
| 477 | /// | |
| 478 | /// See the note on `getCapabilities` regarding `redirect_uri`. | |
| 479 | pub fn discoverCapabilities( | |
| 480 | session: *Session, | |
| 481 | allocator: Allocator, | |
| 482 | redirect_uri: *[]u8, | |
| 483 | ) !void { | |
| 484 | var capability_iterator = try session.getCapabilities(allocator, redirect_uri); | |
| 485 | defer capability_iterator.deinit(); | |
| 486 | while (try capability_iterator.next()) |capability| { | |
| 487 | if (mem.eql(u8, capability.key, "agent")) { | |
| 488 | session.supports_agent = true; | |
| 489 | } else if (mem.eql(u8, capability.key, "fetch")) { | |
| 490 | var feature_iterator = mem.splitScalar(u8, capability.value orelse continue, ' '); | |
| 491 | while (feature_iterator.next()) |feature| { | |
| 492 | if (mem.eql(u8, feature, "shallow")) { | |
| 493 | session.supports_shallow = true; | |
| 494 | } | |
| 495 | } | |
| 496 | } | |
| 497 | } | |
| 498 | } | |
| 499 | ||
| 500 | /// Returns an iterator over capabilities supported by the server. | |
| 501 | /// | |
| 502 | /// If the server redirects the request, `error.Redirected` is returned and | |
| 503 | /// `redirect_uri` is populated with the URI resulting from the redirects. | |
| 504 | /// When this occurs, the value of `redirect_uri` must be freed with | |
| 505 | /// `allocator` when the caller is done with it. | |
| 506 | fn getCapabilities( | |
| 507 | session: Session, | |
| 508 | allocator: Allocator, | |
| 509 | redirect_uri: *[]u8, | |
| 510 | ) !CapabilityIterator { | |
| 511 | var info_refs_uri = session.uri; | |
| 512 | info_refs_uri.path = try std.fs.path.resolvePosix(allocator, &.{ "/", session.uri.path, "info/refs" }); | |
| 513 | defer allocator.free(info_refs_uri.path); | |
| 514 | info_refs_uri.query = "service=git-upload-pack"; | |
| 515 | info_refs_uri.fragment = null; | |
| 516 | ||
| 517 | var headers = std.http.Headers.init(allocator); | |
| 518 | defer headers.deinit(); | |
| 519 | try headers.append("Git-Protocol", "version=2"); | |
| 520 | ||
| 521 | var request = try session.transport.request(.GET, info_refs_uri, headers, .{ | |
| 522 | .max_redirects = 3, | |
| 523 | }); | |
| 524 | errdefer request.deinit(); | |
| 525 | try request.start(.{}); | |
| 526 | try request.finish(); | |
| 527 | ||
| 528 | try request.wait(); | |
| 529 | if (request.response.status != .ok) return error.ProtocolError; | |
| 530 | if (request.redirects_left < 3) { | |
| 531 | if (!mem.endsWith(u8, request.uri.path, "/info/refs")) return error.UnparseableRedirect; | |
| 532 | var new_uri = request.uri; | |
| 533 | new_uri.path = new_uri.path[0 .. new_uri.path.len - "/info/refs".len]; | |
| 534 | new_uri.query = null; | |
| 535 | redirect_uri.* = try std.fmt.allocPrint(allocator, "{+/}", .{new_uri}); | |
| 536 | return error.Redirected; | |
| 537 | } | |
| 538 | ||
| 539 | const reader = request.reader(); | |
| 540 | var buf: [Packet.max_data_length]u8 = undefined; | |
| 541 | var state: enum { response_start, response_content } = .response_start; | |
| 542 | while (true) { | |
| 543 | // Some Git servers (at least GitHub) include an additional | |
| 544 | // '# service=git-upload-pack' informative response before sending | |
| 545 | // the expected 'version 2' packet and capability information. | |
| 546 | // This is not universal: SourceHut, for example, does not do this. | |
| 547 | // Thus, we need to skip any such useless additional responses | |
| 548 | // before we get the one we're actually looking for. The responses | |
| 549 | // will be delimited by flush packets. | |
| 550 | const packet = Packet.read(reader, &buf) catch |e| switch (e) { | |
| 551 | error.EndOfStream => return error.UnsupportedProtocol, // 'version 2' packet not found | |
| 552 | else => |other| return other, | |
| 553 | }; | |
| 554 | switch (packet) { | |
| 555 | .flush => state = .response_start, | |
| 556 | .data => |data| switch (state) { | |
| 557 | .response_start => if (mem.eql(u8, data, "version 2\n")) { | |
| 558 | return .{ .request = request }; | |
| 559 | } else { | |
| 560 | state = .response_content; | |
| 561 | }, | |
| 562 | else => {}, | |
| 563 | }, | |
| 564 | else => return error.UnexpectedPacket, | |
| 565 | } | |
| 566 | } | |
| 567 | } | |
| 568 | ||
| 569 | const CapabilityIterator = struct { | |
| 570 | request: std.http.Client.Request, | |
| 571 | buf: [Packet.max_data_length]u8 = undefined, | |
| 572 | ||
| 573 | const Capability = struct { | |
| 574 | key: []const u8, | |
| 575 | value: ?[]const u8 = null, | |
| 576 | }; | |
| 577 | ||
| 578 | fn deinit(iterator: *CapabilityIterator) void { | |
| 579 | iterator.request.deinit(); | |
| 580 | iterator.* = undefined; | |
| 581 | } | |
| 582 | ||
| 583 | fn next(iterator: *CapabilityIterator) !?Capability { | |
| 584 | switch (try Packet.read(iterator.request.reader(), &iterator.buf)) { | |
| 585 | .flush => return null, | |
| 586 | .data => |data| if (data.len > 0 and data[data.len - 1] == '\n') { | |
| 587 | if (mem.indexOfScalar(u8, data, '=')) |separator_pos| { | |
| 588 | return .{ .key = data[0..separator_pos], .value = data[separator_pos + 1 .. data.len - 1] }; | |
| 589 | } else { | |
| 590 | return .{ .key = data[0 .. data.len - 1] }; | |
| 591 | } | |
| 592 | } else return error.UnexpectedPacket, | |
| 593 | else => return error.UnexpectedPacket, | |
| 594 | } | |
| 595 | } | |
| 596 | }; | |
| 597 | ||
| 598 | const ListRefsOptions = struct { | |
| 599 | /// The ref prefixes (if any) to use to filter the refs available on the | |
| 600 | /// server. Note that the client must still check the returned refs | |
| 601 | /// against its desired filters itself: the server is not required to | |
| 602 | /// respect these prefix filters and may return other refs as well. | |
| 603 | ref_prefixes: []const []const u8 = &.{}, | |
| 604 | /// Whether to include symref targets for returned symbolic refs. | |
| 605 | include_symrefs: bool = false, | |
| 606 | /// Whether to include the peeled object ID for returned tag refs. | |
| 607 | include_peeled: bool = false, | |
| 608 | }; | |
| 609 | ||
| 610 | /// Returns an iterator over refs known to the server. | |
| 611 | pub fn listRefs(session: Session, allocator: Allocator, options: ListRefsOptions) !RefIterator { | |
| 612 | var upload_pack_uri = session.uri; | |
| 613 | upload_pack_uri.path = try std.fs.path.resolvePosix(allocator, &.{ "/", session.uri.path, "git-upload-pack" }); | |
| 614 | defer allocator.free(upload_pack_uri.path); | |
| 615 | upload_pack_uri.query = null; | |
| 616 | upload_pack_uri.fragment = null; | |
| 617 | ||
| 618 | var headers = std.http.Headers.init(allocator); | |
| 619 | defer headers.deinit(); | |
| 620 | try headers.append("Content-Type", "application/x-git-upload-pack-request"); | |
| 621 | try headers.append("Git-Protocol", "version=2"); | |
| 622 | ||
| 623 | var body = std.ArrayListUnmanaged(u8){}; | |
| 624 | defer body.deinit(allocator); | |
| 625 | const body_writer = body.writer(allocator); | |
| 626 | try Packet.write(.{ .data = "command=ls-refs\n" }, body_writer); | |
| 627 | if (session.supports_agent) { | |
| 628 | try Packet.write(.{ .data = agent_capability }, body_writer); | |
| 629 | } | |
| 630 | try Packet.write(.delimiter, body_writer); | |
| 631 | for (options.ref_prefixes) |ref_prefix| { | |
| 632 | const ref_prefix_packet = try std.fmt.allocPrint(allocator, "ref-prefix {s}\n", .{ref_prefix}); | |
| 633 | defer allocator.free(ref_prefix_packet); | |
| 634 | try Packet.write(.{ .data = ref_prefix_packet }, body_writer); | |
| 635 | } | |
| 636 | if (options.include_symrefs) { | |
| 637 | try Packet.write(.{ .data = "symrefs\n" }, body_writer); | |
| 638 | } | |
| 639 | if (options.include_peeled) { | |
| 640 | try Packet.write(.{ .data = "peel\n" }, body_writer); | |
| 641 | } | |
| 642 | try Packet.write(.flush, body_writer); | |
| 643 | ||
| 644 | var request = try session.transport.request(.POST, upload_pack_uri, headers, .{ | |
| 645 | .handle_redirects = false, | |
| 646 | }); | |
| 647 | errdefer request.deinit(); | |
| 648 | request.transfer_encoding = .{ .content_length = body.items.len }; | |
| 649 | try request.start(.{}); | |
| 650 | try request.writeAll(body.items); | |
| 651 | try request.finish(); | |
| 652 | ||
| 653 | try request.wait(); | |
| 654 | if (request.response.status != .ok) return error.ProtocolError; | |
| 655 | ||
| 656 | return .{ .request = request }; | |
| 657 | } | |
| 658 | ||
| 659 | pub const RefIterator = struct { | |
| 660 | request: std.http.Client.Request, | |
| 661 | buf: [Packet.max_data_length]u8 = undefined, | |
| 662 | ||
| 663 | pub const Ref = struct { | |
| 664 | oid: Oid, | |
| 665 | name: []const u8, | |
| 666 | symref_target: ?[]const u8, | |
| 667 | peeled: ?Oid, | |
| 668 | }; | |
| 669 | ||
| 670 | pub fn deinit(iterator: *RefIterator) void { | |
| 671 | iterator.request.deinit(); | |
| 672 | iterator.* = undefined; | |
| 673 | } | |
| 674 | ||
| 675 | pub fn next(iterator: *RefIterator) !?Ref { | |
| 676 | switch (try Packet.read(iterator.request.reader(), &iterator.buf)) { | |
| 677 | .flush => return null, | |
| 678 | .data => |data| { | |
| 679 | const oid_sep_pos = mem.indexOfScalar(u8, data, ' ') orelse return error.InvalidRefPacket; | |
| 680 | const oid = parseOid(data[0..oid_sep_pos]) catch return error.InvalidRefPacket; | |
| 681 | ||
| 682 | const name_sep_pos = mem.indexOfAnyPos(u8, data, oid_sep_pos + 1, " \n") orelse return error.InvalidRefPacket; | |
| 683 | const name = data[oid_sep_pos + 1 .. name_sep_pos]; | |
| 684 | ||
| 685 | var symref_target: ?[]const u8 = null; | |
| 686 | var peeled: ?Oid = null; | |
| 687 | var last_sep_pos = name_sep_pos; | |
| 688 | while (data[last_sep_pos] == ' ') { | |
| 689 | const next_sep_pos = mem.indexOfAnyPos(u8, data, last_sep_pos + 1, " \n") orelse return error.InvalidRefPacket; | |
| 690 | const attribute = data[last_sep_pos + 1 .. next_sep_pos]; | |
| 691 | if (mem.startsWith(u8, attribute, "symref-target:")) { | |
| 692 | symref_target = attribute["symref-target:".len..]; | |
| 693 | } else if (mem.startsWith(u8, attribute, "peeled:")) { | |
| 694 | peeled = parseOid(attribute["peeled:".len..]) catch return error.InvalidRefPacket; | |
| 695 | } | |
| 696 | last_sep_pos = next_sep_pos; | |
| 697 | } | |
| 698 | ||
| 699 | return .{ .oid = oid, .name = name, .symref_target = symref_target, .peeled = peeled }; | |
| 700 | }, | |
| 701 | else => return error.UnexpectedPacket, | |
| 702 | } | |
| 703 | } | |
| 704 | }; | |
| 705 | ||
| 706 | /// Fetches the given refs from the server. A shallow fetch (depth 1) is | |
| 707 | /// performed if the server supports it. | |
| 708 | pub fn fetch(session: Session, allocator: Allocator, wants: []const []const u8) !FetchStream { | |
| 709 | var upload_pack_uri = session.uri; | |
| 710 | upload_pack_uri.path = try std.fs.path.resolvePosix(allocator, &.{ "/", session.uri.path, "git-upload-pack" }); | |
| 711 | defer allocator.free(upload_pack_uri.path); | |
| 712 | upload_pack_uri.query = null; | |
| 713 | upload_pack_uri.fragment = null; | |
| 714 | ||
| 715 | var headers = std.http.Headers.init(allocator); | |
| 716 | defer headers.deinit(); | |
| 717 | try headers.append("Content-Type", "application/x-git-upload-pack-request"); | |
| 718 | try headers.append("Git-Protocol", "version=2"); | |
| 719 | ||
| 720 | var body = std.ArrayListUnmanaged(u8){}; | |
| 721 | defer body.deinit(allocator); | |
| 722 | const body_writer = body.writer(allocator); | |
| 723 | try Packet.write(.{ .data = "command=fetch\n" }, body_writer); | |
| 724 | if (session.supports_agent) { | |
| 725 | try Packet.write(.{ .data = agent_capability }, body_writer); | |
| 726 | } | |
| 727 | try Packet.write(.delimiter, body_writer); | |
| 728 | // Our packfile parser supports the OFS_DELTA object type | |
| 729 | try Packet.write(.{ .data = "ofs-delta\n" }, body_writer); | |
| 730 | // We do not currently convey server progress information to the user | |
| 731 | try Packet.write(.{ .data = "no-progress\n" }, body_writer); | |
| 732 | if (session.supports_shallow) { | |
| 733 | try Packet.write(.{ .data = "deepen 1\n" }, body_writer); | |
| 734 | } | |
| 735 | for (wants) |want| { | |
| 736 | var buf: [Packet.max_data_length]u8 = undefined; | |
| 737 | const arg = std.fmt.bufPrint(&buf, "want {s}\n", .{want}) catch unreachable; | |
| 738 | try Packet.write(.{ .data = arg }, body_writer); | |
| 739 | } | |
| 740 | try Packet.write(.{ .data = "done\n" }, body_writer); | |
| 741 | try Packet.write(.flush, body_writer); | |
| 742 | ||
| 743 | var request = try session.transport.request(.POST, upload_pack_uri, headers, .{ | |
| 744 | .handle_redirects = false, | |
| 745 | }); | |
| 746 | errdefer request.deinit(); | |
| 747 | request.transfer_encoding = .{ .content_length = body.items.len }; | |
| 748 | try request.start(.{}); | |
| 749 | try request.writeAll(body.items); | |
| 750 | try request.finish(); | |
| 751 | ||
| 752 | try request.wait(); | |
| 753 | if (request.response.status != .ok) return error.ProtocolError; | |
| 754 | ||
| 755 | const reader = request.reader(); | |
| 756 | // We are not interested in any of the sections of the returned fetch | |
| 757 | // data other than the packfile section, since we aren't doing anything | |
| 758 | // complex like ref negotiation (this is a fresh clone). | |
| 759 | var state: enum { section_start, section_content } = .section_start; | |
| 760 | while (true) { | |
| 761 | var buf: [Packet.max_data_length]u8 = undefined; | |
| 762 | const packet = try Packet.read(reader, &buf); | |
| 763 | switch (state) { | |
| 764 | .section_start => switch (packet) { | |
| 765 | .data => |data| if (mem.eql(u8, data, "packfile\n")) { | |
| 766 | return .{ .request = request }; | |
| 767 | } else { | |
| 768 | state = .section_content; | |
| 769 | }, | |
| 770 | else => return error.UnexpectedPacket, | |
| 771 | }, | |
| 772 | .section_content => switch (packet) { | |
| 773 | .delimiter => state = .section_start, | |
| 774 | .data => {}, | |
| 775 | else => return error.UnexpectedPacket, | |
| 776 | }, | |
| 777 | } | |
| 778 | } | |
| 779 | } | |
| 780 | ||
| 781 | pub const FetchStream = struct { | |
| 782 | request: std.http.Client.Request, | |
| 783 | buf: [Packet.max_data_length]u8 = undefined, | |
| 784 | pos: usize = 0, | |
| 785 | len: usize = 0, | |
| 786 | ||
| 787 | pub fn deinit(stream: *FetchStream) void { | |
| 788 | stream.request.deinit(); | |
| 789 | } | |
| 790 | ||
| 791 | pub const ReadError = std.http.Client.Request.ReadError || error{ | |
| 792 | InvalidPacket, | |
| 793 | ProtocolError, | |
| 794 | UnexpectedPacket, | |
| 795 | }; | |
| 796 | pub const Reader = std.io.Reader(*FetchStream, ReadError, read); | |
| 797 | ||
| 798 | const StreamCode = enum(u8) { | |
| 799 | pack_data = 1, | |
| 800 | progress = 2, | |
| 801 | fatal_error = 3, | |
| 802 | _, | |
| 803 | }; | |
| 804 | ||
| 805 | pub fn reader(stream: *FetchStream) Reader { | |
| 806 | return .{ .context = stream }; | |
| 807 | } | |
| 808 | ||
| 809 | pub fn read(stream: *FetchStream, buf: []u8) !usize { | |
| 810 | if (stream.pos == stream.len) { | |
| 811 | while (true) { | |
| 812 | switch (try Packet.read(stream.request.reader(), &stream.buf)) { | |
| 813 | .flush => return 0, | |
| 814 | .data => |data| if (data.len > 1) switch (@as(StreamCode, @enumFromInt(data[0]))) { | |
| 815 | .pack_data => { | |
| 816 | stream.pos = 1; | |
| 817 | stream.len = data.len; | |
| 818 | break; | |
| 819 | }, | |
| 820 | .fatal_error => return error.ProtocolError, | |
| 821 | else => {}, | |
| 822 | }, | |
| 823 | else => return error.UnexpectedPacket, | |
| 824 | } | |
| 825 | } | |
| 826 | } | |
| 827 | ||
| 828 | const size = @min(buf.len, stream.len - stream.pos); | |
| 829 | @memcpy(buf[0..size], stream.buf[stream.pos .. stream.pos + size]); | |
| 830 | stream.pos += size; | |
| 831 | return size; | |
| 832 | } | |
| 833 | }; | |
| 834 | }; | |
| 835 | ||
| 836 | const PackHeader = struct { | |
| 837 | total_objects: u32, | |
| 838 | ||
| 839 | const signature = "PACK"; | |
| 840 | const supported_version = 2; | |
| 841 | ||
| 842 | fn read(reader: anytype) !PackHeader { | |
| 843 | const actual_signature = reader.readBytesNoEof(4) catch |e| switch (e) { | |
| 844 | error.EndOfStream => return error.InvalidHeader, | |
| 845 | else => |other| return other, | |
| 846 | }; | |
| 847 | if (!mem.eql(u8, &actual_signature, signature)) return error.InvalidHeader; | |
| 848 | const version = reader.readIntBig(u32) catch |e| switch (e) { | |
| 849 | error.EndOfStream => return error.InvalidHeader, | |
| 850 | else => |other| return other, | |
| 851 | }; | |
| 852 | if (version != supported_version) return error.UnsupportedVersion; | |
| 853 | const total_objects = reader.readIntBig(u32) catch |e| switch (e) { | |
| 854 | error.EndOfStream => return error.InvalidHeader, | |
| 855 | else => |other| return other, | |
| 856 | }; | |
| 857 | return .{ .total_objects = total_objects }; | |
| 858 | } | |
| 859 | }; | |
| 860 | ||
| 861 | const EntryHeader = union(Type) { | |
| 862 | commit: Undeltified, | |
| 863 | tree: Undeltified, | |
| 864 | blob: Undeltified, | |
| 865 | tag: Undeltified, | |
| 866 | ofs_delta: OfsDelta, | |
| 867 | ref_delta: RefDelta, | |
| 868 | ||
| 869 | const Type = enum(u3) { | |
| 870 | commit = 1, | |
| 871 | tree = 2, | |
| 872 | blob = 3, | |
| 873 | tag = 4, | |
| 874 | ofs_delta = 6, | |
| 875 | ref_delta = 7, | |
| 876 | }; | |
| 877 | ||
| 878 | const Undeltified = struct { | |
| 879 | uncompressed_length: u64, | |
| 880 | }; | |
| 881 | ||
| 882 | const OfsDelta = struct { | |
| 883 | offset: u64, | |
| 884 | uncompressed_length: u64, | |
| 885 | }; | |
| 886 | ||
| 887 | const RefDelta = struct { | |
| 888 | base_object: Oid, | |
| 889 | uncompressed_length: u64, | |
| 890 | }; | |
| 891 | ||
| 892 | fn objectType(header: EntryHeader) Object.Type { | |
| 893 | return switch (header) { | |
| 894 | inline .commit, .tree, .blob, .tag => |_, tag| @field(Object.Type, @tagName(tag)), | |
| 895 | else => unreachable, | |
| 896 | }; | |
| 897 | } | |
| 898 | ||
| 899 | fn uncompressedLength(header: EntryHeader) u64 { | |
| 900 | return switch (header) { | |
| 901 | inline else => |entry| entry.uncompressed_length, | |
| 902 | }; | |
| 903 | } | |
| 904 | ||
| 905 | fn read(reader: anytype) !EntryHeader { | |
| 906 | const InitialByte = packed struct { len: u4, type: u3, has_next: bool }; | |
| 907 | const initial: InitialByte = @bitCast(reader.readByte() catch |e| switch (e) { | |
| 908 | error.EndOfStream => return error.InvalidFormat, | |
| 909 | else => |other| return other, | |
| 910 | }); | |
| 911 | const rest_len = if (initial.has_next) try readSizeVarInt(reader) else 0; | |
| 912 | var uncompressed_length: u64 = initial.len; | |
| 913 | uncompressed_length |= std.math.shlExact(u64, rest_len, 4) catch return error.InvalidFormat; | |
| 914 | const @"type" = std.meta.intToEnum(EntryHeader.Type, initial.type) catch return error.InvalidFormat; | |
| 915 | return switch (@"type") { | |
| 916 | inline .commit, .tree, .blob, .tag => |tag| @unionInit(EntryHeader, @tagName(tag), .{ | |
| 917 | .uncompressed_length = uncompressed_length, | |
| 918 | }), | |
| 919 | .ofs_delta => .{ .ofs_delta = .{ | |
| 920 | .offset = try readOffsetVarInt(reader), | |
| 921 | .uncompressed_length = uncompressed_length, | |
| 922 | } }, | |
| 923 | .ref_delta => .{ .ref_delta = .{ | |
| 924 | .base_object = reader.readBytesNoEof(oid_length) catch |e| switch (e) { | |
| 925 | error.EndOfStream => return error.InvalidFormat, | |
| 926 | else => |other| return other, | |
| 927 | }, | |
| 928 | .uncompressed_length = uncompressed_length, | |
| 929 | } }, | |
| 930 | }; | |
| 931 | } | |
| 932 | }; | |
| 933 | ||
| 934 | fn readSizeVarInt(r: anytype) !u64 { | |
| 935 | const Byte = packed struct { value: u7, has_next: bool }; | |
| 936 | var b: Byte = @bitCast(try r.readByte()); | |
| 937 | var value: u64 = b.value; | |
| 938 | var shift: u6 = 0; | |
| 939 | while (b.has_next) { | |
| 940 | b = @bitCast(try r.readByte()); | |
| 941 | shift = std.math.add(u6, shift, 7) catch return error.InvalidFormat; | |
| 942 | value |= @as(u64, b.value) << shift; | |
| 943 | } | |
| 944 | return value; | |
| 945 | } | |
| 946 | ||
| 947 | fn readOffsetVarInt(r: anytype) !u64 { | |
| 948 | const Byte = packed struct { value: u7, has_next: bool }; | |
| 949 | var b: Byte = @bitCast(try r.readByte()); | |
| 950 | var value: u64 = b.value; | |
| 951 | while (b.has_next) { | |
| 952 | b = @bitCast(try r.readByte()); | |
| 953 | value = std.math.shlExact(u64, value + 1, 7) catch return error.InvalidFormat; | |
| 954 | value |= b.value; | |
| 955 | } | |
| 956 | return value; | |
| 957 | } | |
| 958 | ||
| 959 | const IndexHeader = struct { | |
| 960 | fan_out_table: [256]u32, | |
| 961 | ||
| 962 | const signature = "\xFFtOc"; | |
| 963 | const supported_version = 2; | |
| 964 | const size = 4 + 4 + @sizeOf([256]u32); | |
| 965 | ||
| 966 | fn read(reader: anytype) !IndexHeader { | |
| 967 | var header_bytes = try reader.readBytesNoEof(size); | |
| 968 | if (!mem.eql(u8, header_bytes[0..4], signature)) return error.InvalidHeader; | |
| 969 | const version = mem.readIntBig(u32, header_bytes[4..8]); | |
| 970 | if (version != supported_version) return error.UnsupportedVersion; | |
| 971 | ||
| 972 | var fan_out_table: [256]u32 = undefined; | |
| 973 | var fan_out_table_stream = std.io.fixedBufferStream(header_bytes[8..]); | |
| 974 | const fan_out_table_reader = fan_out_table_stream.reader(); | |
| 975 | for (&fan_out_table) |*entry| { | |
| 976 | entry.* = fan_out_table_reader.readIntBig(u32) catch unreachable; | |
| 977 | } | |
| 978 | return .{ .fan_out_table = fan_out_table }; | |
| 979 | } | |
| 980 | }; | |
| 981 | ||
| 982 | const IndexEntry = struct { | |
| 983 | offset: u64, | |
| 984 | crc32: u32, | |
| 985 | }; | |
| 986 | ||
| 987 | /// Writes out a version 2 index for the given packfile, as documented in | |
| 988 | /// [pack-format](https://git-scm.com/docs/pack-format). | |
| 989 | pub fn indexPack(allocator: Allocator, pack: std.fs.File, index_writer: anytype) !void { | |
| 990 | try pack.seekTo(0); | |
| 991 | ||
| 992 | var index_entries = std.AutoHashMapUnmanaged(Oid, IndexEntry){}; | |
| 993 | defer index_entries.deinit(allocator); | |
| 994 | var pending_deltas = std.ArrayListUnmanaged(IndexEntry){}; | |
| 995 | defer pending_deltas.deinit(allocator); | |
| 996 | ||
| 997 | const pack_checksum = try indexPackFirstPass(allocator, pack, &index_entries, &pending_deltas); | |
| 998 | ||
| 999 | var cache: ObjectCache = .{}; | |
| 1000 | defer cache.deinit(allocator); | |
| 1001 | var remaining_deltas = pending_deltas.items.len; | |
| 1002 | while (remaining_deltas > 0) { | |
| 1003 | var i: usize = remaining_deltas; | |
| 1004 | while (i > 0) { | |
| 1005 | i -= 1; | |
| 1006 | const delta = pending_deltas.items[i]; | |
| 1007 | if (try indexPackHashDelta(allocator, pack, delta, index_entries, &cache)) |oid| { | |
| 1008 | try index_entries.put(allocator, oid, delta); | |
| 1009 | _ = pending_deltas.swapRemove(i); | |
| 1010 | } | |
| 1011 | } | |
| 1012 | if (pending_deltas.items.len == remaining_deltas) return error.IncompletePack; | |
| 1013 | remaining_deltas = pending_deltas.items.len; | |
| 1014 | } | |
| 1015 | ||
| 1016 | var oids = std.ArrayListUnmanaged(Oid){}; | |
| 1017 | defer oids.deinit(allocator); | |
| 1018 | try oids.ensureTotalCapacityPrecise(allocator, index_entries.count()); | |
| 1019 | var index_entries_iter = index_entries.iterator(); | |
| 1020 | while (index_entries_iter.next()) |entry| { | |
| 1021 | oids.appendAssumeCapacity(entry.key_ptr.*); | |
| 1022 | } | |
| 1023 | mem.sortUnstable(Oid, oids.items, {}, struct { | |
| 1024 | fn lessThan(_: void, o1: Oid, o2: Oid) bool { | |
| 1025 | return mem.lessThan(u8, &o1, &o2); | |
| 1026 | } | |
| 1027 | }.lessThan); | |
| 1028 | ||
| 1029 | var fan_out_table: [256]u32 = undefined; | |
| 1030 | var count: u32 = 0; | |
| 1031 | var fan_out_index: u8 = 0; | |
| 1032 | for (oids.items) |oid| { | |
| 1033 | if (oid[0] > fan_out_index) { | |
| 1034 | @memset(fan_out_table[fan_out_index..oid[0]], count); | |
| 1035 | fan_out_index = oid[0]; | |
| 1036 | } | |
| 1037 | count += 1; | |
| 1038 | } | |
| 1039 | @memset(fan_out_table[fan_out_index..], count); | |
| 1040 | ||
| 1041 | var index_hashed_writer = hashedWriter(index_writer, Sha1.init(.{})); | |
| 1042 | const writer = index_hashed_writer.writer(); | |
| 1043 | try writer.writeAll(IndexHeader.signature); | |
| 1044 | try writer.writeIntBig(u32, IndexHeader.supported_version); | |
| 1045 | for (fan_out_table) |fan_out_entry| { | |
| 1046 | try writer.writeIntBig(u32, fan_out_entry); | |
| 1047 | } | |
| 1048 | ||
| 1049 | for (oids.items) |oid| { | |
| 1050 | try writer.writeAll(&oid); | |
| 1051 | } | |
| 1052 | ||
| 1053 | for (oids.items) |oid| { | |
| 1054 | try writer.writeIntBig(u32, index_entries.get(oid).?.crc32); | |
| 1055 | } | |
| 1056 | ||
| 1057 | var big_offsets = std.ArrayListUnmanaged(u64){}; | |
| 1058 | defer big_offsets.deinit(allocator); | |
| 1059 | for (oids.items) |oid| { | |
| 1060 | const offset = index_entries.get(oid).?.offset; | |
| 1061 | if (offset <= std.math.maxInt(u31)) { | |
| 1062 | try writer.writeIntBig(u32, @intCast(offset)); | |
| 1063 | } else { | |
| 1064 | const index = big_offsets.items.len; | |
| 1065 | try big_offsets.append(allocator, offset); | |
| 1066 | try writer.writeIntBig(u32, @as(u32, @intCast(index)) | (1 << 31)); | |
| 1067 | } | |
| 1068 | } | |
| 1069 | for (big_offsets.items) |offset| { | |
| 1070 | try writer.writeIntBig(u64, offset); | |
| 1071 | } | |
| 1072 | ||
| 1073 | try writer.writeAll(&pack_checksum); | |
| 1074 | const index_checksum = index_hashed_writer.hasher.finalResult(); | |
| 1075 | try index_writer.writeAll(&index_checksum); | |
| 1076 | } | |
| 1077 | ||
| 1078 | /// Performs the first pass over the packfile data for index construction. | |
| 1079 | /// This will index all non-delta objects, queue delta objects for further | |
| 1080 | /// processing, and return the pack checksum (which is part of the index | |
| 1081 | /// format). | |
| 1082 | fn indexPackFirstPass( | |
| 1083 | allocator: Allocator, | |
| 1084 | pack: std.fs.File, | |
| 1085 | index_entries: *std.AutoHashMapUnmanaged(Oid, IndexEntry), | |
| 1086 | pending_deltas: *std.ArrayListUnmanaged(IndexEntry), | |
| 1087 | ) ![Sha1.digest_length]u8 { | |
| 1088 | var pack_buffered_reader = std.io.bufferedReader(pack.reader()); | |
| 1089 | var pack_counting_reader = std.io.countingReader(pack_buffered_reader.reader()); | |
| 1090 | var pack_hashed_reader = std.compress.hashedReader(pack_counting_reader.reader(), Sha1.init(.{})); | |
| 1091 | const pack_reader = pack_hashed_reader.reader(); | |
| 1092 | ||
| 1093 | const pack_header = try PackHeader.read(pack_reader); | |
| 1094 | ||
| 1095 | var current_entry: u32 = 0; | |
| 1096 | while (current_entry < pack_header.total_objects) : (current_entry += 1) { | |
| 1097 | const entry_offset = pack_counting_reader.bytes_read; | |
| 1098 | var entry_crc32_reader = std.compress.hashedReader(pack_reader, std.hash.Crc32.init()); | |
| 1099 | const entry_header = try EntryHeader.read(entry_crc32_reader.reader()); | |
| 1100 | switch (entry_header) { | |
| 1101 | inline .commit, .tree, .blob, .tag => |object, tag| { | |
| 1102 | var entry_decompress_stream = try std.compress.zlib.decompressStream(allocator, entry_crc32_reader.reader()); | |
| 1103 | defer entry_decompress_stream.deinit(); | |
| 1104 | var entry_counting_reader = std.io.countingReader(entry_decompress_stream.reader()); | |
| 1105 | var entry_hashed_writer = hashedWriter(std.io.null_writer, Sha1.init(.{})); | |
| 1106 | const entry_writer = entry_hashed_writer.writer(); | |
| 1107 | // The object header is not included in the pack data but is | |
| 1108 | // part of the object's ID | |
| 1109 | try entry_writer.print("{s} {}\x00", .{ @tagName(tag), object.uncompressed_length }); | |
| 1110 | var fifo = std.fifo.LinearFifo(u8, .{ .Static = 4096 }).init(); | |
| 1111 | try fifo.pump(entry_counting_reader.reader(), entry_writer); | |
| 1112 | if (entry_counting_reader.bytes_read != object.uncompressed_length) { | |
| 1113 | return error.InvalidObject; | |
| 1114 | } | |
| 1115 | const oid = entry_hashed_writer.hasher.finalResult(); | |
| 1116 | try index_entries.put(allocator, oid, .{ | |
| 1117 | .offset = entry_offset, | |
| 1118 | .crc32 = entry_crc32_reader.hasher.final(), | |
| 1119 | }); | |
| 1120 | }, | |
| 1121 | inline .ofs_delta, .ref_delta => |delta| { | |
| 1122 | var entry_decompress_stream = try std.compress.zlib.decompressStream(allocator, entry_crc32_reader.reader()); | |
| 1123 | defer entry_decompress_stream.deinit(); | |
| 1124 | var entry_counting_reader = std.io.countingReader(entry_decompress_stream.reader()); | |
| 1125 | var fifo = std.fifo.LinearFifo(u8, .{ .Static = 4096 }).init(); | |
| 1126 | try fifo.pump(entry_counting_reader.reader(), std.io.null_writer); | |
| 1127 | if (entry_counting_reader.bytes_read != delta.uncompressed_length) { | |
| 1128 | return error.InvalidObject; | |
| 1129 | } | |
| 1130 | try pending_deltas.append(allocator, .{ | |
| 1131 | .offset = entry_offset, | |
| 1132 | .crc32 = entry_crc32_reader.hasher.final(), | |
| 1133 | }); | |
| 1134 | }, | |
| 1135 | } | |
| 1136 | } | |
| 1137 | ||
| 1138 | const pack_checksum = pack_hashed_reader.hasher.finalResult(); | |
| 1139 | const recorded_checksum = try pack_buffered_reader.reader().readBytesNoEof(Sha1.digest_length); | |
| 1140 | if (!mem.eql(u8, &pack_checksum, &recorded_checksum)) { | |
| 1141 | return error.CorruptedPack; | |
| 1142 | } | |
| 1143 | _ = pack_buffered_reader.reader().readByte() catch |e| switch (e) { | |
| 1144 | error.EndOfStream => return pack_checksum, | |
| 1145 | else => |other| return other, | |
| 1146 | }; | |
| 1147 | return error.InvalidFormat; | |
| 1148 | } | |
| 1149 | ||
| 1150 | /// Attempts to determine the final object ID of the given deltified object. | |
| 1151 | /// May return null if this is not yet possible (if the delta is a ref-based | |
| 1152 | /// delta and we do not yet know the offset of the base object). | |
| 1153 | fn indexPackHashDelta( | |
| 1154 | allocator: Allocator, | |
| 1155 | pack: std.fs.File, | |
| 1156 | delta: IndexEntry, | |
| 1157 | index_entries: std.AutoHashMapUnmanaged(Oid, IndexEntry), | |
| 1158 | cache: *ObjectCache, | |
| 1159 | ) !?Oid { | |
| 1160 | // Figure out the chain of deltas to resolve | |
| 1161 | var base_offset = delta.offset; | |
| 1162 | var base_header: EntryHeader = undefined; | |
| 1163 | var delta_offsets = std.ArrayListUnmanaged(u64){}; | |
| 1164 | defer delta_offsets.deinit(allocator); | |
| 1165 | const base_object = while (true) { | |
| 1166 | if (cache.get(base_offset)) |base_object| break base_object; | |
| 1167 | ||
| 1168 | try pack.seekTo(base_offset); | |
| 1169 | base_header = try EntryHeader.read(pack.reader()); | |
| 1170 | switch (base_header) { | |
| 1171 | .ofs_delta => |ofs_delta| { | |
| 1172 | try delta_offsets.append(allocator, base_offset); | |
| 1173 | base_offset = std.math.sub(u64, base_offset, ofs_delta.offset) catch return error.InvalidObject; | |
| 1174 | }, | |
| 1175 | .ref_delta => |ref_delta| { | |
| 1176 | try delta_offsets.append(allocator, base_offset); | |
| 1177 | base_offset = (index_entries.get(ref_delta.base_object) orelse return null).offset; | |
| 1178 | }, | |
| 1179 | else => { | |
| 1180 | const base_data = try readObjectRaw(allocator, pack.reader(), base_header.uncompressedLength()); | |
| 1181 | errdefer allocator.free(base_data); | |
| 1182 | const base_object: Object = .{ .type = base_header.objectType(), .data = base_data }; | |
| 1183 | try cache.put(allocator, base_offset, base_object); | |
| 1184 | break base_object; | |
| 1185 | }, | |
| 1186 | } | |
| 1187 | }; | |
| 1188 | ||
| 1189 | const base_data = try resolveDeltaChain(allocator, pack, base_object, delta_offsets.items, cache); | |
| 1190 | ||
| 1191 | var entry_hasher = Sha1.init(.{}); | |
| 1192 | var entry_hashed_writer = hashedWriter(std.io.null_writer, &entry_hasher); | |
| 1193 | try entry_hashed_writer.writer().print("{s} {}\x00", .{ @tagName(base_object.type), base_data.len }); | |
| 1194 | entry_hasher.update(base_data); | |
| 1195 | return entry_hasher.finalResult(); | |
| 1196 | } | |
| 1197 | ||
| 1198 | /// Resolves a chain of deltas, returning the final base object data. `pack` is | |
| 1199 | /// assumed to be looking at the start of the object data for the base object of | |
| 1200 | /// the chain, and will then apply the deltas in `delta_offsets` in reverse order | |
| 1201 | /// to obtain the final object. | |
| 1202 | fn resolveDeltaChain( | |
| 1203 | allocator: Allocator, | |
| 1204 | pack: std.fs.File, | |
| 1205 | base_object: Object, | |
| 1206 | delta_offsets: []const u64, | |
| 1207 | cache: *ObjectCache, | |
| 1208 | ) ![]const u8 { | |
| 1209 | var base_data = base_object.data; | |
| 1210 | var i: usize = delta_offsets.len; | |
| 1211 | while (i > 0) { | |
| 1212 | i -= 1; | |
| 1213 | ||
| 1214 | const delta_offset = delta_offsets[i]; | |
| 1215 | try pack.seekTo(delta_offset); | |
| 1216 | const delta_header = try EntryHeader.read(pack.reader()); | |
| 1217 | var delta_data = try readObjectRaw(allocator, pack.reader(), delta_header.uncompressedLength()); | |
| 1218 | defer allocator.free(delta_data); | |
| 1219 | var delta_stream = std.io.fixedBufferStream(delta_data); | |
| 1220 | const delta_reader = delta_stream.reader(); | |
| 1221 | _ = try readSizeVarInt(delta_reader); // base object size | |
| 1222 | const expanded_size = try readSizeVarInt(delta_reader); | |
| 1223 | ||
| 1224 | const expanded_alloc_size = std.math.cast(usize, expanded_size) orelse return error.ObjectTooLarge; | |
| 1225 | var expanded_data = try allocator.alloc(u8, expanded_alloc_size); | |
| 1226 | errdefer allocator.free(expanded_data); | |
| 1227 | var expanded_delta_stream = std.io.fixedBufferStream(expanded_data); | |
| 1228 | var base_stream = std.io.fixedBufferStream(base_data); | |
| 1229 | try expandDelta(&base_stream, delta_reader, expanded_delta_stream.writer()); | |
| 1230 | if (expanded_delta_stream.pos != expanded_size) return error.InvalidObject; | |
| 1231 | ||
| 1232 | try cache.put(allocator, delta_offset, .{ .type = base_object.type, .data = expanded_data }); | |
| 1233 | base_data = expanded_data; | |
| 1234 | } | |
| 1235 | return base_data; | |
| 1236 | } | |
| 1237 | ||
| 1238 | /// Reads the complete contents of an object from `reader`. This function may | |
| 1239 | /// read more bytes than required from `reader`, so the reader position after | |
| 1240 | /// returning is not reliable. | |
| 1241 | fn readObjectRaw(allocator: Allocator, reader: anytype, size: u64) ![]u8 { | |
| 1242 | const alloc_size = std.math.cast(usize, size) orelse return error.ObjectTooLarge; | |
| 1243 | var buffered_reader = std.io.bufferedReader(reader); | |
| 1244 | var decompress_stream = try std.compress.zlib.decompressStream(allocator, buffered_reader.reader()); | |
| 1245 | defer decompress_stream.deinit(); | |
| 1246 | var data = try allocator.alloc(u8, alloc_size); | |
| 1247 | errdefer allocator.free(data); | |
| 1248 | try decompress_stream.reader().readNoEof(data); | |
| 1249 | _ = decompress_stream.reader().readByte() catch |e| switch (e) { | |
| 1250 | error.EndOfStream => return data, | |
| 1251 | else => |other| return other, | |
| 1252 | }; | |
| 1253 | return error.InvalidFormat; | |
| 1254 | } | |
| 1255 | ||
| 1256 | /// Expands delta data from `delta_reader` to `writer`. `base_object` must | |
| 1257 | /// support `reader` and `seekTo` (such as a `std.io.FixedBufferStream`). | |
| 1258 | /// | |
| 1259 | /// The format of the delta data is documented in | |
| 1260 | /// [pack-format](https://git-scm.com/docs/pack-format). | |
| 1261 | fn expandDelta(base_object: anytype, delta_reader: anytype, writer: anytype) !void { | |
| 1262 | while (true) { | |
| 1263 | const inst: packed struct { value: u7, copy: bool } = @bitCast(delta_reader.readByte() catch |e| switch (e) { | |
| 1264 | error.EndOfStream => return, | |
| 1265 | else => |other| return other, | |
| 1266 | }); | |
| 1267 | if (inst.copy) { | |
| 1268 | const available: packed struct { | |
| 1269 | offset1: bool, | |
| 1270 | offset2: bool, | |
| 1271 | offset3: bool, | |
| 1272 | offset4: bool, | |
| 1273 | size1: bool, | |
| 1274 | size2: bool, | |
| 1275 | size3: bool, | |
| 1276 | } = @bitCast(inst.value); | |
| 1277 | var offset_parts: packed struct { offset1: u8, offset2: u8, offset3: u8, offset4: u8 } = .{ | |
| 1278 | .offset1 = if (available.offset1) try delta_reader.readByte() else 0, | |
| 1279 | .offset2 = if (available.offset2) try delta_reader.readByte() else 0, | |
| 1280 | .offset3 = if (available.offset3) try delta_reader.readByte() else 0, | |
| 1281 | .offset4 = if (available.offset4) try delta_reader.readByte() else 0, | |
| 1282 | }; | |
| 1283 | const offset: u32 = @bitCast(offset_parts); | |
| 1284 | var size_parts: packed struct { size1: u8, size2: u8, size3: u8 } = .{ | |
| 1285 | .size1 = if (available.size1) try delta_reader.readByte() else 0, | |
| 1286 | .size2 = if (available.size2) try delta_reader.readByte() else 0, | |
| 1287 | .size3 = if (available.size3) try delta_reader.readByte() else 0, | |
| 1288 | }; | |
| 1289 | var size: u24 = @bitCast(size_parts); | |
| 1290 | if (size == 0) size = 0x10000; | |
| 1291 | try base_object.seekTo(offset); | |
| 1292 | var copy_reader = std.io.limitedReader(base_object.reader(), size); | |
| 1293 | var fifo = std.fifo.LinearFifo(u8, .{ .Static = 4096 }).init(); | |
| 1294 | try fifo.pump(copy_reader.reader(), writer); | |
| 1295 | } else if (inst.value != 0) { | |
| 1296 | var data_reader = std.io.limitedReader(delta_reader, inst.value); | |
| 1297 | var fifo = std.fifo.LinearFifo(u8, .{ .Static = 4096 }).init(); | |
| 1298 | try fifo.pump(data_reader.reader(), writer); | |
| 1299 | } else { | |
| 1300 | return error.InvalidDeltaInstruction; | |
| 1301 | } | |
| 1302 | } | |
| 1303 | } | |
| 1304 | ||
| 1305 | fn HashedWriter( | |
| 1306 | comptime WriterType: anytype, | |
| 1307 | comptime HasherType: anytype, | |
| 1308 | ) type { | |
| 1309 | return struct { | |
| 1310 | child_writer: WriterType, | |
| 1311 | hasher: HasherType, | |
| 1312 | ||
| 1313 | const Error = WriterType.Error; | |
| 1314 | const Writer = std.io.Writer(*@This(), Error, write); | |
| 1315 | ||
| 1316 | fn write(hashed_writer: *@This(), buf: []const u8) Error!usize { | |
| 1317 | const amt = try hashed_writer.child_writer.write(buf); | |
| 1318 | hashed_writer.hasher.update(buf); | |
| 1319 | return amt; | |
| 1320 | } | |
| 1321 | ||
| 1322 | fn writer(hashed_writer: *@This()) Writer { | |
| 1323 | return .{ .context = hashed_writer }; | |
| 1324 | } | |
| 1325 | }; | |
| 1326 | } | |
| 1327 | ||
| 1328 | fn hashedWriter( | |
| 1329 | writer: anytype, | |
| 1330 | hasher: anytype, | |
| 1331 | ) HashedWriter(@TypeOf(writer), @TypeOf(hasher)) { | |
| 1332 | return .{ .child_writer = writer, .hasher = hasher }; | |
| 1333 | } | |
| 1334 | ||
| 1335 | test "packfile indexing and checkout" { | |
| 1336 | // To verify the contents of this packfile without using the code in this | |
| 1337 | // file: | |
| 1338 | // | |
| 1339 | // 1. Create a new empty Git repository (`git init`) | |
| 1340 | // 2. `git unpack-objects <path/to/testdata.pack` | |
| 1341 | // 3. `git fsck` -> note the "dangling commit" ID (which matches the commit | |
| 1342 | // checked out below) | |
| 1343 | // 4. `git checkout dd582c0720819ab7130b103635bd7271b9fd4feb` | |
| 1344 | const testrepo_pack = @embedFile("git/testdata/testrepo.pack"); | |
| 1345 | ||
| 1346 | var git_dir = testing.tmpDir(.{}); | |
| 1347 | defer git_dir.cleanup(); | |
| 1348 | var pack_file = try git_dir.dir.createFile("testrepo.pack", .{ .read = true }); | |
| 1349 | defer pack_file.close(); | |
| 1350 | try pack_file.writeAll(testrepo_pack); | |
| 1351 | ||
| 1352 | var index_file = try git_dir.dir.createFile("testrepo.idx", .{ .read = true }); | |
| 1353 | defer index_file.close(); | |
| 1354 | try indexPack(testing.allocator, pack_file, index_file.writer()); | |
| 1355 | ||
| 1356 | // Arbitrary size limit on files read while checking the repository contents | |
| 1357 | // (all files in the test repo are known to be much smaller than this) | |
| 1358 | const max_file_size = 4096; | |
| 1359 | ||
| 1360 | const index_file_data = try git_dir.dir.readFileAlloc(testing.allocator, "testrepo.idx", max_file_size); | |
| 1361 | defer testing.allocator.free(index_file_data); | |
| 1362 | // testrepo.idx is generated by Git. The index created by this file should | |
| 1363 | // match it exactly. Running `git verify-pack -v testrepo.pack` can verify | |
| 1364 | // this. | |
| 1365 | const testrepo_idx = @embedFile("git/testdata/testrepo.idx"); | |
| 1366 | try testing.expectEqualSlices(u8, testrepo_idx, index_file_data); | |
| 1367 | ||
| 1368 | var repository = try Repository.init(testing.allocator, pack_file, index_file); | |
| 1369 | defer repository.deinit(); | |
| 1370 | ||
| 1371 | var worktree = testing.tmpIterableDir(.{}); | |
| 1372 | defer worktree.cleanup(); | |
| 1373 | ||
| 1374 | const commit_id = try parseOid("dd582c0720819ab7130b103635bd7271b9fd4feb"); | |
| 1375 | try repository.checkout(worktree.iterable_dir.dir, commit_id); | |
| 1376 | ||
| 1377 | const expected_files: []const []const u8 = &.{ | |
| 1378 | "dir/file", | |
| 1379 | "dir/subdir/file", | |
| 1380 | "dir/subdir/file2", | |
| 1381 | "dir2/file", | |
| 1382 | "dir3/file", | |
| 1383 | "dir3/file2", | |
| 1384 | "file", | |
| 1385 | "file2", | |
| 1386 | "file3", | |
| 1387 | "file4", | |
| 1388 | "file5", | |
| 1389 | "file6", | |
| 1390 | "file7", | |
| 1391 | "file8", | |
| 1392 | "file9", | |
| 1393 | }; | |
| 1394 | var actual_files: std.ArrayListUnmanaged([]u8) = .{}; | |
| 1395 | defer actual_files.deinit(testing.allocator); | |
| 1396 | defer for (actual_files.items) |file| testing.allocator.free(file); | |
| 1397 | var walker = try worktree.iterable_dir.walk(testing.allocator); | |
| 1398 | defer walker.deinit(); | |
| 1399 | while (try walker.next()) |entry| { | |
| 1400 | if (entry.kind != .file) continue; | |
| 1401 | var path = try testing.allocator.dupe(u8, entry.path); | |
| 1402 | errdefer testing.allocator.free(path); | |
| 1403 | mem.replaceScalar(u8, path, std.fs.path.sep, '/'); | |
| 1404 | try actual_files.append(testing.allocator, path); | |
| 1405 | } | |
| 1406 | mem.sortUnstable([]u8, actual_files.items, {}, struct { | |
| 1407 | fn lessThan(_: void, a: []u8, b: []u8) bool { | |
| 1408 | return mem.lessThan(u8, a, b); | |
| 1409 | } | |
| 1410 | }.lessThan); | |
| 1411 | try testing.expectEqualDeep(expected_files, actual_files.items); | |
| 1412 | ||
| 1413 | const expected_file_contents = | |
| 1414 | \\revision 1 | |
| 1415 | \\revision 2 | |
| 1416 | \\revision 4 | |
| 1417 | \\revision 5 | |
| 1418 | \\revision 7 | |
| 1419 | \\revision 8 | |
| 1420 | \\revision 9 | |
| 1421 | \\revision 10 | |
| 1422 | \\revision 12 | |
| 1423 | \\revision 13 | |
| 1424 | \\revision 14 | |
| 1425 | \\revision 18 | |
| 1426 | \\revision 19 | |
| 1427 | \\ | |
| 1428 | ; | |
| 1429 | const actual_file_contents = try worktree.iterable_dir.dir.readFileAlloc(testing.allocator, "file", max_file_size); | |
| 1430 | defer testing.allocator.free(actual_file_contents); | |
| 1431 | try testing.expectEqualStrings(expected_file_contents, actual_file_contents); | |
| 1432 | } | |
| 1433 | ||
| 1434 | /// Checks out a commit of a packfile. Intended for experimenting with and | |
| 1435 | /// benchmarking possible optimizations to the indexing and checkout behavior. | |
| 1436 | pub fn main() !void { | |
| 1437 | const allocator = std.heap.c_allocator; | |
| 1438 | ||
| 1439 | const args = try std.process.argsAlloc(allocator); | |
| 1440 | defer std.process.argsFree(allocator, args); | |
| 1441 | if (args.len != 4) { | |
| 1442 | return error.InvalidArguments; // Arguments: packfile commit worktree | |
| 1443 | } | |
| 1444 | ||
| 1445 | var pack_file = try std.fs.cwd().openFile(args[1], .{}); | |
| 1446 | defer pack_file.close(); | |
| 1447 | const commit = try parseOid(args[2]); | |
| 1448 | var worktree = try std.fs.cwd().makeOpenPath(args[3], .{}); | |
| 1449 | defer worktree.close(); | |
| 1450 | ||
| 1451 | var git_dir = try worktree.makeOpenPath(".git", .{}); | |
| 1452 | defer git_dir.close(); | |
| 1453 | ||
| 1454 | std.debug.print("Starting index...\n", .{}); | |
| 1455 | var index_file = try git_dir.createFile("idx", .{ .read = true }); | |
| 1456 | defer index_file.close(); | |
| 1457 | var index_buffered_writer = std.io.bufferedWriter(index_file.writer()); | |
| 1458 | try indexPack(allocator, pack_file, index_buffered_writer.writer()); | |
| 1459 | try index_buffered_writer.flush(); | |
| 1460 | try index_file.sync(); | |
| 1461 | ||
| 1462 | std.debug.print("Starting checkout...\n", .{}); | |
| 1463 | var repository = try Repository.init(allocator, pack_file, index_file); | |
| 1464 | defer repository.deinit(); | |
| 1465 | try repository.checkout(worktree, commit); | |
| 1466 | } |
src/Package/Fetch/git/testdata/testrepo.idx created| Binary files /dev/null and b/src/Package/Fetch/git/testdata/testrepo.idx differ |
src/Package/Fetch/git/testdata/testrepo.pack created| Binary files /dev/null and b/src/Package/Fetch/git/testdata/testrepo.pack differ |
src/Package/Manifest.zig created+564| ... | ... | @@ -0,0 +1,564 @@ |
| 1 | pub const max_bytes = 10 * 1024 * 1024; | |
| 2 | pub const basename = "build.zig.zon"; | |
| 3 | pub const Hash = std.crypto.hash.sha2.Sha256; | |
| 4 | pub const Digest = [Hash.digest_length]u8; | |
| 5 | pub const multihash_len = 1 + 1 + Hash.digest_length; | |
| 6 | pub const multihash_hex_digest_len = 2 * multihash_len; | |
| 7 | pub const MultiHashHexDigest = [multihash_hex_digest_len]u8; | |
| 8 | ||
| 9 | pub const Dependency = struct { | |
| 10 | location: Location, | |
| 11 | location_tok: Ast.TokenIndex, | |
| 12 | hash: ?[]const u8, | |
| 13 | hash_tok: Ast.TokenIndex, | |
| 14 | ||
| 15 | pub const Location = union(enum) { | |
| 16 | url: []const u8, | |
| 17 | path: []const u8, | |
| 18 | }; | |
| 19 | }; | |
| 20 | ||
| 21 | pub const ErrorMessage = struct { | |
| 22 | msg: []const u8, | |
| 23 | tok: Ast.TokenIndex, | |
| 24 | off: u32, | |
| 25 | }; | |
| 26 | ||
| 27 | pub const MultihashFunction = enum(u16) { | |
| 28 | identity = 0x00, | |
| 29 | sha1 = 0x11, | |
| 30 | @"sha2-256" = 0x12, | |
| 31 | @"sha2-512" = 0x13, | |
| 32 | @"sha3-512" = 0x14, | |
| 33 | @"sha3-384" = 0x15, | |
| 34 | @"sha3-256" = 0x16, | |
| 35 | @"sha3-224" = 0x17, | |
| 36 | @"sha2-384" = 0x20, | |
| 37 | @"sha2-256-trunc254-padded" = 0x1012, | |
| 38 | @"sha2-224" = 0x1013, | |
| 39 | @"sha2-512-224" = 0x1014, | |
| 40 | @"sha2-512-256" = 0x1015, | |
| 41 | @"blake2b-256" = 0xb220, | |
| 42 | _, | |
| 43 | }; | |
| 44 | ||
| 45 | pub const multihash_function: MultihashFunction = switch (Hash) { | |
| 46 | std.crypto.hash.sha2.Sha256 => .@"sha2-256", | |
| 47 | else => @compileError("unreachable"), | |
| 48 | }; | |
| 49 | comptime { | |
| 50 | // We avoid unnecessary uleb128 code in hexDigest by asserting here the | |
| 51 | // values are small enough to be contained in the one-byte encoding. | |
| 52 | assert(@intFromEnum(multihash_function) < 127); | |
| 53 | assert(Hash.digest_length < 127); | |
| 54 | } | |
| 55 | ||
| 56 | name: []const u8, | |
| 57 | version: std.SemanticVersion, | |
| 58 | dependencies: std.StringArrayHashMapUnmanaged(Dependency), | |
| 59 | paths: std.StringArrayHashMapUnmanaged(void), | |
| 60 | ||
| 61 | errors: []ErrorMessage, | |
| 62 | arena_state: std.heap.ArenaAllocator.State, | |
| 63 | ||
| 64 | pub const ParseOptions = struct { | |
| 65 | allow_missing_paths_field: bool = false, | |
| 66 | }; | |
| 67 | ||
| 68 | pub const Error = Allocator.Error; | |
| 69 | ||
| 70 | pub fn parse(gpa: Allocator, ast: std.zig.Ast, options: ParseOptions) Error!Manifest { | |
| 71 | const node_tags = ast.nodes.items(.tag); | |
| 72 | const node_datas = ast.nodes.items(.data); | |
| 73 | assert(node_tags[0] == .root); | |
| 74 | const main_node_index = node_datas[0].lhs; | |
| 75 | ||
| 76 | var arena_instance = std.heap.ArenaAllocator.init(gpa); | |
| 77 | errdefer arena_instance.deinit(); | |
| 78 | ||
| 79 | var p: Parse = .{ | |
| 80 | .gpa = gpa, | |
| 81 | .ast = ast, | |
| 82 | .arena = arena_instance.allocator(), | |
| 83 | .errors = .{}, | |
| 84 | ||
| 85 | .name = undefined, | |
| 86 | .version = undefined, | |
| 87 | .dependencies = .{}, | |
| 88 | .paths = .{}, | |
| 89 | .allow_missing_paths_field = options.allow_missing_paths_field, | |
| 90 | .buf = .{}, | |
| 91 | }; | |
| 92 | defer p.buf.deinit(gpa); | |
| 93 | defer p.errors.deinit(gpa); | |
| 94 | defer p.dependencies.deinit(gpa); | |
| 95 | defer p.paths.deinit(gpa); | |
| 96 | ||
| 97 | p.parseRoot(main_node_index) catch |err| switch (err) { | |
| 98 | error.ParseFailure => assert(p.errors.items.len > 0), | |
| 99 | else => |e| return e, | |
| 100 | }; | |
| 101 | ||
| 102 | return .{ | |
| 103 | .name = p.name, | |
| 104 | .version = p.version, | |
| 105 | .dependencies = try p.dependencies.clone(p.arena), | |
| 106 | .paths = try p.paths.clone(p.arena), | |
| 107 | .errors = try p.arena.dupe(ErrorMessage, p.errors.items), | |
| 108 | .arena_state = arena_instance.state, | |
| 109 | }; | |
| 110 | } | |
| 111 | ||
| 112 | pub fn deinit(man: *Manifest, gpa: Allocator) void { | |
| 113 | man.arena_state.promote(gpa).deinit(); | |
| 114 | man.* = undefined; | |
| 115 | } | |
| 116 | ||
| 117 | const hex_charset = "0123456789abcdef"; | |
| 118 | ||
| 119 | pub fn hex64(x: u64) [16]u8 { | |
| 120 | var result: [16]u8 = undefined; | |
| 121 | var i: usize = 0; | |
| 122 | while (i < 8) : (i += 1) { | |
| 123 | const byte = @as(u8, @truncate(x >> @as(u6, @intCast(8 * i)))); | |
| 124 | result[i * 2 + 0] = hex_charset[byte >> 4]; | |
| 125 | result[i * 2 + 1] = hex_charset[byte & 15]; | |
| 126 | } | |
| 127 | return result; | |
| 128 | } | |
| 129 | ||
| 130 | test hex64 { | |
| 131 | const s = "[" ++ hex64(0x12345678_abcdef00) ++ "]"; | |
| 132 | try std.testing.expectEqualStrings("[00efcdab78563412]", s); | |
| 133 | } | |
| 134 | ||
| 135 | pub fn hexDigest(digest: Digest) MultiHashHexDigest { | |
| 136 | var result: MultiHashHexDigest = undefined; | |
| 137 | ||
| 138 | result[0] = hex_charset[@intFromEnum(multihash_function) >> 4]; | |
| 139 | result[1] = hex_charset[@intFromEnum(multihash_function) & 15]; | |
| 140 | ||
| 141 | result[2] = hex_charset[Hash.digest_length >> 4]; | |
| 142 | result[3] = hex_charset[Hash.digest_length & 15]; | |
| 143 | ||
| 144 | for (digest, 0..) |byte, i| { | |
| 145 | result[4 + i * 2] = hex_charset[byte >> 4]; | |
| 146 | result[5 + i * 2] = hex_charset[byte & 15]; | |
| 147 | } | |
| 148 | return result; | |
| 149 | } | |
| 150 | ||
| 151 | const Parse = struct { | |
| 152 | gpa: Allocator, | |
| 153 | ast: std.zig.Ast, | |
| 154 | arena: Allocator, | |
| 155 | buf: std.ArrayListUnmanaged(u8), | |
| 156 | errors: std.ArrayListUnmanaged(ErrorMessage), | |
| 157 | ||
| 158 | name: []const u8, | |
| 159 | version: std.SemanticVersion, | |
| 160 | dependencies: std.StringArrayHashMapUnmanaged(Dependency), | |
| 161 | paths: std.StringArrayHashMapUnmanaged(void), | |
| 162 | allow_missing_paths_field: bool, | |
| 163 | ||
| 164 | const InnerError = error{ ParseFailure, OutOfMemory }; | |
| 165 | ||
| 166 | fn parseRoot(p: *Parse, node: Ast.Node.Index) !void { | |
| 167 | const ast = p.ast; | |
| 168 | const main_tokens = ast.nodes.items(.main_token); | |
| 169 | const main_token = main_tokens[node]; | |
| 170 | ||
| 171 | var buf: [2]Ast.Node.Index = undefined; | |
| 172 | const struct_init = ast.fullStructInit(&buf, node) orelse { | |
| 173 | return fail(p, main_token, "expected top level expression to be a struct", .{}); | |
| 174 | }; | |
| 175 | ||
| 176 | var have_name = false; | |
| 177 | var have_version = false; | |
| 178 | var have_included_paths = false; | |
| 179 | ||
| 180 | for (struct_init.ast.fields) |field_init| { | |
| 181 | const name_token = ast.firstToken(field_init) - 2; | |
| 182 | const field_name = try identifierTokenString(p, name_token); | |
| 183 | // We could get fancy with reflection and comptime logic here but doing | |
| 184 | // things manually provides an opportunity to do any additional verification | |
| 185 | // that is desirable on a per-field basis. | |
| 186 | if (mem.eql(u8, field_name, "dependencies")) { | |
| 187 | try parseDependencies(p, field_init); | |
| 188 | } else if (mem.eql(u8, field_name, "paths")) { | |
| 189 | have_included_paths = true; | |
| 190 | try parseIncludedPaths(p, field_init); | |
| 191 | } else if (mem.eql(u8, field_name, "name")) { | |
| 192 | p.name = try parseString(p, field_init); | |
| 193 | have_name = true; | |
| 194 | } else if (mem.eql(u8, field_name, "version")) { | |
| 195 | const version_text = try parseString(p, field_init); | |
| 196 | p.version = std.SemanticVersion.parse(version_text) catch |err| v: { | |
| 197 | try appendError(p, main_tokens[field_init], "unable to parse semantic version: {s}", .{@errorName(err)}); | |
| 198 | break :v undefined; | |
| 199 | }; | |
| 200 | have_version = true; | |
| 201 | } else { | |
| 202 | // Ignore unknown fields so that we can add fields in future zig | |
| 203 | // versions without breaking older zig versions. | |
| 204 | } | |
| 205 | } | |
| 206 | ||
| 207 | if (!have_name) { | |
| 208 | try appendError(p, main_token, "missing top-level 'name' field", .{}); | |
| 209 | } | |
| 210 | ||
| 211 | if (!have_version) { | |
| 212 | try appendError(p, main_token, "missing top-level 'version' field", .{}); | |
| 213 | } | |
| 214 | ||
| 215 | if (!have_included_paths) { | |
| 216 | if (p.allow_missing_paths_field) { | |
| 217 | try p.paths.put(p.gpa, "", {}); | |
| 218 | } else { | |
| 219 | try appendError(p, main_token, "missing top-level 'paths' field", .{}); | |
| 220 | } | |
| 221 | } | |
| 222 | } | |
| 223 | ||
| 224 | fn parseDependencies(p: *Parse, node: Ast.Node.Index) !void { | |
| 225 | const ast = p.ast; | |
| 226 | const main_tokens = ast.nodes.items(.main_token); | |
| 227 | ||
| 228 | var buf: [2]Ast.Node.Index = undefined; | |
| 229 | const struct_init = ast.fullStructInit(&buf, node) orelse { | |
| 230 | const tok = main_tokens[node]; | |
| 231 | return fail(p, tok, "expected dependencies expression to be a struct", .{}); | |
| 232 | }; | |
| 233 | ||
| 234 | for (struct_init.ast.fields) |field_init| { | |
| 235 | const name_token = ast.firstToken(field_init) - 2; | |
| 236 | const dep_name = try identifierTokenString(p, name_token); | |
| 237 | const dep = try parseDependency(p, field_init); | |
| 238 | try p.dependencies.put(p.gpa, dep_name, dep); | |
| 239 | } | |
| 240 | } | |
| 241 | ||
| 242 | fn parseDependency(p: *Parse, node: Ast.Node.Index) !Dependency { | |
| 243 | const ast = p.ast; | |
| 244 | const main_tokens = ast.nodes.items(.main_token); | |
| 245 | ||
| 246 | var buf: [2]Ast.Node.Index = undefined; | |
| 247 | const struct_init = ast.fullStructInit(&buf, node) orelse { | |
| 248 | const tok = main_tokens[node]; | |
| 249 | return fail(p, tok, "expected dependency expression to be a struct", .{}); | |
| 250 | }; | |
| 251 | ||
| 252 | var dep: Dependency = .{ | |
| 253 | .location = undefined, | |
| 254 | .location_tok = 0, | |
| 255 | .hash = null, | |
| 256 | .hash_tok = 0, | |
| 257 | }; | |
| 258 | var has_location = false; | |
| 259 | ||
| 260 | for (struct_init.ast.fields) |field_init| { | |
| 261 | const name_token = ast.firstToken(field_init) - 2; | |
| 262 | const field_name = try identifierTokenString(p, name_token); | |
| 263 | // We could get fancy with reflection and comptime logic here but doing | |
| 264 | // things manually provides an opportunity to do any additional verification | |
| 265 | // that is desirable on a per-field basis. | |
| 266 | if (mem.eql(u8, field_name, "url")) { | |
| 267 | if (has_location) { | |
| 268 | return fail(p, main_tokens[field_init], "dependency should specify only one of 'url' and 'path' fields.", .{}); | |
| 269 | } | |
| 270 | dep.location = .{ | |
| 271 | .url = parseString(p, field_init) catch |err| switch (err) { | |
| 272 | error.ParseFailure => continue, | |
| 273 | else => |e| return e, | |
| 274 | }, | |
| 275 | }; | |
| 276 | has_location = true; | |
| 277 | dep.location_tok = main_tokens[field_init]; | |
| 278 | } else if (mem.eql(u8, field_name, "path")) { | |
| 279 | if (has_location) { | |
| 280 | return fail(p, main_tokens[field_init], "dependency should specify only one of 'url' and 'path' fields.", .{}); | |
| 281 | } | |
| 282 | dep.location = .{ | |
| 283 | .path = parseString(p, field_init) catch |err| switch (err) { | |
| 284 | error.ParseFailure => continue, | |
| 285 | else => |e| return e, | |
| 286 | }, | |
| 287 | }; | |
| 288 | has_location = true; | |
| 289 | dep.location_tok = main_tokens[field_init]; | |
| 290 | } else if (mem.eql(u8, field_name, "hash")) { | |
| 291 | dep.hash = parseHash(p, field_init) catch |err| switch (err) { | |
| 292 | error.ParseFailure => continue, | |
| 293 | else => |e| return e, | |
| 294 | }; | |
| 295 | dep.hash_tok = main_tokens[field_init]; | |
| 296 | } else { | |
| 297 | // Ignore unknown fields so that we can add fields in future zig | |
| 298 | // versions without breaking older zig versions. | |
| 299 | } | |
| 300 | } | |
| 301 | ||
| 302 | if (!has_location) { | |
| 303 | try appendError(p, main_tokens[node], "dependency requires location field, one of 'url' or 'path'.", .{}); | |
| 304 | } | |
| 305 | ||
| 306 | return dep; | |
| 307 | } | |
| 308 | ||
| 309 | fn parseIncludedPaths(p: *Parse, node: Ast.Node.Index) !void { | |
| 310 | const ast = p.ast; | |
| 311 | const main_tokens = ast.nodes.items(.main_token); | |
| 312 | ||
| 313 | var buf: [2]Ast.Node.Index = undefined; | |
| 314 | const array_init = ast.fullArrayInit(&buf, node) orelse { | |
| 315 | const tok = main_tokens[node]; | |
| 316 | return fail(p, tok, "expected paths expression to be a struct", .{}); | |
| 317 | }; | |
| 318 | ||
| 319 | for (array_init.ast.elements) |elem_node| { | |
| 320 | const path_string = try parseString(p, elem_node); | |
| 321 | const normalized = try std.fs.path.resolve(p.arena, &.{path_string}); | |
| 322 | try p.paths.put(p.gpa, normalized, {}); | |
| 323 | } | |
| 324 | } | |
| 325 | ||
| 326 | fn parseString(p: *Parse, node: Ast.Node.Index) ![]const u8 { | |
| 327 | const ast = p.ast; | |
| 328 | const node_tags = ast.nodes.items(.tag); | |
| 329 | const main_tokens = ast.nodes.items(.main_token); | |
| 330 | if (node_tags[node] != .string_literal) { | |
| 331 | return fail(p, main_tokens[node], "expected string literal", .{}); | |
| 332 | } | |
| 333 | const str_lit_token = main_tokens[node]; | |
| 334 | const token_bytes = ast.tokenSlice(str_lit_token); | |
| 335 | p.buf.clearRetainingCapacity(); | |
| 336 | try parseStrLit(p, str_lit_token, &p.buf, token_bytes, 0); | |
| 337 | const duped = try p.arena.dupe(u8, p.buf.items); | |
| 338 | return duped; | |
| 339 | } | |
| 340 | ||
| 341 | fn parseHash(p: *Parse, node: Ast.Node.Index) ![]const u8 { | |
| 342 | const ast = p.ast; | |
| 343 | const main_tokens = ast.nodes.items(.main_token); | |
| 344 | const tok = main_tokens[node]; | |
| 345 | const h = try parseString(p, node); | |
| 346 | ||
| 347 | if (h.len >= 2) { | |
| 348 | const their_multihash_func = std.fmt.parseInt(u8, h[0..2], 16) catch |err| { | |
| 349 | return fail(p, tok, "invalid multihash value: unable to parse hash function: {s}", .{ | |
| 350 | @errorName(err), | |
| 351 | }); | |
| 352 | }; | |
| 353 | if (@as(MultihashFunction, @enumFromInt(their_multihash_func)) != multihash_function) { | |
| 354 | return fail(p, tok, "unsupported hash function: only sha2-256 is supported", .{}); | |
| 355 | } | |
| 356 | } | |
| 357 | ||
| 358 | if (h.len != multihash_hex_digest_len) { | |
| 359 | return fail(p, tok, "wrong hash size. expected: {d}, found: {d}", .{ | |
| 360 | multihash_hex_digest_len, h.len, | |
| 361 | }); | |
| 362 | } | |
| 363 | ||
| 364 | return h; | |
| 365 | } | |
| 366 | ||
| 367 | /// TODO: try to DRY this with AstGen.identifierTokenString | |
| 368 | fn identifierTokenString(p: *Parse, token: Ast.TokenIndex) InnerError![]const u8 { | |
| 369 | const ast = p.ast; | |
| 370 | const token_tags = ast.tokens.items(.tag); | |
| 371 | assert(token_tags[token] == .identifier); | |
| 372 | const ident_name = ast.tokenSlice(token); | |
| 373 | if (!mem.startsWith(u8, ident_name, "@")) { | |
| 374 | return ident_name; | |
| 375 | } | |
| 376 | p.buf.clearRetainingCapacity(); | |
| 377 | try parseStrLit(p, token, &p.buf, ident_name, 1); | |
| 378 | const duped = try p.arena.dupe(u8, p.buf.items); | |
| 379 | return duped; | |
| 380 | } | |
| 381 | ||
| 382 | /// TODO: try to DRY this with AstGen.parseStrLit | |
| 383 | fn parseStrLit( | |
| 384 | p: *Parse, | |
| 385 | token: Ast.TokenIndex, | |
| 386 | buf: *std.ArrayListUnmanaged(u8), | |
| 387 | bytes: []const u8, | |
| 388 | offset: u32, | |
| 389 | ) InnerError!void { | |
| 390 | const raw_string = bytes[offset..]; | |
| 391 | var buf_managed = buf.toManaged(p.gpa); | |
| 392 | const result = std.zig.string_literal.parseWrite(buf_managed.writer(), raw_string); | |
| 393 | buf.* = buf_managed.moveToUnmanaged(); | |
| 394 | switch (try result) { | |
| 395 | .success => {}, | |
| 396 | .failure => |err| try p.appendStrLitError(err, token, bytes, offset), | |
| 397 | } | |
| 398 | } | |
| 399 | ||
| 400 | /// TODO: try to DRY this with AstGen.failWithStrLitError | |
| 401 | fn appendStrLitError( | |
| 402 | p: *Parse, | |
| 403 | err: std.zig.string_literal.Error, | |
| 404 | token: Ast.TokenIndex, | |
| 405 | bytes: []const u8, | |
| 406 | offset: u32, | |
| 407 | ) Allocator.Error!void { | |
| 408 | const raw_string = bytes[offset..]; | |
| 409 | switch (err) { | |
| 410 | .invalid_escape_character => |bad_index| { | |
| 411 | try p.appendErrorOff( | |
| 412 | token, | |
| 413 | offset + @as(u32, @intCast(bad_index)), | |
| 414 | "invalid escape character: '{c}'", | |
| 415 | .{raw_string[bad_index]}, | |
| 416 | ); | |
| 417 | }, | |
| 418 | .expected_hex_digit => |bad_index| { | |
| 419 | try p.appendErrorOff( | |
| 420 | token, | |
| 421 | offset + @as(u32, @intCast(bad_index)), | |
| 422 | "expected hex digit, found '{c}'", | |
| 423 | .{raw_string[bad_index]}, | |
| 424 | ); | |
| 425 | }, | |
| 426 | .empty_unicode_escape_sequence => |bad_index| { | |
| 427 | try p.appendErrorOff( | |
| 428 | token, | |
| 429 | offset + @as(u32, @intCast(bad_index)), | |
| 430 | "empty unicode escape sequence", | |
| 431 | .{}, | |
| 432 | ); | |
| 433 | }, | |
| 434 | .expected_hex_digit_or_rbrace => |bad_index| { | |
| 435 | try p.appendErrorOff( | |
| 436 | token, | |
| 437 | offset + @as(u32, @intCast(bad_index)), | |
| 438 | "expected hex digit or '}}', found '{c}'", | |
| 439 | .{raw_string[bad_index]}, | |
| 440 | ); | |
| 441 | }, | |
| 442 | .invalid_unicode_codepoint => |bad_index| { | |
| 443 | try p.appendErrorOff( | |
| 444 | token, | |
| 445 | offset + @as(u32, @intCast(bad_index)), | |
| 446 | "unicode escape does not correspond to a valid codepoint", | |
| 447 | .{}, | |
| 448 | ); | |
| 449 | }, | |
| 450 | .expected_lbrace => |bad_index| { | |
| 451 | try p.appendErrorOff( | |
| 452 | token, | |
| 453 | offset + @as(u32, @intCast(bad_index)), | |
| 454 | "expected '{{', found '{c}", | |
| 455 | .{raw_string[bad_index]}, | |
| 456 | ); | |
| 457 | }, | |
| 458 | .expected_rbrace => |bad_index| { | |
| 459 | try p.appendErrorOff( | |
| 460 | token, | |
| 461 | offset + @as(u32, @intCast(bad_index)), | |
| 462 | "expected '}}', found '{c}", | |
| 463 | .{raw_string[bad_index]}, | |
| 464 | ); | |
| 465 | }, | |
| 466 | .expected_single_quote => |bad_index| { | |
| 467 | try p.appendErrorOff( | |
| 468 | token, | |
| 469 | offset + @as(u32, @intCast(bad_index)), | |
| 470 | "expected single quote ('), found '{c}", | |
| 471 | .{raw_string[bad_index]}, | |
| 472 | ); | |
| 473 | }, | |
| 474 | .invalid_character => |bad_index| { | |
| 475 | try p.appendErrorOff( | |
| 476 | token, | |
| 477 | offset + @as(u32, @intCast(bad_index)), | |
| 478 | "invalid byte in string or character literal: '{c}'", | |
| 479 | .{raw_string[bad_index]}, | |
| 480 | ); | |
| 481 | }, | |
| 482 | } | |
| 483 | } | |
| 484 | ||
| 485 | fn fail( | |
| 486 | p: *Parse, | |
| 487 | tok: Ast.TokenIndex, | |
| 488 | comptime fmt: []const u8, | |
| 489 | args: anytype, | |
| 490 | ) InnerError { | |
| 491 | try appendError(p, tok, fmt, args); | |
| 492 | return error.ParseFailure; | |
| 493 | } | |
| 494 | ||
| 495 | fn appendError(p: *Parse, tok: Ast.TokenIndex, comptime fmt: []const u8, args: anytype) !void { | |
| 496 | return appendErrorOff(p, tok, 0, fmt, args); | |
| 497 | } | |
| 498 | ||
| 499 | fn appendErrorOff( | |
| 500 | p: *Parse, | |
| 501 | tok: Ast.TokenIndex, | |
| 502 | byte_offset: u32, | |
| 503 | comptime fmt: []const u8, | |
| 504 | args: anytype, | |
| 505 | ) Allocator.Error!void { | |
| 506 | try p.errors.append(p.gpa, .{ | |
| 507 | .msg = try std.fmt.allocPrint(p.arena, fmt, args), | |
| 508 | .tok = tok, | |
| 509 | .off = byte_offset, | |
| 510 | }); | |
| 511 | } | |
| 512 | }; | |
| 513 | ||
| 514 | const Manifest = @This(); | |
| 515 | const std = @import("std"); | |
| 516 | const mem = std.mem; | |
| 517 | const Allocator = std.mem.Allocator; | |
| 518 | const assert = std.debug.assert; | |
| 519 | const Ast = std.zig.Ast; | |
| 520 | const testing = std.testing; | |
| 521 | ||
| 522 | test "basic" { | |
| 523 | const gpa = testing.allocator; | |
| 524 | ||
| 525 | const example = | |
| 526 | \\.{ | |
| 527 | \\ .name = "foo", | |
| 528 | \\ .version = "3.2.1", | |
| 529 | \\ .dependencies = .{ | |
| 530 | \\ .bar = .{ | |
| 531 | \\ .url = "https://example.com/baz.tar.gz", | |
| 532 | \\ .hash = "1220f1b680b6065fcfc94fe777f22e73bcb7e2767e5f4d99d4255fe76ded69c7a35f", | |
| 533 | \\ }, | |
| 534 | \\ }, | |
| 535 | \\} | |
| 536 | ; | |
| 537 | ||
| 538 | var ast = try std.zig.Ast.parse(gpa, example, .zon); | |
| 539 | defer ast.deinit(gpa); | |
| 540 | ||
| 541 | try testing.expect(ast.errors.len == 0); | |
| 542 | ||
| 543 | var manifest = try Manifest.parse(gpa, ast); | |
| 544 | defer manifest.deinit(gpa); | |
| 545 | ||
| 546 | try testing.expectEqualStrings("foo", manifest.name); | |
| 547 | ||
| 548 | try testing.expectEqual(@as(std.SemanticVersion, .{ | |
| 549 | .major = 3, | |
| 550 | .minor = 2, | |
| 551 | .patch = 1, | |
| 552 | }), manifest.version); | |
| 553 | ||
| 554 | try testing.expect(manifest.dependencies.count() == 1); | |
| 555 | try testing.expectEqualStrings("bar", manifest.dependencies.keys()[0]); | |
| 556 | try testing.expectEqualStrings( | |
| 557 | "https://example.com/baz.tar.gz", | |
| 558 | manifest.dependencies.values()[0].url, | |
| 559 | ); | |
| 560 | try testing.expectEqualStrings( | |
| 561 | "1220f1b680b6065fcfc94fe777f22e73bcb7e2767e5f4d99d4255fe76ded69c7a35f", | |
| 562 | manifest.dependencies.values()[0].hash orelse return error.TestFailed, | |
| 563 | ); | |
| 564 | } |
src/git.zig deleted-1468| ... | ... | @@ -1,1468 +0,0 @@ |
| 1 | //! Git support for package fetching. | |
| 2 | //! | |
| 3 | //! This is not intended to support all features of Git: it is limited to the | |
| 4 | //! basic functionality needed to clone a repository for the purpose of fetching | |
| 5 | //! a package. | |
| 6 | ||
| 7 | const std = @import("std"); | |
| 8 | const mem = std.mem; | |
| 9 | const testing = std.testing; | |
| 10 | const Allocator = mem.Allocator; | |
| 11 | const Sha1 = std.crypto.hash.Sha1; | |
| 12 | const assert = std.debug.assert; | |
| 13 | ||
| 14 | const ProgressReader = @import("Package.zig").ProgressReader; | |
| 15 | ||
| 16 | pub const oid_length = Sha1.digest_length; | |
| 17 | pub const fmt_oid_length = 2 * oid_length; | |
| 18 | /// The ID of a Git object (an SHA-1 hash). | |
| 19 | pub const Oid = [oid_length]u8; | |
| 20 | ||
| 21 | pub fn parseOid(s: []const u8) !Oid { | |
| 22 | if (s.len != fmt_oid_length) return error.InvalidOid; | |
| 23 | var oid: Oid = undefined; | |
| 24 | for (&oid, 0..) |*b, i| { | |
| 25 | b.* = std.fmt.parseUnsigned(u8, s[2 * i ..][0..2], 16) catch return error.InvalidOid; | |
| 26 | } | |
| 27 | return oid; | |
| 28 | } | |
| 29 | ||
| 30 | test parseOid { | |
| 31 | try testing.expectEqualSlices( | |
| 32 | u8, | |
| 33 | &.{ 0xCE, 0x91, 0x9C, 0xCF, 0x45, 0x95, 0x18, 0x56, 0xA7, 0x62, 0xFF, 0xDB, 0x8E, 0xF8, 0x50, 0x30, 0x1C, 0xD8, 0xC5, 0x88 }, | |
| 34 | &try parseOid("ce919ccf45951856a762ffdb8ef850301cd8c588"), | |
| 35 | ); | |
| 36 | try testing.expectError(error.InvalidOid, parseOid("ce919ccf")); | |
| 37 | try testing.expectError(error.InvalidOid, parseOid("master")); | |
| 38 | try testing.expectError(error.InvalidOid, parseOid("HEAD")); | |
| 39 | } | |
| 40 | ||
| 41 | pub const Diagnostics = struct { | |
| 42 | allocator: Allocator, | |
| 43 | errors: std.ArrayListUnmanaged(Error) = .{}, | |
| 44 | ||
| 45 | pub const Error = union(enum) { | |
| 46 | unable_to_create_sym_link: struct { | |
| 47 | code: anyerror, | |
| 48 | file_name: []const u8, | |
| 49 | link_name: []const u8, | |
| 50 | }, | |
| 51 | }; | |
| 52 | ||
| 53 | pub fn deinit(d: *Diagnostics) void { | |
| 54 | for (d.errors.items) |item| { | |
| 55 | switch (item) { | |
| 56 | .unable_to_create_sym_link => |info| { | |
| 57 | d.allocator.free(info.file_name); | |
| 58 | d.allocator.free(info.link_name); | |
| 59 | }, | |
| 60 | } | |
| 61 | } | |
| 62 | d.errors.deinit(d.allocator); | |
| 63 | d.* = undefined; | |
| 64 | } | |
| 65 | }; | |
| 66 | ||
| 67 | pub const Repository = struct { | |
| 68 | odb: Odb, | |
| 69 | ||
| 70 | pub fn init(allocator: Allocator, pack_file: std.fs.File, index_file: std.fs.File) !Repository { | |
| 71 | return .{ .odb = try Odb.init(allocator, pack_file, index_file) }; | |
| 72 | } | |
| 73 | ||
| 74 | pub fn deinit(repository: *Repository) void { | |
| 75 | repository.odb.deinit(); | |
| 76 | repository.* = undefined; | |
| 77 | } | |
| 78 | ||
| 79 | /// Checks out the repository at `commit_oid` to `worktree`. | |
| 80 | pub fn checkout( | |
| 81 | repository: *Repository, | |
| 82 | worktree: std.fs.Dir, | |
| 83 | commit_oid: Oid, | |
| 84 | diagnostics: *Diagnostics, | |
| 85 | ) !void { | |
| 86 | try repository.odb.seekOid(commit_oid); | |
| 87 | const tree_oid = tree_oid: { | |
| 88 | var commit_object = try repository.odb.readObject(); | |
| 89 | if (commit_object.type != .commit) return error.NotACommit; | |
| 90 | break :tree_oid try getCommitTree(commit_object.data); | |
| 91 | }; | |
| 92 | try repository.checkoutTree(worktree, tree_oid, "", diagnostics); | |
| 93 | } | |
| 94 | ||
| 95 | /// Checks out the tree at `tree_oid` to `worktree`. | |
| 96 | fn checkoutTree( | |
| 97 | repository: *Repository, | |
| 98 | dir: std.fs.Dir, | |
| 99 | tree_oid: Oid, | |
| 100 | current_path: []const u8, | |
| 101 | diagnostics: *Diagnostics, | |
| 102 | ) !void { | |
| 103 | try repository.odb.seekOid(tree_oid); | |
| 104 | const tree_object = try repository.odb.readObject(); | |
| 105 | if (tree_object.type != .tree) return error.NotATree; | |
| 106 | // The tree object may be evicted from the object cache while we're | |
| 107 | // iterating over it, so we can make a defensive copy here to make sure | |
| 108 | // it remains valid until we're done with it | |
| 109 | const tree_data = try repository.odb.allocator.dupe(u8, tree_object.data); | |
| 110 | defer repository.odb.allocator.free(tree_data); | |
| 111 | ||
| 112 | var tree_iter: TreeIterator = .{ .data = tree_data }; | |
| 113 | while (try tree_iter.next()) |entry| { | |
| 114 | switch (entry.type) { | |
| 115 | .directory => { | |
| 116 | try dir.makeDir(entry.name); | |
| 117 | var subdir = try dir.openDir(entry.name, .{}); | |
| 118 | defer subdir.close(); | |
| 119 | const sub_path = try std.fs.path.join(repository.odb.allocator, &.{ current_path, entry.name }); | |
| 120 | defer repository.odb.allocator.free(sub_path); | |
| 121 | try repository.checkoutTree(subdir, entry.oid, sub_path, diagnostics); | |
| 122 | }, | |
| 123 | .file => { | |
| 124 | var file = try dir.createFile(entry.name, .{}); | |
| 125 | defer file.close(); | |
| 126 | try repository.odb.seekOid(entry.oid); | |
| 127 | var file_object = try repository.odb.readObject(); | |
| 128 | if (file_object.type != .blob) return error.InvalidFile; | |
| 129 | try file.writeAll(file_object.data); | |
| 130 | try file.sync(); | |
| 131 | }, | |
| 132 | .symlink => { | |
| 133 | try repository.odb.seekOid(entry.oid); | |
| 134 | var symlink_object = try repository.odb.readObject(); | |
| 135 | if (symlink_object.type != .blob) return error.InvalidFile; | |
| 136 | const link_name = symlink_object.data; | |
| 137 | dir.symLink(link_name, entry.name, .{}) catch |e| { | |
| 138 | const file_name = try std.fs.path.join(diagnostics.allocator, &.{ current_path, entry.name }); | |
| 139 | errdefer diagnostics.allocator.free(file_name); | |
| 140 | const link_name_dup = try diagnostics.allocator.dupe(u8, link_name); | |
| 141 | errdefer diagnostics.allocator.free(link_name_dup); | |
| 142 | try diagnostics.errors.append(diagnostics.allocator, .{ .unable_to_create_sym_link = .{ | |
| 143 | .code = e, | |
| 144 | .file_name = file_name, | |
| 145 | .link_name = link_name_dup, | |
| 146 | } }); | |
| 147 | }; | |
| 148 | }, | |
| 149 | .gitlink => { | |
| 150 | // Consistent with git archive behavior, create the directory but | |
| 151 | // do nothing else | |
| 152 | try dir.makeDir(entry.name); | |
| 153 | }, | |
| 154 | } | |
| 155 | } | |
| 156 | } | |
| 157 | ||
| 158 | /// Returns the ID of the tree associated with the given commit (provided as | |
| 159 | /// raw object data). | |
| 160 | fn getCommitTree(commit_data: []const u8) !Oid { | |
| 161 | if (!mem.startsWith(u8, commit_data, "tree ") or | |
| 162 | commit_data.len < "tree ".len + fmt_oid_length + "\n".len or | |
| 163 | commit_data["tree ".len + fmt_oid_length] != '\n') | |
| 164 | { | |
| 165 | return error.InvalidCommit; | |
| 166 | } | |
| 167 | return try parseOid(commit_data["tree ".len..][0..fmt_oid_length]); | |
| 168 | } | |
| 169 | ||
| 170 | const TreeIterator = struct { | |
| 171 | data: []const u8, | |
| 172 | pos: usize = 0, | |
| 173 | ||
| 174 | const Entry = struct { | |
| 175 | type: Type, | |
| 176 | executable: bool, | |
| 177 | name: [:0]const u8, | |
| 178 | oid: Oid, | |
| 179 | ||
| 180 | const Type = enum(u4) { | |
| 181 | directory = 0o4, | |
| 182 | file = 0o10, | |
| 183 | symlink = 0o12, | |
| 184 | gitlink = 0o16, | |
| 185 | }; | |
| 186 | }; | |
| 187 | ||
| 188 | fn next(iterator: *TreeIterator) !?Entry { | |
| 189 | if (iterator.pos == iterator.data.len) return null; | |
| 190 | ||
| 191 | const mode_end = mem.indexOfScalarPos(u8, iterator.data, iterator.pos, ' ') orelse return error.InvalidTree; | |
| 192 | const mode: packed struct { | |
| 193 | permission: u9, | |
| 194 | unused: u3, | |
| 195 | type: u4, | |
| 196 | } = @bitCast(std.fmt.parseUnsigned(u16, iterator.data[iterator.pos..mode_end], 8) catch return error.InvalidTree); | |
| 197 | const @"type" = std.meta.intToEnum(Entry.Type, mode.type) catch return error.InvalidTree; | |
| 198 | const executable = switch (mode.permission) { | |
| 199 | 0 => if (@"type" == .file) return error.InvalidTree else false, | |
| 200 | 0o644 => if (@"type" != .file) return error.InvalidTree else false, | |
| 201 | 0o755 => if (@"type" != .file) return error.InvalidTree else true, | |
| 202 | else => return error.InvalidTree, | |
| 203 | }; | |
| 204 | iterator.pos = mode_end + 1; | |
| 205 | ||
| 206 | const name_end = mem.indexOfScalarPos(u8, iterator.data, iterator.pos, 0) orelse return error.InvalidTree; | |
| 207 | const name = iterator.data[iterator.pos..name_end :0]; | |
| 208 | iterator.pos = name_end + 1; | |
| 209 | ||
| 210 | if (iterator.pos + oid_length > iterator.data.len) return error.InvalidTree; | |
| 211 | const oid = iterator.data[iterator.pos..][0..oid_length].*; | |
| 212 | iterator.pos += oid_length; | |
| 213 | ||
| 214 | return .{ .type = @"type", .executable = executable, .name = name, .oid = oid }; | |
| 215 | } | |
| 216 | }; | |
| 217 | }; | |
| 218 | ||
| 219 | /// A Git object database backed by a packfile. A packfile index is also used | |
| 220 | /// for efficient access to objects in the packfile. | |
| 221 | /// | |
| 222 | /// The format of the packfile and its associated index are documented in | |
| 223 | /// [pack-format](https://git-scm.com/docs/pack-format). | |
| 224 | const Odb = struct { | |
| 225 | pack_file: std.fs.File, | |
| 226 | index_header: IndexHeader, | |
| 227 | index_file: std.fs.File, | |
| 228 | cache: ObjectCache = .{}, | |
| 229 | allocator: Allocator, | |
| 230 | ||
| 231 | /// Initializes the database from open pack and index files. | |
| 232 | fn init(allocator: Allocator, pack_file: std.fs.File, index_file: std.fs.File) !Odb { | |
| 233 | try pack_file.seekTo(0); | |
| 234 | try index_file.seekTo(0); | |
| 235 | const index_header = try IndexHeader.read(index_file.reader()); | |
| 236 | return .{ | |
| 237 | .pack_file = pack_file, | |
| 238 | .index_header = index_header, | |
| 239 | .index_file = index_file, | |
| 240 | .allocator = allocator, | |
| 241 | }; | |
| 242 | } | |
| 243 | ||
| 244 | fn deinit(odb: *Odb) void { | |
| 245 | odb.cache.deinit(odb.allocator); | |
| 246 | odb.* = undefined; | |
| 247 | } | |
| 248 | ||
| 249 | /// Reads the object at the current position in the database. | |
| 250 | fn readObject(odb: *Odb) !Object { | |
| 251 | var base_offset = try odb.pack_file.getPos(); | |
| 252 | var base_header: EntryHeader = undefined; | |
| 253 | var delta_offsets = std.ArrayListUnmanaged(u64){}; | |
| 254 | defer delta_offsets.deinit(odb.allocator); | |
| 255 | const base_object = while (true) { | |
| 256 | if (odb.cache.get(base_offset)) |base_object| break base_object; | |
| 257 | ||
| 258 | base_header = try EntryHeader.read(odb.pack_file.reader()); | |
| 259 | switch (base_header) { | |
| 260 | .ofs_delta => |ofs_delta| { | |
| 261 | try delta_offsets.append(odb.allocator, base_offset); | |
| 262 | base_offset = std.math.sub(u64, base_offset, ofs_delta.offset) catch return error.InvalidFormat; | |
| 263 | try odb.pack_file.seekTo(base_offset); | |
| 264 | }, | |
| 265 | .ref_delta => |ref_delta| { | |
| 266 | try delta_offsets.append(odb.allocator, base_offset); | |
| 267 | try odb.seekOid(ref_delta.base_object); | |
| 268 | base_offset = try odb.pack_file.getPos(); | |
| 269 | }, | |
| 270 | else => { | |
| 271 | const base_data = try readObjectRaw(odb.allocator, odb.pack_file.reader(), base_header.uncompressedLength()); | |
| 272 | errdefer odb.allocator.free(base_data); | |
| 273 | const base_object: Object = .{ .type = base_header.objectType(), .data = base_data }; | |
| 274 | try odb.cache.put(odb.allocator, base_offset, base_object); | |
| 275 | break base_object; | |
| 276 | }, | |
| 277 | } | |
| 278 | }; | |
| 279 | ||
| 280 | const base_data = try resolveDeltaChain( | |
| 281 | odb.allocator, | |
| 282 | odb.pack_file, | |
| 283 | base_object, | |
| 284 | delta_offsets.items, | |
| 285 | &odb.cache, | |
| 286 | ); | |
| 287 | ||
| 288 | return .{ .type = base_object.type, .data = base_data }; | |
| 289 | } | |
| 290 | ||
| 291 | /// Seeks to the beginning of the object with the given ID. | |
| 292 | fn seekOid(odb: *Odb, oid: Oid) !void { | |
| 293 | const key = oid[0]; | |
| 294 | var start_index = if (key > 0) odb.index_header.fan_out_table[key - 1] else 0; | |
| 295 | var end_index = odb.index_header.fan_out_table[key]; | |
| 296 | const found_index = while (start_index < end_index) { | |
| 297 | const mid_index = start_index + (end_index - start_index) / 2; | |
| 298 | try odb.index_file.seekTo(IndexHeader.size + mid_index * oid_length); | |
| 299 | const mid_oid = try odb.index_file.reader().readBytesNoEof(oid_length); | |
| 300 | switch (mem.order(u8, &mid_oid, &oid)) { | |
| 301 | .lt => start_index = mid_index + 1, | |
| 302 | .gt => end_index = mid_index, | |
| 303 | .eq => break mid_index, | |
| 304 | } | |
| 305 | } else return error.ObjectNotFound; | |
| 306 | ||
| 307 | const n_objects = odb.index_header.fan_out_table[255]; | |
| 308 | const offset_values_start = IndexHeader.size + n_objects * (oid_length + 4); | |
| 309 | try odb.index_file.seekTo(offset_values_start + found_index * 4); | |
| 310 | const l1_offset: packed struct { value: u31, big: bool } = @bitCast(try odb.index_file.reader().readIntBig(u32)); | |
| 311 | const pack_offset = pack_offset: { | |
| 312 | if (l1_offset.big) { | |
| 313 | const l2_offset_values_start = offset_values_start + n_objects * 4; | |
| 314 | try odb.index_file.seekTo(l2_offset_values_start + l1_offset.value * 4); | |
| 315 | break :pack_offset try odb.index_file.reader().readIntBig(u64); | |
| 316 | } else { | |
| 317 | break :pack_offset l1_offset.value; | |
| 318 | } | |
| 319 | }; | |
| 320 | ||
| 321 | try odb.pack_file.seekTo(pack_offset); | |
| 322 | } | |
| 323 | }; | |
| 324 | ||
| 325 | const Object = struct { | |
| 326 | type: Type, | |
| 327 | data: []const u8, | |
| 328 | ||
| 329 | const Type = enum { | |
| 330 | commit, | |
| 331 | tree, | |
| 332 | blob, | |
| 333 | tag, | |
| 334 | }; | |
| 335 | }; | |
| 336 | ||
| 337 | /// A cache for object data. | |
| 338 | /// | |
| 339 | /// The purpose of this cache is to speed up resolution of deltas by caching the | |
| 340 | /// results of resolving delta objects, while maintaining a maximum cache size | |
| 341 | /// to avoid excessive memory usage. If the total size of the objects in the | |
| 342 | /// cache exceeds the maximum, the cache will begin evicting the least recently | |
| 343 | /// used objects: when resolving delta chains, the most recently used objects | |
| 344 | /// will likely be more helpful as they will be further along in the chain | |
| 345 | /// (skipping earlier reconstruction steps). | |
| 346 | /// | |
| 347 | /// Object data stored in the cache is managed by the cache. It should not be | |
| 348 | /// freed by the caller at any point after inserting it into the cache. Any | |
| 349 | /// objects remaining in the cache will be freed when the cache itself is freed. | |
| 350 | const ObjectCache = struct { | |
| 351 | objects: std.AutoHashMapUnmanaged(u64, CacheEntry) = .{}, | |
| 352 | lru_nodes: LruList = .{}, | |
| 353 | byte_size: usize = 0, | |
| 354 | ||
| 355 | const max_byte_size = 128 * 1024 * 1024; // 128MiB | |
| 356 | /// A list of offsets stored in the cache, with the most recently used | |
| 357 | /// entries at the end. | |
| 358 | const LruList = std.DoublyLinkedList(u64); | |
| 359 | const CacheEntry = struct { object: Object, lru_node: *LruList.Node }; | |
| 360 | ||
| 361 | fn deinit(cache: *ObjectCache, allocator: Allocator) void { | |
| 362 | var object_iterator = cache.objects.iterator(); | |
| 363 | while (object_iterator.next()) |object| { | |
| 364 | allocator.free(object.value_ptr.object.data); | |
| 365 | allocator.destroy(object.value_ptr.lru_node); | |
| 366 | } | |
| 367 | cache.objects.deinit(allocator); | |
| 368 | cache.* = undefined; | |
| 369 | } | |
| 370 | ||
| 371 | /// Gets an object from the cache, moving it to the most recently used | |
| 372 | /// position if it is present. | |
| 373 | fn get(cache: *ObjectCache, offset: u64) ?Object { | |
| 374 | if (cache.objects.get(offset)) |entry| { | |
| 375 | cache.lru_nodes.remove(entry.lru_node); | |
| 376 | cache.lru_nodes.append(entry.lru_node); | |
| 377 | return entry.object; | |
| 378 | } else { | |
| 379 | return null; | |
| 380 | } | |
| 381 | } | |
| 382 | ||
| 383 | /// Puts an object in the cache, possibly evicting older entries if the | |
| 384 | /// cache exceeds its maximum size. Note that, although old objects may | |
| 385 | /// be evicted, the object just added to the cache with this function | |
| 386 | /// will not be evicted before the next call to `put` or `deinit` even if | |
| 387 | /// it exceeds the maximum cache size. | |
| 388 | fn put(cache: *ObjectCache, allocator: Allocator, offset: u64, object: Object) !void { | |
| 389 | const lru_node = try allocator.create(LruList.Node); | |
| 390 | errdefer allocator.destroy(lru_node); | |
| 391 | lru_node.data = offset; | |
| 392 | ||
| 393 | const gop = try cache.objects.getOrPut(allocator, offset); | |
| 394 | if (gop.found_existing) { | |
| 395 | cache.byte_size -= gop.value_ptr.object.data.len; | |
| 396 | cache.lru_nodes.remove(gop.value_ptr.lru_node); | |
| 397 | allocator.destroy(gop.value_ptr.lru_node); | |
| 398 | allocator.free(gop.value_ptr.object.data); | |
| 399 | } | |
| 400 | gop.value_ptr.* = .{ .object = object, .lru_node = lru_node }; | |
| 401 | cache.byte_size += object.data.len; | |
| 402 | cache.lru_nodes.append(lru_node); | |
| 403 | ||
| 404 | while (cache.byte_size > max_byte_size and cache.lru_nodes.len > 1) { | |
| 405 | // The > 1 check is to make sure that we don't evict the most | |
| 406 | // recently added node, even if it by itself happens to exceed the | |
| 407 | // maximum size of the cache. | |
| 408 | const evict_node = cache.lru_nodes.popFirst().?; | |
| 409 | const evict_offset = evict_node.data; | |
| 410 | allocator.destroy(evict_node); | |
| 411 | const evict_object = cache.objects.get(evict_offset).?.object; | |
| 412 | cache.byte_size -= evict_object.data.len; | |
| 413 | allocator.free(evict_object.data); | |
| 414 | _ = cache.objects.remove(evict_offset); | |
| 415 | } | |
| 416 | } | |
| 417 | }; | |
| 418 | ||
| 419 | /// A single pkt-line in the Git protocol. | |
| 420 | /// | |
| 421 | /// The format of a pkt-line is documented in | |
| 422 | /// [protocol-common](https://git-scm.com/docs/protocol-common). The special | |
| 423 | /// meanings of the delimiter and response-end packets are documented in | |
| 424 | /// [protocol-v2](https://git-scm.com/docs/protocol-v2). | |
| 425 | const Packet = union(enum) { | |
| 426 | flush, | |
| 427 | delimiter, | |
| 428 | response_end, | |
| 429 | data: []const u8, | |
| 430 | ||
| 431 | const max_data_length = 65516; | |
| 432 | ||
| 433 | /// Reads a packet in pkt-line format. | |
| 434 | fn read(reader: anytype, buf: *[max_data_length]u8) !Packet { | |
| 435 | const length = std.fmt.parseUnsigned(u16, &try reader.readBytesNoEof(4), 16) catch return error.InvalidPacket; | |
| 436 | switch (length) { | |
| 437 | 0 => return .flush, | |
| 438 | 1 => return .delimiter, | |
| 439 | 2 => return .response_end, | |
| 440 | 3 => return error.InvalidPacket, | |
| 441 | else => if (length - 4 > max_data_length) return error.InvalidPacket, | |
| 442 | } | |
| 443 | const data = buf[0 .. length - 4]; | |
| 444 | try reader.readNoEof(data); | |
| 445 | return .{ .data = data }; | |
| 446 | } | |
| 447 | ||
| 448 | /// Writes a packet in pkt-line format. | |
| 449 | fn write(packet: Packet, writer: anytype) !void { | |
| 450 | switch (packet) { | |
| 451 | .flush => try writer.writeAll("0000"), | |
| 452 | .delimiter => try writer.writeAll("0001"), | |
| 453 | .response_end => try writer.writeAll("0002"), | |
| 454 | .data => |data| { | |
| 455 | assert(data.len <= max_data_length); | |
| 456 | try writer.print("{x:0>4}", .{data.len + 4}); | |
| 457 | try writer.writeAll(data); | |
| 458 | }, | |
| 459 | } | |
| 460 | } | |
| 461 | }; | |
| 462 | ||
| 463 | /// A client session for the Git protocol, currently limited to an HTTP(S) | |
| 464 | /// transport. Only protocol version 2 is supported, as documented in | |
| 465 | /// [protocol-v2](https://git-scm.com/docs/protocol-v2). | |
| 466 | pub const Session = struct { | |
| 467 | transport: *std.http.Client, | |
| 468 | uri: std.Uri, | |
| 469 | supports_agent: bool = false, | |
| 470 | supports_shallow: bool = false, | |
| 471 | ||
| 472 | const agent = "zig/" ++ @import("builtin").zig_version_string; | |
| 473 | const agent_capability = std.fmt.comptimePrint("agent={s}\n", .{agent}); | |
| 474 | ||
| 475 | /// Discovers server capabilities. This should be called before using any | |
| 476 | /// other client functionality, or the client will be forced to default to | |
| 477 | /// the bare minimum server requirements, which may be considerably less | |
| 478 | /// efficient (e.g. no shallow fetches). | |
| 479 | /// | |
| 480 | /// See the note on `getCapabilities` regarding `redirect_uri`. | |
| 481 | pub fn discoverCapabilities( | |
| 482 | session: *Session, | |
| 483 | allocator: Allocator, | |
| 484 | redirect_uri: *[]u8, | |
| 485 | ) !void { | |
| 486 | var capability_iterator = try session.getCapabilities(allocator, redirect_uri); | |
| 487 | defer capability_iterator.deinit(); | |
| 488 | while (try capability_iterator.next()) |capability| { | |
| 489 | if (mem.eql(u8, capability.key, "agent")) { | |
| 490 | session.supports_agent = true; | |
| 491 | } else if (mem.eql(u8, capability.key, "fetch")) { | |
| 492 | var feature_iterator = mem.splitScalar(u8, capability.value orelse continue, ' '); | |
| 493 | while (feature_iterator.next()) |feature| { | |
| 494 | if (mem.eql(u8, feature, "shallow")) { | |
| 495 | session.supports_shallow = true; | |
| 496 | } | |
| 497 | } | |
| 498 | } | |
| 499 | } | |
| 500 | } | |
| 501 | ||
| 502 | /// Returns an iterator over capabilities supported by the server. | |
| 503 | /// | |
| 504 | /// If the server redirects the request, `error.Redirected` is returned and | |
| 505 | /// `redirect_uri` is populated with the URI resulting from the redirects. | |
| 506 | /// When this occurs, the value of `redirect_uri` must be freed with | |
| 507 | /// `allocator` when the caller is done with it. | |
| 508 | fn getCapabilities( | |
| 509 | session: Session, | |
| 510 | allocator: Allocator, | |
| 511 | redirect_uri: *[]u8, | |
| 512 | ) !CapabilityIterator { | |
| 513 | var info_refs_uri = session.uri; | |
| 514 | info_refs_uri.path = try std.fs.path.resolvePosix(allocator, &.{ "/", session.uri.path, "info/refs" }); | |
| 515 | defer allocator.free(info_refs_uri.path); | |
| 516 | info_refs_uri.query = "service=git-upload-pack"; | |
| 517 | info_refs_uri.fragment = null; | |
| 518 | ||
| 519 | var headers = std.http.Headers.init(allocator); | |
| 520 | defer headers.deinit(); | |
| 521 | try headers.append("Git-Protocol", "version=2"); | |
| 522 | ||
| 523 | var request = try session.transport.request(.GET, info_refs_uri, headers, .{ | |
| 524 | .max_redirects = 3, | |
| 525 | }); | |
| 526 | errdefer request.deinit(); | |
| 527 | try request.start(.{}); | |
| 528 | try request.finish(); | |
| 529 | ||
| 530 | try request.wait(); | |
| 531 | if (request.response.status != .ok) return error.ProtocolError; | |
| 532 | if (request.redirects_left < 3) { | |
| 533 | if (!mem.endsWith(u8, request.uri.path, "/info/refs")) return error.UnparseableRedirect; | |
| 534 | var new_uri = request.uri; | |
| 535 | new_uri.path = new_uri.path[0 .. new_uri.path.len - "/info/refs".len]; | |
| 536 | new_uri.query = null; | |
| 537 | redirect_uri.* = try std.fmt.allocPrint(allocator, "{+/}", .{new_uri}); | |
| 538 | return error.Redirected; | |
| 539 | } | |
| 540 | ||
| 541 | const reader = request.reader(); | |
| 542 | var buf: [Packet.max_data_length]u8 = undefined; | |
| 543 | var state: enum { response_start, response_content } = .response_start; | |
| 544 | while (true) { | |
| 545 | // Some Git servers (at least GitHub) include an additional | |
| 546 | // '# service=git-upload-pack' informative response before sending | |
| 547 | // the expected 'version 2' packet and capability information. | |
| 548 | // This is not universal: SourceHut, for example, does not do this. | |
| 549 | // Thus, we need to skip any such useless additional responses | |
| 550 | // before we get the one we're actually looking for. The responses | |
| 551 | // will be delimited by flush packets. | |
| 552 | const packet = Packet.read(reader, &buf) catch |e| switch (e) { | |
| 553 | error.EndOfStream => return error.UnsupportedProtocol, // 'version 2' packet not found | |
| 554 | else => |other| return other, | |
| 555 | }; | |
| 556 | switch (packet) { | |
| 557 | .flush => state = .response_start, | |
| 558 | .data => |data| switch (state) { | |
| 559 | .response_start => if (mem.eql(u8, data, "version 2\n")) { | |
| 560 | return .{ .request = request }; | |
| 561 | } else { | |
| 562 | state = .response_content; | |
| 563 | }, | |
| 564 | else => {}, | |
| 565 | }, | |
| 566 | else => return error.UnexpectedPacket, | |
| 567 | } | |
| 568 | } | |
| 569 | } | |
| 570 | ||
| 571 | const CapabilityIterator = struct { | |
| 572 | request: std.http.Client.Request, | |
| 573 | buf: [Packet.max_data_length]u8 = undefined, | |
| 574 | ||
| 575 | const Capability = struct { | |
| 576 | key: []const u8, | |
| 577 | value: ?[]const u8 = null, | |
| 578 | }; | |
| 579 | ||
| 580 | fn deinit(iterator: *CapabilityIterator) void { | |
| 581 | iterator.request.deinit(); | |
| 582 | iterator.* = undefined; | |
| 583 | } | |
| 584 | ||
| 585 | fn next(iterator: *CapabilityIterator) !?Capability { | |
| 586 | switch (try Packet.read(iterator.request.reader(), &iterator.buf)) { | |
| 587 | .flush => return null, | |
| 588 | .data => |data| if (data.len > 0 and data[data.len - 1] == '\n') { | |
| 589 | if (mem.indexOfScalar(u8, data, '=')) |separator_pos| { | |
| 590 | return .{ .key = data[0..separator_pos], .value = data[separator_pos + 1 .. data.len - 1] }; | |
| 591 | } else { | |
| 592 | return .{ .key = data[0 .. data.len - 1] }; | |
| 593 | } | |
| 594 | } else return error.UnexpectedPacket, | |
| 595 | else => return error.UnexpectedPacket, | |
| 596 | } | |
| 597 | } | |
| 598 | }; | |
| 599 | ||
| 600 | const ListRefsOptions = struct { | |
| 601 | /// The ref prefixes (if any) to use to filter the refs available on the | |
| 602 | /// server. Note that the client must still check the returned refs | |
| 603 | /// against its desired filters itself: the server is not required to | |
| 604 | /// respect these prefix filters and may return other refs as well. | |
| 605 | ref_prefixes: []const []const u8 = &.{}, | |
| 606 | /// Whether to include symref targets for returned symbolic refs. | |
| 607 | include_symrefs: bool = false, | |
| 608 | /// Whether to include the peeled object ID for returned tag refs. | |
| 609 | include_peeled: bool = false, | |
| 610 | }; | |
| 611 | ||
| 612 | /// Returns an iterator over refs known to the server. | |
| 613 | pub fn listRefs(session: Session, allocator: Allocator, options: ListRefsOptions) !RefIterator { | |
| 614 | var upload_pack_uri = session.uri; | |
| 615 | upload_pack_uri.path = try std.fs.path.resolvePosix(allocator, &.{ "/", session.uri.path, "git-upload-pack" }); | |
| 616 | defer allocator.free(upload_pack_uri.path); | |
| 617 | upload_pack_uri.query = null; | |
| 618 | upload_pack_uri.fragment = null; | |
| 619 | ||
| 620 | var headers = std.http.Headers.init(allocator); | |
| 621 | defer headers.deinit(); | |
| 622 | try headers.append("Content-Type", "application/x-git-upload-pack-request"); | |
| 623 | try headers.append("Git-Protocol", "version=2"); | |
| 624 | ||
| 625 | var body = std.ArrayListUnmanaged(u8){}; | |
| 626 | defer body.deinit(allocator); | |
| 627 | const body_writer = body.writer(allocator); | |
| 628 | try Packet.write(.{ .data = "command=ls-refs\n" }, body_writer); | |
| 629 | if (session.supports_agent) { | |
| 630 | try Packet.write(.{ .data = agent_capability }, body_writer); | |
| 631 | } | |
| 632 | try Packet.write(.delimiter, body_writer); | |
| 633 | for (options.ref_prefixes) |ref_prefix| { | |
| 634 | const ref_prefix_packet = try std.fmt.allocPrint(allocator, "ref-prefix {s}\n", .{ref_prefix}); | |
| 635 | defer allocator.free(ref_prefix_packet); | |
| 636 | try Packet.write(.{ .data = ref_prefix_packet }, body_writer); | |
| 637 | } | |
| 638 | if (options.include_symrefs) { | |
| 639 | try Packet.write(.{ .data = "symrefs\n" }, body_writer); | |
| 640 | } | |
| 641 | if (options.include_peeled) { | |
| 642 | try Packet.write(.{ .data = "peel\n" }, body_writer); | |
| 643 | } | |
| 644 | try Packet.write(.flush, body_writer); | |
| 645 | ||
| 646 | var request = try session.transport.request(.POST, upload_pack_uri, headers, .{ | |
| 647 | .handle_redirects = false, | |
| 648 | }); | |
| 649 | errdefer request.deinit(); | |
| 650 | request.transfer_encoding = .{ .content_length = body.items.len }; | |
| 651 | try request.start(.{}); | |
| 652 | try request.writeAll(body.items); | |
| 653 | try request.finish(); | |
| 654 | ||
| 655 | try request.wait(); | |
| 656 | if (request.response.status != .ok) return error.ProtocolError; | |
| 657 | ||
| 658 | return .{ .request = request }; | |
| 659 | } | |
| 660 | ||
| 661 | pub const RefIterator = struct { | |
| 662 | request: std.http.Client.Request, | |
| 663 | buf: [Packet.max_data_length]u8 = undefined, | |
| 664 | ||
| 665 | pub const Ref = struct { | |
| 666 | oid: Oid, | |
| 667 | name: []const u8, | |
| 668 | symref_target: ?[]const u8, | |
| 669 | peeled: ?Oid, | |
| 670 | }; | |
| 671 | ||
| 672 | pub fn deinit(iterator: *RefIterator) void { | |
| 673 | iterator.request.deinit(); | |
| 674 | iterator.* = undefined; | |
| 675 | } | |
| 676 | ||
| 677 | pub fn next(iterator: *RefIterator) !?Ref { | |
| 678 | switch (try Packet.read(iterator.request.reader(), &iterator.buf)) { | |
| 679 | .flush => return null, | |
| 680 | .data => |data| { | |
| 681 | const oid_sep_pos = mem.indexOfScalar(u8, data, ' ') orelse return error.InvalidRefPacket; | |
| 682 | const oid = parseOid(data[0..oid_sep_pos]) catch return error.InvalidRefPacket; | |
| 683 | ||
| 684 | const name_sep_pos = mem.indexOfAnyPos(u8, data, oid_sep_pos + 1, " \n") orelse return error.InvalidRefPacket; | |
| 685 | const name = data[oid_sep_pos + 1 .. name_sep_pos]; | |
| 686 | ||
| 687 | var symref_target: ?[]const u8 = null; | |
| 688 | var peeled: ?Oid = null; | |
| 689 | var last_sep_pos = name_sep_pos; | |
| 690 | while (data[last_sep_pos] == ' ') { | |
| 691 | const next_sep_pos = mem.indexOfAnyPos(u8, data, last_sep_pos + 1, " \n") orelse return error.InvalidRefPacket; | |
| 692 | const attribute = data[last_sep_pos + 1 .. next_sep_pos]; | |
| 693 | if (mem.startsWith(u8, attribute, "symref-target:")) { | |
| 694 | symref_target = attribute["symref-target:".len..]; | |
| 695 | } else if (mem.startsWith(u8, attribute, "peeled:")) { | |
| 696 | peeled = parseOid(attribute["peeled:".len..]) catch return error.InvalidRefPacket; | |
| 697 | } | |
| 698 | last_sep_pos = next_sep_pos; | |
| 699 | } | |
| 700 | ||
| 701 | return .{ .oid = oid, .name = name, .symref_target = symref_target, .peeled = peeled }; | |
| 702 | }, | |
| 703 | else => return error.UnexpectedPacket, | |
| 704 | } | |
| 705 | } | |
| 706 | }; | |
| 707 | ||
| 708 | /// Fetches the given refs from the server. A shallow fetch (depth 1) is | |
| 709 | /// performed if the server supports it. | |
| 710 | pub fn fetch(session: Session, allocator: Allocator, wants: []const []const u8) !FetchStream { | |
| 711 | var upload_pack_uri = session.uri; | |
| 712 | upload_pack_uri.path = try std.fs.path.resolvePosix(allocator, &.{ "/", session.uri.path, "git-upload-pack" }); | |
| 713 | defer allocator.free(upload_pack_uri.path); | |
| 714 | upload_pack_uri.query = null; | |
| 715 | upload_pack_uri.fragment = null; | |
| 716 | ||
| 717 | var headers = std.http.Headers.init(allocator); | |
| 718 | defer headers.deinit(); | |
| 719 | try headers.append("Content-Type", "application/x-git-upload-pack-request"); | |
| 720 | try headers.append("Git-Protocol", "version=2"); | |
| 721 | ||
| 722 | var body = std.ArrayListUnmanaged(u8){}; | |
| 723 | defer body.deinit(allocator); | |
| 724 | const body_writer = body.writer(allocator); | |
| 725 | try Packet.write(.{ .data = "command=fetch\n" }, body_writer); | |
| 726 | if (session.supports_agent) { | |
| 727 | try Packet.write(.{ .data = agent_capability }, body_writer); | |
| 728 | } | |
| 729 | try Packet.write(.delimiter, body_writer); | |
| 730 | // Our packfile parser supports the OFS_DELTA object type | |
| 731 | try Packet.write(.{ .data = "ofs-delta\n" }, body_writer); | |
| 732 | // We do not currently convey server progress information to the user | |
| 733 | try Packet.write(.{ .data = "no-progress\n" }, body_writer); | |
| 734 | if (session.supports_shallow) { | |
| 735 | try Packet.write(.{ .data = "deepen 1\n" }, body_writer); | |
| 736 | } | |
| 737 | for (wants) |want| { | |
| 738 | var buf: [Packet.max_data_length]u8 = undefined; | |
| 739 | const arg = std.fmt.bufPrint(&buf, "want {s}\n", .{want}) catch unreachable; | |
| 740 | try Packet.write(.{ .data = arg }, body_writer); | |
| 741 | } | |
| 742 | try Packet.write(.{ .data = "done\n" }, body_writer); | |
| 743 | try Packet.write(.flush, body_writer); | |
| 744 | ||
| 745 | var request = try session.transport.request(.POST, upload_pack_uri, headers, .{ | |
| 746 | .handle_redirects = false, | |
| 747 | }); | |
| 748 | errdefer request.deinit(); | |
| 749 | request.transfer_encoding = .{ .content_length = body.items.len }; | |
| 750 | try request.start(.{}); | |
| 751 | try request.writeAll(body.items); | |
| 752 | try request.finish(); | |
| 753 | ||
| 754 | try request.wait(); | |
| 755 | if (request.response.status != .ok) return error.ProtocolError; | |
| 756 | ||
| 757 | const reader = request.reader(); | |
| 758 | // We are not interested in any of the sections of the returned fetch | |
| 759 | // data other than the packfile section, since we aren't doing anything | |
| 760 | // complex like ref negotiation (this is a fresh clone). | |
| 761 | var state: enum { section_start, section_content } = .section_start; | |
| 762 | while (true) { | |
| 763 | var buf: [Packet.max_data_length]u8 = undefined; | |
| 764 | const packet = try Packet.read(reader, &buf); | |
| 765 | switch (state) { | |
| 766 | .section_start => switch (packet) { | |
| 767 | .data => |data| if (mem.eql(u8, data, "packfile\n")) { | |
| 768 | return .{ .request = request }; | |
| 769 | } else { | |
| 770 | state = .section_content; | |
| 771 | }, | |
| 772 | else => return error.UnexpectedPacket, | |
| 773 | }, | |
| 774 | .section_content => switch (packet) { | |
| 775 | .delimiter => state = .section_start, | |
| 776 | .data => {}, | |
| 777 | else => return error.UnexpectedPacket, | |
| 778 | }, | |
| 779 | } | |
| 780 | } | |
| 781 | } | |
| 782 | ||
| 783 | pub const FetchStream = struct { | |
| 784 | request: std.http.Client.Request, | |
| 785 | buf: [Packet.max_data_length]u8 = undefined, | |
| 786 | pos: usize = 0, | |
| 787 | len: usize = 0, | |
| 788 | ||
| 789 | pub fn deinit(stream: *FetchStream) void { | |
| 790 | stream.request.deinit(); | |
| 791 | } | |
| 792 | ||
| 793 | pub const ReadError = std.http.Client.Request.ReadError || error{ | |
| 794 | InvalidPacket, | |
| 795 | ProtocolError, | |
| 796 | UnexpectedPacket, | |
| 797 | }; | |
| 798 | pub const Reader = std.io.Reader(*FetchStream, ReadError, read); | |
| 799 | ||
| 800 | const StreamCode = enum(u8) { | |
| 801 | pack_data = 1, | |
| 802 | progress = 2, | |
| 803 | fatal_error = 3, | |
| 804 | _, | |
| 805 | }; | |
| 806 | ||
| 807 | pub fn reader(stream: *FetchStream) Reader { | |
| 808 | return .{ .context = stream }; | |
| 809 | } | |
| 810 | ||
| 811 | pub fn read(stream: *FetchStream, buf: []u8) !usize { | |
| 812 | if (stream.pos == stream.len) { | |
| 813 | while (true) { | |
| 814 | switch (try Packet.read(stream.request.reader(), &stream.buf)) { | |
| 815 | .flush => return 0, | |
| 816 | .data => |data| if (data.len > 1) switch (@as(StreamCode, @enumFromInt(data[0]))) { | |
| 817 | .pack_data => { | |
| 818 | stream.pos = 1; | |
| 819 | stream.len = data.len; | |
| 820 | break; | |
| 821 | }, | |
| 822 | .fatal_error => return error.ProtocolError, | |
| 823 | else => {}, | |
| 824 | }, | |
| 825 | else => return error.UnexpectedPacket, | |
| 826 | } | |
| 827 | } | |
| 828 | } | |
| 829 | ||
| 830 | const size = @min(buf.len, stream.len - stream.pos); | |
| 831 | @memcpy(buf[0..size], stream.buf[stream.pos .. stream.pos + size]); | |
| 832 | stream.pos += size; | |
| 833 | return size; | |
| 834 | } | |
| 835 | }; | |
| 836 | }; | |
| 837 | ||
| 838 | const PackHeader = struct { | |
| 839 | total_objects: u32, | |
| 840 | ||
| 841 | const signature = "PACK"; | |
| 842 | const supported_version = 2; | |
| 843 | ||
| 844 | fn read(reader: anytype) !PackHeader { | |
| 845 | const actual_signature = reader.readBytesNoEof(4) catch |e| switch (e) { | |
| 846 | error.EndOfStream => return error.InvalidHeader, | |
| 847 | else => |other| return other, | |
| 848 | }; | |
| 849 | if (!mem.eql(u8, &actual_signature, signature)) return error.InvalidHeader; | |
| 850 | const version = reader.readIntBig(u32) catch |e| switch (e) { | |
| 851 | error.EndOfStream => return error.InvalidHeader, | |
| 852 | else => |other| return other, | |
| 853 | }; | |
| 854 | if (version != supported_version) return error.UnsupportedVersion; | |
| 855 | const total_objects = reader.readIntBig(u32) catch |e| switch (e) { | |
| 856 | error.EndOfStream => return error.InvalidHeader, | |
| 857 | else => |other| return other, | |
| 858 | }; | |
| 859 | return .{ .total_objects = total_objects }; | |
| 860 | } | |
| 861 | }; | |
| 862 | ||
| 863 | const EntryHeader = union(Type) { | |
| 864 | commit: Undeltified, | |
| 865 | tree: Undeltified, | |
| 866 | blob: Undeltified, | |
| 867 | tag: Undeltified, | |
| 868 | ofs_delta: OfsDelta, | |
| 869 | ref_delta: RefDelta, | |
| 870 | ||
| 871 | const Type = enum(u3) { | |
| 872 | commit = 1, | |
| 873 | tree = 2, | |
| 874 | blob = 3, | |
| 875 | tag = 4, | |
| 876 | ofs_delta = 6, | |
| 877 | ref_delta = 7, | |
| 878 | }; | |
| 879 | ||
| 880 | const Undeltified = struct { | |
| 881 | uncompressed_length: u64, | |
| 882 | }; | |
| 883 | ||
| 884 | const OfsDelta = struct { | |
| 885 | offset: u64, | |
| 886 | uncompressed_length: u64, | |
| 887 | }; | |
| 888 | ||
| 889 | const RefDelta = struct { | |
| 890 | base_object: Oid, | |
| 891 | uncompressed_length: u64, | |
| 892 | }; | |
| 893 | ||
| 894 | fn objectType(header: EntryHeader) Object.Type { | |
| 895 | return switch (header) { | |
| 896 | inline .commit, .tree, .blob, .tag => |_, tag| @field(Object.Type, @tagName(tag)), | |
| 897 | else => unreachable, | |
| 898 | }; | |
| 899 | } | |
| 900 | ||
| 901 | fn uncompressedLength(header: EntryHeader) u64 { | |
| 902 | return switch (header) { | |
| 903 | inline else => |entry| entry.uncompressed_length, | |
| 904 | }; | |
| 905 | } | |
| 906 | ||
| 907 | fn read(reader: anytype) !EntryHeader { | |
| 908 | const InitialByte = packed struct { len: u4, type: u3, has_next: bool }; | |
| 909 | const initial: InitialByte = @bitCast(reader.readByte() catch |e| switch (e) { | |
| 910 | error.EndOfStream => return error.InvalidFormat, | |
| 911 | else => |other| return other, | |
| 912 | }); | |
| 913 | const rest_len = if (initial.has_next) try readSizeVarInt(reader) else 0; | |
| 914 | var uncompressed_length: u64 = initial.len; | |
| 915 | uncompressed_length |= std.math.shlExact(u64, rest_len, 4) catch return error.InvalidFormat; | |
| 916 | const @"type" = std.meta.intToEnum(EntryHeader.Type, initial.type) catch return error.InvalidFormat; | |
| 917 | return switch (@"type") { | |
| 918 | inline .commit, .tree, .blob, .tag => |tag| @unionInit(EntryHeader, @tagName(tag), .{ | |
| 919 | .uncompressed_length = uncompressed_length, | |
| 920 | }), | |
| 921 | .ofs_delta => .{ .ofs_delta = .{ | |
| 922 | .offset = try readOffsetVarInt(reader), | |
| 923 | .uncompressed_length = uncompressed_length, | |
| 924 | } }, | |
| 925 | .ref_delta => .{ .ref_delta = .{ | |
| 926 | .base_object = reader.readBytesNoEof(oid_length) catch |e| switch (e) { | |
| 927 | error.EndOfStream => return error.InvalidFormat, | |
| 928 | else => |other| return other, | |
| 929 | }, | |
| 930 | .uncompressed_length = uncompressed_length, | |
| 931 | } }, | |
| 932 | }; | |
| 933 | } | |
| 934 | }; | |
| 935 | ||
| 936 | fn readSizeVarInt(r: anytype) !u64 { | |
| 937 | const Byte = packed struct { value: u7, has_next: bool }; | |
| 938 | var b: Byte = @bitCast(try r.readByte()); | |
| 939 | var value: u64 = b.value; | |
| 940 | var shift: u6 = 0; | |
| 941 | while (b.has_next) { | |
| 942 | b = @bitCast(try r.readByte()); | |
| 943 | shift = std.math.add(u6, shift, 7) catch return error.InvalidFormat; | |
| 944 | value |= @as(u64, b.value) << shift; | |
| 945 | } | |
| 946 | return value; | |
| 947 | } | |
| 948 | ||
| 949 | fn readOffsetVarInt(r: anytype) !u64 { | |
| 950 | const Byte = packed struct { value: u7, has_next: bool }; | |
| 951 | var b: Byte = @bitCast(try r.readByte()); | |
| 952 | var value: u64 = b.value; | |
| 953 | while (b.has_next) { | |
| 954 | b = @bitCast(try r.readByte()); | |
| 955 | value = std.math.shlExact(u64, value + 1, 7) catch return error.InvalidFormat; | |
| 956 | value |= b.value; | |
| 957 | } | |
| 958 | return value; | |
| 959 | } | |
| 960 | ||
| 961 | const IndexHeader = struct { | |
| 962 | fan_out_table: [256]u32, | |
| 963 | ||
| 964 | const signature = "\xFFtOc"; | |
| 965 | const supported_version = 2; | |
| 966 | const size = 4 + 4 + @sizeOf([256]u32); | |
| 967 | ||
| 968 | fn read(reader: anytype) !IndexHeader { | |
| 969 | var header_bytes = try reader.readBytesNoEof(size); | |
| 970 | if (!mem.eql(u8, header_bytes[0..4], signature)) return error.InvalidHeader; | |
| 971 | const version = mem.readIntBig(u32, header_bytes[4..8]); | |
| 972 | if (version != supported_version) return error.UnsupportedVersion; | |
| 973 | ||
| 974 | var fan_out_table: [256]u32 = undefined; | |
| 975 | var fan_out_table_stream = std.io.fixedBufferStream(header_bytes[8..]); | |
| 976 | const fan_out_table_reader = fan_out_table_stream.reader(); | |
| 977 | for (&fan_out_table) |*entry| { | |
| 978 | entry.* = fan_out_table_reader.readIntBig(u32) catch unreachable; | |
| 979 | } | |
| 980 | return .{ .fan_out_table = fan_out_table }; | |
| 981 | } | |
| 982 | }; | |
| 983 | ||
| 984 | const IndexEntry = struct { | |
| 985 | offset: u64, | |
| 986 | crc32: u32, | |
| 987 | }; | |
| 988 | ||
| 989 | /// Writes out a version 2 index for the given packfile, as documented in | |
| 990 | /// [pack-format](https://git-scm.com/docs/pack-format). | |
| 991 | pub fn indexPack(allocator: Allocator, pack: std.fs.File, index_writer: anytype) !void { | |
| 992 | try pack.seekTo(0); | |
| 993 | ||
| 994 | var index_entries = std.AutoHashMapUnmanaged(Oid, IndexEntry){}; | |
| 995 | defer index_entries.deinit(allocator); | |
| 996 | var pending_deltas = std.ArrayListUnmanaged(IndexEntry){}; | |
| 997 | defer pending_deltas.deinit(allocator); | |
| 998 | ||
| 999 | const pack_checksum = try indexPackFirstPass(allocator, pack, &index_entries, &pending_deltas); | |
| 1000 | ||
| 1001 | var cache: ObjectCache = .{}; | |
| 1002 | defer cache.deinit(allocator); | |
| 1003 | var remaining_deltas = pending_deltas.items.len; | |
| 1004 | while (remaining_deltas > 0) { | |
| 1005 | var i: usize = remaining_deltas; | |
| 1006 | while (i > 0) { | |
| 1007 | i -= 1; | |
| 1008 | const delta = pending_deltas.items[i]; | |
| 1009 | if (try indexPackHashDelta(allocator, pack, delta, index_entries, &cache)) |oid| { | |
| 1010 | try index_entries.put(allocator, oid, delta); | |
| 1011 | _ = pending_deltas.swapRemove(i); | |
| 1012 | } | |
| 1013 | } | |
| 1014 | if (pending_deltas.items.len == remaining_deltas) return error.IncompletePack; | |
| 1015 | remaining_deltas = pending_deltas.items.len; | |
| 1016 | } | |
| 1017 | ||
| 1018 | var oids = std.ArrayListUnmanaged(Oid){}; | |
| 1019 | defer oids.deinit(allocator); | |
| 1020 | try oids.ensureTotalCapacityPrecise(allocator, index_entries.count()); | |
| 1021 | var index_entries_iter = index_entries.iterator(); | |
| 1022 | while (index_entries_iter.next()) |entry| { | |
| 1023 | oids.appendAssumeCapacity(entry.key_ptr.*); | |
| 1024 | } | |
| 1025 | mem.sortUnstable(Oid, oids.items, {}, struct { | |
| 1026 | fn lessThan(_: void, o1: Oid, o2: Oid) bool { | |
| 1027 | return mem.lessThan(u8, &o1, &o2); | |
| 1028 | } | |
| 1029 | }.lessThan); | |
| 1030 | ||
| 1031 | var fan_out_table: [256]u32 = undefined; | |
| 1032 | var count: u32 = 0; | |
| 1033 | var fan_out_index: u8 = 0; | |
| 1034 | for (oids.items) |oid| { | |
| 1035 | if (oid[0] > fan_out_index) { | |
| 1036 | @memset(fan_out_table[fan_out_index..oid[0]], count); | |
| 1037 | fan_out_index = oid[0]; | |
| 1038 | } | |
| 1039 | count += 1; | |
| 1040 | } | |
| 1041 | @memset(fan_out_table[fan_out_index..], count); | |
| 1042 | ||
| 1043 | var index_hashed_writer = hashedWriter(index_writer, Sha1.init(.{})); | |
| 1044 | const writer = index_hashed_writer.writer(); | |
| 1045 | try writer.writeAll(IndexHeader.signature); | |
| 1046 | try writer.writeIntBig(u32, IndexHeader.supported_version); | |
| 1047 | for (fan_out_table) |fan_out_entry| { | |
| 1048 | try writer.writeIntBig(u32, fan_out_entry); | |
| 1049 | } | |
| 1050 | ||
| 1051 | for (oids.items) |oid| { | |
| 1052 | try writer.writeAll(&oid); | |
| 1053 | } | |
| 1054 | ||
| 1055 | for (oids.items) |oid| { | |
| 1056 | try writer.writeIntBig(u32, index_entries.get(oid).?.crc32); | |
| 1057 | } | |
| 1058 | ||
| 1059 | var big_offsets = std.ArrayListUnmanaged(u64){}; | |
| 1060 | defer big_offsets.deinit(allocator); | |
| 1061 | for (oids.items) |oid| { | |
| 1062 | const offset = index_entries.get(oid).?.offset; | |
| 1063 | if (offset <= std.math.maxInt(u31)) { | |
| 1064 | try writer.writeIntBig(u32, @intCast(offset)); | |
| 1065 | } else { | |
| 1066 | const index = big_offsets.items.len; | |
| 1067 | try big_offsets.append(allocator, offset); | |
| 1068 | try writer.writeIntBig(u32, @as(u32, @intCast(index)) | (1 << 31)); | |
| 1069 | } | |
| 1070 | } | |
| 1071 | for (big_offsets.items) |offset| { | |
| 1072 | try writer.writeIntBig(u64, offset); | |
| 1073 | } | |
| 1074 | ||
| 1075 | try writer.writeAll(&pack_checksum); | |
| 1076 | const index_checksum = index_hashed_writer.hasher.finalResult(); | |
| 1077 | try index_writer.writeAll(&index_checksum); | |
| 1078 | } | |
| 1079 | ||
| 1080 | /// Performs the first pass over the packfile data for index construction. | |
| 1081 | /// This will index all non-delta objects, queue delta objects for further | |
| 1082 | /// processing, and return the pack checksum (which is part of the index | |
| 1083 | /// format). | |
| 1084 | fn indexPackFirstPass( | |
| 1085 | allocator: Allocator, | |
| 1086 | pack: std.fs.File, | |
| 1087 | index_entries: *std.AutoHashMapUnmanaged(Oid, IndexEntry), | |
| 1088 | pending_deltas: *std.ArrayListUnmanaged(IndexEntry), | |
| 1089 | ) ![Sha1.digest_length]u8 { | |
| 1090 | var pack_buffered_reader = std.io.bufferedReader(pack.reader()); | |
| 1091 | var pack_counting_reader = std.io.countingReader(pack_buffered_reader.reader()); | |
| 1092 | var pack_hashed_reader = std.compress.hashedReader(pack_counting_reader.reader(), Sha1.init(.{})); | |
| 1093 | const pack_reader = pack_hashed_reader.reader(); | |
| 1094 | ||
| 1095 | const pack_header = try PackHeader.read(pack_reader); | |
| 1096 | ||
| 1097 | var current_entry: u32 = 0; | |
| 1098 | while (current_entry < pack_header.total_objects) : (current_entry += 1) { | |
| 1099 | const entry_offset = pack_counting_reader.bytes_read; | |
| 1100 | var entry_crc32_reader = std.compress.hashedReader(pack_reader, std.hash.Crc32.init()); | |
| 1101 | const entry_header = try EntryHeader.read(entry_crc32_reader.reader()); | |
| 1102 | switch (entry_header) { | |
| 1103 | inline .commit, .tree, .blob, .tag => |object, tag| { | |
| 1104 | var entry_decompress_stream = try std.compress.zlib.decompressStream(allocator, entry_crc32_reader.reader()); | |
| 1105 | defer entry_decompress_stream.deinit(); | |
| 1106 | var entry_counting_reader = std.io.countingReader(entry_decompress_stream.reader()); | |
| 1107 | var entry_hashed_writer = hashedWriter(std.io.null_writer, Sha1.init(.{})); | |
| 1108 | const entry_writer = entry_hashed_writer.writer(); | |
| 1109 | // The object header is not included in the pack data but is | |
| 1110 | // part of the object's ID | |
| 1111 | try entry_writer.print("{s} {}\x00", .{ @tagName(tag), object.uncompressed_length }); | |
| 1112 | var fifo = std.fifo.LinearFifo(u8, .{ .Static = 4096 }).init(); | |
| 1113 | try fifo.pump(entry_counting_reader.reader(), entry_writer); | |
| 1114 | if (entry_counting_reader.bytes_read != object.uncompressed_length) { | |
| 1115 | return error.InvalidObject; | |
| 1116 | } | |
| 1117 | const oid = entry_hashed_writer.hasher.finalResult(); | |
| 1118 | try index_entries.put(allocator, oid, .{ | |
| 1119 | .offset = entry_offset, | |
| 1120 | .crc32 = entry_crc32_reader.hasher.final(), | |
| 1121 | }); | |
| 1122 | }, | |
| 1123 | inline .ofs_delta, .ref_delta => |delta| { | |
| 1124 | var entry_decompress_stream = try std.compress.zlib.decompressStream(allocator, entry_crc32_reader.reader()); | |
| 1125 | defer entry_decompress_stream.deinit(); | |
| 1126 | var entry_counting_reader = std.io.countingReader(entry_decompress_stream.reader()); | |
| 1127 | var fifo = std.fifo.LinearFifo(u8, .{ .Static = 4096 }).init(); | |
| 1128 | try fifo.pump(entry_counting_reader.reader(), std.io.null_writer); | |
| 1129 | if (entry_counting_reader.bytes_read != delta.uncompressed_length) { | |
| 1130 | return error.InvalidObject; | |
| 1131 | } | |
| 1132 | try pending_deltas.append(allocator, .{ | |
| 1133 | .offset = entry_offset, | |
| 1134 | .crc32 = entry_crc32_reader.hasher.final(), | |
| 1135 | }); | |
| 1136 | }, | |
| 1137 | } | |
| 1138 | } | |
| 1139 | ||
| 1140 | const pack_checksum = pack_hashed_reader.hasher.finalResult(); | |
| 1141 | const recorded_checksum = try pack_buffered_reader.reader().readBytesNoEof(Sha1.digest_length); | |
| 1142 | if (!mem.eql(u8, &pack_checksum, &recorded_checksum)) { | |
| 1143 | return error.CorruptedPack; | |
| 1144 | } | |
| 1145 | _ = pack_buffered_reader.reader().readByte() catch |e| switch (e) { | |
| 1146 | error.EndOfStream => return pack_checksum, | |
| 1147 | else => |other| return other, | |
| 1148 | }; | |
| 1149 | return error.InvalidFormat; | |
| 1150 | } | |
| 1151 | ||
| 1152 | /// Attempts to determine the final object ID of the given deltified object. | |
| 1153 | /// May return null if this is not yet possible (if the delta is a ref-based | |
| 1154 | /// delta and we do not yet know the offset of the base object). | |
| 1155 | fn indexPackHashDelta( | |
| 1156 | allocator: Allocator, | |
| 1157 | pack: std.fs.File, | |
| 1158 | delta: IndexEntry, | |
| 1159 | index_entries: std.AutoHashMapUnmanaged(Oid, IndexEntry), | |
| 1160 | cache: *ObjectCache, | |
| 1161 | ) !?Oid { | |
| 1162 | // Figure out the chain of deltas to resolve | |
| 1163 | var base_offset = delta.offset; | |
| 1164 | var base_header: EntryHeader = undefined; | |
| 1165 | var delta_offsets = std.ArrayListUnmanaged(u64){}; | |
| 1166 | defer delta_offsets.deinit(allocator); | |
| 1167 | const base_object = while (true) { | |
| 1168 | if (cache.get(base_offset)) |base_object| break base_object; | |
| 1169 | ||
| 1170 | try pack.seekTo(base_offset); | |
| 1171 | base_header = try EntryHeader.read(pack.reader()); | |
| 1172 | switch (base_header) { | |
| 1173 | .ofs_delta => |ofs_delta| { | |
| 1174 | try delta_offsets.append(allocator, base_offset); | |
| 1175 | base_offset = std.math.sub(u64, base_offset, ofs_delta.offset) catch return error.InvalidObject; | |
| 1176 | }, | |
| 1177 | .ref_delta => |ref_delta| { | |
| 1178 | try delta_offsets.append(allocator, base_offset); | |
| 1179 | base_offset = (index_entries.get(ref_delta.base_object) orelse return null).offset; | |
| 1180 | }, | |
| 1181 | else => { | |
| 1182 | const base_data = try readObjectRaw(allocator, pack.reader(), base_header.uncompressedLength()); | |
| 1183 | errdefer allocator.free(base_data); | |
| 1184 | const base_object: Object = .{ .type = base_header.objectType(), .data = base_data }; | |
| 1185 | try cache.put(allocator, base_offset, base_object); | |
| 1186 | break base_object; | |
| 1187 | }, | |
| 1188 | } | |
| 1189 | }; | |
| 1190 | ||
| 1191 | const base_data = try resolveDeltaChain(allocator, pack, base_object, delta_offsets.items, cache); | |
| 1192 | ||
| 1193 | var entry_hasher = Sha1.init(.{}); | |
| 1194 | var entry_hashed_writer = hashedWriter(std.io.null_writer, &entry_hasher); | |
| 1195 | try entry_hashed_writer.writer().print("{s} {}\x00", .{ @tagName(base_object.type), base_data.len }); | |
| 1196 | entry_hasher.update(base_data); | |
| 1197 | return entry_hasher.finalResult(); | |
| 1198 | } | |
| 1199 | ||
| 1200 | /// Resolves a chain of deltas, returning the final base object data. `pack` is | |
| 1201 | /// assumed to be looking at the start of the object data for the base object of | |
| 1202 | /// the chain, and will then apply the deltas in `delta_offsets` in reverse order | |
| 1203 | /// to obtain the final object. | |
| 1204 | fn resolveDeltaChain( | |
| 1205 | allocator: Allocator, | |
| 1206 | pack: std.fs.File, | |
| 1207 | base_object: Object, | |
| 1208 | delta_offsets: []const u64, | |
| 1209 | cache: *ObjectCache, | |
| 1210 | ) ![]const u8 { | |
| 1211 | var base_data = base_object.data; | |
| 1212 | var i: usize = delta_offsets.len; | |
| 1213 | while (i > 0) { | |
| 1214 | i -= 1; | |
| 1215 | ||
| 1216 | const delta_offset = delta_offsets[i]; | |
| 1217 | try pack.seekTo(delta_offset); | |
| 1218 | const delta_header = try EntryHeader.read(pack.reader()); | |
| 1219 | var delta_data = try readObjectRaw(allocator, pack.reader(), delta_header.uncompressedLength()); | |
| 1220 | defer allocator.free(delta_data); | |
| 1221 | var delta_stream = std.io.fixedBufferStream(delta_data); | |
| 1222 | const delta_reader = delta_stream.reader(); | |
| 1223 | _ = try readSizeVarInt(delta_reader); // base object size | |
| 1224 | const expanded_size = try readSizeVarInt(delta_reader); | |
| 1225 | ||
| 1226 | const expanded_alloc_size = std.math.cast(usize, expanded_size) orelse return error.ObjectTooLarge; | |
| 1227 | var expanded_data = try allocator.alloc(u8, expanded_alloc_size); | |
| 1228 | errdefer allocator.free(expanded_data); | |
| 1229 | var expanded_delta_stream = std.io.fixedBufferStream(expanded_data); | |
| 1230 | var base_stream = std.io.fixedBufferStream(base_data); | |
| 1231 | try expandDelta(&base_stream, delta_reader, expanded_delta_stream.writer()); | |
| 1232 | if (expanded_delta_stream.pos != expanded_size) return error.InvalidObject; | |
| 1233 | ||
| 1234 | try cache.put(allocator, delta_offset, .{ .type = base_object.type, .data = expanded_data }); | |
| 1235 | base_data = expanded_data; | |
| 1236 | } | |
| 1237 | return base_data; | |
| 1238 | } | |
| 1239 | ||
| 1240 | /// Reads the complete contents of an object from `reader`. This function may | |
| 1241 | /// read more bytes than required from `reader`, so the reader position after | |
| 1242 | /// returning is not reliable. | |
| 1243 | fn readObjectRaw(allocator: Allocator, reader: anytype, size: u64) ![]u8 { | |
| 1244 | const alloc_size = std.math.cast(usize, size) orelse return error.ObjectTooLarge; | |
| 1245 | var buffered_reader = std.io.bufferedReader(reader); | |
| 1246 | var decompress_stream = try std.compress.zlib.decompressStream(allocator, buffered_reader.reader()); | |
| 1247 | defer decompress_stream.deinit(); | |
| 1248 | var data = try allocator.alloc(u8, alloc_size); | |
| 1249 | errdefer allocator.free(data); | |
| 1250 | try decompress_stream.reader().readNoEof(data); | |
| 1251 | _ = decompress_stream.reader().readByte() catch |e| switch (e) { | |
| 1252 | error.EndOfStream => return data, | |
| 1253 | else => |other| return other, | |
| 1254 | }; | |
| 1255 | return error.InvalidFormat; | |
| 1256 | } | |
| 1257 | ||
| 1258 | /// Expands delta data from `delta_reader` to `writer`. `base_object` must | |
| 1259 | /// support `reader` and `seekTo` (such as a `std.io.FixedBufferStream`). | |
| 1260 | /// | |
| 1261 | /// The format of the delta data is documented in | |
| 1262 | /// [pack-format](https://git-scm.com/docs/pack-format). | |
| 1263 | fn expandDelta(base_object: anytype, delta_reader: anytype, writer: anytype) !void { | |
| 1264 | while (true) { | |
| 1265 | const inst: packed struct { value: u7, copy: bool } = @bitCast(delta_reader.readByte() catch |e| switch (e) { | |
| 1266 | error.EndOfStream => return, | |
| 1267 | else => |other| return other, | |
| 1268 | }); | |
| 1269 | if (inst.copy) { | |
| 1270 | const available: packed struct { | |
| 1271 | offset1: bool, | |
| 1272 | offset2: bool, | |
| 1273 | offset3: bool, | |
| 1274 | offset4: bool, | |
| 1275 | size1: bool, | |
| 1276 | size2: bool, | |
| 1277 | size3: bool, | |
| 1278 | } = @bitCast(inst.value); | |
| 1279 | var offset_parts: packed struct { offset1: u8, offset2: u8, offset3: u8, offset4: u8 } = .{ | |
| 1280 | .offset1 = if (available.offset1) try delta_reader.readByte() else 0, | |
| 1281 | .offset2 = if (available.offset2) try delta_reader.readByte() else 0, | |
| 1282 | .offset3 = if (available.offset3) try delta_reader.readByte() else 0, | |
| 1283 | .offset4 = if (available.offset4) try delta_reader.readByte() else 0, | |
| 1284 | }; | |
| 1285 | const offset: u32 = @bitCast(offset_parts); | |
| 1286 | var size_parts: packed struct { size1: u8, size2: u8, size3: u8 } = .{ | |
| 1287 | .size1 = if (available.size1) try delta_reader.readByte() else 0, | |
| 1288 | .size2 = if (available.size2) try delta_reader.readByte() else 0, | |
| 1289 | .size3 = if (available.size3) try delta_reader.readByte() else 0, | |
| 1290 | }; | |
| 1291 | var size: u24 = @bitCast(size_parts); | |
| 1292 | if (size == 0) size = 0x10000; | |
| 1293 | try base_object.seekTo(offset); | |
| 1294 | var copy_reader = std.io.limitedReader(base_object.reader(), size); | |
| 1295 | var fifo = std.fifo.LinearFifo(u8, .{ .Static = 4096 }).init(); | |
| 1296 | try fifo.pump(copy_reader.reader(), writer); | |
| 1297 | } else if (inst.value != 0) { | |
| 1298 | var data_reader = std.io.limitedReader(delta_reader, inst.value); | |
| 1299 | var fifo = std.fifo.LinearFifo(u8, .{ .Static = 4096 }).init(); | |
| 1300 | try fifo.pump(data_reader.reader(), writer); | |
| 1301 | } else { | |
| 1302 | return error.InvalidDeltaInstruction; | |
| 1303 | } | |
| 1304 | } | |
| 1305 | } | |
| 1306 | ||
| 1307 | fn HashedWriter( | |
| 1308 | comptime WriterType: anytype, | |
| 1309 | comptime HasherType: anytype, | |
| 1310 | ) type { | |
| 1311 | return struct { | |
| 1312 | child_writer: WriterType, | |
| 1313 | hasher: HasherType, | |
| 1314 | ||
| 1315 | const Error = WriterType.Error; | |
| 1316 | const Writer = std.io.Writer(*@This(), Error, write); | |
| 1317 | ||
| 1318 | fn write(hashed_writer: *@This(), buf: []const u8) Error!usize { | |
| 1319 | const amt = try hashed_writer.child_writer.write(buf); | |
| 1320 | hashed_writer.hasher.update(buf); | |
| 1321 | return amt; | |
| 1322 | } | |
| 1323 | ||
| 1324 | fn writer(hashed_writer: *@This()) Writer { | |
| 1325 | return .{ .context = hashed_writer }; | |
| 1326 | } | |
| 1327 | }; | |
| 1328 | } | |
| 1329 | ||
| 1330 | fn hashedWriter( | |
| 1331 | writer: anytype, | |
| 1332 | hasher: anytype, | |
| 1333 | ) HashedWriter(@TypeOf(writer), @TypeOf(hasher)) { | |
| 1334 | return .{ .child_writer = writer, .hasher = hasher }; | |
| 1335 | } | |
| 1336 | ||
| 1337 | test "packfile indexing and checkout" { | |
| 1338 | // To verify the contents of this packfile without using the code in this | |
| 1339 | // file: | |
| 1340 | // | |
| 1341 | // 1. Create a new empty Git repository (`git init`) | |
| 1342 | // 2. `git unpack-objects <path/to/testdata.pack` | |
| 1343 | // 3. `git fsck` -> note the "dangling commit" ID (which matches the commit | |
| 1344 | // checked out below) | |
| 1345 | // 4. `git checkout dd582c0720819ab7130b103635bd7271b9fd4feb` | |
| 1346 | const testrepo_pack = @embedFile("git/testdata/testrepo.pack"); | |
| 1347 | ||
| 1348 | var git_dir = testing.tmpDir(.{}); | |
| 1349 | defer git_dir.cleanup(); | |
| 1350 | var pack_file = try git_dir.dir.createFile("testrepo.pack", .{ .read = true }); | |
| 1351 | defer pack_file.close(); | |
| 1352 | try pack_file.writeAll(testrepo_pack); | |
| 1353 | ||
| 1354 | var index_file = try git_dir.dir.createFile("testrepo.idx", .{ .read = true }); | |
| 1355 | defer index_file.close(); | |
| 1356 | try indexPack(testing.allocator, pack_file, index_file.writer()); | |
| 1357 | ||
| 1358 | // Arbitrary size limit on files read while checking the repository contents | |
| 1359 | // (all files in the test repo are known to be much smaller than this) | |
| 1360 | const max_file_size = 4096; | |
| 1361 | ||
| 1362 | const index_file_data = try git_dir.dir.readFileAlloc(testing.allocator, "testrepo.idx", max_file_size); | |
| 1363 | defer testing.allocator.free(index_file_data); | |
| 1364 | // testrepo.idx is generated by Git. The index created by this file should | |
| 1365 | // match it exactly. Running `git verify-pack -v testrepo.pack` can verify | |
| 1366 | // this. | |
| 1367 | const testrepo_idx = @embedFile("git/testdata/testrepo.idx"); | |
| 1368 | try testing.expectEqualSlices(u8, testrepo_idx, index_file_data); | |
| 1369 | ||
| 1370 | var repository = try Repository.init(testing.allocator, pack_file, index_file); | |
| 1371 | defer repository.deinit(); | |
| 1372 | ||
| 1373 | var worktree = testing.tmpIterableDir(.{}); | |
| 1374 | defer worktree.cleanup(); | |
| 1375 | ||
| 1376 | const commit_id = try parseOid("dd582c0720819ab7130b103635bd7271b9fd4feb"); | |
| 1377 | try repository.checkout(worktree.iterable_dir.dir, commit_id); | |
| 1378 | ||
| 1379 | const expected_files: []const []const u8 = &.{ | |
| 1380 | "dir/file", | |
| 1381 | "dir/subdir/file", | |
| 1382 | "dir/subdir/file2", | |
| 1383 | "dir2/file", | |
| 1384 | "dir3/file", | |
| 1385 | "dir3/file2", | |
| 1386 | "file", | |
| 1387 | "file2", | |
| 1388 | "file3", | |
| 1389 | "file4", | |
| 1390 | "file5", | |
| 1391 | "file6", | |
| 1392 | "file7", | |
| 1393 | "file8", | |
| 1394 | "file9", | |
| 1395 | }; | |
| 1396 | var actual_files: std.ArrayListUnmanaged([]u8) = .{}; | |
| 1397 | defer actual_files.deinit(testing.allocator); | |
| 1398 | defer for (actual_files.items) |file| testing.allocator.free(file); | |
| 1399 | var walker = try worktree.iterable_dir.walk(testing.allocator); | |
| 1400 | defer walker.deinit(); | |
| 1401 | while (try walker.next()) |entry| { | |
| 1402 | if (entry.kind != .file) continue; | |
| 1403 | var path = try testing.allocator.dupe(u8, entry.path); | |
| 1404 | errdefer testing.allocator.free(path); | |
| 1405 | mem.replaceScalar(u8, path, std.fs.path.sep, '/'); | |
| 1406 | try actual_files.append(testing.allocator, path); | |
| 1407 | } | |
| 1408 | mem.sortUnstable([]u8, actual_files.items, {}, struct { | |
| 1409 | fn lessThan(_: void, a: []u8, b: []u8) bool { | |
| 1410 | return mem.lessThan(u8, a, b); | |
| 1411 | } | |
| 1412 | }.lessThan); | |
| 1413 | try testing.expectEqualDeep(expected_files, actual_files.items); | |
| 1414 | ||
| 1415 | const expected_file_contents = | |
| 1416 | \\revision 1 | |
| 1417 | \\revision 2 | |
| 1418 | \\revision 4 | |
| 1419 | \\revision 5 | |
| 1420 | \\revision 7 | |
| 1421 | \\revision 8 | |
| 1422 | \\revision 9 | |
| 1423 | \\revision 10 | |
| 1424 | \\revision 12 | |
| 1425 | \\revision 13 | |
| 1426 | \\revision 14 | |
| 1427 | \\revision 18 | |
| 1428 | \\revision 19 | |
| 1429 | \\ | |
| 1430 | ; | |
| 1431 | const actual_file_contents = try worktree.iterable_dir.dir.readFileAlloc(testing.allocator, "file", max_file_size); | |
| 1432 | defer testing.allocator.free(actual_file_contents); | |
| 1433 | try testing.expectEqualStrings(expected_file_contents, actual_file_contents); | |
| 1434 | } | |
| 1435 | ||
| 1436 | /// Checks out a commit of a packfile. Intended for experimenting with and | |
| 1437 | /// benchmarking possible optimizations to the indexing and checkout behavior. | |
| 1438 | pub fn main() !void { | |
| 1439 | const allocator = std.heap.c_allocator; | |
| 1440 | ||
| 1441 | const args = try std.process.argsAlloc(allocator); | |
| 1442 | defer std.process.argsFree(allocator, args); | |
| 1443 | if (args.len != 4) { | |
| 1444 | return error.InvalidArguments; // Arguments: packfile commit worktree | |
| 1445 | } | |
| 1446 | ||
| 1447 | var pack_file = try std.fs.cwd().openFile(args[1], .{}); | |
| 1448 | defer pack_file.close(); | |
| 1449 | const commit = try parseOid(args[2]); | |
| 1450 | var worktree = try std.fs.cwd().makeOpenPath(args[3], .{}); | |
| 1451 | defer worktree.close(); | |
| 1452 | ||
| 1453 | var git_dir = try worktree.makeOpenPath(".git", .{}); | |
| 1454 | defer git_dir.close(); | |
| 1455 | ||
| 1456 | std.debug.print("Starting index...\n", .{}); | |
| 1457 | var index_file = try git_dir.createFile("idx", .{ .read = true }); | |
| 1458 | defer index_file.close(); | |
| 1459 | var index_buffered_writer = std.io.bufferedWriter(index_file.writer()); | |
| 1460 | try indexPack(allocator, pack_file, index_buffered_writer.writer()); | |
| 1461 | try index_buffered_writer.flush(); | |
| 1462 | try index_file.sync(); | |
| 1463 | ||
| 1464 | std.debug.print("Starting checkout...\n", .{}); | |
| 1465 | var repository = try Repository.init(allocator, pack_file, index_file); | |
| 1466 | defer repository.deinit(); | |
| 1467 | try repository.checkout(worktree, commit); | |
| 1468 | } |
src/git/testdata/testrepo.idx deleted| Binary files a/src/git/testdata/testrepo.idx and /dev/null differ |
src/git/testdata/testrepo.pack deleted| Binary files a/src/git/testdata/testrepo.pack and /dev/null differ |