Doing git tree lookups based on the SHA-1 of the Message-ID
is expensive as trees get larger, instead, use the SHA-1
object ID directly. This drastically reduces the amount
of time spent in the "git cat-file --batch" process for
fetching the /$INBOX/all.mbox.gz endpoint on the ~800MB
git@vger.kernel.org mirror
This retains backwards compatibility and allows existing
indices to be transparently upgraded without performance
degradation.
+sub msg_by_smsg ($$;$) {
+ my ($self, $smsg, $ref) = @_;
+
+ # backwards compat to fallback to msg_by_mid
+ # TODO: remove if we bump SCHEMA_VERSION in Search.pm:
+ defined(my $blob = $smsg->blob) or return msg_by_mid($self, $smsg->mid);
+
+ my $str = git($self)->cat_file($blob, $ref);
+ $$str =~ s/\A[\r\n]*From [^\r\n]*\r?\n//s if $str;
+ $str;
+}
+
sub path_check {
my ($self, $path) = @_;
git($self)->check('HEAD:'.$path);
sub path_check {
my ($self, $path) = @_;
git($self)->check('HEAD:'.$path);
my $gz = $self->{gz};
do {
while (defined(my $smsg = shift @{$self->{msgs}})) {
my $gz = $self->{gz};
do {
while (defined(my $smsg = shift @{$self->{msgs}})) {
- my $msg = eval { $ibx->msg_by_mid($smsg->mid) } or next;
+ my $msg = eval { $ibx->msg_by_smsg($smsg) } or next;
$msg = Email::Simple->new($msg);
$gz->write(PublicInbox::Mbox::msg_str($ctx, $msg));
my $bref = $self->{buf};
$msg = Email::Simple->new($msg);
$gz->write(PublicInbox::Mbox::msg_str($ctx, $msg));
my $bref = $self->{buf};
- my ($self, $mime, $bytes, $num) = @_; # mime = Email::MIME object
+ my ($self, $mime, $bytes, $num, $blob) = @_; # mime = Email::MIME object
my $db = $self->{xdb};
my ($doc_id, $old_tid);
my $db = $self->{xdb};
my ($doc_id, $old_tid);
});
link_message($self, $smsg, $old_tid);
});
link_message($self, $smsg, $old_tid);
- $doc->set_data($smsg->to_doc_data);
+ $doc->set_data($smsg->to_doc_data($blob));
if (defined $doc_id) {
$db->replace_document($doc_id, $doc);
} else {
if (defined $doc_id) {
$db->replace_document($doc_id, $doc);
} else {
- my ($self, $git, $mime, $bytes, $num) = @_;
- $self->add_message($mime, $bytes, $num);
+ my ($self, $git, $mime, $bytes, $num, $blob) = @_;
+ $self->add_message($mime, $bytes, $num, $blob);
- my ($self, $git, $mime, $bytes) = @_;
+ my ($self, $git, $mime, $bytes, $blob) = @_;
my $num = $self->{mm}->num_for(mid_clean(mid_mime($mime)));
my $num = $self->{mm}->num_for(mid_clean(mid_mime($mime)));
- index_blob($self, $git, $mime, $bytes, $num);
+ index_blob($self, $git, $mime, $bytes, $num, $blob);
- my ($self, $git, $mime, $bytes) = @_;
+ my ($self, $git, $mime, $bytes, $blob) = @_;
my $num = index_mm($self, $git, $mime);
my $num = index_mm($self, $git, $mime);
- index_blob($self, $git, $mime, $bytes, $num);
+ index_blob($self, $git, $mime, $bytes, $num, $blob);
my $line;
while (defined($line = <$log>)) {
if ($line =~ /$addmsg/o) {
my $line;
while (defined($line = <$log>)) {
if ($line =~ /$addmsg/o) {
- my $mime = do_cat_mail($git, $1, \$bytes) or next;
- $add_cb->($self, $git, $mime, $bytes);
+ my $blob = $1;
+ my $mime = do_cat_mail($git, $blob, \$bytes) or next;
+ $add_cb->($self, $git, $mime, $bytes, $blob);
} elsif ($line =~ /$delmsg/o) {
} elsif ($line =~ /$delmsg/o) {
- my $mime = do_cat_mail($git, $1) or next;
+ my $blob = $1;
+ my $mime = do_cat_mail($git, $blob) or next;
$del_cb->($self, $git, $mime);
} elsif ($line =~ /^commit ($h40)/o) {
if (defined $max && --$max <= 0) {
$del_cb->($self, $git, $mime);
} elsif ($line =~ /^commit ($h40)/o) {
if (defined $max && --$max <= 0) {
my $data = $doc->get_data or return;
my $ts = get_val($doc, &PublicInbox::Search::TS);
utf8::decode($data);
my $data = $doc->get_data or return;
my $ts = get_val($doc, &PublicInbox::Search::TS);
utf8::decode($data);
- my ($subj, $from, $refs, $to, $cc) = split(/\n/, $data);
+ my ($subj, $from, $refs, $to, $cc, $blob) = split(/\n/, $data);
bless {
doc => $doc,
subject => $subj,
bless {
doc => $doc,
subject => $subj,
references => $refs,
to => $to,
cc => $cc,
references => $refs,
to => $to,
cc => $cc,
- my ($self) = @_;
- join("\n", $self->subject, $self->from, $self->references,
- $self->to, $self->cc);
+ my ($self, $blob) = @_;
+ my @rows = ($self->subject, $self->from, $self->references,
+ $self->to, $self->cc);
+ push @rows, $blob if defined $blob;
+ join("\n", @rows);
sub _extract_mid { mid_clean(mid_mime($_[0]->mime)) }
sub _extract_mid { mid_clean(mid_mime($_[0]->mime)) }
+sub blob {
+ my ($self, $x40) = @_;
+ if (defined $x40) {
+ $self->{blob} = $x40;
+ } else {
+ $self->{blob};
+ }
+}
+
sub mime {
my ($self, $mime) = @_;
if (defined $mime) {
sub mime {
my ($self, $mime) = @_;
if (defined $mime) {