| // Copyright 2010 Google Inc. |
| // |
| // Licensed under the Apache License, Version 2.0 (the "License"); |
| // you may not use this file except in compliance with the License. |
| // You may obtain a copy of the License at |
| // |
| // http://www.apache.org/licenses/LICENSE-2.0 |
| // |
| // Unless required by applicable law or agreed to in writing, software |
| // distributed under the License is distributed on an "AS IS" BASIS, |
| // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. |
| // See the License for the specific language governing permissions and |
| // limitations under the License. |
| // |
| // Author: jmarantz@google.com (Joshua Marantz) |
| // lsong@google.com (Libo Song) |
| |
| #include "net/instaweb/apache/instaweb_context.h" |
| #include "net/instaweb/apache/header_util.h" |
| #include "net/instaweb/util/public/content_type.h" |
| #include "net/instaweb/util/public/gzip_inflater.h" |
| #include "net/instaweb/util/stack_buffer.h" |
| #include "http_config.h" |
| |
| extern "C" { |
| extern module AP_MODULE_DECLARE_DATA pagespeed_module; |
| } |
| |
| namespace net_instaweb { |
| |
| const char InstawebContext::kRepairHeadersFilterName[] = |
| "MOD_PAGESPEED_REPAIR_HEADERS"; |
| |
| InstawebContext::InstawebContext(request_rec* request, |
| const ContentType& content_type, |
| ApacheRewriteDriverFactory* factory, |
| const std::string& absolute_url, |
| bool use_custom_options, |
| const RewriteOptions& custom_options) |
| : content_encoding_(kNone), |
| content_type_(content_type), |
| factory_(factory), |
| string_writer_(&output_), |
| inflater_(NULL), |
| content_detection_state_(kStart), |
| absolute_url_(absolute_url) { |
| if (use_custom_options) { |
| // TODO(jmarantz): this is a temporary hack until we sort out better |
| // memory management of RewriteOptions. This will drag on performance. |
| // We need to do this because we are changing RewriteDriver to keep |
| // a reference to its options throughout its lifetime to refer to the |
| // domain lawyer and other options. |
| rewrite_options_.CopyFrom(custom_options); |
| custom_rewriter_.reset(factory->NewCustomRewriteDriver(rewrite_options_)); |
| rewrite_driver_ = custom_rewriter_.get(); |
| } else { |
| rewrite_driver_ = factory->NewRewriteDriver(); |
| } |
| |
| ComputeContentEncoding(request); |
| apr_pool_cleanup_register(request->pool, this, Cleanup, |
| apr_pool_cleanup_null); |
| |
| bucket_brigade_ = apr_brigade_create(request->pool, |
| request->connection->bucket_alloc); |
| |
| if (content_encoding_ == kGzip || content_encoding_ == kDeflate) { |
| // TODO(jmarantz): consider keeping a pool of these if they are expensive |
| // to initialize. |
| if (content_encoding_ == kGzip) { |
| inflater_.reset(new GzipInflater(GzipInflater::kGzip)); |
| } else { |
| inflater_.reset(new GzipInflater(GzipInflater::kDeflate)); |
| } |
| inflater_->Init(); |
| } |
| |
| |
| const char* user_agent = apr_table_get(request->headers_in, |
| HttpAttributes::kUserAgent); |
| rewrite_driver_->SetUserAgent(user_agent); |
| // TODO(lsong): Bypass the string buffer, writer data directly to the next |
| // apache bucket. |
| rewrite_driver_->SetWriter(&string_writer_); |
| } |
| |
| InstawebContext::~InstawebContext() { |
| if (custom_rewriter_ == NULL) { |
| factory_->ReleaseRewriteDriver(rewrite_driver_); |
| } |
| } |
| |
| void InstawebContext::Rewrite(const char* input, int size) { |
| if (inflater_.get() != NULL) { |
| char buf[kStackBufferSize]; |
| inflater_->SetInput(input, size); |
| while (inflater_->HasUnconsumedInput()) { |
| int num_inflated_bytes = inflater_->InflateBytes(buf, kStackBufferSize); |
| ProcessBytes(buf, num_inflated_bytes); |
| } |
| } else { |
| ProcessBytes(input, size); |
| } |
| } |
| |
| namespace { |
| |
| // http://en.wikipedia.org/wiki/Byte_order_mark |
| // |
| // The byte-order marker sequence will typically appear at the beginning of |
| // an HTML or XML file. We probably should be order-sensitive but for now |
| // we will just treat all such characters as allowable characters preceding |
| // the HTML. Note the use of unsigned char here to avoid sign-extending when |
| // comparing to the int constants. |
| inline bool IsByteOrderMarkerCharacter(unsigned char c) { |
| return ((c == 0xef) || (c == 0xbb) || (c == 0xbf)); |
| } |
| |
| } // namespace |
| |
| void InstawebContext::ProcessBytes(const char* input, int size) { |
| // Try to figure out whether this looks like HTML or not, if we haven't |
| // figured it out already. We just scan past whitespace for '<'. |
| for (int i = 0; (content_detection_state_ == kStart) && (i < size); ++i) { |
| char c = input[i]; |
| if (c == '<') { |
| content_detection_state_ = kHtml; |
| rewrite_driver_->html_parse()->StartParseWithType(absolute_url_, |
| content_type_); |
| } else if (!isspace(c) && !IsByteOrderMarkerCharacter(c)) { |
| // TODO(jmarantz): figure out whether it's possible to remove our |
| // filter from the chain entirely. |
| // |
| // TODO(jmarantz): look for 'gzip' data. We do not expect to see |
| // this if the Content-Encoding header is set upstream of mod_pagespeed, |
| // but we have heard evidence from the field that WordPress plugins and |
| // possibly other modules send compressed data through without that |
| // header. |
| content_detection_state_ = kNotHtml; |
| } |
| } |
| |
| switch (content_detection_state_) { |
| case kStart: |
| // Handle the corner where the first buffer of text contains |
| // only whitespace, which we will retain for the next call. |
| buffer_.append(input, size); |
| break; |
| |
| case kHtml: |
| // Looks like HTML: send it through the HTML rewriter. |
| if (!buffer_.empty()) { |
| rewrite_driver_->html_parse()->ParseText( |
| buffer_.data(), buffer_.size()); |
| buffer_.clear(); |
| } |
| rewrite_driver_->html_parse()->ParseText(input, size); |
| break; |
| |
| case kNotHtml: |
| // Looks like something that's not HTML. Send it directly to the |
| // output buffer. |
| output_.append(buffer_.data(), buffer_.size()); |
| buffer_.clear(); |
| output_.append(input, size); |
| break; |
| } |
| } |
| |
| apr_status_t InstawebContext::Cleanup(void* object) { |
| InstawebContext* ic = static_cast<InstawebContext*>(object); |
| delete ic; |
| return APR_SUCCESS; |
| } |
| |
| void InstawebContext::ComputeContentEncoding(request_rec* request) { |
| // Check if the content is gzipped. Steal from mod_deflate. |
| const char* encoding = apr_table_get( |
| request->headers_out, HttpAttributes::kContentEncoding); |
| if (encoding) { |
| const char* err_enc = apr_table_get(request->err_headers_out, |
| HttpAttributes::kContentEncoding); |
| if (err_enc) { |
| // We don't properly handle stacked encodings now. |
| content_encoding_ = kOther; |
| } |
| } else { |
| encoding = apr_table_get(request->err_headers_out, |
| HttpAttributes::kContentEncoding); |
| } |
| |
| if (encoding) { |
| if (strcasecmp(encoding, HttpAttributes::kGzip) == 0) { |
| content_encoding_ = kGzip; |
| } else if (strcasecmp(encoding, HttpAttributes::kDeflate) == 0) { |
| content_encoding_ = kDeflate; |
| } else { |
| content_encoding_ = kOther; |
| } |
| } |
| |
| // Copy the output headers coming into our own filter into response_headers_. |
| // This is purely for debugging context. |
| ApacheHeaderToMetaData(request->headers_out, request->status, |
| request->proto_num, &response_headers_); |
| } |
| |
| ApacheRewriteDriverFactory* InstawebContext::Factory(server_rec* server) { |
| return static_cast<ApacheRewriteDriverFactory*> |
| ap_get_module_config(server->module_config, &pagespeed_module); |
| } |
| |
| } // namespace net_instaweb |