Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,10 @@

import java.io.IOException;
import java.io.InputStream;
import java.net.URLEncoder;
import java.nio.charset.StandardCharsets;
import java.sql.SQLException;
import java.text.Normalizer;
import java.util.List;
import java.util.Objects;
import java.util.UUID;
Expand Down Expand Up @@ -110,7 +113,7 @@ public void downloadFileZip(@PathVariable UUID uuid, @RequestParam("handleId") S

Item item = (Item) dso;
name = item.getName() + ".zip";
response.setHeader(HttpHeaders.CONTENT_DISPOSITION, String.format("attachment;filename=\"%s\"", name));
response.setHeader(HttpHeaders.CONTENT_DISPOSITION, buildContentDisposition(name));
response.setContentType("application/zip");
List<Bundle> bundles = item.getBundles("ORIGINAL");
Comment thread
milanmajchrak marked this conversation as resolved.

Expand All @@ -134,4 +137,49 @@ public void downloadFileZip(@PathVariable UUID uuid, @RequestParam("handleId") S
zip.close();
response.getOutputStream().flush();
}

/**
* Build the Content-Disposition value the way vanilla's HttpHeadersInitializer does: an ASCII
* fallback in {@code filename} for clients that predate RFC 5987, plus the real UTF-8 name in
* {@code filename*} for everyone else. This endpoint has no upstream counterpart, so the logic
* is copied from vanilla rather than shared, to keep it tracking upstream's behaviour.
*/
private String buildContentDisposition(String name) {
return String.format("attachment; filename=\"%s\"; filename*=UTF-8''%s",
createFallbackAsciiName(name), createEncodedUtf8Name(name));
}

/**
* Creates a safe ASCII-only fallback filename by removing diacritics (accents)
* and replacing any remaining non-ASCII characters.
* E.g., "ä-ö-é.pdf" becomes "a-o-e.pdf".
* @param originalFilename The original filename.
* @return A string containing only ASCII characters.
*/
private String createFallbackAsciiName(String originalFilename) {
if (originalFilename == null) {
return "";
}
String normalized = Normalizer.normalize(originalFilename, Normalizer.Form.NFD);
String withoutAccents = normalized.replaceAll("\\p{InCombiningDiacriticalMarks}+", "");
// Deviates from vanilla by escaping \ and ": the value is a quoted-string, and an item name
// containing a quote closes it early. That is the bug #1267 fixed; vanilla still has it.
return withoutAccents.replaceAll("[^\\x00-\\x7F]", "")
.replace("\\", "\\\\")
.replace("\"", "\\\"");
}

/**
* Creates a percent-encoded UTF-8 filename according to RFC 5987.
* This is for the `filename*` parameter.
* E.g., "ä ö é.pdf" becomes "%C3%A4%20%C3%B6%20%C3%A9.pdf".
* @param originalFilename The original filename.
* @return A percent-encoded string.
*/
private String createEncodedUtf8Name(String originalFilename) {
if (originalFilename == null) {
return "";
}
return URLEncoder.encode(originalFilename, StandardCharsets.UTF_8).replace("+", "%20");
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -9,9 +9,11 @@

import static java.util.Objects.isNull;
import static java.util.Objects.nonNull;
import static javax.mail.internet.MimeUtility.encodeText;

import java.io.IOException;
import java.net.URLEncoder;
import java.nio.charset.StandardCharsets;
import java.text.Normalizer;
import java.util.Arrays;
import java.util.Collections;
import javax.servlet.http.HttpServletRequest;
Expand Down Expand Up @@ -165,9 +167,16 @@ public HttpHeaders initialiseHeaders() throws IOException {

httpHeaders.put(HttpHeaders.ACCESS_CONTROL_EXPOSE_HEADERS,
Collections.singletonList(HttpHeaders.ACCEPT_RANGES));
httpHeaders.put(CONTENT_DISPOSITION, Collections.singletonList(String.format(CONTENT_DISPOSITION_FORMAT,
disposition,
encodeText(fileName))));
String fallbackAsciiName = createFallbackAsciiName(this.fileName);
String encodedUtf8Name = createEncodedUtf8Name(this.fileName);

String headerValue = String.format(
"%s; filename=\"%s\"; filename*=UTF-8''%s",
disposition,
fallbackAsciiName,
encodedUtf8Name
);
httpHeaders.put(CONTENT_DISPOSITION, Collections.singletonList(headerValue));
log.debug("Content-Disposition : {}", disposition);

// Content phase
Expand Down Expand Up @@ -260,4 +269,38 @@ private static boolean matches(String matchHeader, String toMatch) {
return Arrays.binarySearch(matchValues, toMatch) > -1 || Arrays.binarySearch(matchValues, "*") > -1;
}

/**
* Creates a safe ASCII-only fallback filename by removing diacritics (accents)
* and replacing any remaining non-ASCII characters.
* E.g., "ä-ö-é.pdf" becomes "a-o-e.pdf".
* @param originalFilename The original filename.
* @return A string containing only ASCII characters.
*/
private String createFallbackAsciiName(String originalFilename) {
if (originalFilename == null) {
return "";
}
String normalized = Normalizer.normalize(originalFilename, Normalizer.Form.NFD);
String withoutAccents = normalized.replaceAll("\\p{InCombiningDiacriticalMarks}+", "");
// Escape \ and ": the value is placed in a quoted-string, so a filename containing a quote
// would close it early and break the header. Mirrors MetadataBitstreamController (bug #1267).
return withoutAccents.replaceAll("[^\\x00-\\x7F]", "")
.replace("\\", "\\\\")
.replace("\"", "\\\"");
}

/**
* Creates a percent-encoded UTF-8 filename according to RFC 5987.
* This is for the `filename*` parameter.
* E.g., "ä ö é.pdf" becomes "%C3%A4%20%C3%B6%20%C3%A9.pdf".
* @param originalFilename The original filename.
* @return A percent-encoded string.
*/
private String createEncodedUtf8Name(String originalFilename) {
if (originalFilename == null) {
return "";
}
return URLEncoder.encode(originalFilename, StandardCharsets.UTF_8).replace("+", "%20");
}

}
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,6 @@
package org.dspace.app.rest;

import static java.util.UUID.randomUUID;
import static javax.mail.internet.MimeUtility.encodeText;
import static org.apache.commons.codec.CharEncoding.UTF_8;
import static org.apache.commons.collections.CollectionUtils.isEmpty;
import static org.apache.commons.io.IOUtils.toInputStream;
Expand Down Expand Up @@ -332,7 +331,11 @@ public void testBitstreamName() throws Exception {
//2. A public item with a bitstream

String bitstreamContent = "0123456789";
String bitstreamName = "ภาษาไทย";
String bitstreamName = "ภาษาไทย-com-acentuação.pdf";
String expectedAscii = "-com-acentuacao.pdf";
String expectedUtf8Encoded =
"%E0%B8%A0%E0%B8%B2%E0%B8%A9%E0%B8%B2%E0%B9%84%E0%B8%97%E0%B8%A2-"
+ "com-acentua%C3%A7%C3%A3o.pdf";

try (InputStream is = IOUtils.toInputStream(bitstreamContent, CharEncoding.UTF_8)) {

Expand All @@ -356,7 +359,9 @@ public void testBitstreamName() throws Exception {
//We expect the content disposition to have the encoded bitstream name
.andExpect(header().string(
"Content-Disposition",
"attachment;filename=\"" + encodeText(bitstreamName) + "\""
String.format("attachment; filename=\"%s\"; filename*=UTF-8''%s",
expectedAscii,
expectedUtf8Encoded)
));
}

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@

import static org.springframework.test.web.servlet.request.MockMvcRequestBuilders.get;
import static org.springframework.test.web.servlet.result.MockMvcResultMatchers.content;
import static org.springframework.test.web.servlet.result.MockMvcResultMatchers.header;
import static org.springframework.test.web.servlet.result.MockMvcResultMatchers.status;

import java.io.ByteArrayOutputStream;
Expand Down Expand Up @@ -97,4 +98,62 @@ public void downloadAllZip() throws Exception {
.andExpect(content().bytes(byteArrayOutputStream.toByteArray()));

}

@Test
public void downloadAllZipWithDoubleQuotesInItemName() throws Exception {
context.turnOffAuthorisationSystem();

// Double quotes in the name used to close the header's quoted-string early, which browsers
// reported as ERR_RESPONSE_HEADERS_MULTIPLE_CONTENT_DISPOSITION.
Item itemWithQuotes = ItemBuilder.createItem(context, col)
.withTitle("Supported data for manuscript \"Thermally-induced evolution\"")
.withAuthor(AUTHOR)
.build();

try (InputStream is = IOUtils.toInputStream("QuotedItemContent", CharEncoding.UTF_8)) {
BitstreamBuilder.createBitstream(context, itemWithQuotes, is)
.withName("data.csv")
.withMimeType("text/csv")
.build();
}
context.restoreAuthSystemState();

String token = getAuthToken(admin.getEmail(), password);
getClient(token).perform(get(METADATABITSTREAM_ENDPOINT + "/" + itemWithQuotes.getID() +
"/" + ALL_ZIP_PATH).param(HANDLE_PARAM, itemWithQuotes.getHandle()))
.andExpect(status().isOk())
.andExpect(header().string("Content-Disposition",
"attachment; filename=\"Supported data for manuscript"
+ " \\\"Thermally-induced evolution\\\".zip\";"
+ " filename*=UTF-8''Supported%20data%20for%20manuscript"
+ "%20%22Thermally-induced%20evolution%22.zip"));
}

@Test
public void downloadAllZipWithNonAsciiItemName() throws Exception {
context.turnOffAuthorisationSystem();

Item itemWithDiacritics = ItemBuilder.createItem(context, col)
.withTitle("Příliš žluťoučký kůň")
.withAuthor(AUTHOR)
.build();

try (InputStream is = IOUtils.toInputStream("DiacriticsContent", CharEncoding.UTF_8)) {
BitstreamBuilder.createBitstream(context, itemWithDiacritics, is)
.withName("file.txt")
.withMimeType("text/plain")
.build();
}
context.restoreAuthSystemState();

String token = getAuthToken(admin.getEmail(), password);
getClient(token).perform(get(METADATABITSTREAM_ENDPOINT + "/" + itemWithDiacritics.getID() +
"/" + ALL_ZIP_PATH).param(HANDLE_PARAM, itemWithDiacritics.getHandle()))
.andExpect(status().isOk())
// fallback transliterates the diacritics away; filename* carries the real name
.andExpect(header().string("Content-Disposition",
"attachment; filename=\"Prilis zlutoucky kun.zip\";"
+ " filename*=UTF-8''P%C5%99%C3%ADli%C5%A1%20%C5%BElu%C5%A5ou%C4%8Dk%C3%BD"
+ "%20k%C5%AF%C5%88.zip"));
}
}
Loading